From 350684cd6b29f01681b9d0fa849863ba4d470fe6 Mon Sep 17 00:00:00 2001 From: Trenton H <797416+stumpylog@users.noreply.github.com> Date: Wed, 29 Jul 2026 12:41:09 -0700 Subject: [PATCH] Fix: consolidate born-digital PDF detection between archive decision and OCR (#13409) * Fix: unify born-digital PDF detection between archive decision and OCR should_produce_archive() and RasterisedDocumentParser.parse() each reimplemented the "does this PDF have real text" check independently, using different normalization of pdftotext output. Raw pdftotext output can be non-empty (whitespace/form-feed layout padding) even when there is no real content, so the two checks could disagree: consumer.py treated a tagged-but-textless PDF as born-digital and skipped the archive, while the parser's own (stricter, normalized) check found no text and ran OCR anyway, leaving the document with no archive despite real OCR text (GH #13387). Both call sites now share one predicate, pdf_born_digital_text() in paperless/parsers/utils.py, so they can no longer drift apart. * Fix: restore extract_text seam for born-digital detection in parse() parse() had switched to calling pdf_born_digital_text() directly for its initial text/born-digital check, bypassing the parser's own extract_text instance method. That broke test mockability (tests patch tesseract_parser.extract_text to control the born-digital decision) and caused CI failures with mismatched OCR call counts and text. Split pdf_born_digital_text() into is_born_digital_text(text, path, log) - a pure decision function - and a thin pdf_born_digital_text() wrapper for callers without text in hand (consumer.should_produce_archive). parse() now extracts via self.extract_text(None, document_path) and passes the result to is_born_digital_text(), restoring the seam with no change to production behavior. Co-Authored-By: Claude Sonnet 5 * Cleanup: simplify born-digital detection, close #13387 test gap Simplification pass over the born-digital detection consolidation: - is_born_digital_text(): drop the has_text temp for an early return. - consumer.py: standardize the archive-decision log lines on plain hyphens (was a mix of em-dash and hyphen) and hoist the duplicated text_length computation. - Parametrize TestPdfBornDigitalText instead of four near-identical tests. Code review follow-up: the existing tests only ever exercised pdf_born_digital_text() through mocks, so the actual #13387 scenario (a tagged PDF whose only "text" is layout padding) was never checked against real pdftotext/pikepdf output - a regression in the normalize-before-decide logic itself would have gone undetected. Moved tagged_no_text_pdf_file from parsers/conftest.py up to the shared paperless/tests/conftest.py (it was previously only visible to tests under parsers/) and added a non-mocked regression test against the real sample file. Also fixed two stale comments in test_consumer.py referencing a _extract_text_for_archive_check helper that no longer exists. Co-Authored-By: Claude Sonnet 5 --------- Co-authored-by: Claude Sonnet 5 --- src/documents/consumer.py | 40 +++---- src/documents/tests/test_consumer.py | 4 +- src/documents/tests/test_consumer_archive.py | 54 +++------ src/paperless/parsers/tesseract.py | 27 +---- src/paperless/parsers/utils.py | 82 +++++++++++++ src/paperless/tests/conftest.py | 17 +++ .../tests/parsers/test_tesseract_parser.py | 46 ++------ .../samples/tesseract/tagged-but-no-text.pdf | Bin 0 -> 5506 bytes src/paperless/tests/test_parser_utils.py | 110 ++++++++++++++++++ 9 files changed, 253 insertions(+), 127 deletions(-) create mode 100644 src/paperless/tests/samples/tesseract/tagged-but-no-text.pdf diff --git a/src/documents/consumer.py b/src/documents/consumer.py index e7f026e719..43185c19a1 100644 --- a/src/documents/consumer.py +++ b/src/documents/consumer.py @@ -57,9 +57,7 @@ from paperless.models import ArchiveFileGenerationChoices from paperless.parsers import ParserContext from paperless.parsers import ParserProtocol from paperless.parsers.registry import get_parser_registry -from paperless.parsers.utils import PDF_TEXT_MIN_LENGTH -from paperless.parsers.utils import extract_pdf_text -from paperless.parsers.utils import is_tagged_pdf +from paperless.parsers.utils import pdf_born_digital_text LOGGING_NAME: Final[str] = "paperless.consumer" @@ -138,53 +136,45 @@ def should_produce_archive( # Must produce a PDF so the frontend can display the original format at all. if parser.requires_pdf_rendition: - _log.debug("Archive: yes — parser requires PDF rendition for frontend display") + _log.debug("Archive: yes - parser requires PDF rendition for frontend display") return True # Parser cannot produce an archive (e.g. TextDocumentParser). if not parser.can_produce_archive: - _log.debug("Archive: no — parser cannot produce archives") + _log.debug("Archive: no - parser cannot produce archives") return False generation = OcrConfig().archive_file_generation if generation == ArchiveFileGenerationChoices.ALWAYS: - _log.debug("Archive: yes — ARCHIVE_FILE_GENERATION=always") + _log.debug("Archive: yes - ARCHIVE_FILE_GENERATION=always") return True if generation == ArchiveFileGenerationChoices.NEVER: - _log.debug("Archive: no — ARCHIVE_FILE_GENERATION=never") + _log.debug("Archive: no - ARCHIVE_FILE_GENERATION=never") return False # auto: produce archives for scanned/image documents; skip for born-digital PDFs. if mime_type.startswith("image/"): - _log.debug("Archive: yes — image document, ARCHIVE_FILE_GENERATION=auto") + _log.debug("Archive: yes - image document, ARCHIVE_FILE_GENERATION=auto") return True if mime_type == "application/pdf": - text = extract_pdf_text(document_path) - has_text = text is not None and len(text) > 0 - if has_text and is_tagged_pdf(document_path): + text, born_digital = pdf_born_digital_text(document_path, log=_log) + text_length = len(text) if text else 0 + if born_digital: _log.debug( - "Archive: no — born-digital PDF (structure tags detected)," + "Archive: no - born-digital PDF (text_length=%d)," " ARCHIVE_FILE_GENERATION=auto", + text_length, ) return False - if text is None or len(text) <= PDF_TEXT_MIN_LENGTH: - _log.debug( - "Archive: yes — scanned PDF (text_length=%d ≤ %d)," - " ARCHIVE_FILE_GENERATION=auto", - len(text) if text else 0, - PDF_TEXT_MIN_LENGTH, - ) - return True _log.debug( - "Archive: no — born-digital PDF (text_length=%d > %d)," + "Archive: yes - scanned/textless PDF (text_length=%d)," " ARCHIVE_FILE_GENERATION=auto", - len(text), - PDF_TEXT_MIN_LENGTH, + text_length, ) - return False + return True _log.debug( - "Archive: no — MIME type %r not eligible for auto archive generation", + "Archive: no - MIME type %r not eligible for auto archive generation", mime_type, ) return False diff --git a/src/documents/tests/test_consumer.py b/src/documents/tests/test_consumer.py index fccc736f65..e5b988e1d1 100644 --- a/src/documents/tests/test_consumer.py +++ b/src/documents/tests/test_consumer.py @@ -1329,7 +1329,7 @@ class PreConsumeTestCase(DirectoriesMixin, GetConsumerMixin, TestCase): with self.get_consumer(self.test_file) as c: c.run() # Verify no pre-consume script subprocess was invoked - # (run_subprocess may still be called by _extract_text_for_archive_check) + # (run_subprocess may still be called by pdf_born_digital_text via pdftotext) script_calls = [ call for call in m.call_args_list @@ -1354,7 +1354,7 @@ class PreConsumeTestCase(DirectoriesMixin, GetConsumerMixin, TestCase): self.assertTrue(m.called) # Find the call that invoked the pre-consume script - # (run_subprocess may also be called by _extract_text_for_archive_check) + # (run_subprocess may also be called by pdf_born_digital_text via pdftotext) script_call = next( call for call in m.call_args_list diff --git a/src/documents/tests/test_consumer_archive.py b/src/documents/tests/test_consumer_archive.py index c179162db7..59de1ce42b 100644 --- a/src/documents/tests/test_consumer_archive.py +++ b/src/documents/tests/test_consumer_archive.py @@ -134,60 +134,32 @@ class TestShouldProduceArchive: assert should_produce_archive(parser, mime, Path("/tmp/doc")) is expected @pytest.mark.parametrize( - ("extracted_text", "expected"), + ("born_digital", "expected"), [ - pytest.param( - "This is a born-digital PDF with lots of text content. " * 10, - False, - id="born-digital-long-text-skips-archive", - ), - pytest.param(None, True, id="no-text-scanned-produces-archive"), - pytest.param("tiny", True, id="short-text-treated-as-scanned"), + pytest.param(True, False, id="born-digital-skips-archive"), + pytest.param(False, True, id="not-born-digital-produces-archive"), ], ) def test_auto_pdf_archive_decision( self, mocker: MockerFixture, settings, - extracted_text: str | None, + born_digital: bool, # noqa: FBT001 expected: bool, # noqa: FBT001 ) -> None: - settings.ARCHIVE_FILE_GENERATION = "auto" - mocker.patch("documents.consumer.is_tagged_pdf", return_value=False) - mocker.patch("documents.consumer.extract_pdf_text", return_value=extracted_text) - parser = _parser_instance(can_produce=True, requires_rendition=False) - assert ( - should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf")) - is expected - ) + """Archive decision tracks pdf_born_digital_text()'s verdict exactly. - def test_tagged_pdf_skips_archive_in_auto_mode( - self, - mocker: MockerFixture, - settings, - ) -> None: - """Tagged PDFs (e.g. Word exports) with real text are treated as born-digital, even below PDF_TEXT_MIN_LENGTH.""" + should_produce_archive() defers entirely to pdf_born_digital_text() + for the has-real-text decision, so both callers of that predicate + (this function and RasterisedDocumentParser.parse()) always agree. + """ settings.ARCHIVE_FILE_GENERATION = "auto" - mocker.patch("documents.consumer.is_tagged_pdf", return_value=True) - mocker.patch("documents.consumer.extract_pdf_text", return_value="tiny") - parser = _parser_instance(can_produce=True, requires_rendition=False) - assert ( - should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf")) - is False + mocker.patch( + "documents.consumer.pdf_born_digital_text", + return_value=("some text", born_digital), ) - - def test_tagged_pdf_without_text_produces_archive( - self, - mocker: MockerFixture, - settings, - ) -> None: - """A tagged PDF with no actual extractable text (e.g. some scanner firmware) is not - trusted as born-digital — the tag alone must not bypass OCR.""" - settings.ARCHIVE_FILE_GENERATION = "auto" - mocker.patch("documents.consumer.is_tagged_pdf", return_value=True) - mocker.patch("documents.consumer.extract_pdf_text", return_value=None) parser = _parser_instance(can_produce=True, requires_rendition=False) assert ( should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf")) - is True + is expected ) diff --git a/src/paperless/parsers/tesseract.py b/src/paperless/parsers/tesseract.py index 78ce4c5c40..f3ebf9925f 100644 --- a/src/paperless/parsers/tesseract.py +++ b/src/paperless/parsers/tesseract.py @@ -3,7 +3,6 @@ from __future__ import annotations import importlib.resources import logging import os -import re import shutil import tempfile from pathlib import Path @@ -25,9 +24,9 @@ from paperless.config import OcrConfig from paperless.models import CleanChoices from paperless.models import ModeChoices from paperless.models import OutputTypeChoices -from paperless.parsers.utils import PDF_TEXT_MIN_LENGTH from paperless.parsers.utils import extract_pdf_text -from paperless.parsers.utils import is_tagged_pdf +from paperless.parsers.utils import is_born_digital_text +from paperless.parsers.utils import post_process_text from paperless.parsers.utils import read_file_handle_unicode_errors from paperless.version import __full_version_str__ @@ -510,10 +509,10 @@ class RasterisedDocumentParser: if mime_type == "application/pdf": text_original = self.extract_text(None, document_path) - has_text = text_original is not None and len(text_original) > 0 - original_has_text = has_text and ( - is_tagged_pdf(document_path, log=self.log) - or len(text_original) > PDF_TEXT_MIN_LENGTH + original_has_text = is_born_digital_text( + text_original, + document_path, + log=self.log, ) else: text_original = None @@ -658,17 +657,3 @@ class RasterisedDocumentParser: f"No text was found in {document_path}, the content will be empty.", ) self.text = "" - - -def post_process_text(text: str | None) -> str | None: - if not text: - return None - - collapsed_spaces = re.sub(r"([^\S\r\n]+)", " ", text) - no_leading_whitespace = re.sub(r"([\n\r]+)([^\S\n\r]+)", "\\1", collapsed_spaces) - no_trailing_whitespace = re.sub(r"([^\S\n\r]+)$", "", no_leading_whitespace) - - # TODO: this needs a rework - # replace \0 prevents issues with saving to postgres. - # text may contain \0 when this character is present in PDF files. - return no_trailing_whitespace.strip().replace("\0", " ") diff --git a/src/paperless/parsers/utils.py b/src/paperless/parsers/utils.py index 0257ab7366..9fe1d49081 100644 --- a/src/paperless/parsers/utils.py +++ b/src/paperless/parsers/utils.py @@ -111,6 +111,88 @@ def extract_pdf_text( return None +def post_process_text(text: str | None) -> str | None: + """Normalize extracted PDF/OCR text: collapse whitespace, strip padding. + + Returns ``None`` for ``None`` or whitespace-only input, so callers can + treat "no text" and "only layout padding" the same way. + """ + if not text: + return None + + collapsed_spaces = re.sub(r"([^\S\r\n]+)", " ", text) + no_leading_whitespace = re.sub(r"([\n\r]+)([^\S\n\r]+)", "\\1", collapsed_spaces) + no_trailing_whitespace = re.sub(r"([^\S\n\r]+)$", "", no_leading_whitespace) + + # replace \0 prevents issues with saving to postgres. + # text may contain \0 when this character is present in PDF files. + result = no_trailing_whitespace.strip().replace("\0", " ") + return result or None + + +def is_born_digital_text( + text: str | None, + path: Path, + log: logging.Logger | None = None, +) -> bool: + """Decide whether already-extracted, normalized PDF text counts as born-digital. + + This is the single source of truth for "does this PDF already have real + text", used both to decide whether to produce an archive file and to + decide whether OCR can be skipped. Both decisions must agree, or a + tagged-but-textless PDF can end up with no archive AND a forced OCR pass + (see GH #13387): raw ``pdftotext -layout`` output can be non-empty + (whitespace/form-feed padding) even when there is no real content, so + *text* must already be normalized via :func:`post_process_text`, not the + raw extraction. + + Parameters + ---------- + text: + The normalized extracted text (or ``None``) to evaluate. + path: + Absolute path to the PDF file, used for the tagged-PDF check. + log: + Logger for warnings. Falls back to the module-level logger when omitted. + + Returns + ------- + bool + Whether the PDF counts as born-digital (has real text, and is either + tagged or exceeds ``PDF_TEXT_MIN_LENGTH``). + """ + if not text: + return False + return is_tagged_pdf(path, log=log) or len(text) > PDF_TEXT_MIN_LENGTH + + +def pdf_born_digital_text( + path: Path, + log: logging.Logger | None = None, +) -> tuple[str | None, bool]: + """Extract a PDF's text and decide whether it should be treated as born-digital. + + Convenience wrapper around :func:`is_born_digital_text` for callers that + don't already have the PDF's text extracted (e.g. the archive-generation + decision, which runs before any parser has touched the file). + + Parameters + ---------- + path: + Absolute path to the PDF file. + log: + Logger for warnings. Falls back to the module-level logger when omitted. + + Returns + ------- + tuple[str | None, bool] + The normalized extracted text (or ``None``), and whether the PDF + counts as born-digital. + """ + text = post_process_text(extract_pdf_text(path, log=log)) + return text, is_born_digital_text(text, path, log=log) + + def read_file_handle_unicode_errors( filepath: Path, log: logging.Logger | None = None, diff --git a/src/paperless/tests/conftest.py b/src/paperless/tests/conftest.py index b016191c45..c436595062 100644 --- a/src/paperless/tests/conftest.py +++ b/src/paperless/tests/conftest.py @@ -36,6 +36,23 @@ def samples_dir() -> Path: return (Path(__file__).parent / "samples").resolve() +@pytest.fixture(scope="session") +def tagged_no_text_pdf_file(samples_dir: Path) -> Path: + """Path to a tagged PDF whose only "text" is pdftotext layout padding. + + Reproduces GH #13387: ``/MarkInfo /Marked true`` is set, but the only + extractable content is a form-feed byte, not real text. Lives here + rather than in parsers/conftest.py so both parser tests and + paperless/tests/test_parser_utils.py can use it. + + Returns + ------- + Path + Absolute path to ``tesseract/tagged-but-no-text.pdf``. + """ + return samples_dir / "tesseract" / "tagged-but-no-text.pdf" + + @pytest.fixture(autouse=True) def clean_registry() -> Generator[None, None, None]: """Reset the parser registry before and after every test. diff --git a/src/paperless/tests/parsers/test_tesseract_parser.py b/src/paperless/tests/parsers/test_tesseract_parser.py index e56992d0bc..f25efccc73 100644 --- a/src/paperless/tests/parsers/test_tesseract_parser.py +++ b/src/paperless/tests/parsers/test_tesseract_parser.py @@ -21,7 +21,7 @@ from documents.parsers import run_convert from paperless.models import ModeChoices from paperless.parsers import ParserProtocol from paperless.parsers.tesseract import RasterisedDocumentParser -from paperless.parsers.tesseract import post_process_text +from paperless.parsers.utils import is_tagged_pdf if TYPE_CHECKING: from pathlib import Path @@ -151,36 +151,6 @@ class TestRasterisedDocumentParserLifecycle: assert tempdir is not None and not tempdir.exists() -# --------------------------------------------------------------------------- -# post_process_text -# --------------------------------------------------------------------------- - - -class TestPostProcessText: - @pytest.mark.parametrize( - ("source", "expected"), - [ - pytest.param( - "simple string", - "simple string", - id="collapse-spaces", - ), - pytest.param( - "simple newline\n testing string", - "simple newline\ntesting string", - id="preserve-newline", - ), - pytest.param( - "utf-8 строка с пробелами в конце ", # noqa: RUF001 - "utf-8 строка с пробелами в конце", # noqa: RUF001 - id="utf8-trailing-spaces", - ), - ], - ) - def test_post_process_text(self, source: str, expected: str) -> None: - assert post_process_text(source) == expected - - # --------------------------------------------------------------------------- # Page count # --------------------------------------------------------------------------- @@ -910,25 +880,25 @@ class TestSkipArchive: self, mocker: MockerFixture, tesseract_parser: RasterisedDocumentParser, - tesseract_samples_dir: Path, + tagged_no_text_pdf_file: Path, ) -> None: """ GIVEN: - - A PDF that reports itself as tagged (/MarkInfo /Marked true) but - has no actual extractable text (some scanner firmware produces - this — see GitHub issue #13349) + - A real PDF that reports itself as tagged (/MarkInfo /Marked + true) but whose only pdftotext output is layout padding (a + lone form-feed byte), not real text (see GitHub issue #13387, + originally reported against #13349's tagged-PDF handling) - Mode: auto, produce_archive=False WHEN: - Document is parsed THEN: - The tag alone is not trusted as "has text"; OCRmyPDF still runs """ + assert is_tagged_pdf(tagged_no_text_pdf_file) is True tesseract_parser.settings.mode = ModeChoices.AUTO - mocker.patch("paperless.parsers.tesseract.is_tagged_pdf", return_value=True) - mocker.patch.object(tesseract_parser, "extract_text", return_value=None) mock_ocr = mocker.patch("ocrmypdf.ocr") tesseract_parser.parse( - tesseract_samples_dir / "multi-page-images.pdf", + tagged_no_text_pdf_file, "application/pdf", produce_archive=False, ) diff --git a/src/paperless/tests/samples/tesseract/tagged-but-no-text.pdf b/src/paperless/tests/samples/tesseract/tagged-but-no-text.pdf new file mode 100644 index 0000000000000000000000000000000000000000..5a652345a1c9e1fbe0ab1a90d183194ad344d49c GIT binary patch literal 5506 zc-qZa2~-o;8U|~{rcy0^b%9m}EH2PwCV_+`#!x^|kWKbQ@Wo^TAt*q@ZAsA|Pr-lma3(D6$C(Dk3VPpn`7_kVf#liqD?cKPTth`{(W8p|s zg1(4_ISSb%RjIrk9dMunEDDqc0p<`*X)lqYFiVJIScHjLY(6XnHo<{B(JI6K<0lLO z9%q%I7ugYV6kEeVytNT1>=xnV&W_--X&gffCXRt}xOCUmwm_IrAd=9r*($>zsZ>k{ z!O+l9LZ~SLK?6Y|jYb0@5=bK9F$7)`E|Rjs@FK|)Lmcoz#vFtR#S;!h5&{cb8cbj# zLNJUa28jd+G-P5Xs;H}B2^-~!r94CgDCewANQiWmp_)rzl5;VK%Tg{#EkPEKCsd`V z{0)l;@;Kms@$h=-VF`|fqKjFmlu2ZO>eH71Yy{hTRLBysVW#B-PF*_&#{$!>v3-&v zs5^oPm~K3wSO5cVY?cThLj;HcPDV_!gI8l;CJ7=@@Q^v4WbRHP(jf|+NHu{-bV%KP z4Ecw-Bf{Zv!(Yc-4XI{6UWBjZKpm^xXK>O+V+99}R~%EVVlfJ=G7MGLBQ_eZ=NGgv zok9y0e{kSB9!CQB0c0hQ0Dl~4jf9A#08yP%~ zNo72c8oBL+s_eXY9E{y0Gb)4w*TTHOAS_cdL{)khAgCMm%rNX~cngmWuR&SiIM9kG zmAJsDH6j!vBIUwJ6K8}eg^C&4@&r;C1wdPYvOHim!hzLJj@MpDC0gn1CSLJdL)+HI@(U+ESr7o>s8)@!1Py zcysbW)ZgQmw5&*v-U>>3)V%yyhy3Ty?vy_qu^K5UEjl`TEx*3eJ8i>_2Y0vn^ycb& zJ&sS5XBQ-$>CmdEr8rs#w{Cr$-{iBk3=fj7X&*Za@=bl~4EiZ=4*Zz`w zv6{M@#vg!W){(7T{7#def|YSL$j1Smt=@@Aa?MnlAm*$M@u1trye$jjzbDi+%aWEI zzh4V!9bvo=>wNf87xn7YxT3=J+ky4TMWw`AlW2Xlz ztRjwoaHDXcd-M|-Gm)IXU!L2(Cuf1ry3CY~zm4SF8tJ}PmSDJV@!X49HG^8$&U^MX zX9&2#IjzT*uv28U?ZY^~RC&OLtfsG*h39fBs>n~8j}l_Kf?et&Zrr+Squ9Hk|C19I zW!1h%91K?WRqN{MopAc*P;d0L0|9Y0SB@cdigyA(bvl19Y+J(*ZR-k>2ik@=omj-k ztD{D3?PetQ);HyfioCAI;LFhqqJlHj70nfjxT1TG)SI;_!WPqQakbBSg^CAs>1f7# z-iuOK&ii^^bFt@9&9t87qe~ff3kLUOmi6yKP949=AGACnqL3FC_hcd|mrliBt$GBM zCCKZ_>k_*6St>4v%@Jn%=WQyj z)jrw$sCS>U)xg2#pL$cd4B?~S2~`(D6rH^g6oal1Rmz-^Bc z+C}%(ZcJ@l8rMVIBYFSxtA4`)S+Nlw1rOP_g0c^UPZYcWy`KWlLDB(mg+Qq2`7oBd;4|lrT<*D!#q0C$ zJj;-@+-Z5lZum;8_4fkBn!7B<#sk{2iz!95zSI?s9s`97n{4(6=*P_g2o4J--uILz!gpIaA|vJJm@ zaDM%%*z)qIkI|^xYYftM95{y0pPil_rB`~b?3eelhgbaEeI;YzLBNZA?(uwZ?~~Y~ zbLU2H4tHJ;Zwpwd8`l|xep^HtuDN$Fe4t~*7YWuS6z)YGYZsGy*KP_tSRoJG_Zz`8t+$S9e%$0XPeD~E@9g8`CT~fz>^qfv>)3GEkX}j{J?1!=X ze+av=BunS8QJ{Tga%91@i{u{9R~bCA;muw<+v9Jx&-(P~(pNP1)G1x@%U5RV9kjU` zcD%baGgE70H{-_Iaiyrb^Rc#kV#*fcR1N)V1h@VbshFv`AHRnImQTK>S)ra#7z%T7 z5J1C069Q1kWK%N01txSvGBEawXhKIcB~8)MNT!o@Wa3mEM59j85s9QJ{>TuCI$38% zqniDp!%0yVPXME;^}6vQU~CLSx@{8mJHV@Vtv+pPxt7!-)gB5~MM5{*bUHD^N< p7&fPIspc@roXXss>HqN%V{{UHdVjKVf literal 0 Hc-jL100001 diff --git a/src/paperless/tests/test_parser_utils.py b/src/paperless/tests/test_parser_utils.py index c6bb3e34a2..f0a63ea44d 100644 --- a/src/paperless/tests/test_parser_utils.py +++ b/src/paperless/tests/test_parser_utils.py @@ -4,10 +4,18 @@ from __future__ import annotations import codecs from pathlib import Path +from typing import TYPE_CHECKING + +import pytest from paperless.parsers.utils import is_tagged_pdf +from paperless.parsers.utils import pdf_born_digital_text +from paperless.parsers.utils import post_process_text from paperless.parsers.utils import read_file_handle_unicode_errors +if TYPE_CHECKING: + from pytest_mock import MockerFixture + SAMPLES = Path(__file__).parent / "samples" / "tesseract" @@ -60,3 +68,105 @@ class TestIsTaggedPdf: bad = tmp_path / "bad.pdf" bad.write_bytes(b"not a pdf") assert is_tagged_pdf(bad) is False + + +class TestPostProcessText: + @pytest.mark.parametrize( + ("source", "expected"), + [ + pytest.param( + "simple string", + "simple string", + id="collapse-spaces", + ), + pytest.param( + "simple newline\n testing string", + "simple newline\ntesting string", + id="preserve-newline", + ), + pytest.param( + "utf-8 строка с пробелами в конце ", # noqa: RUF001 + "utf-8 строка с пробелами в конце", # noqa: RUF001 + id="utf8-trailing-spaces", + ), + pytest.param(None, None, id="none-input"), + pytest.param("", None, id="empty-string"), + pytest.param(" \n\x0c \n ", None, id="whitespace-and-formfeed-only"), + ], + ) + def test_post_process_text( + self, + source: str | None, + expected: str | None, + ) -> None: + assert post_process_text(source) == expected + + +class TestPdfBornDigitalText: + """Regression coverage for GH #13387. + + should_produce_archive() and RasterisedDocumentParser.parse() must agree + on whether a PDF has real text, so both go through this one function. + """ + + @pytest.mark.parametrize( + ("extracted", "tagged", "expected_text", "expected_born_digital"), + [ + pytest.param("tiny", True, "tiny", True, id="tagged-with-real-text"), + pytest.param("tiny", False, "tiny", False, id="untagged-below-min-length"), + pytest.param( + "x" * 51, + False, + "x" * 51, + True, + id="untagged-above-min-length", + ), + pytest.param(None, True, None, False, id="tagged-but-no-text"), + ], + ) + def test_born_digital_decision( + self, + mocker: MockerFixture, + tmp_path: Path, + extracted: str | None, + tagged: bool, # noqa: FBT001 + expected_text: str | None, + expected_born_digital: bool, # noqa: FBT001 + ) -> None: + """ + GIVEN: + - A PDF whose pdftotext output and /MarkInfo tag status vary + WHEN: + - pdf_born_digital_text() is called + THEN: + - The normalized text and born-digital verdict match; the tag + alone never counts as "has text" + """ + mocker.patch( + "paperless.parsers.utils.extract_pdf_text", + return_value=extracted, + ) + mocker.patch("paperless.parsers.utils.is_tagged_pdf", return_value=tagged) + text, born_digital = pdf_born_digital_text(tmp_path / "doc.pdf") + assert text == expected_text + assert born_digital is expected_born_digital + + def test_tagged_but_textless_pdf_is_not_born_digital( + self, + tagged_no_text_pdf_file: Path, + ) -> None: + """ + GIVEN: + - A real PDF that is tagged (/MarkInfo /Marked true) but whose + only "text" is layout padding (a stray form-feed byte) + WHEN: + - pdf_born_digital_text() is called with no mocking + THEN: + - The normalized text is None and the PDF is not treated as + born-digital. The raw, unnormalized pdftotext output is + non-empty for this file, which is exactly what caused the + archive decision to disagree with the OCR decision in #13387. + """ + text, born_digital = pdf_born_digital_text(tagged_no_text_pdf_file) + assert text is None + assert born_digital is False -- 2.47.3