]> git.ipfire.org Git - thirdparty/paperless-ngx.git/commitdiff
Fix: consolidate born-digital PDF detection between archive decision and OCR (#13409) dev
authorTrenton H <797416+stumpylog@users.noreply.github.com>
Wed, 29 Jul 2026 19:41:09 +0000 (12:41 -0700)
committerGitHub <noreply@github.com>
Wed, 29 Jul 2026 19:41:09 +0000 (12:41 -0700)
* Fix: unify born-digital PDF detection between archive decision and OCR

should_produce_archive() and RasterisedDocumentParser.parse() each
reimplemented the "does this PDF have real text" check independently,
using different normalization of pdftotext output. Raw pdftotext output
can be non-empty (whitespace/form-feed layout padding) even when there
is no real content, so the two checks could disagree: consumer.py
treated a tagged-but-textless PDF as born-digital and skipped the
archive, while the parser's own (stricter, normalized) check found no
text and ran OCR anyway, leaving the document with no archive despite
real OCR text (GH #13387).

Both call sites now share one predicate, pdf_born_digital_text() in
paperless/parsers/utils.py, so they can no longer drift apart.

* Fix: restore extract_text seam for born-digital detection in parse()

parse() had switched to calling pdf_born_digital_text() directly for
its initial text/born-digital check, bypassing the parser's own
extract_text instance method. That broke test mockability (tests patch
tesseract_parser.extract_text to control the born-digital decision)
and caused CI failures with mismatched OCR call counts and text.

Split pdf_born_digital_text() into is_born_digital_text(text, path,
log) - a pure decision function - and a thin pdf_born_digital_text()
wrapper for callers without text in hand (consumer.should_produce_archive).
parse() now extracts via self.extract_text(None, document_path) and
passes the result to is_born_digital_text(), restoring the seam with
no change to production behavior.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
* Cleanup: simplify born-digital detection, close #13387 test gap

Simplification pass over the born-digital detection consolidation:
- is_born_digital_text(): drop the has_text temp for an early return.
- consumer.py: standardize the archive-decision log lines on plain
  hyphens (was a mix of em-dash and hyphen) and hoist the duplicated
  text_length computation.
- Parametrize TestPdfBornDigitalText instead of four near-identical
  tests.

Code review follow-up: the existing tests only ever exercised
pdf_born_digital_text() through mocks, so the actual #13387 scenario
(a tagged PDF whose only "text" is layout padding) was never checked
against real pdftotext/pikepdf output - a regression in the
normalize-before-decide logic itself would have gone undetected.
Moved tagged_no_text_pdf_file from parsers/conftest.py up to the
shared paperless/tests/conftest.py (it was previously only visible to
tests under parsers/) and added a non-mocked regression test against
the real sample file.

Also fixed two stale comments in test_consumer.py referencing a
_extract_text_for_archive_check helper that no longer exists.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
---------

Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>
src/documents/consumer.py
src/documents/tests/test_consumer.py
src/documents/tests/test_consumer_archive.py
src/paperless/parsers/tesseract.py
src/paperless/parsers/utils.py
src/paperless/tests/conftest.py
src/paperless/tests/parsers/test_tesseract_parser.py
src/paperless/tests/samples/tesseract/tagged-but-no-text.pdf [new file with mode: 0644]
src/paperless/tests/test_parser_utils.py

index e7f026e719906bd4954557b51b7b5ca3aaca1523..43185c19a17e109e3aeff430c1ddf1088ab66bc8 100644 (file)
@@ -57,9 +57,7 @@ from paperless.models import ArchiveFileGenerationChoices
 from paperless.parsers import ParserContext
 from paperless.parsers import ParserProtocol
 from paperless.parsers.registry import get_parser_registry
-from paperless.parsers.utils import PDF_TEXT_MIN_LENGTH
-from paperless.parsers.utils import extract_pdf_text
-from paperless.parsers.utils import is_tagged_pdf
+from paperless.parsers.utils import pdf_born_digital_text
 
 LOGGING_NAME: Final[str] = "paperless.consumer"
 
@@ -138,53 +136,45 @@ def should_produce_archive(
 
     # Must produce a PDF so the frontend can display the original format at all.
     if parser.requires_pdf_rendition:
-        _log.debug("Archive: yes  parser requires PDF rendition for frontend display")
+        _log.debug("Archive: yes - parser requires PDF rendition for frontend display")
         return True
 
     # Parser cannot produce an archive (e.g. TextDocumentParser).
     if not parser.can_produce_archive:
-        _log.debug("Archive: no  parser cannot produce archives")
+        _log.debug("Archive: no - parser cannot produce archives")
         return False
 
     generation = OcrConfig().archive_file_generation
 
     if generation == ArchiveFileGenerationChoices.ALWAYS:
-        _log.debug("Archive: yes  ARCHIVE_FILE_GENERATION=always")
+        _log.debug("Archive: yes - ARCHIVE_FILE_GENERATION=always")
         return True
     if generation == ArchiveFileGenerationChoices.NEVER:
-        _log.debug("Archive: no  ARCHIVE_FILE_GENERATION=never")
+        _log.debug("Archive: no - ARCHIVE_FILE_GENERATION=never")
         return False
 
     # auto: produce archives for scanned/image documents; skip for born-digital PDFs.
     if mime_type.startswith("image/"):
-        _log.debug("Archive: yes  image document, ARCHIVE_FILE_GENERATION=auto")
+        _log.debug("Archive: yes - image document, ARCHIVE_FILE_GENERATION=auto")
         return True
     if mime_type == "application/pdf":
-        text = extract_pdf_text(document_path)
-        has_text = text is not None and len(text) > 0
-        if has_text and is_tagged_pdf(document_path):
+        text, born_digital = pdf_born_digital_text(document_path, log=_log)
+        text_length = len(text) if text else 0
+        if born_digital:
             _log.debug(
-                "Archive: no — born-digital PDF (structure tags detected),"
+                "Archive: no - born-digital PDF (text_length=%d),"
                 " ARCHIVE_FILE_GENERATION=auto",
+                text_length,
             )
             return False
-        if text is None or len(text) <= PDF_TEXT_MIN_LENGTH:
-            _log.debug(
-                "Archive: yes — scanned PDF (text_length=%d ≤ %d),"
-                " ARCHIVE_FILE_GENERATION=auto",
-                len(text) if text else 0,
-                PDF_TEXT_MIN_LENGTH,
-            )
-            return True
         _log.debug(
-            "Archive: no — born-digital PDF (text_length=%d > %d),"
+            "Archive: yes - scanned/textless PDF (text_length=%d),"
             " ARCHIVE_FILE_GENERATION=auto",
-            len(text),
-            PDF_TEXT_MIN_LENGTH,
+            text_length,
         )
-        return False
+        return True
     _log.debug(
-        "Archive: no  MIME type %r not eligible for auto archive generation",
+        "Archive: no - MIME type %r not eligible for auto archive generation",
         mime_type,
     )
     return False
index fccc736f652a59dcb969f4f1c4728e844e55eaf5..e5b988e1d1b2f151286e915d06634004218ccdcd 100644 (file)
@@ -1329,7 +1329,7 @@ class PreConsumeTestCase(DirectoriesMixin, GetConsumerMixin, TestCase):
         with self.get_consumer(self.test_file) as c:
             c.run()
             # Verify no pre-consume script subprocess was invoked
-            # (run_subprocess may still be called by _extract_text_for_archive_check)
+            # (run_subprocess may still be called by pdf_born_digital_text via pdftotext)
             script_calls = [
                 call
                 for call in m.call_args_list
@@ -1354,7 +1354,7 @@ class PreConsumeTestCase(DirectoriesMixin, GetConsumerMixin, TestCase):
                     self.assertTrue(m.called)
 
                     # Find the call that invoked the pre-consume script
-                    # (run_subprocess may also be called by _extract_text_for_archive_check)
+                    # (run_subprocess may also be called by pdf_born_digital_text via pdftotext)
                     script_call = next(
                         call
                         for call in m.call_args_list
index c179162db7abb87a8b1cfb8edd37293c44e3bd35..59de1ce42bb01b87b616744491164eeecadbe6c2 100644 (file)
@@ -134,60 +134,32 @@ class TestShouldProduceArchive:
         assert should_produce_archive(parser, mime, Path("/tmp/doc")) is expected
 
     @pytest.mark.parametrize(
-        ("extracted_text", "expected"),
+        ("born_digital", "expected"),
         [
-            pytest.param(
-                "This is a born-digital PDF with lots of text content. " * 10,
-                False,
-                id="born-digital-long-text-skips-archive",
-            ),
-            pytest.param(None, True, id="no-text-scanned-produces-archive"),
-            pytest.param("tiny", True, id="short-text-treated-as-scanned"),
+            pytest.param(True, False, id="born-digital-skips-archive"),
+            pytest.param(False, True, id="not-born-digital-produces-archive"),
         ],
     )
     def test_auto_pdf_archive_decision(
         self,
         mocker: MockerFixture,
         settings,
-        extracted_text: str | None,
+        born_digital: bool,  # noqa: FBT001
         expected: bool,  # noqa: FBT001
     ) -> None:
-        settings.ARCHIVE_FILE_GENERATION = "auto"
-        mocker.patch("documents.consumer.is_tagged_pdf", return_value=False)
-        mocker.patch("documents.consumer.extract_pdf_text", return_value=extracted_text)
-        parser = _parser_instance(can_produce=True, requires_rendition=False)
-        assert (
-            should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
-            is expected
-        )
+        """Archive decision tracks pdf_born_digital_text()'s verdict exactly.
 
-    def test_tagged_pdf_skips_archive_in_auto_mode(
-        self,
-        mocker: MockerFixture,
-        settings,
-    ) -> None:
-        """Tagged PDFs (e.g. Word exports) with real text are treated as born-digital, even below PDF_TEXT_MIN_LENGTH."""
+        should_produce_archive() defers entirely to pdf_born_digital_text()
+        for the has-real-text decision, so both callers of that predicate
+        (this function and RasterisedDocumentParser.parse()) always agree.
+        """
         settings.ARCHIVE_FILE_GENERATION = "auto"
-        mocker.patch("documents.consumer.is_tagged_pdf", return_value=True)
-        mocker.patch("documents.consumer.extract_pdf_text", return_value="tiny")
-        parser = _parser_instance(can_produce=True, requires_rendition=False)
-        assert (
-            should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
-            is False
+        mocker.patch(
+            "documents.consumer.pdf_born_digital_text",
+            return_value=("some text", born_digital),
         )
-
-    def test_tagged_pdf_without_text_produces_archive(
-        self,
-        mocker: MockerFixture,
-        settings,
-    ) -> None:
-        """A tagged PDF with no actual extractable text (e.g. some scanner firmware) is not
-        trusted as born-digital — the tag alone must not bypass OCR."""
-        settings.ARCHIVE_FILE_GENERATION = "auto"
-        mocker.patch("documents.consumer.is_tagged_pdf", return_value=True)
-        mocker.patch("documents.consumer.extract_pdf_text", return_value=None)
         parser = _parser_instance(can_produce=True, requires_rendition=False)
         assert (
             should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
-            is True
+            is expected
         )
index 78ce4c5c408be7ed953bbac483e76af58226293d..f3ebf9925fd2755eecc2ed1d58e1e761cb45676f 100644 (file)
@@ -3,7 +3,6 @@ from __future__ import annotations
 import importlib.resources
 import logging
 import os
-import re
 import shutil
 import tempfile
 from pathlib import Path
@@ -25,9 +24,9 @@ from paperless.config import OcrConfig
 from paperless.models import CleanChoices
 from paperless.models import ModeChoices
 from paperless.models import OutputTypeChoices
-from paperless.parsers.utils import PDF_TEXT_MIN_LENGTH
 from paperless.parsers.utils import extract_pdf_text
-from paperless.parsers.utils import is_tagged_pdf
+from paperless.parsers.utils import is_born_digital_text
+from paperless.parsers.utils import post_process_text
 from paperless.parsers.utils import read_file_handle_unicode_errors
 from paperless.version import __full_version_str__
 
@@ -510,10 +509,10 @@ class RasterisedDocumentParser:
 
         if mime_type == "application/pdf":
             text_original = self.extract_text(None, document_path)
-            has_text = text_original is not None and len(text_original) > 0
-            original_has_text = has_text and (
-                is_tagged_pdf(document_path, log=self.log)
-                or len(text_original) > PDF_TEXT_MIN_LENGTH
+            original_has_text = is_born_digital_text(
+                text_original,
+                document_path,
+                log=self.log,
             )
         else:
             text_original = None
@@ -658,17 +657,3 @@ class RasterisedDocumentParser:
                     f"No text was found in {document_path}, the content will be empty.",
                 )
                 self.text = ""
-
-
-def post_process_text(text: str | None) -> str | None:
-    if not text:
-        return None
-
-    collapsed_spaces = re.sub(r"([^\S\r\n]+)", " ", text)
-    no_leading_whitespace = re.sub(r"([\n\r]+)([^\S\n\r]+)", "\\1", collapsed_spaces)
-    no_trailing_whitespace = re.sub(r"([^\S\n\r]+)$", "", no_leading_whitespace)
-
-    # TODO: this needs a rework
-    # replace \0 prevents issues with saving to postgres.
-    # text may contain \0 when this character is present in PDF files.
-    return no_trailing_whitespace.strip().replace("\0", " ")
index 0257ab7366e82504524978480bfe496c3ad1deb9..9fe1d490812bd80f95d8f64255f6e016f3f9b90f 100644 (file)
@@ -111,6 +111,88 @@ def extract_pdf_text(
         return None
 
 
+def post_process_text(text: str | None) -> str | None:
+    """Normalize extracted PDF/OCR text: collapse whitespace, strip padding.
+
+    Returns ``None`` for ``None`` or whitespace-only input, so callers can
+    treat "no text" and "only layout padding" the same way.
+    """
+    if not text:
+        return None
+
+    collapsed_spaces = re.sub(r"([^\S\r\n]+)", " ", text)
+    no_leading_whitespace = re.sub(r"([\n\r]+)([^\S\n\r]+)", "\\1", collapsed_spaces)
+    no_trailing_whitespace = re.sub(r"([^\S\n\r]+)$", "", no_leading_whitespace)
+
+    # replace \0 prevents issues with saving to postgres.
+    # text may contain \0 when this character is present in PDF files.
+    result = no_trailing_whitespace.strip().replace("\0", " ")
+    return result or None
+
+
+def is_born_digital_text(
+    text: str | None,
+    path: Path,
+    log: logging.Logger | None = None,
+) -> bool:
+    """Decide whether already-extracted, normalized PDF text counts as born-digital.
+
+    This is the single source of truth for "does this PDF already have real
+    text", used both to decide whether to produce an archive file and to
+    decide whether OCR can be skipped. Both decisions must agree, or a
+    tagged-but-textless PDF can end up with no archive AND a forced OCR pass
+    (see GH #13387): raw ``pdftotext -layout`` output can be non-empty
+    (whitespace/form-feed padding) even when there is no real content, so
+    *text* must already be normalized via :func:`post_process_text`, not the
+    raw extraction.
+
+    Parameters
+    ----------
+    text:
+        The normalized extracted text (or ``None``) to evaluate.
+    path:
+        Absolute path to the PDF file, used for the tagged-PDF check.
+    log:
+        Logger for warnings.  Falls back to the module-level logger when omitted.
+
+    Returns
+    -------
+    bool
+        Whether the PDF counts as born-digital (has real text, and is either
+        tagged or exceeds ``PDF_TEXT_MIN_LENGTH``).
+    """
+    if not text:
+        return False
+    return is_tagged_pdf(path, log=log) or len(text) > PDF_TEXT_MIN_LENGTH
+
+
+def pdf_born_digital_text(
+    path: Path,
+    log: logging.Logger | None = None,
+) -> tuple[str | None, bool]:
+    """Extract a PDF's text and decide whether it should be treated as born-digital.
+
+    Convenience wrapper around :func:`is_born_digital_text` for callers that
+    don't already have the PDF's text extracted (e.g. the archive-generation
+    decision, which runs before any parser has touched the file).
+
+    Parameters
+    ----------
+    path:
+        Absolute path to the PDF file.
+    log:
+        Logger for warnings.  Falls back to the module-level logger when omitted.
+
+    Returns
+    -------
+    tuple[str | None, bool]
+        The normalized extracted text (or ``None``), and whether the PDF
+        counts as born-digital.
+    """
+    text = post_process_text(extract_pdf_text(path, log=log))
+    return text, is_born_digital_text(text, path, log=log)
+
+
 def read_file_handle_unicode_errors(
     filepath: Path,
     log: logging.Logger | None = None,
index b016191c455b3ee84a6d7bceac27178e789c0c29..c436595062140652f5e995c9887d34cb712657c0 100644 (file)
@@ -36,6 +36,23 @@ def samples_dir() -> Path:
     return (Path(__file__).parent / "samples").resolve()
 
 
+@pytest.fixture(scope="session")
+def tagged_no_text_pdf_file(samples_dir: Path) -> Path:
+    """Path to a tagged PDF whose only "text" is pdftotext layout padding.
+
+    Reproduces GH #13387: ``/MarkInfo /Marked true`` is set, but the only
+    extractable content is a form-feed byte, not real text. Lives here
+    rather than in parsers/conftest.py so both parser tests and
+    paperless/tests/test_parser_utils.py can use it.
+
+    Returns
+    -------
+    Path
+        Absolute path to ``tesseract/tagged-but-no-text.pdf``.
+    """
+    return samples_dir / "tesseract" / "tagged-but-no-text.pdf"
+
+
 @pytest.fixture(autouse=True)
 def clean_registry() -> Generator[None, None, None]:
     """Reset the parser registry before and after every test.
index e56992d0bcc203b0838cdcac73f0cded0640e13c..f25efccc7308fedaa52d491bb5e9fb48dd04c62b 100644 (file)
@@ -21,7 +21,7 @@ from documents.parsers import run_convert
 from paperless.models import ModeChoices
 from paperless.parsers import ParserProtocol
 from paperless.parsers.tesseract import RasterisedDocumentParser
-from paperless.parsers.tesseract import post_process_text
+from paperless.parsers.utils import is_tagged_pdf
 
 if TYPE_CHECKING:
     from pathlib import Path
@@ -151,36 +151,6 @@ class TestRasterisedDocumentParserLifecycle:
         assert tempdir is not None and not tempdir.exists()
 
 
-# ---------------------------------------------------------------------------
-# post_process_text
-# ---------------------------------------------------------------------------
-
-
-class TestPostProcessText:
-    @pytest.mark.parametrize(
-        ("source", "expected"),
-        [
-            pytest.param(
-                "simple     string",
-                "simple string",
-                id="collapse-spaces",
-            ),
-            pytest.param(
-                "simple    newline\n   testing string",
-                "simple newline\ntesting string",
-                id="preserve-newline",
-            ),
-            pytest.param(
-                "utf-8   строка с пробелами в конце  ",  # noqa: RUF001
-                "utf-8 строка с пробелами в конце",  # noqa: RUF001
-                id="utf8-trailing-spaces",
-            ),
-        ],
-    )
-    def test_post_process_text(self, source: str, expected: str) -> None:
-        assert post_process_text(source) == expected
-
-
 # ---------------------------------------------------------------------------
 # Page count
 # ---------------------------------------------------------------------------
@@ -910,25 +880,25 @@ class TestSkipArchive:
         self,
         mocker: MockerFixture,
         tesseract_parser: RasterisedDocumentParser,
-        tesseract_samples_dir: Path,
+        tagged_no_text_pdf_file: Path,
     ) -> None:
         """
         GIVEN:
-            - A PDF that reports itself as tagged (/MarkInfo /Marked true) but
-              has no actual extractable text (some scanner firmware produces
-              this — see GitHub issue #13349)
+            - A real PDF that reports itself as tagged (/MarkInfo /Marked
+              true) but whose only pdftotext output is layout padding (a
+              lone form-feed byte), not real text (see GitHub issue #13387,
+              originally reported against #13349's tagged-PDF handling)
             - Mode: auto, produce_archive=False
         WHEN:
             - Document is parsed
         THEN:
             - The tag alone is not trusted as "has text"; OCRmyPDF still runs
         """
+        assert is_tagged_pdf(tagged_no_text_pdf_file) is True
         tesseract_parser.settings.mode = ModeChoices.AUTO
-        mocker.patch("paperless.parsers.tesseract.is_tagged_pdf", return_value=True)
-        mocker.patch.object(tesseract_parser, "extract_text", return_value=None)
         mock_ocr = mocker.patch("ocrmypdf.ocr")
         tesseract_parser.parse(
-            tesseract_samples_dir / "multi-page-images.pdf",
+            tagged_no_text_pdf_file,
             "application/pdf",
             produce_archive=False,
         )
diff --git a/src/paperless/tests/samples/tesseract/tagged-but-no-text.pdf b/src/paperless/tests/samples/tesseract/tagged-but-no-text.pdf
new file mode 100644 (file)
index 0000000..5a65234
Binary files /dev/null and b/src/paperless/tests/samples/tesseract/tagged-but-no-text.pdf differ
index c6bb3e34a2314fb843779b260004052ed106498f..f0a63ea44d910e310807681055d627d893c332ac 100644 (file)
@@ -4,10 +4,18 @@ from __future__ import annotations
 
 import codecs
 from pathlib import Path
+from typing import TYPE_CHECKING
+
+import pytest
 
 from paperless.parsers.utils import is_tagged_pdf
+from paperless.parsers.utils import pdf_born_digital_text
+from paperless.parsers.utils import post_process_text
 from paperless.parsers.utils import read_file_handle_unicode_errors
 
+if TYPE_CHECKING:
+    from pytest_mock import MockerFixture
+
 SAMPLES = Path(__file__).parent / "samples" / "tesseract"
 
 
@@ -60,3 +68,105 @@ class TestIsTaggedPdf:
         bad = tmp_path / "bad.pdf"
         bad.write_bytes(b"not a pdf")
         assert is_tagged_pdf(bad) is False
+
+
+class TestPostProcessText:
+    @pytest.mark.parametrize(
+        ("source", "expected"),
+        [
+            pytest.param(
+                "simple     string",
+                "simple string",
+                id="collapse-spaces",
+            ),
+            pytest.param(
+                "simple    newline\n   testing string",
+                "simple newline\ntesting string",
+                id="preserve-newline",
+            ),
+            pytest.param(
+                "utf-8   строка с пробелами в конце  ",  # noqa: RUF001
+                "utf-8 строка с пробелами в конце",  # noqa: RUF001
+                id="utf8-trailing-spaces",
+            ),
+            pytest.param(None, None, id="none-input"),
+            pytest.param("", None, id="empty-string"),
+            pytest.param("   \n\x0c  \n ", None, id="whitespace-and-formfeed-only"),
+        ],
+    )
+    def test_post_process_text(
+        self,
+        source: str | None,
+        expected: str | None,
+    ) -> None:
+        assert post_process_text(source) == expected
+
+
+class TestPdfBornDigitalText:
+    """Regression coverage for GH #13387.
+
+    should_produce_archive() and RasterisedDocumentParser.parse() must agree
+    on whether a PDF has real text, so both go through this one function.
+    """
+
+    @pytest.mark.parametrize(
+        ("extracted", "tagged", "expected_text", "expected_born_digital"),
+        [
+            pytest.param("tiny", True, "tiny", True, id="tagged-with-real-text"),
+            pytest.param("tiny", False, "tiny", False, id="untagged-below-min-length"),
+            pytest.param(
+                "x" * 51,
+                False,
+                "x" * 51,
+                True,
+                id="untagged-above-min-length",
+            ),
+            pytest.param(None, True, None, False, id="tagged-but-no-text"),
+        ],
+    )
+    def test_born_digital_decision(
+        self,
+        mocker: MockerFixture,
+        tmp_path: Path,
+        extracted: str | None,
+        tagged: bool,  # noqa: FBT001
+        expected_text: str | None,
+        expected_born_digital: bool,  # noqa: FBT001
+    ) -> None:
+        """
+        GIVEN:
+            - A PDF whose pdftotext output and /MarkInfo tag status vary
+        WHEN:
+            - pdf_born_digital_text() is called
+        THEN:
+            - The normalized text and born-digital verdict match; the tag
+              alone never counts as "has text"
+        """
+        mocker.patch(
+            "paperless.parsers.utils.extract_pdf_text",
+            return_value=extracted,
+        )
+        mocker.patch("paperless.parsers.utils.is_tagged_pdf", return_value=tagged)
+        text, born_digital = pdf_born_digital_text(tmp_path / "doc.pdf")
+        assert text == expected_text
+        assert born_digital is expected_born_digital
+
+    def test_tagged_but_textless_pdf_is_not_born_digital(
+        self,
+        tagged_no_text_pdf_file: Path,
+    ) -> None:
+        """
+        GIVEN:
+            - A real PDF that is tagged (/MarkInfo /Marked true) but whose
+              only "text" is layout padding (a stray form-feed byte)
+        WHEN:
+            - pdf_born_digital_text() is called with no mocking
+        THEN:
+            - The normalized text is None and the PDF is not treated as
+              born-digital. The raw, unnormalized pdftotext output is
+              non-empty for this file, which is exactly what caused the
+              archive decision to disagree with the OCR decision in #13387.
+        """
+        text, born_digital = pdf_born_digital_text(tagged_no_text_pdf_file)
+        assert text is None
+        assert born_digital is False