mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-07-30 15:45:58 +00:00
* Fix: unify born-digital PDF detection between archive decision and OCR should_produce_archive() and RasterisedDocumentParser.parse() each reimplemented the "does this PDF have real text" check independently, using different normalization of pdftotext output. Raw pdftotext output can be non-empty (whitespace/form-feed layout padding) even when there is no real content, so the two checks could disagree: consumer.py treated a tagged-but-textless PDF as born-digital and skipped the archive, while the parser's own (stricter, normalized) check found no text and ran OCR anyway, leaving the document with no archive despite real OCR text (GH #13387). Both call sites now share one predicate, pdf_born_digital_text() in paperless/parsers/utils.py, so they can no longer drift apart. * Fix: restore extract_text seam for born-digital detection in parse() parse() had switched to calling pdf_born_digital_text() directly for its initial text/born-digital check, bypassing the parser's own extract_text instance method. That broke test mockability (tests patch tesseract_parser.extract_text to control the born-digital decision) and caused CI failures with mismatched OCR call counts and text. Split pdf_born_digital_text() into is_born_digital_text(text, path, log) - a pure decision function - and a thin pdf_born_digital_text() wrapper for callers without text in hand (consumer.should_produce_archive). parse() now extracts via self.extract_text(None, document_path) and passes the result to is_born_digital_text(), restoring the seam with no change to production behavior. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> * Cleanup: simplify born-digital detection, close #13387 test gap Simplification pass over the born-digital detection consolidation: - is_born_digital_text(): drop the has_text temp for an early return. - consumer.py: standardize the archive-decision log lines on plain hyphens (was a mix of em-dash and hyphen) and hoist the duplicated text_length computation. - Parametrize TestPdfBornDigitalText instead of four near-identical tests. Code review follow-up: the existing tests only ever exercised pdf_born_digital_text() through mocks, so the actual #13387 scenario (a tagged PDF whose only "text" is layout padding) was never checked against real pdftotext/pikepdf output - a regression in the normalize-before-decide logic itself would have gone undetected. Moved tagged_no_text_pdf_file from parsers/conftest.py up to the shared paperless/tests/conftest.py (it was previously only visible to tests under parsers/) and added a non-mocked regression test against the real sample file. Also fixed two stale comments in test_consumer.py referencing a _extract_text_for_archive_check helper that no longer exists. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> --------- Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>
166 lines
4.8 KiB
Python
166 lines
4.8 KiB
Python
"""Tests for should_produce_archive()."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
from typing import TYPE_CHECKING
|
|
from unittest.mock import MagicMock
|
|
|
|
import pytest
|
|
|
|
from documents.consumer import should_produce_archive
|
|
|
|
if TYPE_CHECKING:
|
|
from pytest_mock import MockerFixture
|
|
|
|
|
|
def _parser_instance(
|
|
*,
|
|
can_produce: bool = True,
|
|
requires_rendition: bool = False,
|
|
) -> MagicMock:
|
|
"""Return a mock parser instance with the given capability flags."""
|
|
instance = MagicMock()
|
|
instance.can_produce_archive = can_produce
|
|
instance.requires_pdf_rendition = requires_rendition
|
|
return instance
|
|
|
|
|
|
@pytest.fixture()
|
|
def null_app_config(mocker) -> MagicMock:
|
|
"""Mock ApplicationConfiguration with all fields None → falls back to Django settings."""
|
|
return mocker.MagicMock(
|
|
output_type=None,
|
|
pages=None,
|
|
language=None,
|
|
mode=None,
|
|
archive_file_generation=None,
|
|
image_dpi=None,
|
|
unpaper_clean=None,
|
|
deskew=None,
|
|
rotate_pages=None,
|
|
rotate_pages_threshold=None,
|
|
max_image_pixels=None,
|
|
color_conversion_strategy=None,
|
|
user_args=None,
|
|
)
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def patch_app_config(mocker, null_app_config):
|
|
"""Patch BaseConfig._get_config_instance for all tests in this module."""
|
|
mocker.patch(
|
|
"paperless.config.BaseConfig._get_config_instance",
|
|
return_value=null_app_config,
|
|
)
|
|
|
|
|
|
class TestShouldProduceArchive:
|
|
@pytest.mark.parametrize(
|
|
("generation", "can_produce", "requires_rendition", "mime", "expected"),
|
|
[
|
|
pytest.param(
|
|
"never",
|
|
True,
|
|
False,
|
|
"application/pdf",
|
|
False,
|
|
id="never-returns-false",
|
|
),
|
|
pytest.param(
|
|
"always",
|
|
True,
|
|
False,
|
|
"application/pdf",
|
|
True,
|
|
id="always-returns-true",
|
|
),
|
|
pytest.param(
|
|
"never",
|
|
True,
|
|
True,
|
|
"application/pdf",
|
|
True,
|
|
id="requires-rendition-overrides-never",
|
|
),
|
|
pytest.param(
|
|
"always",
|
|
False,
|
|
False,
|
|
"text/plain",
|
|
False,
|
|
id="cannot-produce-overrides-always",
|
|
),
|
|
pytest.param(
|
|
"always",
|
|
False,
|
|
True,
|
|
"application/pdf",
|
|
True,
|
|
id="requires-rendition-wins-even-if-cannot-produce",
|
|
),
|
|
pytest.param(
|
|
"auto",
|
|
True,
|
|
False,
|
|
"image/tiff",
|
|
True,
|
|
id="auto-image-returns-true",
|
|
),
|
|
pytest.param(
|
|
"auto",
|
|
True,
|
|
False,
|
|
"message/rfc822",
|
|
False,
|
|
id="auto-non-pdf-non-image-returns-false",
|
|
),
|
|
],
|
|
)
|
|
def test_generation_setting(
|
|
self,
|
|
settings,
|
|
generation: str,
|
|
can_produce: bool, # noqa: FBT001
|
|
requires_rendition: bool, # noqa: FBT001
|
|
mime: str,
|
|
expected: bool, # noqa: FBT001
|
|
) -> None:
|
|
settings.ARCHIVE_FILE_GENERATION = generation
|
|
parser = _parser_instance(
|
|
can_produce=can_produce,
|
|
requires_rendition=requires_rendition,
|
|
)
|
|
assert should_produce_archive(parser, mime, Path("/tmp/doc")) is expected
|
|
|
|
@pytest.mark.parametrize(
|
|
("born_digital", "expected"),
|
|
[
|
|
pytest.param(True, False, id="born-digital-skips-archive"),
|
|
pytest.param(False, True, id="not-born-digital-produces-archive"),
|
|
],
|
|
)
|
|
def test_auto_pdf_archive_decision(
|
|
self,
|
|
mocker: MockerFixture,
|
|
settings,
|
|
born_digital: bool, # noqa: FBT001
|
|
expected: bool, # noqa: FBT001
|
|
) -> None:
|
|
"""Archive decision tracks pdf_born_digital_text()'s verdict exactly.
|
|
|
|
should_produce_archive() defers entirely to pdf_born_digital_text()
|
|
for the has-real-text decision, so both callers of that predicate
|
|
(this function and RasterisedDocumentParser.parse()) always agree.
|
|
"""
|
|
settings.ARCHIVE_FILE_GENERATION = "auto"
|
|
mocker.patch(
|
|
"documents.consumer.pdf_born_digital_text",
|
|
return_value=("some text", born_digital),
|
|
)
|
|
parser = _parser_instance(can_produce=True, requires_rendition=False)
|
|
assert (
|
|
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
|
|
is expected
|
|
)
|