diff --git a/src/documents/parsers.py b/src/documents/parsers.py index 69ee4e285..6af499f41 100644 --- a/src/documents/parsers.py +++ b/src/documents/parsers.py @@ -127,42 +127,160 @@ def get_default_thumbnail() -> Path: return (Path(__file__).parent / "resources" / "document.webp").resolve() -def make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group=None) -> Path: - out_path: Path = Path(temp_dir) / "convert_gs.webp" +_THUMBNAIL_MAX_WIDTH = 500 +_THUMBNAIL_MAX_HEIGHT = 5000 +# Used only when the page geometry cannot be read +_THUMBNAIL_FALLBACK_DPI = 150 - # if convert fails, fall back to extracting - # the first PDF page as a PNG using Ghostscript - logger.warning( - "Thumbnail generation with ImageMagick failed, falling back " - "to ghostscript. Check your /etc/ImageMagick-x/policy.xml!", - extra={"group": logging_group}, - ) - # Ghostscript doesn't handle WebP outputs - gs_out_path: Path = Path(temp_dir) / "gs_out.png" - cmd = [settings.GS_BINARY, "-q", "-sDEVICE=pngalpha", "-o", gs_out_path, in_path] + +def rasterize_pdf_page_to_png( + in_path: Path, + out_path: Path, + *, + dpi: int, + use_cropbox: bool = True, + logging_group=None, +) -> None: + """ + Rasterizes page 1 of a PDF to a PNG at out_path via pdftoppm (Poppler), + at the given DPI. pdftoppm honors the page's /Rotate on its own. + """ + # -singlefile stops pdftoppm appending a page number to the output name, + # and with -png it appends ".png" itself, so it is given the path without + # its suffix to write exactly out_path + args = [ + "pdftoppm", + "-f", + "1", + "-l", + "1", + "-r", + str(dpi), + "-png", + "-singlefile", + ] + if use_cropbox: + args.append("-cropbox") + args += [str(in_path), str(out_path.with_suffix(""))] + + logger.debug("Execute: " + " ".join(args), extra={"group": logging_group}) try: - try: - run_subprocess(cmd, logger=logger) - except subprocess.CalledProcessError as e: - raise ParseError(f"Thumbnail (gs) failed at {cmd}") from e - # then run convert on the output from gs to make WebP - run_convert( - density=300, - scale="500x5000>", - alpha="remove", - strip=True, - trim=False, - auto_orient=True, - input_file=gs_out_path, - output_file=out_path, - logging_group=logging_group, + run_subprocess(args, logger=logger) + except subprocess.CalledProcessError as e: + raise ParseError(f"pdftoppm failed at {args}") from e + except Exception as e: # pragma: no cover + raise ParseError("Unknown error running pdftoppm") from e + + +def encode_thumbnail_webp( + png_path: Path, + out_path: Path, + *, + max_width: int = _THUMBNAIL_MAX_WIDTH, + max_height: int = _THUMBNAIL_MAX_HEIGHT, +) -> None: + """ + Flattens any alpha onto white and saves the image as WebP. + + max_width/max_height are only a safety-net clamp for DPI rounding, since + the render is already sized by the computed DPI. The image is never + enlarged, matching the previous "-scale WxH>" behavior. + """ + from PIL import Image + + try: + with Image.open(png_path) as im: + if im.mode in ("RGBA", "LA"): + flattened = Image.new("RGB", im.size, (255, 255, 255)) + flattened.paste(im, mask=im.split()[-1]) + else: + flattened = im.convert("RGB") + + flattened.thumbnail((max_width, max_height)) + flattened.save(out_path, format="WEBP") + except OSError as e: + raise ParseError(f"Unable to encode thumbnail from {png_path}") from e + + +def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> int: + """ + Computes the DPI which renders the first page of the PDF to fit within the + thumbnail size in one pass, never above the page's natural 72 DPI size. + """ + from paperless.parsers.utils import get_pdf_first_page_size_points + + size = get_pdf_first_page_size_points(in_path) + if size is None: + logger.debug( + "Could not read PDF page size, using fallback DPI", + extra={"group": logging_group}, ) + return _THUMBNAIL_FALLBACK_DPI + + width_pts, height_pts = size + dpi_for_width = _THUMBNAIL_MAX_WIDTH * 72 / width_pts + dpi_for_height = _THUMBNAIL_MAX_HEIGHT * 72 / height_pts + # Capping at 72 (1px per point) keeps the shrink-only behavior: a page + # already smaller than the thumbnail size is never enlarged + return max(1, round(min(72, dpi_for_width, dpi_for_height))) + + +def _render_pdf_thumbnail( + in_path: Path, + png_path: Path, + out_path: Path, + logging_group=None, +) -> None: + dpi = _compute_thumbnail_dpi(in_path, logging_group=logging_group) + rasterize_pdf_page_to_png( + in_path, + png_path, + dpi=dpi, + use_cropbox=True, + logging_group=logging_group, + ) + encode_thumbnail_webp(png_path, out_path) + + +def make_thumbnail_from_pdf_qpdf_fallback( + in_path: Path, + temp_dir: Path, + logging_group=None, +) -> Path: + png_path: Path = Path(temp_dir) / "page1_repaired.png" + out_path: Path = Path(temp_dir) / "convert_qpdf.webp" + repaired_path: Path = Path(temp_dir) / "repaired.pdf" + + logger.warning( + "Thumbnail generation with pdftoppm failed, attempting qpdf repair and retry.", + extra={"group": logging_group}, + ) + + try: + # qpdf rewrites in place, so work on a copy and leave the original alone. + # qpdf exits 3 when it had to repair the file, which is the expected + # outcome here, so warnings must not count as failure. + try: + shutil.copy(in_path, repaired_path) + run_subprocess( + [ + "qpdf", + "--warning-exit-0", + "--replace-input", + str(repaired_path), + ], + logger=logger, + ) + except (subprocess.CalledProcessError, OSError) as e: + raise ParseError(f"qpdf repair failed for {in_path}") from e + + _render_pdf_thumbnail(repaired_path, png_path, out_path, logging_group) return out_path except ParseError as e: - logger.error(f"Unable to make thumbnail with Ghostscript: {e}") + logger.error(f"Unable to make thumbnail after qpdf repair: {e}") # The caller might expect a generated thumbnail that can be moved, # so we need to copy it before it gets moved. # https://github.com/paperless-ngx/paperless-ngx/issues/3631 @@ -175,25 +293,18 @@ def make_thumbnail_from_pdf(in_path: Path, temp_dir: Path, logging_group=None) - """ The thumbnail of a PDF is just a 500px wide image of the first page. """ + png_path: Path = temp_dir / "page1.png" out_path: Path = temp_dir / "convert.webp" - # Run convert to get a decent thumbnail try: - run_convert( - density=300, - scale="500x5000>", - alpha="remove", - strip=True, - trim=False, - auto_orient=True, - use_cropbox=True, - input_file=f"{in_path}[0]", - output_file=str(out_path), - logging_group=logging_group, - ) + _render_pdf_thumbnail(in_path, png_path, out_path, logging_group) except ParseError as e: - logger.error(f"Unable to make thumbnail with convert: {e}") - out_path = make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group) + logger.error(f"Unable to make thumbnail with pdftoppm: {e}") + out_path = make_thumbnail_from_pdf_qpdf_fallback( + in_path, + temp_dir, + logging_group, + ) return out_path diff --git a/src/documents/tests/test_parsers.py b/src/documents/tests/test_parsers.py index 5f3893112..ebc8eb6cf 100644 --- a/src/documents/tests/test_parsers.py +++ b/src/documents/tests/test_parsers.py @@ -1,11 +1,19 @@ from collections.abc import Generator +from pathlib import Path +import pikepdf import pytest +from PIL import Image from pytest_django.fixtures import Settings +from pytest_mock import MockerFixture +from documents.parsers import ParseError +from documents.parsers import _compute_thumbnail_dpi +from documents.parsers import encode_thumbnail_webp from documents.parsers import get_default_file_extension from documents.parsers import get_supported_file_extensions from documents.parsers import is_file_ext_supported +from documents.parsers import rasterize_pdf_page_to_png from paperless.parsers.registry import get_parser_registry from paperless.parsers.registry import reset_parser_registry from paperless.parsers.tesseract import RasterisedDocumentParser @@ -125,3 +133,183 @@ class TestParserAvailability: assert is_file_ext_supported(".pdf") assert not is_file_ext_supported(".hsdfh") assert not is_file_ext_supported("") + + +class TestComputeThumbnailDpi: + @pytest.mark.parametrize( + ("size", "expected_dpi"), + [ + pytest.param((612.0, 792.0), 59, id="letter-width-bound"), + pytest.param((792.0, 612.0), 45, id="landscape-width-bound"), + pytest.param((612.0, 100000.0), 4, id="tall-strip-height-bound"), + pytest.param((200.0, 300.0), 72, id="small-page-never-enlarged"), + pytest.param((1000000.0, 1000000.0), 1, id="huge-page-minimum-one"), + pytest.param(None, 150, id="unreadable-geometry-fallback"), + ], + ) + def test_dpi_from_page_size( + self, + mocker: MockerFixture, + tmp_path: Path, + size: tuple[float, float] | None, + expected_dpi: int, + ) -> None: + """ + GIVEN: + - A PDF whose first page has the given size in points (or whose + geometry cannot be read) + WHEN: + - The thumbnail DPI is computed + THEN: + - The DPI fits the page into 500x5000 without ever exceeding 72, + is at least 1, and falls back to 150 when the size is unknown + """ + mocker.patch( + "paperless.parsers.utils.get_pdf_first_page_size_points", + return_value=size, + ) + assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected_dpi + + +class TestRasterizePdfPageToPng: + @staticmethod + def _write_pdf( + path: Path, + *, + crop_box: tuple[float, float, float, float] | None = None, + rotate: int | None = None, + ) -> Path: + pdf = pikepdf.new() + pdf.add_blank_page(page_size=(144, 72)) + pdf.add_blank_page(page_size=(300, 300)) + page = pdf.pages[0] + if crop_box is not None: + page.obj.CropBox = pikepdf.Array(crop_box) + if rotate is not None: + page.obj.Rotate = rotate + pdf.save(path) + return path + + @pytest.mark.parametrize( + ("crop_box", "rotate", "expected_size"), + [ + pytest.param(None, None, (144, 72), id="plain"), + pytest.param(None, 90, (72, 144), id="rotated-90"), + pytest.param((0, 0, 72, 36), None, (72, 36), id="crop-box"), + ], + ) + def test_renders_first_page_to_exact_path( + self, + tmp_path: Path, + crop_box: tuple[float, float, float, float] | None, + rotate: int | None, + expected_size: tuple[int, int], + ) -> None: + """ + GIVEN: + - A two page PDF whose first page is 144x72 points, optionally + cropped or rotated + WHEN: + - The first page is rasterized at 72 DPI + THEN: + - Exactly the requested output path is written, and the image + matches the first page's cropped, rotated size + """ + pdf_path = self._write_pdf( + tmp_path / "in.pdf", + crop_box=crop_box, + rotate=rotate, + ) + out_dir = tmp_path / "out" + out_dir.mkdir() + out_path = out_dir / "page1.png" + + rasterize_pdf_page_to_png(pdf_path, out_path, dpi=72) + + assert list(out_dir.iterdir()) == [out_path] + with Image.open(out_path) as im: + assert im.format == "PNG" + assert im.size == expected_size + + def test_failure_raises_parse_error(self, tmp_path: Path) -> None: + """ + GIVEN: + - A file that is not a PDF + WHEN: + - Rasterization is attempted + THEN: + - A ParseError is raised + """ + bad = tmp_path / "bad.pdf" + bad.write_bytes(b"not a pdf") + + with pytest.raises(ParseError): + rasterize_pdf_page_to_png(bad, tmp_path / "page1.png", dpi=72) + + +class TestEncodeThumbnailWebp: + @pytest.mark.parametrize( + ("mode", "color"), + [ + pytest.param("RGBA", (0, 0, 0, 0), id="rgba"), + pytest.param("LA", (0, 0), id="la"), + ], + ) + def test_alpha_flattened_onto_white( + self, + tmp_path: Path, + mode: str, + color: tuple[int, ...], + ) -> None: + """ + GIVEN: + - A fully transparent PNG with an alpha channel + WHEN: + - It is encoded as a thumbnail + THEN: + - The WebP output is RGB with the transparency flattened to white + """ + png_path = tmp_path / "in.png" + Image.new(mode, (20, 10), color).save(png_path) + out_path = tmp_path / "out.webp" + + encode_thumbnail_webp(png_path, out_path) + + with Image.open(out_path) as im: + assert im.format == "WEBP" + assert im.mode == "RGB" + assert im.size == (20, 10) + red, green, blue = im.getpixel((10, 5)) + assert min(red, green, blue) >= 250 + + @pytest.mark.parametrize( + ("in_size", "expected_size"), + [ + pytest.param((1000, 2000), (500, 1000), id="too-wide-shrunk"), + pytest.param((100, 10000), (50, 5000), id="too-tall-shrunk"), + pytest.param((100, 200), (100, 200), id="small-not-enlarged"), + ], + ) + def test_size_clamped_without_enlarging( + self, + tmp_path: Path, + in_size: tuple[int, int], + expected_size: tuple[int, int], + ) -> None: + """ + GIVEN: + - A rendered page image of the given size + WHEN: + - It is encoded as a thumbnail + THEN: + - It is shrunk to fit 500x5000 keeping aspect ratio, and never + enlarged + """ + png_path = tmp_path / "in.png" + Image.new("RGB", in_size, (255, 255, 255)).save(png_path) + out_path = tmp_path / "out.webp" + + encode_thumbnail_webp(png_path, out_path) + + with Image.open(out_path) as im: + assert im.size == expected_size diff --git a/src/paperless/parsers/utils.py b/src/paperless/parsers/utils.py index 9fe1d4908..3ed2d8da0 100644 --- a/src/paperless/parsers/utils.py +++ b/src/paperless/parsers/utils.py @@ -265,6 +265,57 @@ def get_page_count_for_pdf( return None +def get_pdf_first_page_size_points( + path: Path, + log: logging.Logger | None = None, +) -> tuple[float, float] | None: + """Return the first page's (width, height) in PDF points, post-rotation. + + Uses ``page.cropbox``, which falls back to the MediaBox when the page has + no ``/CropBox`` of its own. This must match the box the renderer is told + to use (pdftoppm's ``-cropbox`` flag), or a DPI computed from it will + target the wrong box's dimensions. + + Width and height are swapped when the page's effective rotation is 90 or + 270, since that is the orientation the page is rendered in. + ``page.rotation`` resolves a ``/Rotate`` inherited from an ancestor + ``/Pages`` node and normalizes it to ``[0, 360)``, which a raw + ``/Rotate`` lookup on the page dictionary would not. + + Parameters + ---------- + path: + Absolute path to the PDF file. + log: + Logger for warnings. Falls back to the module-level logger when omitted. + + Returns + ------- + tuple[float, float] | None + ``(width_points, height_points)``, or ``None`` if the file cannot be + opened, has no pages, or the page box is degenerate. + """ + import pikepdf + + _log = log or logger + try: + with pikepdf.Pdf.open(path) as pdf: + if len(pdf.pages) == 0: + return None + page = pdf.pages[0] + llx, lly, urx, ury = (float(v) for v in page.cropbox) + width = abs(urx - llx) + height = abs(ury - lly) + if width <= 0 or height <= 0: + return None + if page.rotation in (90, 270): + width, height = height, width + return width, height + except Exception: + _log.warning("Could not determine PDF page size for %s", path, exc_info=True) + return None + + def extract_pdf_metadata( document_path: Path, log: logging.Logger | None = None, diff --git a/src/paperless/tests/parsers/test_tesseract_parser.py b/src/paperless/tests/parsers/test_tesseract_parser.py index bb6a85031..2effa0a8a 100644 --- a/src/paperless/tests/parsers/test_tesseract_parser.py +++ b/src/paperless/tests/parsers/test_tesseract_parser.py @@ -15,9 +15,10 @@ from typing import TYPE_CHECKING import pytest from ocrmypdf import SubprocessOutputError +from PIL import Image +import documents.parsers from documents.parsers import ParseError -from documents.parsers import run_convert from paperless.models import ModeChoices from paperless.parsers import ParserProtocol from paperless.parsers.tesseract import RasterisedDocumentParser @@ -280,24 +281,74 @@ class TestGetThumbnail: ) assert thumb.is_file() - def test_thumbnail_fallback_on_convert_error( + @pytest.mark.parametrize( + ("filename", "expected_height"), + [ + pytest.param("simple-digital.pdf", 647, id="portrait-letter"), + pytest.param("rotated.pdf", 386, id="landscape"), + ], + ) + def test_thumbnail_is_correct_format_and_size( + self, + tesseract_parser: RasterisedDocumentParser, + tesseract_samples_dir: Path, + filename: str, + expected_height: int, + ) -> None: + """ + GIVEN: + - A PDF whose first page is wider than the thumbnail width + WHEN: + - A thumbnail is generated + THEN: + - The thumbnail is a WebP, 500px wide, keeping the page's aspect + ratio (within rounding of the DPI computation) + """ + thumb = tesseract_parser.get_thumbnail( + tesseract_samples_dir / filename, + "application/pdf", + ) + with Image.open(thumb) as im: + assert im.format == "WEBP" + assert im.width == 500 + assert im.height == pytest.approx(expected_height, abs=2) + + def test_thumbnail_fallback_on_pdftoppm_error( self, mocker: MockerFixture, tesseract_parser: RasterisedDocumentParser, tesseract_samples_dir: Path, ) -> None: - def _raise_on_pdf(input_file, output_file, **kwargs) -> None: - if ".pdf" in str(input_file): + """ + GIVEN: + - Rasterizing the original PDF fails + WHEN: + - A thumbnail is generated + THEN: + - The PDF is repaired with qpdf and rasterized again, producing a + real thumbnail rather than the default placeholder + """ + real_rasterize = documents.parsers.rasterize_pdf_page_to_png + original = tesseract_samples_dir / "simple-digital.pdf" + + def _fail_on_original(in_path: Path, out_path: Path, **kwargs) -> None: + if in_path == original: raise ParseError("Does not compute.") - run_convert(input_file=input_file, output_file=output_file, **kwargs) + real_rasterize(in_path, out_path, **kwargs) - mocker.patch("documents.parsers.run_convert", side_effect=_raise_on_pdf) - - thumb = tesseract_parser.get_thumbnail( - tesseract_samples_dir / "simple-digital.pdf", - "application/pdf", + rasterize = mocker.patch( + "documents.parsers.rasterize_pdf_page_to_png", + side_effect=_fail_on_original, ) + + thumb = tesseract_parser.get_thumbnail(original, "application/pdf") + + assert rasterize.call_count == 2 assert thumb.is_file() + assert thumb.name == "convert_qpdf.webp" + with Image.open(thumb) as im: + assert im.format == "WEBP" + assert im.width == 500 def test_thumbnail_encrypted_pdf( self, diff --git a/src/paperless/tests/test_parser_utils.py b/src/paperless/tests/test_parser_utils.py index f0a63ea44..2205a008d 100644 --- a/src/paperless/tests/test_parser_utils.py +++ b/src/paperless/tests/test_parser_utils.py @@ -6,8 +6,10 @@ import codecs from pathlib import Path from typing import TYPE_CHECKING +import pikepdf import pytest +from paperless.parsers.utils import get_pdf_first_page_size_points from paperless.parsers.utils import is_tagged_pdf from paperless.parsers.utils import pdf_born_digital_text from paperless.parsers.utils import post_process_text @@ -70,6 +72,151 @@ class TestIsTaggedPdf: assert is_tagged_pdf(bad) is False +class TestGetPdfFirstPageSizePoints: + @staticmethod + def _write_pdf( + path: Path, + *, + media_box: tuple[float, float, float, float] = (0, 0, 600, 800), + crop_box: tuple[float, float, float, float] | None = None, + page_rotate: int | None = None, + inherited_rotate: int | None = None, + ) -> Path: + pdf = pikepdf.new() + pdf.add_blank_page(page_size=(media_box[2], media_box[3])) + page = pdf.pages[0] + page.obj.MediaBox = pikepdf.Array(media_box) + if crop_box is not None: + page.obj.CropBox = pikepdf.Array(crop_box) + if page_rotate is not None: + page.obj.Rotate = page_rotate + if inherited_rotate is not None: + pdf.Root.Pages.Rotate = inherited_rotate + pdf.save(path) + return path + + def test_letter_sample(self) -> None: + """ + GIVEN: + - A US Letter sample PDF with no CropBox and no rotation + WHEN: + - The first page size is requested + THEN: + - The MediaBox size in points is returned + """ + assert get_pdf_first_page_size_points(SAMPLES / "simple-digital.pdf") == ( + 612.0, + 792.0, + ) + + @pytest.mark.parametrize( + ("rotate", "expected"), + [ + pytest.param(0, (600.0, 800.0), id="rotate-0"), + pytest.param(90, (800.0, 600.0), id="rotate-90"), + pytest.param(180, (600.0, 800.0), id="rotate-180"), + pytest.param(270, (800.0, 600.0), id="rotate-270"), + pytest.param(-90, (800.0, 600.0), id="rotate-negative-90"), + ], + ) + def test_page_rotation_swaps_dimensions( + self, + tmp_path: Path, + rotate: int, + expected: tuple[float, float], + ) -> None: + """ + GIVEN: + - A portrait PDF page with /Rotate set directly on the page + WHEN: + - The first page size is requested + THEN: + - Width and height are swapped for quarter-turn rotations only + """ + pdf_path = self._write_pdf(tmp_path / "rotated.pdf", page_rotate=rotate) + assert get_pdf_first_page_size_points(pdf_path) == expected + + def test_inherited_rotation_swaps_dimensions(self, tmp_path: Path) -> None: + """ + GIVEN: + - A portrait PDF page whose /Rotate 90 is set on the /Pages node, + not on the page itself + WHEN: + - The first page size is requested + THEN: + - The inherited rotation is honored and width/height are swapped + """ + pdf_path = self._write_pdf(tmp_path / "inherited.pdf", inherited_rotate=90) + assert get_pdf_first_page_size_points(pdf_path) == (800.0, 600.0) + + def test_crop_box_preferred_over_media_box(self, tmp_path: Path) -> None: + """ + GIVEN: + - A PDF page with a CropBox smaller than its MediaBox + WHEN: + - The first page size is requested + THEN: + - The CropBox dimensions are returned, matching what pdftoppm + renders with -cropbox + """ + pdf_path = self._write_pdf( + tmp_path / "cropped.pdf", + crop_box=(50, 100, 350, 500), + ) + assert get_pdf_first_page_size_points(pdf_path) == (300.0, 400.0) + + def test_degenerate_box_returns_none(self, tmp_path: Path) -> None: + """ + GIVEN: + - A PDF page whose box has zero width + WHEN: + - The first page size is requested + THEN: + - None is returned instead of a size that would break DPI math + """ + pdf_path = self._write_pdf( + tmp_path / "degenerate.pdf", + media_box=(0, 0, 600, 800), + crop_box=(100, 0, 100, 800), + ) + assert get_pdf_first_page_size_points(pdf_path) is None + + def test_nonexistent_path_returns_none(self) -> None: + """ + GIVEN: + - A path that does not exist + WHEN: + - The first page size is requested + THEN: + - None is returned and nothing is raised + """ + assert get_pdf_first_page_size_points(Path("/nonexistent/file.pdf")) is None + + def test_corrupt_pdf_returns_none(self, tmp_path: Path) -> None: + """ + GIVEN: + - A file that is not a PDF + WHEN: + - The first page size is requested + THEN: + - None is returned and nothing is raised + """ + bad = tmp_path / "bad.pdf" + bad.write_bytes(b"not a pdf") + assert get_pdf_first_page_size_points(bad) is None + + def test_encrypted_pdf_returns_none(self) -> None: + """ + GIVEN: + - A password protected PDF + WHEN: + - The first page size is requested + THEN: + - None is returned and nothing is raised + """ + assert get_pdf_first_page_size_points(SAMPLES / "encrypted.pdf") is None + + class TestPostProcessText: @pytest.mark.parametrize( ("source", "expected"),