diff --git a/src/documents/parsers.py b/src/documents/parsers.py index 6af499f41..f526b4c74 100644 --- a/src/documents/parsers.py +++ b/src/documents/parsers.py @@ -1,6 +1,7 @@ from __future__ import annotations import logging +import math import mimetypes import os import shutil @@ -131,6 +132,8 @@ _THUMBNAIL_MAX_WIDTH = 500 _THUMBNAIL_MAX_HEIGHT = 5000 # Used only when the page geometry cannot be read _THUMBNAIL_FALLBACK_DPI = 150 +# Matches the density the thumbnail was previously rendered at before scaling +_THUMBNAIL_MAX_DPI = 300 def rasterize_pdf_page_to_png( @@ -183,9 +186,9 @@ def encode_thumbnail_webp( """ Flattens any alpha onto white and saves the image as WebP. - max_width/max_height are only a safety-net clamp for DPI rounding, since - the render is already sized by the computed DPI. The image is never - enlarged, matching the previous "-scale WxH>" behavior. + max_width/max_height trim the render to the exact thumbnail size, since + the computed DPI is rounded up and lands at or slightly above it. The + image is never enlarged, matching the previous "-scale WxH>" behavior. """ from PIL import Image @@ -205,8 +208,9 @@ def encode_thumbnail_webp( def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> int: """ - Computes the DPI which renders the first page of the PDF to fit within the - thumbnail size in one pass, never above the page's natural 72 DPI size. + Computes the DPI which renders the first page of the PDF at or just above + the thumbnail size in one pass, never above the 300 DPI the thumbnail was + previously rendered at before being scaled down. """ from paperless.parsers.utils import get_pdf_first_page_size_points @@ -221,9 +225,15 @@ def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> int: width_pts, height_pts = size dpi_for_width = _THUMBNAIL_MAX_WIDTH * 72 / width_pts dpi_for_height = _THUMBNAIL_MAX_HEIGHT * 72 / height_pts - # Capping at 72 (1px per point) keeps the shrink-only behavior: a page - # already smaller than the thumbnail size is never enlarged - return max(1, round(min(72, dpi_for_width, dpi_for_height))) + # The old pipeline rendered at 300 DPI and then shrank to fit, so only + # pages too small to reach the thumbnail size even at 300 DPI end up + # smaller than it. Rounding up keeps the render at or above the target, + # so the shrink-only clamp in encode_thumbnail_webp trims it to exactly + # the thumbnail size instead of leaving it a few pixels short. + return max( + 1, + math.ceil(min(_THUMBNAIL_MAX_DPI, dpi_for_width, dpi_for_height)), + ) def _render_pdf_thumbnail( diff --git a/src/documents/tests/test_parsers.py b/src/documents/tests/test_parsers.py index ebc8eb6cf..053eee476 100644 --- a/src/documents/tests/test_parsers.py +++ b/src/documents/tests/test_parsers.py @@ -13,6 +13,7 @@ from documents.parsers import encode_thumbnail_webp from documents.parsers import get_default_file_extension from documents.parsers import get_supported_file_extensions from documents.parsers import is_file_ext_supported +from documents.parsers import make_thumbnail_from_pdf from documents.parsers import rasterize_pdf_page_to_png from paperless.parsers.registry import get_parser_registry from paperless.parsers.registry import reset_parser_registry @@ -140,9 +141,10 @@ class TestComputeThumbnailDpi: ("size", "expected_dpi"), [ pytest.param((612.0, 792.0), 59, id="letter-width-bound"), - pytest.param((792.0, 612.0), 45, id="landscape-width-bound"), + pytest.param((792.0, 612.0), 46, id="landscape-rounded-up"), pytest.param((612.0, 100000.0), 4, id="tall-strip-height-bound"), - pytest.param((200.0, 300.0), 72, id="small-page-never-enlarged"), + pytest.param((200.0, 300.0), 180, id="small-page-width-bound"), + pytest.param((72.0, 72.0), 300, id="tiny-page-capped-at-300"), pytest.param((1000000.0, 1000000.0), 1, id="huge-page-minimum-one"), pytest.param(None, 150, id="unreadable-geometry-fallback"), ], @@ -161,8 +163,9 @@ class TestComputeThumbnailDpi: WHEN: - The thumbnail DPI is computed THEN: - - The DPI fits the page into 500x5000 without ever exceeding 72, - is at least 1, and falls back to 150 when the size is unknown + - The DPI is rounded up so the render reaches 500x5000, never + exceeds 300, is at least 1, and falls back to 150 when the + size is unknown """ mocker.patch( "paperless.parsers.utils.get_pdf_first_page_size_points", @@ -171,6 +174,47 @@ class TestComputeThumbnailDpi: assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected_dpi +class TestMakeThumbnailFromPdf: + @pytest.mark.parametrize( + ("page_size", "expected_width"), + [ + pytest.param((612, 792), 500, id="letter"), + pytest.param((792, 612), 500, id="landscape-letter"), + pytest.param((595, 842), 500, id="a4"), + pytest.param((200, 300), 500, id="small-page"), + pytest.param((72, 72), 300, id="tiny-page-capped"), + ], + ) + def test_thumbnail_width( + self, + tmp_path: Path, + page_size: tuple[int, int], + expected_width: int, + ) -> None: + """ + GIVEN: + - A PDF whose first page has the given size in points + WHEN: + - A thumbnail is made from it + THEN: + - The WebP thumbnail is exactly 500px wide, unless the page is + too small to reach that even at 300 DPI + """ + pdf = pikepdf.new() + pdf.add_blank_page(page_size=page_size) + pdf_path = tmp_path / "in.pdf" + pdf.save(pdf_path) + work_dir = tmp_path / "work" + work_dir.mkdir() + + thumb = make_thumbnail_from_pdf(pdf_path, work_dir) + + assert thumb == work_dir / "convert.webp" + with Image.open(thumb) as im: + assert im.format == "WEBP" + assert im.width == expected_width + + class TestRasterizePdfPageToPng: @staticmethod def _write_pdf( diff --git a/src/paperless/parsers/utils.py b/src/paperless/parsers/utils.py index 3ed2d8da0..a02ac1ecb 100644 --- a/src/paperless/parsers/utils.py +++ b/src/paperless/parsers/utils.py @@ -311,8 +311,8 @@ def get_pdf_first_page_size_points( if page.rotation in (90, 270): width, height = height, width return width, height - except Exception: - _log.warning("Could not determine PDF page size for %s", path, exc_info=True) + except Exception as e: + _log.warning("Could not determine PDF page size for %s: %s", path, e) return None