diff --git a/src/documents/parsers.py b/src/documents/parsers.py index bce2dcc88..75256bbed 100644 --- a/src/documents/parsers.py +++ b/src/documents/parsers.py @@ -81,6 +81,10 @@ _THUMBNAIL_MAX_HEIGHT = 5000 _THUMBNAIL_FALLBACK_DPI = 150 # Matches the density the thumbnail was previously rendered at before scaling _THUMBNAIL_MAX_DPI = 300 +# Pages with known geometry are rendered at this multiple of the computed DPI +# and downsampled with Lanczos, which keeps text noticeably crisper than +# rasterizing straight at the target size +_THUMBNAIL_SUPERSAMPLE = 2 def rasterize_pdf_page_to_png( @@ -129,13 +133,16 @@ def encode_thumbnail_webp( *, max_width: int = _THUMBNAIL_MAX_WIDTH, max_height: int = _THUMBNAIL_MAX_HEIGHT, + supersample: int = 1, ) -> None: """ Flattens any alpha onto white and saves the image as WebP. - max_width/max_height trim the render to the exact thumbnail size, since - the computed DPI is rounded up and lands at or slightly above it. The - image is never enlarged, matching the previous "-scale WxH>" behavior. + A render made at supersample times the target density is first + downsampled by that factor with Lanczos. max_width/max_height then trim + the result to the exact thumbnail size, since the computed DPI is rounded + up and lands at or slightly above it. The image is never enlarged, + matching the previous "-scale WxH>" behavior. """ from PIL import Image @@ -147,17 +154,32 @@ def encode_thumbnail_webp( else: flattened = im.convert("RGB") + if supersample > 1: + flattened = flattened.resize( + ( + max(1, round(flattened.width / supersample)), + max(1, round(flattened.height / supersample)), + ), + Image.Resampling.LANCZOS, + ) + flattened.thumbnail((max_width, max_height)) flattened.save(out_path, format="WEBP") except (OSError, Image.DecompressionBombError) as e: raise ParseError(f"Unable to encode thumbnail from {png_path}") from e -def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> int: +def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> tuple[int, int]: """ - Computes the DPI which renders the first page of the PDF at or just above - the thumbnail size in one pass, never above the 300 DPI the thumbnail was - previously rendered at before being scaled down. + Computes the DPI at which the first page of the PDF reaches at or just + above the thumbnail size, never above the 300 DPI the thumbnail was + previously rendered at before being scaled down, and the supersampling + factor to render with. + + Returns (dpi, supersample). The page is rendered at supersample * dpi and + downsampled by supersample afterwards. When the page geometry cannot be + read the fixed fallback DPI is used without supersampling, as that render + is not bounded by the thumbnail size. """ from paperless.parsers.utils import get_pdf_first_page_size_points @@ -167,7 +189,7 @@ def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> int: "Could not read PDF page size, using fallback DPI", extra={"group": logging_group}, ) - return _THUMBNAIL_FALLBACK_DPI + return _THUMBNAIL_FALLBACK_DPI, 1 width_pts, height_pts = size dpi_for_width = _THUMBNAIL_MAX_WIDTH * 72 / width_pts @@ -177,10 +199,11 @@ def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> int: # smaller than it. Rounding up keeps the render at or above the target, # so the shrink-only clamp in encode_thumbnail_webp trims it to exactly # the thumbnail size instead of leaving it a few pixels short. - return max( + dpi = max( 1, math.ceil(min(_THUMBNAIL_MAX_DPI, dpi_for_width, dpi_for_height)), ) + return dpi, _THUMBNAIL_SUPERSAMPLE def _render_pdf_thumbnail( @@ -189,15 +212,15 @@ def _render_pdf_thumbnail( out_path: Path, logging_group=None, ) -> None: - dpi = _compute_thumbnail_dpi(in_path, logging_group=logging_group) + dpi, supersample = _compute_thumbnail_dpi(in_path, logging_group=logging_group) rasterize_pdf_page_to_png( in_path, png_path, - dpi=dpi, + dpi=dpi * supersample, use_cropbox=True, logging_group=logging_group, ) - encode_thumbnail_webp(png_path, out_path) + encode_thumbnail_webp(png_path, out_path, supersample=supersample) def make_thumbnail_from_pdf_qpdf_fallback( diff --git a/src/documents/tests/test_parsers.py b/src/documents/tests/test_parsers.py index 6ba10eebd..3531cec92 100644 --- a/src/documents/tests/test_parsers.py +++ b/src/documents/tests/test_parsers.py @@ -140,15 +140,15 @@ class TestParserAvailability: class TestComputeThumbnailDpi: @pytest.mark.parametrize( - ("size", "expected_dpi"), + ("size", "expected"), [ - pytest.param((612.0, 792.0), 59, id="letter-width-bound"), - pytest.param((792.0, 612.0), 46, id="landscape-rounded-up"), - pytest.param((612.0, 100000.0), 4, id="tall-strip-height-bound"), - pytest.param((200.0, 300.0), 180, id="small-page-width-bound"), - pytest.param((72.0, 72.0), 300, id="tiny-page-capped-at-300"), - pytest.param((1000000.0, 1000000.0), 1, id="huge-page-minimum-one"), - pytest.param(None, 150, id="unreadable-geometry-fallback"), + pytest.param((612.0, 792.0), (59, 2), id="letter-width-bound"), + pytest.param((792.0, 612.0), (46, 2), id="landscape-rounded-up"), + pytest.param((612.0, 100000.0), (4, 2), id="tall-strip-height-bound"), + pytest.param((200.0, 300.0), (180, 2), id="small-page-width-bound"), + pytest.param((72.0, 72.0), (300, 2), id="tiny-page-capped-at-300"), + pytest.param((1000000.0, 1000000.0), (1, 2), id="huge-page-minimum-one"), + pytest.param(None, (150, 1), id="unreadable-geometry-fallback"), ], ) def test_dpi_from_page_size( @@ -156,7 +156,7 @@ class TestComputeThumbnailDpi: mocker: MockerFixture, tmp_path: Path, size: tuple[float, float] | None, - expected_dpi: int, + expected: tuple[int, int], ) -> None: """ GIVEN: @@ -166,17 +166,56 @@ class TestComputeThumbnailDpi: - The thumbnail DPI is computed THEN: - The DPI is rounded up so the render reaches 500x5000, never - exceeds 300, is at least 1, and falls back to 150 when the - size is unknown + exceeds 300, is at least 1, and is supersampled 2x; an unknown + size gives the plain 150 DPI fallback without supersampling """ mocker.patch( "paperless.parsers.utils.get_pdf_first_page_size_points", return_value=size, ) - assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected_dpi + assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected class TestMakeThumbnailFromPdf: + @pytest.mark.parametrize( + ("size", "expected_dpi", "expected_supersample"), + [ + pytest.param((612.0, 792.0), 118, 2, id="known-geometry-2x"), + pytest.param(None, 150, 1, id="unreadable-geometry-plain-fallback"), + ], + ) + def test_render_dpi_requested( + self, + mocker: MockerFixture, + tmp_path: Path, + size: tuple[float, float] | None, + expected_dpi: int, + expected_supersample: int, + ) -> None: + """ + GIVEN: + - A PDF whose page geometry is either readable or not + WHEN: + - A thumbnail is made from it + THEN: + - The page is rasterized at twice the computed DPI when the + geometry is known, and at the plain 150 DPI fallback otherwise + - The encode step is told the matching downsample factor + """ + mocker.patch( + "paperless.parsers.utils.get_pdf_first_page_size_points", + return_value=size, + ) + rasterize = mocker.patch("documents.parsers.rasterize_pdf_page_to_png") + encode = mocker.patch("documents.parsers.encode_thumbnail_webp") + work_dir = tmp_path / "work" + work_dir.mkdir() + + make_thumbnail_from_pdf(tmp_path / "in.pdf", work_dir) + + assert rasterize.call_args.kwargs["dpi"] == expected_dpi + assert encode.call_args.kwargs["supersample"] == expected_supersample + @pytest.mark.parametrize( ("page_size", "expected_width"), [ @@ -445,6 +484,39 @@ class TestEncodeThumbnailWebp: with Image.open(out_path) as im: assert im.size == expected_size + @pytest.mark.parametrize( + ("in_size", "supersample", "expected_size"), + [ + pytest.param((1000, 1400), 2, (500, 700), id="2x-halved"), + pytest.param((1001, 1401), 2, (500, 700), id="2x-odd-rounded"), + pytest.param((1000, 1400), 1, (500, 700), id="no-supersample-clamped"), + pytest.param((600, 800), 2, (300, 400), id="2x-small-not-enlarged"), + ], + ) + def test_supersampled_render_downsampled( + self, + tmp_path: Path, + in_size: tuple[int, int], + supersample: int, + expected_size: tuple[int, int], + ) -> None: + """ + GIVEN: + - A rendered page image made at a supersampling factor + WHEN: + - It is encoded as a thumbnail with that factor + THEN: + - It is downsampled by the factor before the 500x5000 clamp + """ + png_path = tmp_path / "in.png" + Image.new("RGB", in_size, (255, 255, 255)).save(png_path) + out_path = tmp_path / "out.webp" + + encode_thumbnail_webp(png_path, out_path, supersample=supersample) + + with Image.open(out_path) as im: + assert im.size == expected_size + @pytest.mark.parametrize( "error", [