From 26efad77ac8a04f93bd9202ba3b3d6bb8c761092 Mon Sep 17 00:00:00 2001 From: stumpylog <797416+stumpylog@users.noreply.github.com> Date: Wed, 7 Oct 2026 14:30:50 -0700 Subject: [PATCH] Chore: Supersample the PDF thumbnail render 2x Rendering page 1 straight at the thumbnail density left text slightly soft, since pdftoppm antialiases at the final size. The page is now rendered at twice the computed DPI and downsampled by that factor with Lanczos before the existing 500x5000 clamp, which gives crisper text at about the same file size for a negligible memory cost. When the page geometry cannot be read, the fixed 150 DPI fallback is kept as a single unsupersampled render, because doubling a render that is not bounded by the thumbnail size would quadruple its pixel count. The qpdf repair path shares the same render step and gets the same behavior. --- src/documents/parsers.py | 47 ++++++++++---- src/documents/tests/test_parsers.py | 96 +++++++++++++++++++++++++---- 2 files changed, 119 insertions(+), 24 deletions(-) diff --git a/src/documents/parsers.py b/src/documents/parsers.py index bce2dcc88..75256bbed 100644 --- a/src/documents/parsers.py +++ b/src/documents/parsers.py @@ -81,6 +81,10 @@ _THUMBNAIL_MAX_HEIGHT = 5000 _THUMBNAIL_FALLBACK_DPI = 150 # Matches the density the thumbnail was previously rendered at before scaling _THUMBNAIL_MAX_DPI = 300 +# Pages with known geometry are rendered at this multiple of the computed DPI +# and downsampled with Lanczos, which keeps text noticeably crisper than +# rasterizing straight at the target size +_THUMBNAIL_SUPERSAMPLE = 2 def rasterize_pdf_page_to_png( @@ -129,13 +133,16 @@ def encode_thumbnail_webp( *, max_width: int = _THUMBNAIL_MAX_WIDTH, max_height: int = _THUMBNAIL_MAX_HEIGHT, + supersample: int = 1, ) -> None: """ Flattens any alpha onto white and saves the image as WebP. - max_width/max_height trim the render to the exact thumbnail size, since - the computed DPI is rounded up and lands at or slightly above it. The - image is never enlarged, matching the previous "-scale WxH>" behavior. + A render made at supersample times the target density is first + downsampled by that factor with Lanczos. max_width/max_height then trim + the result to the exact thumbnail size, since the computed DPI is rounded + up and lands at or slightly above it. The image is never enlarged, + matching the previous "-scale WxH>" behavior. """ from PIL import Image @@ -147,17 +154,32 @@ def encode_thumbnail_webp( else: flattened = im.convert("RGB") + if supersample > 1: + flattened = flattened.resize( + ( + max(1, round(flattened.width / supersample)), + max(1, round(flattened.height / supersample)), + ), + Image.Resampling.LANCZOS, + ) + flattened.thumbnail((max_width, max_height)) flattened.save(out_path, format="WEBP") except (OSError, Image.DecompressionBombError) as e: raise ParseError(f"Unable to encode thumbnail from {png_path}") from e -def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> int: +def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> tuple[int, int]: """ - Computes the DPI which renders the first page of the PDF at or just above - the thumbnail size in one pass, never above the 300 DPI the thumbnail was - previously rendered at before being scaled down. + Computes the DPI at which the first page of the PDF reaches at or just + above the thumbnail size, never above the 300 DPI the thumbnail was + previously rendered at before being scaled down, and the supersampling + factor to render with. + + Returns (dpi, supersample). The page is rendered at supersample * dpi and + downsampled by supersample afterwards. When the page geometry cannot be + read the fixed fallback DPI is used without supersampling, as that render + is not bounded by the thumbnail size. """ from paperless.parsers.utils import get_pdf_first_page_size_points @@ -167,7 +189,7 @@ def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> int: "Could not read PDF page size, using fallback DPI", extra={"group": logging_group}, ) - return _THUMBNAIL_FALLBACK_DPI + return _THUMBNAIL_FALLBACK_DPI, 1 width_pts, height_pts = size dpi_for_width = _THUMBNAIL_MAX_WIDTH * 72 / width_pts @@ -177,10 +199,11 @@ def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> int: # smaller than it. Rounding up keeps the render at or above the target, # so the shrink-only clamp in encode_thumbnail_webp trims it to exactly # the thumbnail size instead of leaving it a few pixels short. - return max( + dpi = max( 1, math.ceil(min(_THUMBNAIL_MAX_DPI, dpi_for_width, dpi_for_height)), ) + return dpi, _THUMBNAIL_SUPERSAMPLE def _render_pdf_thumbnail( @@ -189,15 +212,15 @@ def _render_pdf_thumbnail( out_path: Path, logging_group=None, ) -> None: - dpi = _compute_thumbnail_dpi(in_path, logging_group=logging_group) + dpi, supersample = _compute_thumbnail_dpi(in_path, logging_group=logging_group) rasterize_pdf_page_to_png( in_path, png_path, - dpi=dpi, + dpi=dpi * supersample, use_cropbox=True, logging_group=logging_group, ) - encode_thumbnail_webp(png_path, out_path) + encode_thumbnail_webp(png_path, out_path, supersample=supersample) def make_thumbnail_from_pdf_qpdf_fallback( diff --git a/src/documents/tests/test_parsers.py b/src/documents/tests/test_parsers.py index 6ba10eebd..3531cec92 100644 --- a/src/documents/tests/test_parsers.py +++ b/src/documents/tests/test_parsers.py @@ -140,15 +140,15 @@ class TestParserAvailability: class TestComputeThumbnailDpi: @pytest.mark.parametrize( - ("size", "expected_dpi"), + ("size", "expected"), [ - pytest.param((612.0, 792.0), 59, id="letter-width-bound"), - pytest.param((792.0, 612.0), 46, id="landscape-rounded-up"), - pytest.param((612.0, 100000.0), 4, id="tall-strip-height-bound"), - pytest.param((200.0, 300.0), 180, id="small-page-width-bound"), - pytest.param((72.0, 72.0), 300, id="tiny-page-capped-at-300"), - pytest.param((1000000.0, 1000000.0), 1, id="huge-page-minimum-one"), - pytest.param(None, 150, id="unreadable-geometry-fallback"), + pytest.param((612.0, 792.0), (59, 2), id="letter-width-bound"), + pytest.param((792.0, 612.0), (46, 2), id="landscape-rounded-up"), + pytest.param((612.0, 100000.0), (4, 2), id="tall-strip-height-bound"), + pytest.param((200.0, 300.0), (180, 2), id="small-page-width-bound"), + pytest.param((72.0, 72.0), (300, 2), id="tiny-page-capped-at-300"), + pytest.param((1000000.0, 1000000.0), (1, 2), id="huge-page-minimum-one"), + pytest.param(None, (150, 1), id="unreadable-geometry-fallback"), ], ) def test_dpi_from_page_size( @@ -156,7 +156,7 @@ class TestComputeThumbnailDpi: mocker: MockerFixture, tmp_path: Path, size: tuple[float, float] | None, - expected_dpi: int, + expected: tuple[int, int], ) -> None: """ GIVEN: @@ -166,17 +166,56 @@ class TestComputeThumbnailDpi: - The thumbnail DPI is computed THEN: - The DPI is rounded up so the render reaches 500x5000, never - exceeds 300, is at least 1, and falls back to 150 when the - size is unknown + exceeds 300, is at least 1, and is supersampled 2x; an unknown + size gives the plain 150 DPI fallback without supersampling """ mocker.patch( "paperless.parsers.utils.get_pdf_first_page_size_points", return_value=size, ) - assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected_dpi + assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected class TestMakeThumbnailFromPdf: + @pytest.mark.parametrize( + ("size", "expected_dpi", "expected_supersample"), + [ + pytest.param((612.0, 792.0), 118, 2, id="known-geometry-2x"), + pytest.param(None, 150, 1, id="unreadable-geometry-plain-fallback"), + ], + ) + def test_render_dpi_requested( + self, + mocker: MockerFixture, + tmp_path: Path, + size: tuple[float, float] | None, + expected_dpi: int, + expected_supersample: int, + ) -> None: + """ + GIVEN: + - A PDF whose page geometry is either readable or not + WHEN: + - A thumbnail is made from it + THEN: + - The page is rasterized at twice the computed DPI when the + geometry is known, and at the plain 150 DPI fallback otherwise + - The encode step is told the matching downsample factor + """ + mocker.patch( + "paperless.parsers.utils.get_pdf_first_page_size_points", + return_value=size, + ) + rasterize = mocker.patch("documents.parsers.rasterize_pdf_page_to_png") + encode = mocker.patch("documents.parsers.encode_thumbnail_webp") + work_dir = tmp_path / "work" + work_dir.mkdir() + + make_thumbnail_from_pdf(tmp_path / "in.pdf", work_dir) + + assert rasterize.call_args.kwargs["dpi"] == expected_dpi + assert encode.call_args.kwargs["supersample"] == expected_supersample + @pytest.mark.parametrize( ("page_size", "expected_width"), [ @@ -445,6 +484,39 @@ class TestEncodeThumbnailWebp: with Image.open(out_path) as im: assert im.size == expected_size + @pytest.mark.parametrize( + ("in_size", "supersample", "expected_size"), + [ + pytest.param((1000, 1400), 2, (500, 700), id="2x-halved"), + pytest.param((1001, 1401), 2, (500, 700), id="2x-odd-rounded"), + pytest.param((1000, 1400), 1, (500, 700), id="no-supersample-clamped"), + pytest.param((600, 800), 2, (300, 400), id="2x-small-not-enlarged"), + ], + ) + def test_supersampled_render_downsampled( + self, + tmp_path: Path, + in_size: tuple[int, int], + supersample: int, + expected_size: tuple[int, int], + ) -> None: + """ + GIVEN: + - A rendered page image made at a supersampling factor + WHEN: + - It is encoded as a thumbnail with that factor + THEN: + - It is downsampled by the factor before the 500x5000 clamp + """ + png_path = tmp_path / "in.png" + Image.new("RGB", in_size, (255, 255, 255)).save(png_path) + out_path = tmp_path / "out.webp" + + encode_thumbnail_webp(png_path, out_path, supersample=supersample) + + with Image.open(out_path) as im: + assert im.size == expected_size + @pytest.mark.parametrize( "error", [