Chore: Keep PDF thumbnails at 500px wide for small and odd-sized pages

The new pdftoppm thumbnail path capped its computed DPI at 72, on the
assumption that the old "-scale 500x5000>" never enlarged a page past its
natural size. The old pipeline actually rendered at 300 DPI before
shrinking, so any page wider than about 120pt used to produce a 500px wide
thumbnail, while the 72 DPI cap made A5 pages, receipts and other small
documents come out noticeably narrower. Rounding the DPI to the nearest
integer could also land a few pixels short of 500px (a landscape Letter page
rendered 495px wide at 45 DPI), and the Pillow clamp only ever shrinks.

The DPI is now capped at 300 to match the old render density, and rounded
up so the render always lands at or just above the target, letting the
shrink-only clamp trim it to exactly 500px. The page size helper also no
longer logs a full traceback for every encrypted or corrupt PDF, matching
the page count helper.
This commit is contained in:
stumpylog committed 2026-10-07 12:46:49 -07:00
1 parent ae3d8680fa
commit 1a9b08394c
3 files changed
+68 -14

No files matched your search

+18 -8
View File
@@ -1,6 +1,7 @@
from __future__ import annotations
import logging
import math
import mimetypes
import os
import shutil
@@ -131,6 +132,8 @@ _THUMBNAIL_MAX_WIDTH = 500
_THUMBNAIL_MAX_HEIGHT = 5000
# Used only when the page geometry cannot be read
_THUMBNAIL_FALLBACK_DPI = 150
# Matches the density the thumbnail was previously rendered at before scaling
_THUMBNAIL_MAX_DPI = 300
def rasterize_pdf_page_to_png(
@@ -183,9 +186,9 @@ def encode_thumbnail_webp(
"""
Flattens any alpha onto white and saves the image as WebP.
max_width/max_height are only a safety-net clamp for DPI rounding, since
the render is already sized by the computed DPI. The image is never
enlarged, matching the previous "-scale WxH>" behavior.
max_width/max_height trim the render to the exact thumbnail size, since
the computed DPI is rounded up and lands at or slightly above it. The
image is never enlarged, matching the previous "-scale WxH>" behavior.
"""
from PIL import Image
@@ -205,8 +208,9 @@ def encode_thumbnail_webp(
def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> int:
"""
Computes the DPI which renders the first page of the PDF to fit within the
thumbnail size in one pass, never above the page's natural 72 DPI size.
Computes the DPI which renders the first page of the PDF at or just above
the thumbnail size in one pass, never above the 300 DPI the thumbnail was
previously rendered at before being scaled down.
"""
from paperless.parsers.utils import get_pdf_first_page_size_points
@@ -221,9 +225,15 @@ def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> int:
width_pts, height_pts = size
dpi_for_width = _THUMBNAIL_MAX_WIDTH * 72 / width_pts
dpi_for_height = _THUMBNAIL_MAX_HEIGHT * 72 / height_pts
# Capping at 72 (1px per point) keeps the shrink-only behavior: a page
# already smaller than the thumbnail size is never enlarged
return max(1, round(min(72, dpi_for_width, dpi_for_height)))
# The old pipeline rendered at 300 DPI and then shrank to fit, so only
# pages too small to reach the thumbnail size even at 300 DPI end up
# smaller than it. Rounding up keeps the render at or above the target,
# so the shrink-only clamp in encode_thumbnail_webp trims it to exactly
# the thumbnail size instead of leaving it a few pixels short.
return max(
1,
math.ceil(min(_THUMBNAIL_MAX_DPI, dpi_for_width, dpi_for_height)),
)
def _render_pdf_thumbnail(
+48 -4
View File
@@ -13,6 +13,7 @@ from documents.parsers import encode_thumbnail_webp
from documents.parsers import get_default_file_extension
from documents.parsers import get_supported_file_extensions
from documents.parsers import is_file_ext_supported
from documents.parsers import make_thumbnail_from_pdf
from documents.parsers import rasterize_pdf_page_to_png
from paperless.parsers.registry import get_parser_registry
from paperless.parsers.registry import reset_parser_registry
@@ -140,9 +141,10 @@ class TestComputeThumbnailDpi:
("size", "expected_dpi"),
[
pytest.param((612.0, 792.0), 59, id="letter-width-bound"),
pytest.param((792.0, 612.0), 45, id="landscape-width-bound"),
pytest.param((792.0, 612.0), 46, id="landscape-rounded-up"),
pytest.param((612.0, 100000.0), 4, id="tall-strip-height-bound"),
pytest.param((200.0, 300.0), 72, id="small-page-never-enlarged"),
pytest.param((200.0, 300.0), 180, id="small-page-width-bound"),
pytest.param((72.0, 72.0), 300, id="tiny-page-capped-at-300"),
pytest.param((1000000.0, 1000000.0), 1, id="huge-page-minimum-one"),
pytest.param(None, 150, id="unreadable-geometry-fallback"),
],
@@ -161,8 +163,9 @@ class TestComputeThumbnailDpi:
WHEN:
- The thumbnail DPI is computed
THEN:
- The DPI fits the page into 500x5000 without ever exceeding 72,
is at least 1, and falls back to 150 when the size is unknown
- The DPI is rounded up so the render reaches 500x5000, never
exceeds 300, is at least 1, and falls back to 150 when the
size is unknown
"""
mocker.patch(
"paperless.parsers.utils.get_pdf_first_page_size_points",
@@ -171,6 +174,47 @@ class TestComputeThumbnailDpi:
assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected_dpi
class TestMakeThumbnailFromPdf:
@pytest.mark.parametrize(
("page_size", "expected_width"),
[
pytest.param((612, 792), 500, id="letter"),
pytest.param((792, 612), 500, id="landscape-letter"),
pytest.param((595, 842), 500, id="a4"),
pytest.param((200, 300), 500, id="small-page"),
pytest.param((72, 72), 300, id="tiny-page-capped"),
],
)
def test_thumbnail_width(
self,
tmp_path: Path,
page_size: tuple[int, int],
expected_width: int,
) -> None:
"""
GIVEN:
- A PDF whose first page has the given size in points
WHEN:
- A thumbnail is made from it
THEN:
- The WebP thumbnail is exactly 500px wide, unless the page is
too small to reach that even at 300 DPI
"""
pdf = pikepdf.new()
pdf.add_blank_page(page_size=page_size)
pdf_path = tmp_path / "in.pdf"
pdf.save(pdf_path)
work_dir = tmp_path / "work"
work_dir.mkdir()
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
assert thumb == work_dir / "convert.webp"
with Image.open(thumb) as im:
assert im.format == "WEBP"
assert im.width == expected_width
class TestRasterizePdfPageToPng:
@staticmethod
def _write_pdf(
+2 -2
View File
@@ -311,8 +311,8 @@ def get_pdf_first_page_size_points(
if page.rotation in (90, 270):
width, height = height, width
return width, height
except Exception:
_log.warning("Could not determine PDF page size for %s", path, exc_info=True)
except Exception as e:
_log.warning("Could not determine PDF page size for %s: %s", path, e)
return None