Chore: Trim thumbnail comments and docstrings

The comments and docstrings around PDF thumbnail generation were wordier than the code they describe. This cuts them back to the essential facts.
This commit is contained in:
stumpylog committed 2026-10-07 14:56:47 -07:00
1 parent f1e85fff3e
commit 2a8df9dca0
7 files changed
+43 -88

No files matched your search

+4 -6
View File
@@ -1317,17 +1317,15 @@ valid crontab(5) expression describing when to run.
!!! warning
This option is deprecated and has no effect. It only applied to PDF
thumbnail generation via ImageMagick, which no longer happens. It will be
removed in a future release and can be removed from your configuration now.
Deprecated and has no effect, since PDF thumbnails no longer use
ImageMagick. It will be removed in a future release.
#### [`PAPERLESS_CONVERT_TMPDIR=<path>`](#PAPERLESS_CONVERT_TMPDIR) {#PAPERLESS_CONVERT_TMPDIR}
!!! warning
This option is deprecated and has no effect. It only applied to PDF
thumbnail generation via ImageMagick, which no longer happens. It will be
removed in a future release and can be removed from your configuration now.
Deprecated and has no effect, since PDF thumbnails no longer use
ImageMagick. It will be removed in a future release.
#### [`PAPERLESS_APPS=<string>`](#PAPERLESS_APPS) {#PAPERLESS_APPS}
+1 -4
View File
@@ -417,10 +417,7 @@ to a positive number to enable polling and disable native filesystem notificatio
`ExecStart=/opt/paperless/.local/bin/celery --app paperless worker --loglevel INFO`
12. Harden ImageMagick by disabling formats that Paperless-ngx does not use.
Most distributions disable PDF processing by default, since PDF documents
can contain malware. Paperless-ngx no longer passes PDF documents to
ImageMagick, so enabling PDF processing is not required and should be left
disabled.
PDF processing is not needed and should stay disabled.
Configure the active ImageMagick policy file (commonly
`/etc/ImageMagick-6/policy.xml` or `/etc/ImageMagick-7/policy.xml`) and
+9 -21
View File
@@ -79,12 +79,9 @@ _THUMBNAIL_MAX_WIDTH = 500
_THUMBNAIL_MAX_HEIGHT = 5000
# Used only when the page geometry cannot be read
_THUMBNAIL_FALLBACK_DPI = 150
# Cap on the target density, applied before supersampling, so a tiny page is
# not blown up past the size a plain 300 DPI render would give
# Applied before supersampling, so tiny pages are not enlarged
_THUMBNAIL_MAX_DPI = 300
# Pages with known geometry are rendered at this multiple of the computed DPI
# and downsampled with Lanczos, which keeps text noticeably crisper than
# rasterizing straight at the target size
# Rendering at a multiple and downsampling keeps text crisper
_THUMBNAIL_SUPERSAMPLE = 2
@@ -96,12 +93,10 @@ def rasterize_pdf_page_to_png(
logging_group=None,
) -> None:
"""
Rasterizes page 1 of a PDF to a PNG at out_path via pdftoppm (Poppler),
at the given DPI. pdftoppm honors the page's /Rotate on its own.
Rasterizes the first page of a PDF to a PNG with pdftoppm.
"""
# -singlefile stops pdftoppm appending a page number to the output name,
# and with -png it appends ".png" itself, so it is given the path without
# its suffix to write exactly out_path
# -singlefile drops the page number and -png appends ".png", so pass the
# path without its suffix
args = [
"pdftoppm",
"-f",
@@ -134,12 +129,7 @@ def encode_thumbnail_webp(
supersample: int = 1,
) -> None:
"""
Flattens any alpha onto white and saves the image as WebP.
A render made at supersample times the target density is first
downsampled by that factor with Lanczos. The result is then trimmed to the
thumbnail size, since the computed DPI is rounded up and lands at or
slightly above it. The image is never enlarged.
Flattens alpha onto white, undoes supersampling, shrinks to fit and saves as WebP.
"""
from PIL import Image
@@ -168,8 +158,8 @@ def encode_thumbnail_webp(
def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> tuple[int, int]:
"""
Returns (dpi, supersample): render at dpi * supersample, then downsample
by supersample. Unknown page geometry gives the fallback DPI, unsupersampled.
Returns (dpi, supersample). Unknown geometry is not supersampled, since
the render size cannot be bounded.
"""
from paperless.parsers.utils import get_pdf_first_page_size_points
@@ -211,9 +201,7 @@ def _render_pdf_thumbnail(
def _repair_pdf_with_qpdf(in_path: Path, out_path: Path) -> None:
# qpdf rewrites in place, so work on a copy and leave the original alone.
# qpdf exits 3 when it had to repair the file, which is the expected
# outcome here, so warnings must not count as failure.
# qpdf exits 3 after a repair; --warning-exit-0 keeps that from failing
try:
shutil.copy(in_path, out_path)
run_subprocess(
+18 -37
View File
@@ -164,16 +164,11 @@ class TestComputeThumbnailDpi:
) -> None:
"""
GIVEN:
- A PDF whose first page has the given size in points (or whose
geometry cannot be read)
- A first page of the given size, or unreadable geometry
WHEN:
- The thumbnail DPI is computed
THEN:
- The DPI is rounded up so the render reaches 500x5000, is capped
at 300 and floored at 1, and is supersampled 2x except at the
1 DPI floor
- An unknown size gives the plain 150 DPI fallback without
supersampling
- The expected DPI and supersample factor are returned
"""
mocker.patch(
"paperless.parsers.utils.get_pdf_first_page_size_points",
@@ -214,13 +209,11 @@ class TestMakeThumbnailFromPdf:
) -> None:
"""
GIVEN:
- A PDF whose page geometry is either readable or not
- Readable or unreadable page geometry
WHEN:
- A thumbnail is made from it
- A thumbnail is made
THEN:
- The page is rasterized at twice the computed DPI when the
geometry is known, and at the plain 150 DPI fallback otherwise
- The encode step is told the matching downsample factor
- Rasterize and encode get the matching DPI and supersample factor
"""
mocker.patch(
"paperless.parsers.utils.get_pdf_first_page_size_points",
@@ -257,8 +250,7 @@ class TestMakeThumbnailFromPdf:
WHEN:
- A thumbnail is made from it
THEN:
- The WebP thumbnail is exactly 500px wide, unless the page is
too small to reach that even at 300 DPI
- The thumbnail has the expected width
"""
pdf_path = self._write_blank_pdf(tmp_path / "in.pdf", page_size)
@@ -272,8 +264,7 @@ class TestMakeThumbnailFromPdf:
@classmethod
def _write_pdf_without_xref(cls, path: Path) -> Path:
"""
Writes a valid one page PDF, then cuts off its cross reference table
and trailer, which poppler cannot recover from but qpdf can.
Cuts off the xref and trailer, which pdftoppm cannot recover from but qpdf can.
"""
cls._write_blank_pdf(path)
data = path.read_bytes()
@@ -287,15 +278,12 @@ class TestMakeThumbnailFromPdf:
) -> None:
"""
GIVEN:
- A PDF without a cross reference table or trailer, which
pdftoppm cannot render but qpdf can repair
- A PDF with its xref table and trailer cut off
WHEN:
- A thumbnail is made from it
THEN:
- The unrepaired file really cannot be rasterized
- The thumbnail comes from the repaired copy and is a rendered
500px wide page, not the default placeholder
- The original file is left untouched
- The thumbnail is rendered from a qpdf repaired copy
- The original file is unchanged
"""
pdf_path = self._write_pdf_without_xref(tmp_path / "broken.pdf")
original_bytes = pdf_path.read_bytes()
@@ -327,13 +315,11 @@ class TestMakeThumbnailFromPdf:
) -> None:
"""
GIVEN:
- A PDF which cannot be rasterized, either because qpdf cannot
repair it or because the repaired copy still cannot be rendered
- A PDF that cannot be rendered, even after qpdf repair
WHEN:
- A thumbnail is made from it
THEN:
- The result is a copy of the default thumbnail, so the caller
can move it without consuming the shared resource
- A copy of the default thumbnail is returned
"""
mocker.patch(
"documents.parsers.rasterize_pdf_page_to_png",
@@ -385,13 +371,11 @@ class TestRasterizePdfPageToPng:
) -> None:
"""
GIVEN:
- A two page PDF whose first page is 144x72 points, optionally
cropped or rotated
- A two page PDF, first page optionally cropped or rotated
WHEN:
- The first page is rasterized at 72 DPI
THEN:
- Exactly the requested output path is written, and the image
matches the first page's cropped, rotated size
- Only out_path is written, sized to the first page's crop and rotation
"""
pdf_path = self._write_pdf(
tmp_path / "in.pdf",
@@ -481,13 +465,11 @@ class TestEncodeThumbnailWebp:
) -> None:
"""
GIVEN:
- A rendered page image of the given size, made at the given
supersampling factor
- A rendered image and its supersample factor
WHEN:
- It is encoded as a thumbnail with that factor
THEN:
- It is downsampled by the factor, then shrunk to fit 500x5000
keeping aspect ratio, and never enlarged
- It is downsampled, fit within 500x5000 and never enlarged
"""
png_path = tmp_path / "in.png"
Image.new("RGB", in_size, (255, 255, 255)).save(png_path)
@@ -513,12 +495,11 @@ class TestEncodeThumbnailWebp:
) -> None:
"""
GIVEN:
- Opening the rendered image fails with an OSError or a
DecompressionBombError
- Opening the rendered image fails
WHEN:
- It is encoded as a thumbnail
THEN:
- A ParseError is raised so the default thumbnail is used
- A ParseError is raised
"""
png_path = tmp_path / "in.png"
Image.new("RGB", (10, 10)).save(png_path)
+4 -9
View File
@@ -271,16 +271,11 @@ def get_pdf_first_page_size_points(
) -> tuple[float, float] | None:
"""Return the first page's (width, height) in PDF points, post-rotation.
Uses ``page.cropbox``, which falls back to the MediaBox when the page has
no ``/CropBox`` of its own. This must match the box the renderer is told
to use (pdftoppm's ``-cropbox`` flag), or a DPI computed from it will
target the wrong box's dimensions.
Uses the CropBox (MediaBox if absent), which must match pdftoppm's
``-cropbox`` or the computed DPI targets the wrong box.
Width and height are swapped when the page's effective rotation is 90 or
270, since that is the orientation the page is rendered in.
``page.rotation`` resolves a ``/Rotate`` inherited from an ancestor
``/Pages`` node and normalizes it to ``[0, 360)``, which a raw
``/Rotate`` lookup on the page dictionary would not.
Swaps width and height for 90/270 rotation. ``page.rotation`` resolves
inherited and negative ``/Rotate`` values, a raw lookup does not.
Parameters
----------
@@ -297,12 +297,11 @@ class TestGetThumbnail:
) -> None:
"""
GIVEN:
- A PDF whose first page is wider than the thumbnail width
- A portrait or landscape PDF
WHEN:
- A thumbnail is generated
THEN:
- The thumbnail is a WebP, 500px wide, keeping the page's aspect
ratio (within rounding of the DPI computation)
- A 500px wide WebP keeping the page's aspect ratio
"""
thumb = tesseract_parser.get_thumbnail(
tesseract_samples_dir / filename,
@@ -325,8 +324,7 @@ class TestGetThumbnail:
WHEN:
- A thumbnail is generated
THEN:
- The PDF is repaired with qpdf and rasterized again, producing a
real thumbnail rather than the default placeholder
- The PDF is repaired with qpdf and a real thumbnail is rendered
"""
original = tesseract_samples_dir / "simple-digital.pdf"
+4 -6
View File
@@ -139,12 +139,11 @@ class TestGetPdfFirstPageSizePoints:
def test_inherited_rotation_swaps_dimensions(self, tmp_path: Path) -> None:
"""
GIVEN:
- A portrait PDF page whose /Rotate 90 is set on the /Pages node,
not on the page itself
- A page inheriting /Rotate 90 from the /Pages node
WHEN:
- The first page size is requested
THEN:
- The inherited rotation is honored and width/height are swapped
- Width and height are swapped
"""
pdf_path = self._write_pdf(tmp_path / "inherited.pdf", inherited_rotate=90)
assert get_pdf_first_page_size_points(pdf_path) == (800.0, 600.0)
@@ -156,8 +155,7 @@ class TestGetPdfFirstPageSizePoints:
WHEN:
- The first page size is requested
THEN:
- The CropBox dimensions are returned, matching what pdftoppm
renders with -cropbox
- The CropBox size is returned
"""
pdf_path = self._write_pdf(
tmp_path / "cropped.pdf",
@@ -172,7 +170,7 @@ class TestGetPdfFirstPageSizePoints:
WHEN:
- The first page size is requested
THEN:
- None is returned instead of a size that would break DPI math
- None is returned
"""
pdf_path = self._write_pdf(
tmp_path / "degenerate.pdf",