From 2a8df9dca0540f0c12e48178e93aeb997d5b76b6 Mon Sep 17 00:00:00 2001 From: stumpylog <797416+stumpylog@users.noreply.github.com> Date: Wed, 7 Oct 2026 14:56:47 -0700 Subject: [PATCH] Chore: Trim thumbnail comments and docstrings The comments and docstrings around PDF thumbnail generation were wordier than the code they describe. This cuts them back to the essential facts. --- docs/configuration.md | 10 ++-- docs/setup.md | 5 +- src/documents/parsers.py | 30 +++------- src/documents/tests/test_parsers.py | 55 ++++++------------- src/paperless/parsers/utils.py | 13 ++--- .../tests/parsers/test_tesseract_parser.py | 8 +-- src/paperless/tests/test_parser_utils.py | 10 ++-- 7 files changed, 43 insertions(+), 88 deletions(-) diff --git a/docs/configuration.md b/docs/configuration.md index b42397da1..80aecaa60 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -1317,17 +1317,15 @@ valid crontab(5) expression describing when to run. !!! warning - This option is deprecated and has no effect. It only applied to PDF - thumbnail generation via ImageMagick, which no longer happens. It will be - removed in a future release and can be removed from your configuration now. + Deprecated and has no effect, since PDF thumbnails no longer use + ImageMagick. It will be removed in a future release. #### [`PAPERLESS_CONVERT_TMPDIR=`](#PAPERLESS_CONVERT_TMPDIR) {#PAPERLESS_CONVERT_TMPDIR} !!! warning - This option is deprecated and has no effect. It only applied to PDF - thumbnail generation via ImageMagick, which no longer happens. It will be - removed in a future release and can be removed from your configuration now. + Deprecated and has no effect, since PDF thumbnails no longer use + ImageMagick. It will be removed in a future release. #### [`PAPERLESS_APPS=`](#PAPERLESS_APPS) {#PAPERLESS_APPS} diff --git a/docs/setup.md b/docs/setup.md index ba6e90af3..514cf424c 100644 --- a/docs/setup.md +++ b/docs/setup.md @@ -417,10 +417,7 @@ to a positive number to enable polling and disable native filesystem notificatio `ExecStart=/opt/paperless/.local/bin/celery --app paperless worker --loglevel INFO` 12. Harden ImageMagick by disabling formats that Paperless-ngx does not use. - Most distributions disable PDF processing by default, since PDF documents - can contain malware. Paperless-ngx no longer passes PDF documents to - ImageMagick, so enabling PDF processing is not required and should be left - disabled. + PDF processing is not needed and should stay disabled. Configure the active ImageMagick policy file (commonly `/etc/ImageMagick-6/policy.xml` or `/etc/ImageMagick-7/policy.xml`) and diff --git a/src/documents/parsers.py b/src/documents/parsers.py index 2e26eb4fc..d6a08ca9f 100644 --- a/src/documents/parsers.py +++ b/src/documents/parsers.py @@ -79,12 +79,9 @@ _THUMBNAIL_MAX_WIDTH = 500 _THUMBNAIL_MAX_HEIGHT = 5000 # Used only when the page geometry cannot be read _THUMBNAIL_FALLBACK_DPI = 150 -# Cap on the target density, applied before supersampling, so a tiny page is -# not blown up past the size a plain 300 DPI render would give +# Applied before supersampling, so tiny pages are not enlarged _THUMBNAIL_MAX_DPI = 300 -# Pages with known geometry are rendered at this multiple of the computed DPI -# and downsampled with Lanczos, which keeps text noticeably crisper than -# rasterizing straight at the target size +# Rendering at a multiple and downsampling keeps text crisper _THUMBNAIL_SUPERSAMPLE = 2 @@ -96,12 +93,10 @@ def rasterize_pdf_page_to_png( logging_group=None, ) -> None: """ - Rasterizes page 1 of a PDF to a PNG at out_path via pdftoppm (Poppler), - at the given DPI. pdftoppm honors the page's /Rotate on its own. + Rasterizes the first page of a PDF to a PNG with pdftoppm. """ - # -singlefile stops pdftoppm appending a page number to the output name, - # and with -png it appends ".png" itself, so it is given the path without - # its suffix to write exactly out_path + # -singlefile drops the page number and -png appends ".png", so pass the + # path without its suffix args = [ "pdftoppm", "-f", @@ -134,12 +129,7 @@ def encode_thumbnail_webp( supersample: int = 1, ) -> None: """ - Flattens any alpha onto white and saves the image as WebP. - - A render made at supersample times the target density is first - downsampled by that factor with Lanczos. The result is then trimmed to the - thumbnail size, since the computed DPI is rounded up and lands at or - slightly above it. The image is never enlarged. + Flattens alpha onto white, undoes supersampling, shrinks to fit and saves as WebP. """ from PIL import Image @@ -168,8 +158,8 @@ def encode_thumbnail_webp( def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> tuple[int, int]: """ - Returns (dpi, supersample): render at dpi * supersample, then downsample - by supersample. Unknown page geometry gives the fallback DPI, unsupersampled. + Returns (dpi, supersample). Unknown geometry is not supersampled, since + the render size cannot be bounded. """ from paperless.parsers.utils import get_pdf_first_page_size_points @@ -211,9 +201,7 @@ def _render_pdf_thumbnail( def _repair_pdf_with_qpdf(in_path: Path, out_path: Path) -> None: - # qpdf rewrites in place, so work on a copy and leave the original alone. - # qpdf exits 3 when it had to repair the file, which is the expected - # outcome here, so warnings must not count as failure. + # qpdf exits 3 after a repair; --warning-exit-0 keeps that from failing try: shutil.copy(in_path, out_path) run_subprocess( diff --git a/src/documents/tests/test_parsers.py b/src/documents/tests/test_parsers.py index 835aea2a5..9b8b47908 100644 --- a/src/documents/tests/test_parsers.py +++ b/src/documents/tests/test_parsers.py @@ -164,16 +164,11 @@ class TestComputeThumbnailDpi: ) -> None: """ GIVEN: - - A PDF whose first page has the given size in points (or whose - geometry cannot be read) + - A first page of the given size, or unreadable geometry WHEN: - The thumbnail DPI is computed THEN: - - The DPI is rounded up so the render reaches 500x5000, is capped - at 300 and floored at 1, and is supersampled 2x except at the - 1 DPI floor - - An unknown size gives the plain 150 DPI fallback without - supersampling + - The expected DPI and supersample factor are returned """ mocker.patch( "paperless.parsers.utils.get_pdf_first_page_size_points", @@ -214,13 +209,11 @@ class TestMakeThumbnailFromPdf: ) -> None: """ GIVEN: - - A PDF whose page geometry is either readable or not + - Readable or unreadable page geometry WHEN: - - A thumbnail is made from it + - A thumbnail is made THEN: - - The page is rasterized at twice the computed DPI when the - geometry is known, and at the plain 150 DPI fallback otherwise - - The encode step is told the matching downsample factor + - Rasterize and encode get the matching DPI and supersample factor """ mocker.patch( "paperless.parsers.utils.get_pdf_first_page_size_points", @@ -257,8 +250,7 @@ class TestMakeThumbnailFromPdf: WHEN: - A thumbnail is made from it THEN: - - The WebP thumbnail is exactly 500px wide, unless the page is - too small to reach that even at 300 DPI + - The thumbnail has the expected width """ pdf_path = self._write_blank_pdf(tmp_path / "in.pdf", page_size) @@ -272,8 +264,7 @@ class TestMakeThumbnailFromPdf: @classmethod def _write_pdf_without_xref(cls, path: Path) -> Path: """ - Writes a valid one page PDF, then cuts off its cross reference table - and trailer, which poppler cannot recover from but qpdf can. + Cuts off the xref and trailer, which pdftoppm cannot recover from but qpdf can. """ cls._write_blank_pdf(path) data = path.read_bytes() @@ -287,15 +278,12 @@ class TestMakeThumbnailFromPdf: ) -> None: """ GIVEN: - - A PDF without a cross reference table or trailer, which - pdftoppm cannot render but qpdf can repair + - A PDF with its xref table and trailer cut off WHEN: - A thumbnail is made from it THEN: - - The unrepaired file really cannot be rasterized - - The thumbnail comes from the repaired copy and is a rendered - 500px wide page, not the default placeholder - - The original file is left untouched + - The thumbnail is rendered from a qpdf repaired copy + - The original file is unchanged """ pdf_path = self._write_pdf_without_xref(tmp_path / "broken.pdf") original_bytes = pdf_path.read_bytes() @@ -327,13 +315,11 @@ class TestMakeThumbnailFromPdf: ) -> None: """ GIVEN: - - A PDF which cannot be rasterized, either because qpdf cannot - repair it or because the repaired copy still cannot be rendered + - A PDF that cannot be rendered, even after qpdf repair WHEN: - A thumbnail is made from it THEN: - - The result is a copy of the default thumbnail, so the caller - can move it without consuming the shared resource + - A copy of the default thumbnail is returned """ mocker.patch( "documents.parsers.rasterize_pdf_page_to_png", @@ -385,13 +371,11 @@ class TestRasterizePdfPageToPng: ) -> None: """ GIVEN: - - A two page PDF whose first page is 144x72 points, optionally - cropped or rotated + - A two page PDF, first page optionally cropped or rotated WHEN: - The first page is rasterized at 72 DPI THEN: - - Exactly the requested output path is written, and the image - matches the first page's cropped, rotated size + - Only out_path is written, sized to the first page's crop and rotation """ pdf_path = self._write_pdf( tmp_path / "in.pdf", @@ -481,13 +465,11 @@ class TestEncodeThumbnailWebp: ) -> None: """ GIVEN: - - A rendered page image of the given size, made at the given - supersampling factor + - A rendered image and its supersample factor WHEN: - It is encoded as a thumbnail with that factor THEN: - - It is downsampled by the factor, then shrunk to fit 500x5000 - keeping aspect ratio, and never enlarged + - It is downsampled, fit within 500x5000 and never enlarged """ png_path = tmp_path / "in.png" Image.new("RGB", in_size, (255, 255, 255)).save(png_path) @@ -513,12 +495,11 @@ class TestEncodeThumbnailWebp: ) -> None: """ GIVEN: - - Opening the rendered image fails with an OSError or a - DecompressionBombError + - Opening the rendered image fails WHEN: - It is encoded as a thumbnail THEN: - - A ParseError is raised so the default thumbnail is used + - A ParseError is raised """ png_path = tmp_path / "in.png" Image.new("RGB", (10, 10)).save(png_path) diff --git a/src/paperless/parsers/utils.py b/src/paperless/parsers/utils.py index a02ac1ecb..f9e299d21 100644 --- a/src/paperless/parsers/utils.py +++ b/src/paperless/parsers/utils.py @@ -271,16 +271,11 @@ def get_pdf_first_page_size_points( ) -> tuple[float, float] | None: """Return the first page's (width, height) in PDF points, post-rotation. - Uses ``page.cropbox``, which falls back to the MediaBox when the page has - no ``/CropBox`` of its own. This must match the box the renderer is told - to use (pdftoppm's ``-cropbox`` flag), or a DPI computed from it will - target the wrong box's dimensions. + Uses the CropBox (MediaBox if absent), which must match pdftoppm's + ``-cropbox`` or the computed DPI targets the wrong box. - Width and height are swapped when the page's effective rotation is 90 or - 270, since that is the orientation the page is rendered in. - ``page.rotation`` resolves a ``/Rotate`` inherited from an ancestor - ``/Pages`` node and normalizes it to ``[0, 360)``, which a raw - ``/Rotate`` lookup on the page dictionary would not. + Swaps width and height for 90/270 rotation. ``page.rotation`` resolves + inherited and negative ``/Rotate`` values, a raw lookup does not. Parameters ---------- diff --git a/src/paperless/tests/parsers/test_tesseract_parser.py b/src/paperless/tests/parsers/test_tesseract_parser.py index c47c5a2f9..f057ad086 100644 --- a/src/paperless/tests/parsers/test_tesseract_parser.py +++ b/src/paperless/tests/parsers/test_tesseract_parser.py @@ -297,12 +297,11 @@ class TestGetThumbnail: ) -> None: """ GIVEN: - - A PDF whose first page is wider than the thumbnail width + - A portrait or landscape PDF WHEN: - A thumbnail is generated THEN: - - The thumbnail is a WebP, 500px wide, keeping the page's aspect - ratio (within rounding of the DPI computation) + - A 500px wide WebP keeping the page's aspect ratio """ thumb = tesseract_parser.get_thumbnail( tesseract_samples_dir / filename, @@ -325,8 +324,7 @@ class TestGetThumbnail: WHEN: - A thumbnail is generated THEN: - - The PDF is repaired with qpdf and rasterized again, producing a - real thumbnail rather than the default placeholder + - The PDF is repaired with qpdf and a real thumbnail is rendered """ original = tesseract_samples_dir / "simple-digital.pdf" diff --git a/src/paperless/tests/test_parser_utils.py b/src/paperless/tests/test_parser_utils.py index 2205a008d..4cd67297d 100644 --- a/src/paperless/tests/test_parser_utils.py +++ b/src/paperless/tests/test_parser_utils.py @@ -139,12 +139,11 @@ class TestGetPdfFirstPageSizePoints: def test_inherited_rotation_swaps_dimensions(self, tmp_path: Path) -> None: """ GIVEN: - - A portrait PDF page whose /Rotate 90 is set on the /Pages node, - not on the page itself + - A page inheriting /Rotate 90 from the /Pages node WHEN: - The first page size is requested THEN: - - The inherited rotation is honored and width/height are swapped + - Width and height are swapped """ pdf_path = self._write_pdf(tmp_path / "inherited.pdf", inherited_rotate=90) assert get_pdf_first_page_size_points(pdf_path) == (800.0, 600.0) @@ -156,8 +155,7 @@ class TestGetPdfFirstPageSizePoints: WHEN: - The first page size is requested THEN: - - The CropBox dimensions are returned, matching what pdftoppm - renders with -cropbox + - The CropBox size is returned """ pdf_path = self._write_pdf( tmp_path / "cropped.pdf", @@ -172,7 +170,7 @@ class TestGetPdfFirstPageSizePoints: WHEN: - The first page size is requested THEN: - - None is returned instead of a size that would break DPI math + - None is returned """ pdf_path = self._write_pdf( tmp_path / "degenerate.pdf",