diff --git a/.github/workflows/ci-backend.yml b/.github/workflows/ci-backend.yml index 1e0ba23b1..331168ba2 100644 --- a/.github/workflows/ci-backend.yml +++ b/.github/workflows/ci-backend.yml @@ -111,7 +111,7 @@ jobs: timeout-minutes: 12 uses: $/.github/actions/apt-install with: - packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils + packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils qpdf - name: Configure ImageMagick run: | sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-7/policy.xml diff --git a/docs/setup.md b/docs/setup.md index e666cb58d..313ebf3d6 100644 --- a/docs/setup.md +++ b/docs/setup.md @@ -182,7 +182,7 @@ to a positive number to enable polling and disable native filesystem notificatio - `libpq-dev` for PostgreSQL - `libmagic-dev` for mime type detection - `mariadb-client` for MariaDB compile time - - `poppler-utils` for barcode detection + - `poppler-utils` for thumbnail generation and barcode detection Use this list for your preferred package management: @@ -419,8 +419,8 @@ to a positive number to enable polling and disable native filesystem notificatio 12. Configure ImageMagick to allow processing of PDF documents and disable formats that Paperless-ngx does not use. Most distributions disable PDF processing by default, since PDF documents can contain malware. If you - don't enable it, Paperless-ngx will fall back to Ghostscript for certain - steps such as thumbnail generation. + don't enable it, steps that still rely on ImageMagick, such as TIFF to + PDF conversion, may fail. PDF thumbnails no longer use ImageMagick. Configure the active ImageMagick policy file (commonly `/etc/ImageMagick-6/policy.xml` or `/etc/ImageMagick-7/policy.xml`) and diff --git a/src/documents/parsers.py b/src/documents/parsers.py index fe683be00..bce2dcc88 100644 --- a/src/documents/parsers.py +++ b/src/documents/parsers.py @@ -149,7 +149,7 @@ def encode_thumbnail_webp( flattened.thumbnail((max_width, max_height)) flattened.save(out_path, format="WEBP") - except OSError as e: + except (OSError, Image.DecompressionBombError) as e: raise ParseError(f"Unable to encode thumbnail from {png_path}") from e diff --git a/src/documents/tests/test_parsers.py b/src/documents/tests/test_parsers.py index 66c9a8f12..6ba10eebd 100644 --- a/src/documents/tests/test_parsers.py +++ b/src/documents/tests/test_parsers.py @@ -444,3 +444,32 @@ class TestEncodeThumbnailWebp: with Image.open(out_path) as im: assert im.size == expected_size + + @pytest.mark.parametrize( + "error", + [ + pytest.param(OSError("broken image"), id="os-error"), + pytest.param(Image.DecompressionBombError("too large"), id="bomb"), + ], + ) + def test_decode_failure_raises_parse_error( + self, + tmp_path: Path, + mocker: MockerFixture, + error: Exception, + ) -> None: + """ + GIVEN: + - Opening the rendered image fails with an OSError or a + DecompressionBombError + WHEN: + - It is encoded as a thumbnail + THEN: + - A ParseError is raised so the default thumbnail is used + """ + png_path = tmp_path / "in.png" + Image.new("RGB", (10, 10)).save(png_path) + mocker.patch("PIL.Image.open", side_effect=error) + + with pytest.raises(ParseError): + encode_thumbnail_webp(png_path, tmp_path / "out.webp")