mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-10-08 09:07:13 +00:00
Chore: Generate PDF thumbnails with pdftoppm and Pillow instead of ImageMagick
PDF thumbnails were produced by handing the original PDF to ImageMagick's convert, which delegates PDF handling to Ghostscript. When that failed, a fallback invoked gs directly and then ran convert a second time just to encode the WebP, so a single thumbnail could take three subprocess calls through two general purpose tools for what is only "render page one small". The first page is now rasterized with Poppler's pdftoppm, which is already installed for pdftotext, at a DPI computed from the page's own CropBox and effective rotation as read by pikepdf. The DPI is chosen so the page fits 500x5000 pixels in a single render and is capped at 72 so small pages are never enlarged, matching the previous shrink-only scale. Pillow then flattens any alpha onto white, applies a no-enlarge safety clamp and saves the WebP in process. If the geometry cannot be read, a fixed 150 DPI is used and the clamp keeps the output in bounds. The Ghostscript fallback is replaced with a qpdf repair and retry: the PDF is copied, repaired in place with qpdf (treating its "repaired with warnings" exit status as success), and rasterized again. If that also fails, the default thumbnail is used as before.
This commit is contained in:
1 parent
ee34a6598e
commit
ae3d8680fa
5 files changed
+601
-53
No files matched your search
@@ -1,11 +1,19 @@
|
||||
from collections.abc import Generator
|
||||
from pathlib import Path
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
from PIL import Image
|
||||
from pytest_django.fixtures import Settings
|
||||
from pytest_mock import MockerFixture
|
||||
|
||||
from documents.parsers import ParseError
|
||||
from documents.parsers import _compute_thumbnail_dpi
|
||||
from documents.parsers import encode_thumbnail_webp
|
||||
from documents.parsers import get_default_file_extension
|
||||
from documents.parsers import get_supported_file_extensions
|
||||
from documents.parsers import is_file_ext_supported
|
||||
from documents.parsers import rasterize_pdf_page_to_png
|
||||
from paperless.parsers.registry import get_parser_registry
|
||||
from paperless.parsers.registry import reset_parser_registry
|
||||
from paperless.parsers.tesseract import RasterisedDocumentParser
|
||||
@@ -125,3 +133,183 @@ class TestParserAvailability:
|
||||
assert is_file_ext_supported(".pdf")
|
||||
assert not is_file_ext_supported(".hsdfh")
|
||||
assert not is_file_ext_supported("")
|
||||
|
||||
|
||||
class TestComputeThumbnailDpi:
|
||||
@pytest.mark.parametrize(
|
||||
("size", "expected_dpi"),
|
||||
[
|
||||
pytest.param((612.0, 792.0), 59, id="letter-width-bound"),
|
||||
pytest.param((792.0, 612.0), 45, id="landscape-width-bound"),
|
||||
pytest.param((612.0, 100000.0), 4, id="tall-strip-height-bound"),
|
||||
pytest.param((200.0, 300.0), 72, id="small-page-never-enlarged"),
|
||||
pytest.param((1000000.0, 1000000.0), 1, id="huge-page-minimum-one"),
|
||||
pytest.param(None, 150, id="unreadable-geometry-fallback"),
|
||||
],
|
||||
)
|
||||
def test_dpi_from_page_size(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tmp_path: Path,
|
||||
size: tuple[float, float] | None,
|
||||
expected_dpi: int,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF whose first page has the given size in points (or whose
|
||||
geometry cannot be read)
|
||||
WHEN:
|
||||
- The thumbnail DPI is computed
|
||||
THEN:
|
||||
- The DPI fits the page into 500x5000 without ever exceeding 72,
|
||||
is at least 1, and falls back to 150 when the size is unknown
|
||||
"""
|
||||
mocker.patch(
|
||||
"paperless.parsers.utils.get_pdf_first_page_size_points",
|
||||
return_value=size,
|
||||
)
|
||||
assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected_dpi
|
||||
|
||||
|
||||
class TestRasterizePdfPageToPng:
|
||||
@staticmethod
|
||||
def _write_pdf(
|
||||
path: Path,
|
||||
*,
|
||||
crop_box: tuple[float, float, float, float] | None = None,
|
||||
rotate: int | None = None,
|
||||
) -> Path:
|
||||
pdf = pikepdf.new()
|
||||
pdf.add_blank_page(page_size=(144, 72))
|
||||
pdf.add_blank_page(page_size=(300, 300))
|
||||
page = pdf.pages[0]
|
||||
if crop_box is not None:
|
||||
page.obj.CropBox = pikepdf.Array(crop_box)
|
||||
if rotate is not None:
|
||||
page.obj.Rotate = rotate
|
||||
pdf.save(path)
|
||||
return path
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("crop_box", "rotate", "expected_size"),
|
||||
[
|
||||
pytest.param(None, None, (144, 72), id="plain"),
|
||||
pytest.param(None, 90, (72, 144), id="rotated-90"),
|
||||
pytest.param((0, 0, 72, 36), None, (72, 36), id="crop-box"),
|
||||
],
|
||||
)
|
||||
def test_renders_first_page_to_exact_path(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
crop_box: tuple[float, float, float, float] | None,
|
||||
rotate: int | None,
|
||||
expected_size: tuple[int, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A two page PDF whose first page is 144x72 points, optionally
|
||||
cropped or rotated
|
||||
WHEN:
|
||||
- The first page is rasterized at 72 DPI
|
||||
THEN:
|
||||
- Exactly the requested output path is written, and the image
|
||||
matches the first page's cropped, rotated size
|
||||
"""
|
||||
pdf_path = self._write_pdf(
|
||||
tmp_path / "in.pdf",
|
||||
crop_box=crop_box,
|
||||
rotate=rotate,
|
||||
)
|
||||
out_dir = tmp_path / "out"
|
||||
out_dir.mkdir()
|
||||
out_path = out_dir / "page1.png"
|
||||
|
||||
rasterize_pdf_page_to_png(pdf_path, out_path, dpi=72)
|
||||
|
||||
assert list(out_dir.iterdir()) == [out_path]
|
||||
with Image.open(out_path) as im:
|
||||
assert im.format == "PNG"
|
||||
assert im.size == expected_size
|
||||
|
||||
def test_failure_raises_parse_error(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A file that is not a PDF
|
||||
WHEN:
|
||||
- Rasterization is attempted
|
||||
THEN:
|
||||
- A ParseError is raised
|
||||
"""
|
||||
bad = tmp_path / "bad.pdf"
|
||||
bad.write_bytes(b"not a pdf")
|
||||
|
||||
with pytest.raises(ParseError):
|
||||
rasterize_pdf_page_to_png(bad, tmp_path / "page1.png", dpi=72)
|
||||
|
||||
|
||||
class TestEncodeThumbnailWebp:
|
||||
@pytest.mark.parametrize(
|
||||
("mode", "color"),
|
||||
[
|
||||
pytest.param("RGBA", (0, 0, 0, 0), id="rgba"),
|
||||
pytest.param("LA", (0, 0), id="la"),
|
||||
],
|
||||
)
|
||||
def test_alpha_flattened_onto_white(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
mode: str,
|
||||
color: tuple[int, ...],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A fully transparent PNG with an alpha channel
|
||||
WHEN:
|
||||
- It is encoded as a thumbnail
|
||||
THEN:
|
||||
- The WebP output is RGB with the transparency flattened to white
|
||||
"""
|
||||
png_path = tmp_path / "in.png"
|
||||
Image.new(mode, (20, 10), color).save(png_path)
|
||||
out_path = tmp_path / "out.webp"
|
||||
|
||||
encode_thumbnail_webp(png_path, out_path)
|
||||
|
||||
with Image.open(out_path) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.mode == "RGB"
|
||||
assert im.size == (20, 10)
|
||||
red, green, blue = im.getpixel((10, 5))
|
||||
assert min(red, green, blue) >= 250
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("in_size", "expected_size"),
|
||||
[
|
||||
pytest.param((1000, 2000), (500, 1000), id="too-wide-shrunk"),
|
||||
pytest.param((100, 10000), (50, 5000), id="too-tall-shrunk"),
|
||||
pytest.param((100, 200), (100, 200), id="small-not-enlarged"),
|
||||
],
|
||||
)
|
||||
def test_size_clamped_without_enlarging(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
in_size: tuple[int, int],
|
||||
expected_size: tuple[int, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A rendered page image of the given size
|
||||
WHEN:
|
||||
- It is encoded as a thumbnail
|
||||
THEN:
|
||||
- It is shrunk to fit 500x5000 keeping aspect ratio, and never
|
||||
enlarged
|
||||
"""
|
||||
png_path = tmp_path / "in.png"
|
||||
Image.new("RGB", in_size, (255, 255, 255)).save(png_path)
|
||||
out_path = tmp_path / "out.webp"
|
||||
|
||||
encode_thumbnail_webp(png_path, out_path)
|
||||
|
||||
with Image.open(out_path) as im:
|
||||
assert im.size == expected_size
|
||||
Reference in new issue
Block a user