mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-10-08 00:57:14 +00:00
Chore: Generate PDF thumbnails with pdftoppm and Pillow instead of ImageMagick
PDF thumbnails were produced by handing the original PDF to ImageMagick's convert, which delegates PDF handling to Ghostscript. When that failed, a fallback invoked gs directly and then ran convert a second time just to encode the WebP, so a single thumbnail could take three subprocess calls through two general purpose tools for what is only "render page one small". The first page is now rasterized with Poppler's pdftoppm, which is already installed for pdftotext, at a DPI computed from the page's own CropBox and effective rotation as read by pikepdf. The DPI is chosen so the page fits 500x5000 pixels in a single render and is capped at 72 so small pages are never enlarged, matching the previous shrink-only scale. Pillow then flattens any alpha onto white, applies a no-enlarge safety clamp and saves the WebP in process. If the geometry cannot be read, a fixed 150 DPI is used and the clamp keeps the output in bounds. The Ghostscript fallback is replaced with a qpdf repair and retry: the PDF is copied, repaired in place with qpdf (treating its "repaired with warnings" exit status as success), and rasterized again. If that also fails, the default thumbnail is used as before.
This commit is contained in:
1 parent
ee34a6598e
commit
ae3d8680fa
5 files changed
+601
-53
No files matched your search
+154
-43
@@ -127,42 +127,160 @@ def get_default_thumbnail() -> Path:
|
||||
return (Path(__file__).parent / "resources" / "document.webp").resolve()
|
||||
|
||||
|
||||
def make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group=None) -> Path:
|
||||
out_path: Path = Path(temp_dir) / "convert_gs.webp"
|
||||
_THUMBNAIL_MAX_WIDTH = 500
|
||||
_THUMBNAIL_MAX_HEIGHT = 5000
|
||||
# Used only when the page geometry cannot be read
|
||||
_THUMBNAIL_FALLBACK_DPI = 150
|
||||
|
||||
# if convert fails, fall back to extracting
|
||||
# the first PDF page as a PNG using Ghostscript
|
||||
logger.warning(
|
||||
"Thumbnail generation with ImageMagick failed, falling back "
|
||||
"to ghostscript. Check your /etc/ImageMagick-x/policy.xml!",
|
||||
extra={"group": logging_group},
|
||||
)
|
||||
# Ghostscript doesn't handle WebP outputs
|
||||
gs_out_path: Path = Path(temp_dir) / "gs_out.png"
|
||||
cmd = [settings.GS_BINARY, "-q", "-sDEVICE=pngalpha", "-o", gs_out_path, in_path]
|
||||
|
||||
def rasterize_pdf_page_to_png(
|
||||
in_path: Path,
|
||||
out_path: Path,
|
||||
*,
|
||||
dpi: int,
|
||||
use_cropbox: bool = True,
|
||||
logging_group=None,
|
||||
) -> None:
|
||||
"""
|
||||
Rasterizes page 1 of a PDF to a PNG at out_path via pdftoppm (Poppler),
|
||||
at the given DPI. pdftoppm honors the page's /Rotate on its own.
|
||||
"""
|
||||
# -singlefile stops pdftoppm appending a page number to the output name,
|
||||
# and with -png it appends ".png" itself, so it is given the path without
|
||||
# its suffix to write exactly out_path
|
||||
args = [
|
||||
"pdftoppm",
|
||||
"-f",
|
||||
"1",
|
||||
"-l",
|
||||
"1",
|
||||
"-r",
|
||||
str(dpi),
|
||||
"-png",
|
||||
"-singlefile",
|
||||
]
|
||||
if use_cropbox:
|
||||
args.append("-cropbox")
|
||||
args += [str(in_path), str(out_path.with_suffix(""))]
|
||||
|
||||
logger.debug("Execute: " + " ".join(args), extra={"group": logging_group})
|
||||
|
||||
try:
|
||||
try:
|
||||
run_subprocess(cmd, logger=logger)
|
||||
except subprocess.CalledProcessError as e:
|
||||
raise ParseError(f"Thumbnail (gs) failed at {cmd}") from e
|
||||
# then run convert on the output from gs to make WebP
|
||||
run_convert(
|
||||
density=300,
|
||||
scale="500x5000>",
|
||||
alpha="remove",
|
||||
strip=True,
|
||||
trim=False,
|
||||
auto_orient=True,
|
||||
input_file=gs_out_path,
|
||||
output_file=out_path,
|
||||
logging_group=logging_group,
|
||||
run_subprocess(args, logger=logger)
|
||||
except subprocess.CalledProcessError as e:
|
||||
raise ParseError(f"pdftoppm failed at {args}") from e
|
||||
except Exception as e: # pragma: no cover
|
||||
raise ParseError("Unknown error running pdftoppm") from e
|
||||
|
||||
|
||||
def encode_thumbnail_webp(
|
||||
png_path: Path,
|
||||
out_path: Path,
|
||||
*,
|
||||
max_width: int = _THUMBNAIL_MAX_WIDTH,
|
||||
max_height: int = _THUMBNAIL_MAX_HEIGHT,
|
||||
) -> None:
|
||||
"""
|
||||
Flattens any alpha onto white and saves the image as WebP.
|
||||
|
||||
max_width/max_height are only a safety-net clamp for DPI rounding, since
|
||||
the render is already sized by the computed DPI. The image is never
|
||||
enlarged, matching the previous "-scale WxH>" behavior.
|
||||
"""
|
||||
from PIL import Image
|
||||
|
||||
try:
|
||||
with Image.open(png_path) as im:
|
||||
if im.mode in ("RGBA", "LA"):
|
||||
flattened = Image.new("RGB", im.size, (255, 255, 255))
|
||||
flattened.paste(im, mask=im.split()[-1])
|
||||
else:
|
||||
flattened = im.convert("RGB")
|
||||
|
||||
flattened.thumbnail((max_width, max_height))
|
||||
flattened.save(out_path, format="WEBP")
|
||||
except OSError as e:
|
||||
raise ParseError(f"Unable to encode thumbnail from {png_path}") from e
|
||||
|
||||
|
||||
def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> int:
|
||||
"""
|
||||
Computes the DPI which renders the first page of the PDF to fit within the
|
||||
thumbnail size in one pass, never above the page's natural 72 DPI size.
|
||||
"""
|
||||
from paperless.parsers.utils import get_pdf_first_page_size_points
|
||||
|
||||
size = get_pdf_first_page_size_points(in_path)
|
||||
if size is None:
|
||||
logger.debug(
|
||||
"Could not read PDF page size, using fallback DPI",
|
||||
extra={"group": logging_group},
|
||||
)
|
||||
return _THUMBNAIL_FALLBACK_DPI
|
||||
|
||||
width_pts, height_pts = size
|
||||
dpi_for_width = _THUMBNAIL_MAX_WIDTH * 72 / width_pts
|
||||
dpi_for_height = _THUMBNAIL_MAX_HEIGHT * 72 / height_pts
|
||||
# Capping at 72 (1px per point) keeps the shrink-only behavior: a page
|
||||
# already smaller than the thumbnail size is never enlarged
|
||||
return max(1, round(min(72, dpi_for_width, dpi_for_height)))
|
||||
|
||||
|
||||
def _render_pdf_thumbnail(
|
||||
in_path: Path,
|
||||
png_path: Path,
|
||||
out_path: Path,
|
||||
logging_group=None,
|
||||
) -> None:
|
||||
dpi = _compute_thumbnail_dpi(in_path, logging_group=logging_group)
|
||||
rasterize_pdf_page_to_png(
|
||||
in_path,
|
||||
png_path,
|
||||
dpi=dpi,
|
||||
use_cropbox=True,
|
||||
logging_group=logging_group,
|
||||
)
|
||||
encode_thumbnail_webp(png_path, out_path)
|
||||
|
||||
|
||||
def make_thumbnail_from_pdf_qpdf_fallback(
|
||||
in_path: Path,
|
||||
temp_dir: Path,
|
||||
logging_group=None,
|
||||
) -> Path:
|
||||
png_path: Path = Path(temp_dir) / "page1_repaired.png"
|
||||
out_path: Path = Path(temp_dir) / "convert_qpdf.webp"
|
||||
repaired_path: Path = Path(temp_dir) / "repaired.pdf"
|
||||
|
||||
logger.warning(
|
||||
"Thumbnail generation with pdftoppm failed, attempting qpdf repair and retry.",
|
||||
extra={"group": logging_group},
|
||||
)
|
||||
|
||||
try:
|
||||
# qpdf rewrites in place, so work on a copy and leave the original alone.
|
||||
# qpdf exits 3 when it had to repair the file, which is the expected
|
||||
# outcome here, so warnings must not count as failure.
|
||||
try:
|
||||
shutil.copy(in_path, repaired_path)
|
||||
run_subprocess(
|
||||
[
|
||||
"qpdf",
|
||||
"--warning-exit-0",
|
||||
"--replace-input",
|
||||
str(repaired_path),
|
||||
],
|
||||
logger=logger,
|
||||
)
|
||||
except (subprocess.CalledProcessError, OSError) as e:
|
||||
raise ParseError(f"qpdf repair failed for {in_path}") from e
|
||||
|
||||
_render_pdf_thumbnail(repaired_path, png_path, out_path, logging_group)
|
||||
|
||||
return out_path
|
||||
|
||||
except ParseError as e:
|
||||
logger.error(f"Unable to make thumbnail with Ghostscript: {e}")
|
||||
logger.error(f"Unable to make thumbnail after qpdf repair: {e}")
|
||||
# The caller might expect a generated thumbnail that can be moved,
|
||||
# so we need to copy it before it gets moved.
|
||||
# https://github.com/paperless-ngx/paperless-ngx/issues/3631
|
||||
@@ -175,25 +293,18 @@ def make_thumbnail_from_pdf(in_path: Path, temp_dir: Path, logging_group=None) -
|
||||
"""
|
||||
The thumbnail of a PDF is just a 500px wide image of the first page.
|
||||
"""
|
||||
png_path: Path = temp_dir / "page1.png"
|
||||
out_path: Path = temp_dir / "convert.webp"
|
||||
|
||||
# Run convert to get a decent thumbnail
|
||||
try:
|
||||
run_convert(
|
||||
density=300,
|
||||
scale="500x5000>",
|
||||
alpha="remove",
|
||||
strip=True,
|
||||
trim=False,
|
||||
auto_orient=True,
|
||||
use_cropbox=True,
|
||||
input_file=f"{in_path}[0]",
|
||||
output_file=str(out_path),
|
||||
logging_group=logging_group,
|
||||
)
|
||||
_render_pdf_thumbnail(in_path, png_path, out_path, logging_group)
|
||||
except ParseError as e:
|
||||
logger.error(f"Unable to make thumbnail with convert: {e}")
|
||||
out_path = make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group)
|
||||
logger.error(f"Unable to make thumbnail with pdftoppm: {e}")
|
||||
out_path = make_thumbnail_from_pdf_qpdf_fallback(
|
||||
in_path,
|
||||
temp_dir,
|
||||
logging_group,
|
||||
)
|
||||
|
||||
return out_path
|
||||
|
||||
|
||||
@@ -1,11 +1,19 @@
|
||||
from collections.abc import Generator
|
||||
from pathlib import Path
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
from PIL import Image
|
||||
from pytest_django.fixtures import Settings
|
||||
from pytest_mock import MockerFixture
|
||||
|
||||
from documents.parsers import ParseError
|
||||
from documents.parsers import _compute_thumbnail_dpi
|
||||
from documents.parsers import encode_thumbnail_webp
|
||||
from documents.parsers import get_default_file_extension
|
||||
from documents.parsers import get_supported_file_extensions
|
||||
from documents.parsers import is_file_ext_supported
|
||||
from documents.parsers import rasterize_pdf_page_to_png
|
||||
from paperless.parsers.registry import get_parser_registry
|
||||
from paperless.parsers.registry import reset_parser_registry
|
||||
from paperless.parsers.tesseract import RasterisedDocumentParser
|
||||
@@ -125,3 +133,183 @@ class TestParserAvailability:
|
||||
assert is_file_ext_supported(".pdf")
|
||||
assert not is_file_ext_supported(".hsdfh")
|
||||
assert not is_file_ext_supported("")
|
||||
|
||||
|
||||
class TestComputeThumbnailDpi:
|
||||
@pytest.mark.parametrize(
|
||||
("size", "expected_dpi"),
|
||||
[
|
||||
pytest.param((612.0, 792.0), 59, id="letter-width-bound"),
|
||||
pytest.param((792.0, 612.0), 45, id="landscape-width-bound"),
|
||||
pytest.param((612.0, 100000.0), 4, id="tall-strip-height-bound"),
|
||||
pytest.param((200.0, 300.0), 72, id="small-page-never-enlarged"),
|
||||
pytest.param((1000000.0, 1000000.0), 1, id="huge-page-minimum-one"),
|
||||
pytest.param(None, 150, id="unreadable-geometry-fallback"),
|
||||
],
|
||||
)
|
||||
def test_dpi_from_page_size(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tmp_path: Path,
|
||||
size: tuple[float, float] | None,
|
||||
expected_dpi: int,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF whose first page has the given size in points (or whose
|
||||
geometry cannot be read)
|
||||
WHEN:
|
||||
- The thumbnail DPI is computed
|
||||
THEN:
|
||||
- The DPI fits the page into 500x5000 without ever exceeding 72,
|
||||
is at least 1, and falls back to 150 when the size is unknown
|
||||
"""
|
||||
mocker.patch(
|
||||
"paperless.parsers.utils.get_pdf_first_page_size_points",
|
||||
return_value=size,
|
||||
)
|
||||
assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected_dpi
|
||||
|
||||
|
||||
class TestRasterizePdfPageToPng:
|
||||
@staticmethod
|
||||
def _write_pdf(
|
||||
path: Path,
|
||||
*,
|
||||
crop_box: tuple[float, float, float, float] | None = None,
|
||||
rotate: int | None = None,
|
||||
) -> Path:
|
||||
pdf = pikepdf.new()
|
||||
pdf.add_blank_page(page_size=(144, 72))
|
||||
pdf.add_blank_page(page_size=(300, 300))
|
||||
page = pdf.pages[0]
|
||||
if crop_box is not None:
|
||||
page.obj.CropBox = pikepdf.Array(crop_box)
|
||||
if rotate is not None:
|
||||
page.obj.Rotate = rotate
|
||||
pdf.save(path)
|
||||
return path
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("crop_box", "rotate", "expected_size"),
|
||||
[
|
||||
pytest.param(None, None, (144, 72), id="plain"),
|
||||
pytest.param(None, 90, (72, 144), id="rotated-90"),
|
||||
pytest.param((0, 0, 72, 36), None, (72, 36), id="crop-box"),
|
||||
],
|
||||
)
|
||||
def test_renders_first_page_to_exact_path(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
crop_box: tuple[float, float, float, float] | None,
|
||||
rotate: int | None,
|
||||
expected_size: tuple[int, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A two page PDF whose first page is 144x72 points, optionally
|
||||
cropped or rotated
|
||||
WHEN:
|
||||
- The first page is rasterized at 72 DPI
|
||||
THEN:
|
||||
- Exactly the requested output path is written, and the image
|
||||
matches the first page's cropped, rotated size
|
||||
"""
|
||||
pdf_path = self._write_pdf(
|
||||
tmp_path / "in.pdf",
|
||||
crop_box=crop_box,
|
||||
rotate=rotate,
|
||||
)
|
||||
out_dir = tmp_path / "out"
|
||||
out_dir.mkdir()
|
||||
out_path = out_dir / "page1.png"
|
||||
|
||||
rasterize_pdf_page_to_png(pdf_path, out_path, dpi=72)
|
||||
|
||||
assert list(out_dir.iterdir()) == [out_path]
|
||||
with Image.open(out_path) as im:
|
||||
assert im.format == "PNG"
|
||||
assert im.size == expected_size
|
||||
|
||||
def test_failure_raises_parse_error(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A file that is not a PDF
|
||||
WHEN:
|
||||
- Rasterization is attempted
|
||||
THEN:
|
||||
- A ParseError is raised
|
||||
"""
|
||||
bad = tmp_path / "bad.pdf"
|
||||
bad.write_bytes(b"not a pdf")
|
||||
|
||||
with pytest.raises(ParseError):
|
||||
rasterize_pdf_page_to_png(bad, tmp_path / "page1.png", dpi=72)
|
||||
|
||||
|
||||
class TestEncodeThumbnailWebp:
|
||||
@pytest.mark.parametrize(
|
||||
("mode", "color"),
|
||||
[
|
||||
pytest.param("RGBA", (0, 0, 0, 0), id="rgba"),
|
||||
pytest.param("LA", (0, 0), id="la"),
|
||||
],
|
||||
)
|
||||
def test_alpha_flattened_onto_white(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
mode: str,
|
||||
color: tuple[int, ...],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A fully transparent PNG with an alpha channel
|
||||
WHEN:
|
||||
- It is encoded as a thumbnail
|
||||
THEN:
|
||||
- The WebP output is RGB with the transparency flattened to white
|
||||
"""
|
||||
png_path = tmp_path / "in.png"
|
||||
Image.new(mode, (20, 10), color).save(png_path)
|
||||
out_path = tmp_path / "out.webp"
|
||||
|
||||
encode_thumbnail_webp(png_path, out_path)
|
||||
|
||||
with Image.open(out_path) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.mode == "RGB"
|
||||
assert im.size == (20, 10)
|
||||
red, green, blue = im.getpixel((10, 5))
|
||||
assert min(red, green, blue) >= 250
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("in_size", "expected_size"),
|
||||
[
|
||||
pytest.param((1000, 2000), (500, 1000), id="too-wide-shrunk"),
|
||||
pytest.param((100, 10000), (50, 5000), id="too-tall-shrunk"),
|
||||
pytest.param((100, 200), (100, 200), id="small-not-enlarged"),
|
||||
],
|
||||
)
|
||||
def test_size_clamped_without_enlarging(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
in_size: tuple[int, int],
|
||||
expected_size: tuple[int, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A rendered page image of the given size
|
||||
WHEN:
|
||||
- It is encoded as a thumbnail
|
||||
THEN:
|
||||
- It is shrunk to fit 500x5000 keeping aspect ratio, and never
|
||||
enlarged
|
||||
"""
|
||||
png_path = tmp_path / "in.png"
|
||||
Image.new("RGB", in_size, (255, 255, 255)).save(png_path)
|
||||
out_path = tmp_path / "out.webp"
|
||||
|
||||
encode_thumbnail_webp(png_path, out_path)
|
||||
|
||||
with Image.open(out_path) as im:
|
||||
assert im.size == expected_size
|
||||
@@ -265,6 +265,57 @@ def get_page_count_for_pdf(
|
||||
return None
|
||||
|
||||
|
||||
def get_pdf_first_page_size_points(
|
||||
path: Path,
|
||||
log: logging.Logger | None = None,
|
||||
) -> tuple[float, float] | None:
|
||||
"""Return the first page's (width, height) in PDF points, post-rotation.
|
||||
|
||||
Uses ``page.cropbox``, which falls back to the MediaBox when the page has
|
||||
no ``/CropBox`` of its own. This must match the box the renderer is told
|
||||
to use (pdftoppm's ``-cropbox`` flag), or a DPI computed from it will
|
||||
target the wrong box's dimensions.
|
||||
|
||||
Width and height are swapped when the page's effective rotation is 90 or
|
||||
270, since that is the orientation the page is rendered in.
|
||||
``page.rotation`` resolves a ``/Rotate`` inherited from an ancestor
|
||||
``/Pages`` node and normalizes it to ``[0, 360)``, which a raw
|
||||
``/Rotate`` lookup on the page dictionary would not.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
path:
|
||||
Absolute path to the PDF file.
|
||||
log:
|
||||
Logger for warnings. Falls back to the module-level logger when omitted.
|
||||
|
||||
Returns
|
||||
-------
|
||||
tuple[float, float] | None
|
||||
``(width_points, height_points)``, or ``None`` if the file cannot be
|
||||
opened, has no pages, or the page box is degenerate.
|
||||
"""
|
||||
import pikepdf
|
||||
|
||||
_log = log or logger
|
||||
try:
|
||||
with pikepdf.Pdf.open(path) as pdf:
|
||||
if len(pdf.pages) == 0:
|
||||
return None
|
||||
page = pdf.pages[0]
|
||||
llx, lly, urx, ury = (float(v) for v in page.cropbox)
|
||||
width = abs(urx - llx)
|
||||
height = abs(ury - lly)
|
||||
if width <= 0 or height <= 0:
|
||||
return None
|
||||
if page.rotation in (90, 270):
|
||||
width, height = height, width
|
||||
return width, height
|
||||
except Exception:
|
||||
_log.warning("Could not determine PDF page size for %s", path, exc_info=True)
|
||||
return None
|
||||
|
||||
|
||||
def extract_pdf_metadata(
|
||||
document_path: Path,
|
||||
log: logging.Logger | None = None,
|
||||
|
||||
@@ -15,9 +15,10 @@ from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
from ocrmypdf import SubprocessOutputError
|
||||
from PIL import Image
|
||||
|
||||
import documents.parsers
|
||||
from documents.parsers import ParseError
|
||||
from documents.parsers import run_convert
|
||||
from paperless.models import ModeChoices
|
||||
from paperless.parsers import ParserProtocol
|
||||
from paperless.parsers.tesseract import RasterisedDocumentParser
|
||||
@@ -280,24 +281,74 @@ class TestGetThumbnail:
|
||||
)
|
||||
assert thumb.is_file()
|
||||
|
||||
def test_thumbnail_fallback_on_convert_error(
|
||||
@pytest.mark.parametrize(
|
||||
("filename", "expected_height"),
|
||||
[
|
||||
pytest.param("simple-digital.pdf", 647, id="portrait-letter"),
|
||||
pytest.param("rotated.pdf", 386, id="landscape"),
|
||||
],
|
||||
)
|
||||
def test_thumbnail_is_correct_format_and_size(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
tesseract_samples_dir: Path,
|
||||
filename: str,
|
||||
expected_height: int,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF whose first page is wider than the thumbnail width
|
||||
WHEN:
|
||||
- A thumbnail is generated
|
||||
THEN:
|
||||
- The thumbnail is a WebP, 500px wide, keeping the page's aspect
|
||||
ratio (within rounding of the DPI computation)
|
||||
"""
|
||||
thumb = tesseract_parser.get_thumbnail(
|
||||
tesseract_samples_dir / filename,
|
||||
"application/pdf",
|
||||
)
|
||||
with Image.open(thumb) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.width == 500
|
||||
assert im.height == pytest.approx(expected_height, abs=2)
|
||||
|
||||
def test_thumbnail_fallback_on_pdftoppm_error(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
def _raise_on_pdf(input_file, output_file, **kwargs) -> None:
|
||||
if ".pdf" in str(input_file):
|
||||
"""
|
||||
GIVEN:
|
||||
- Rasterizing the original PDF fails
|
||||
WHEN:
|
||||
- A thumbnail is generated
|
||||
THEN:
|
||||
- The PDF is repaired with qpdf and rasterized again, producing a
|
||||
real thumbnail rather than the default placeholder
|
||||
"""
|
||||
real_rasterize = documents.parsers.rasterize_pdf_page_to_png
|
||||
original = tesseract_samples_dir / "simple-digital.pdf"
|
||||
|
||||
def _fail_on_original(in_path: Path, out_path: Path, **kwargs) -> None:
|
||||
if in_path == original:
|
||||
raise ParseError("Does not compute.")
|
||||
run_convert(input_file=input_file, output_file=output_file, **kwargs)
|
||||
real_rasterize(in_path, out_path, **kwargs)
|
||||
|
||||
mocker.patch("documents.parsers.run_convert", side_effect=_raise_on_pdf)
|
||||
|
||||
thumb = tesseract_parser.get_thumbnail(
|
||||
tesseract_samples_dir / "simple-digital.pdf",
|
||||
"application/pdf",
|
||||
rasterize = mocker.patch(
|
||||
"documents.parsers.rasterize_pdf_page_to_png",
|
||||
side_effect=_fail_on_original,
|
||||
)
|
||||
|
||||
thumb = tesseract_parser.get_thumbnail(original, "application/pdf")
|
||||
|
||||
assert rasterize.call_count == 2
|
||||
assert thumb.is_file()
|
||||
assert thumb.name == "convert_qpdf.webp"
|
||||
with Image.open(thumb) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.width == 500
|
||||
|
||||
def test_thumbnail_encrypted_pdf(
|
||||
self,
|
||||
|
||||
@@ -6,8 +6,10 @@ import codecs
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
|
||||
from paperless.parsers.utils import get_pdf_first_page_size_points
|
||||
from paperless.parsers.utils import is_tagged_pdf
|
||||
from paperless.parsers.utils import pdf_born_digital_text
|
||||
from paperless.parsers.utils import post_process_text
|
||||
@@ -70,6 +72,151 @@ class TestIsTaggedPdf:
|
||||
assert is_tagged_pdf(bad) is False
|
||||
|
||||
|
||||
class TestGetPdfFirstPageSizePoints:
|
||||
@staticmethod
|
||||
def _write_pdf(
|
||||
path: Path,
|
||||
*,
|
||||
media_box: tuple[float, float, float, float] = (0, 0, 600, 800),
|
||||
crop_box: tuple[float, float, float, float] | None = None,
|
||||
page_rotate: int | None = None,
|
||||
inherited_rotate: int | None = None,
|
||||
) -> Path:
|
||||
pdf = pikepdf.new()
|
||||
pdf.add_blank_page(page_size=(media_box[2], media_box[3]))
|
||||
page = pdf.pages[0]
|
||||
page.obj.MediaBox = pikepdf.Array(media_box)
|
||||
if crop_box is not None:
|
||||
page.obj.CropBox = pikepdf.Array(crop_box)
|
||||
if page_rotate is not None:
|
||||
page.obj.Rotate = page_rotate
|
||||
if inherited_rotate is not None:
|
||||
pdf.Root.Pages.Rotate = inherited_rotate
|
||||
pdf.save(path)
|
||||
return path
|
||||
|
||||
def test_letter_sample(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A US Letter sample PDF with no CropBox and no rotation
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- The MediaBox size in points is returned
|
||||
"""
|
||||
assert get_pdf_first_page_size_points(SAMPLES / "simple-digital.pdf") == (
|
||||
612.0,
|
||||
792.0,
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("rotate", "expected"),
|
||||
[
|
||||
pytest.param(0, (600.0, 800.0), id="rotate-0"),
|
||||
pytest.param(90, (800.0, 600.0), id="rotate-90"),
|
||||
pytest.param(180, (600.0, 800.0), id="rotate-180"),
|
||||
pytest.param(270, (800.0, 600.0), id="rotate-270"),
|
||||
pytest.param(-90, (800.0, 600.0), id="rotate-negative-90"),
|
||||
],
|
||||
)
|
||||
def test_page_rotation_swaps_dimensions(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
rotate: int,
|
||||
expected: tuple[float, float],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A portrait PDF page with /Rotate set directly on the page
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- Width and height are swapped for quarter-turn rotations only
|
||||
"""
|
||||
pdf_path = self._write_pdf(tmp_path / "rotated.pdf", page_rotate=rotate)
|
||||
assert get_pdf_first_page_size_points(pdf_path) == expected
|
||||
|
||||
def test_inherited_rotation_swaps_dimensions(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A portrait PDF page whose /Rotate 90 is set on the /Pages node,
|
||||
not on the page itself
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- The inherited rotation is honored and width/height are swapped
|
||||
"""
|
||||
pdf_path = self._write_pdf(tmp_path / "inherited.pdf", inherited_rotate=90)
|
||||
assert get_pdf_first_page_size_points(pdf_path) == (800.0, 600.0)
|
||||
|
||||
def test_crop_box_preferred_over_media_box(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF page with a CropBox smaller than its MediaBox
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- The CropBox dimensions are returned, matching what pdftoppm
|
||||
renders with -cropbox
|
||||
"""
|
||||
pdf_path = self._write_pdf(
|
||||
tmp_path / "cropped.pdf",
|
||||
crop_box=(50, 100, 350, 500),
|
||||
)
|
||||
assert get_pdf_first_page_size_points(pdf_path) == (300.0, 400.0)
|
||||
|
||||
def test_degenerate_box_returns_none(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF page whose box has zero width
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- None is returned instead of a size that would break DPI math
|
||||
"""
|
||||
pdf_path = self._write_pdf(
|
||||
tmp_path / "degenerate.pdf",
|
||||
media_box=(0, 0, 600, 800),
|
||||
crop_box=(100, 0, 100, 800),
|
||||
)
|
||||
assert get_pdf_first_page_size_points(pdf_path) is None
|
||||
|
||||
def test_nonexistent_path_returns_none(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A path that does not exist
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- None is returned and nothing is raised
|
||||
"""
|
||||
assert get_pdf_first_page_size_points(Path("/nonexistent/file.pdf")) is None
|
||||
|
||||
def test_corrupt_pdf_returns_none(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A file that is not a PDF
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- None is returned and nothing is raised
|
||||
"""
|
||||
bad = tmp_path / "bad.pdf"
|
||||
bad.write_bytes(b"not a pdf")
|
||||
assert get_pdf_first_page_size_points(bad) is None
|
||||
|
||||
def test_encrypted_pdf_returns_none(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A password protected PDF
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- None is returned and nothing is raised
|
||||
"""
|
||||
assert get_pdf_first_page_size_points(SAMPLES / "encrypted.pdf") is None
|
||||
|
||||
|
||||
class TestPostProcessText:
|
||||
@pytest.mark.parametrize(
|
||||
("source", "expected"),
|
||||
|
||||
Reference in new issue
Block a user