Compare commits

..
18 changed files with 1218 additions and 1259 deletions

No files matched your search

+1 -1
View File
@@ -111,7 +111,7 @@ jobs:
timeout-minutes: 12
uses: $/.github/actions/apt-install
with:
packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils qpdf
packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils
- name: Configure ImageMagick
run: |
sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-7/policy.xml
-1
View File
@@ -38,7 +38,6 @@ src/documents/bulk_edit.py:0: error: Incompatible types in assignment (expressio
src/documents/bulk_edit.py:0: error: Invalid index type "str" for "dict[FieldDataType, str]"; expected type "FieldDataType" [index]
src/documents/bulk_edit.py:0: error: List comprehension has incompatible type List[tuple[int, Any]]; expected List[int] [misc]
src/documents/bulk_edit.py:0: error: List comprehension has incompatible type List[tuple[int, None]]; expected List[int] [misc]
src/documents/bulk_edit.py:0: error: Missing named argument "p" for "remove" of "PageList" [call-arg]
src/documents/bulk_edit.py:0: error: Missing type arguments for generic type "dict" [type-arg]
src/documents/bulk_edit.py:0: error: Missing type arguments for generic type "dict" [type-arg]
src/documents/bulk_edit.py:0: error: Need type annotation for "to_create" (hint: "to_create: list[<type>] = ...") [var-annotated]
-7
View File
@@ -91,13 +91,6 @@
"concise_description": "Argument `list[int]` is not assignable to parameter `args` with type `tuple[Any, ...] | None` in function `celery.app.task.Task.apply_async`",
"severity": "error"
},
{
"column": 33,
"path": "src/documents/bulk_edit.py",
"name": "missing-argument",
"concise_description": "Missing argument `p` in function `pikepdf._core.PageList.remove`",
"severity": "error"
},
{
"column": 25,
"path": "src/documents/caching.py",
+18 -6
View File
@@ -1315,17 +1315,29 @@ valid crontab(5) expression describing when to run.
#### [`PAPERLESS_CONVERT_MEMORY_LIMIT=<num>`](#PAPERLESS_CONVERT_MEMORY_LIMIT) {#PAPERLESS_CONVERT_MEMORY_LIMIT}
!!! warning
: On smaller systems, or even in the case of Very Large Documents, the
consumer may explode, complaining about how it's "unable to extend
pixel cache". In such cases, try setting this to a reasonably low
value, like 32. The default is to use whatever is necessary to do
everything without writing to disk, and units are in megabytes.
Deprecated and has no effect, since PDF thumbnails no longer use
ImageMagick. It will be removed in a future release.
For more information on how to use this value, you should search the
web for "MAGICK_MEMORY_LIMIT".
Defaults to 0, which disables the limit.
#### [`PAPERLESS_CONVERT_TMPDIR=<path>`](#PAPERLESS_CONVERT_TMPDIR) {#PAPERLESS_CONVERT_TMPDIR}
!!! warning
: Similar to the memory limit, if you've got a small system and your
OS mounts /tmp as tmpfs, you should set this to a path that's on a
physical disk, like /home/your_user/tmp or something. ImageMagick
will use this as scratch space when crunching through very large
documents.
Deprecated and has no effect, since PDF thumbnails no longer use
ImageMagick. It will be removed in a future release.
For more information on how to use this value, you should search the
web for "MAGICK_TMPDIR".
Default is none, which disables the temporary directory.
#### [`PAPERLESS_APPS=<string>`](#PAPERLESS_APPS) {#PAPERLESS_APPS}
+7 -4
View File
@@ -177,12 +177,12 @@ to a positive number to enable polling and disable native filesystem notificatio
- `pkg-config` for mysqlclient (python dependency)
- `fonts-liberation` for generating thumbnails for plain text
files
- `imagemagick` >= 6 for image alpha handling
- `imagemagick` >= 6 for PDF conversion
- `gnupg` for decrypting GPG-encrypted email
- `libpq-dev` for PostgreSQL
- `libmagic-dev` for mime type detection
- `mariadb-client` for MariaDB compile time
- `poppler-utils` for thumbnail generation and barcode detection
- `poppler-utils` for barcode detection
Use this list for your preferred package management:
@@ -416,8 +416,11 @@ to a positive number to enable polling and disable native filesystem notificatio
You may need to change the path in the files. Example:
`ExecStart=/opt/paperless/.local/bin/celery --app paperless worker --loglevel INFO`
12. Harden ImageMagick by disabling formats that Paperless-ngx does not use.
PDF processing is not needed and should stay disabled.
12. Configure ImageMagick to allow processing of PDF documents and disable
formats that Paperless-ngx does not use. Most distributions disable PDF
processing by default, since PDF documents can contain malware. If you
don't enable it, Paperless-ngx will fall back to Ghostscript for certain
steps such as thumbnail generation.
Configure the active ImageMagick policy file (commonly
`/etc/ImageMagick-6/policy.xml` or `/etc/ImageMagick-7/policy.xml`) and
+2
View File
@@ -50,6 +50,8 @@ PAPERLESS_SECRET_KEY=change-me
#PAPERLESS_OCR_ROTATE_PAGES=true
#PAPERLESS_OCR_ROTATE_PAGES_THRESHOLD=12.0
#PAPERLESS_OCR_USER_ARGS={}
#PAPERLESS_CONVERT_MEMORY_LIMIT=0
#PAPERLESS_CONVERT_TMPDIR=/var/tmp/paperless
# Software tweaks
+184 -219
View File
@@ -3,6 +3,7 @@ from __future__ import annotations
import logging
import tempfile
import uuid
from functools import partial
from pathlib import Path
from typing import TYPE_CHECKING
from typing import Literal
@@ -17,6 +18,7 @@ from django.db.models import Max
from django.db.models import Q
from django.utils import timezone
from documents import pdf_ops
from documents.data_models import ConsumableDocument
from documents.data_models import DocumentMetadataOverrides
from documents.data_models import DocumentSource
@@ -116,6 +118,11 @@ def _resolve_root_and_source_doc(
)
def _scratch_path(name: str) -> Path:
"""A path inside a fresh directory under SCRATCH_DIR."""
return Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR)) / name
def set_correspondent(
doc_ids: list[int],
correspondent: Correspondent,
@@ -474,8 +481,6 @@ def rotate(
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
docs_by_root_id.setdefault(pair.root_doc.id, pair)
import pikepdf
for pair in docs_by_root_id.values():
if pair.source_doc.mime_type != "application/pdf":
logger.warning(
@@ -488,11 +493,7 @@ def rotate(
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_rotated.pdf"
)
with pikepdf.open(pair.source_doc.source_path) as pdf:
for page in pdf.pages:
page.rotate(degrees, relative=True)
pdf.remove_unreferenced_resources()
pdf.save(filepath)
pdf_ops.rotate_pdf(pair.source_doc.source_path, filepath, degrees)
# Preserve metadata/permissions via overrides; mark as new version
overrides = DocumentMetadataOverrides().from_document(pair.root_doc)
@@ -535,48 +536,45 @@ def merge(
qs = Document.objects.select_related("root_document").filter(id__in=doc_ids)
docs_by_id = {doc.id: doc for doc in qs}
affected_docs: list[int] = []
import pikepdf
merged_pdf = pikepdf.new()
version: str = merged_pdf.pdf_version
handoff_asn: int | None = None
# use doc_ids to preserve order
for doc_id in doc_ids:
doc = docs_by_id.get(doc_id)
if doc is None:
continue
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
try:
doc_path = (
pair.source_doc.archive_path
if archive_fallback
and pair.source_doc.mime_type != "application/pdf"
and pair.source_doc.has_archive_version
else pair.source_doc.source_path
)
with pikepdf.open(str(doc_path)) as pdf:
version = max(version, pdf.pdf_version)
merged_pdf.pages.extend(pdf.pages)
affected_docs.append(doc.id)
if handoff_asn is None and doc.archive_serial_number is not None:
handoff_asn = doc.archive_serial_number
except Exception as e:
logger.exception(
f"Error merging document {doc.id}, it will not be included in the merge: {e}",
)
if len(affected_docs) == 0:
logger.warning("No documents were merged")
return "OK"
with pdf_ops.PdfMerger() as merger:
# use doc_ids to preserve order
for doc_id in doc_ids:
doc = docs_by_id.get(doc_id)
if doc is None:
continue
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
try:
# archive_path is None when there is no archive version
archive_path = (
pair.source_doc.archive_path
if archive_fallback
and pair.source_doc.mime_type != "application/pdf"
else None
)
merger.add(
archive_path
if archive_path is not None
else pair.source_doc.source_path,
)
affected_docs.append(doc.id)
if handoff_asn is None and doc.archive_serial_number is not None:
handoff_asn = doc.archive_serial_number
except Exception as e:
logger.exception(
f"Error merging document {doc.id}, it will not be included in the merge: {e}",
)
if len(affected_docs) == 0:
logger.warning("No documents were merged")
return "OK"
filepath = (
Path(
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
filepath = (
Path(
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
)
/ f"{'_'.join([str(doc_id) for doc_id in affected_docs])[:100]}_merged.pdf"
)
/ f"{'_'.join([str(doc_id) for doc_id in affected_docs])[:100]}_merged.pdf"
)
merged_pdf.remove_unreferenced_resources()
merged_pdf.save(filepath, min_version=version)
merged_pdf.close()
merger.save(filepath)
if metadata_document_id:
metadata_document = qs.get(id=metadata_document_id)
@@ -752,64 +750,60 @@ def split(
)
doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
import pikepdf
consume_tasks = []
try:
with pikepdf.open(pair.source_doc.source_path) as pdf:
for idx, split_doc in enumerate(pages):
dst: pikepdf.Pdf = pikepdf.new()
for page in split_doc:
dst.pages.append(pdf.pages[page - 1])
filepath: Path = (
Path(
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
)
/ f"{doc.id}_{split_doc[0]}-{split_doc[-1]}.pdf"
)
dst.remove_unreferenced_resources()
dst.save(filepath)
dst.close()
outputs = [
(
[pdf_ops.PageSpec(page) for page in split_doc],
partial(_scratch_path, f"{doc.id}_{split_doc[0]}-{split_doc[-1]}.pdf"),
)
for split_doc in pages
]
filepaths = pdf_ops.build_pdfs(pair.source_doc.source_path, outputs)
overrides: DocumentMetadataOverrides = (
DocumentMetadataOverrides().from_document(doc)
)
overrides.title = f"{doc.title} (split {idx + 1})"
if user is not None:
overrides.owner_id = user.id
if not delete_originals:
overrides.skip_asn_if_exists = True
logger.info(
f"Adding split document with pages {split_doc} to the task queue.",
)
consume_tasks.append(
consume_file.s(
input_doc=ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
),
overrides=overrides,
).set(headers={"trigger_source": trigger_source}),
)
for idx, (split_doc, filepath) in enumerate(
zip(pages, filepaths, strict=True),
):
overrides: DocumentMetadataOverrides = (
DocumentMetadataOverrides().from_document(doc)
)
overrides.title = f"{doc.title} (split {idx + 1})"
if user is not None:
overrides.owner_id = user.id
if not delete_originals:
overrides.skip_asn_if_exists = True
logger.info(
f"Adding split document with pages {split_doc} to the task queue.",
)
consume_tasks.append(
consume_file.s(
input_doc=ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
),
overrides=overrides,
).set(headers={"trigger_source": trigger_source}),
)
if delete_originals:
backup = release_archive_serial_numbers([doc.id])
logger.info(
"Queueing removal of original document after consumption of the split documents",
)
try:
chord(
header=consume_tasks,
body=delete.si([doc.id]),
).on_error(
restore_archive_serial_numbers_task.s(backup),
).apply_async()
except Exception:
restore_archive_serial_numbers(backup)
raise
else:
group(consume_tasks).delay()
if delete_originals:
backup = release_archive_serial_numbers([doc.id])
logger.info(
"Queueing removal of original document after consumption of the split documents",
)
try:
chord(
header=consume_tasks,
body=delete.si([doc.id]),
).on_error(
restore_archive_serial_numbers_task.s(backup),
).apply_async()
except Exception:
restore_archive_serial_numbers(backup)
raise
else:
group(consume_tasks).delay()
except Exception as e:
logger.exception(f"Error splitting document {doc.id}: {e}")
@@ -830,8 +824,6 @@ def delete_pages(
)
doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
pages = sorted(pages) # sort pages to avoid index issues
import pikepdf
try:
# Produce edited PDF to a temp file and create a new version
@@ -839,13 +831,7 @@ def delete_pages(
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_pages_deleted.pdf"
)
with pikepdf.open(pair.source_doc.source_path) as pdf:
offset = 1 # pages are 1-indexed
for page_num in pages:
pdf.pages.remove(pdf.pages[page_num - offset])
offset += 1 # remove() changes the index of the pages
pdf.remove_unreferenced_resources()
pdf.save(filepath)
pdf_ops.remove_pages(pair.source_doc.source_path, filepath, pages)
overrides = DocumentMetadataOverrides().from_document(pair.root_doc)
if user is not None:
@@ -894,47 +880,28 @@ def edit_pdf(
)
doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
import pikepdf
pdf_docs: list[pikepdf.Pdf] = []
try:
if not operations:
raise ValueError("Output document index is out of bounds")
max_idx = max(op.get("doc", 0) for op in operations)
if update_document and max_idx > 0:
logger.error(
"Update requested but multiple output documents specified",
output_count = pdf_ops.validate_page_operations(
operations,
single_output=update_document,
)
page_specs: list[list[pdf_ops.PageSpec]] = [[] for _ in range(output_count)]
for op in operations:
page_specs[op.get("doc", 0)].append(
pdf_ops.PageSpec(op["page"], op.get("rotate", 0)),
)
raise ValueError("Multiple output documents specified")
if any(
op.get("doc", 0) < 0 or op.get("doc", 0) >= len(operations)
for op in operations
):
raise ValueError("Output document index is out of bounds")
with pikepdf.open(pair.source_doc.source_path) as src:
# prepare output documents
pdf_docs = [pikepdf.new() for _ in range(max_idx + 1)]
for op in operations:
dst = pdf_docs[op.get("doc", 0)]
page = src.pages[op["page"] - 1]
dst.pages.append(page)
if op.get("rotate"):
dst.pages[-1].rotate(op["rotate"], relative=True)
if update_document:
# Create a new version from the edited PDF rather than replacing in-place
pdf = pdf_docs[0]
pdf.remove_unreferenced_resources()
filepath: Path = (
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_edited.pdf"
(filepath,) = pdf_ops.build_pdfs(
pair.source_doc.source_path,
[
(
page_specs[0],
partial(_scratch_path, f"{pair.root_doc.id}_edited.pdf"),
),
],
)
pdf.save(filepath)
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata
@@ -955,6 +922,19 @@ def edit_pdf(
headers={"trigger_source": trigger_source},
)
else:
version_filepaths = pdf_ops.build_pdfs(
pair.source_doc.source_path,
[
(
specs,
partial(
_scratch_path,
f"{pair.root_doc.id}_edit_{idx}.pdf",
),
)
for idx, specs in enumerate(page_specs, start=1)
],
)
consume_tasks = []
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
@@ -966,15 +946,9 @@ def edit_pdf(
overrides.actor_id = user.id
if not delete_original:
overrides.skip_asn_if_exists = True
if delete_original and len(pdf_docs) == 1:
if delete_original and output_count == 1:
overrides.asn = pair.root_doc.archive_serial_number
for idx, pdf in enumerate(pdf_docs, start=1):
version_filepath: Path = (
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_edit_{idx}.pdf"
)
pdf.remove_unreferenced_resources()
pdf.save(version_filepath)
for version_filepath in version_filepaths:
consume_tasks.append(
consume_file.s(
input_doc=ConsumableDocument(
@@ -1024,8 +998,6 @@ def remove_password(
"""
Remove password protection from PDF documents.
"""
import pikepdf
for doc_id in doc_ids:
doc = Document.objects.select_related("root_document").get(id=doc_id)
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
@@ -1039,76 +1011,69 @@ def remove_password(
doc.id,
pair.source_doc.source_path,
)
try:
with pikepdf.open(source_path) as pdf:
if not pdf.is_encrypted:
logger.info(
"Skipping password removal for document %s because the "
"source PDF is not encrypted",
pair.root_doc.id,
)
continue
except pikepdf.PasswordError:
# Password-protected PDFs need the supplied password below.
pass
with pikepdf.open(source_path, password=password) as pdf:
filepath: Path = (
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_unprotected.pdf"
if not pdf_ops.needs_decrypt(source_path):
logger.info(
"Skipping password removal for document %s because the "
"source PDF is not encrypted",
pair.root_doc.id,
)
pdf.remove_unreferenced_resources()
pdf.save(filepath)
continue
if update_document:
# Create a new version rather than modifying the root/original in place.
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata
else DocumentMetadataOverrides()
)
if user is not None:
overrides.owner_id = user.id
overrides.actor_id = user.id
consume_file.apply_async(
kwargs={
"input_doc": ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
root_document_id=pair.root_doc.id,
),
"overrides": overrides,
},
headers={"trigger_source": trigger_source},
)
filepath = pdf_ops.decrypt_pdf(
source_path,
partial(_scratch_path, f"{pair.root_doc.id}_unprotected.pdf"),
password,
)
if update_document:
# Create a new version rather than modifying the root/original in place.
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata
else DocumentMetadataOverrides()
)
if user is not None:
overrides.owner_id = user.id
overrides.actor_id = user.id
consume_file.apply_async(
kwargs={
"input_doc": ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
root_document_id=pair.root_doc.id,
),
"overrides": overrides,
},
headers={"trigger_source": trigger_source},
)
else:
consume_tasks = []
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata
else DocumentMetadataOverrides()
)
if user is not None:
overrides.owner_id = user.id
overrides.actor_id = user.id
consume_tasks.append(
consume_file.s(
input_doc=ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
),
overrides=overrides,
).set(headers={"trigger_source": trigger_source}),
)
if delete_original:
chord(
header=consume_tasks,
body=delete.si([doc.id]),
).delay()
else:
consume_tasks = []
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata
else DocumentMetadataOverrides()
)
if user is not None:
overrides.owner_id = user.id
overrides.actor_id = user.id
consume_tasks.append(
consume_file.s(
input_doc=ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
),
overrides=overrides,
).set(headers={"trigger_source": trigger_source}),
)
if delete_original:
chord(
header=consume_tasks,
body=delete.si([doc.id]),
).delay()
else:
group(consume_tasks).delay()
group(consume_tasks).delay()
except Exception as e:
logger.exception(
+96 -159
View File
@@ -1,8 +1,8 @@
from __future__ import annotations
import logging
import math
import mimetypes
import os
import shutil
import subprocess
import tempfile
@@ -68,6 +68,58 @@ def get_supported_file_extensions() -> set[str]:
return extensions
def run_convert(
input_file,
output_file,
*,
density=None,
scale=None,
alpha=None,
strip=False,
trim=False,
type=None,
depth=None,
auto_orient=False,
use_cropbox=False,
extra=None,
logging_group=None,
) -> None:
environment = os.environ.copy()
if settings.CONVERT_MEMORY_LIMIT:
# MAGICK_MEMORY_LIMIT sets the maximum amount of RAM the pixel cache can use.
# MAGICK_MAP_LIMIT sets the maximum amount of memory-mapped I/O allowed.
#
# For large-format documents ImageMagick will hit the RAM limit and
# immediately try to "map" the remaining data. If MAGICK_MAP_LIMIT isn't
# also set, the process may trigger an OOM kill because the default
# system/policy map limit is often too restrictive for these massive bitmaps.
environment["MAGICK_MEMORY_LIMIT"] = settings.CONVERT_MEMORY_LIMIT
environment["MAGICK_MAP_LIMIT"] = settings.CONVERT_MEMORY_LIMIT
if settings.CONVERT_TMPDIR:
environment["MAGICK_TMPDIR"] = settings.CONVERT_TMPDIR
args = [settings.CONVERT_BINARY]
args += ["-density", str(density)] if density else []
args += ["-scale", str(scale)] if scale else []
args += ["-alpha", str(alpha)] if alpha else []
args += ["-strip"] if strip else []
args += ["-trim"] if trim else []
args += ["-type", str(type)] if type else []
args += ["-depth", str(depth)] if depth else []
args += ["-auto-orient"] if auto_orient else []
args += ["-define", "pdf:use-cropbox=true"] if use_cropbox else []
args += [str(input_file), str(output_file)]
logger.debug("Execute: " + " ".join(args), extra={"group": logging_group})
try:
run_subprocess(args, environment, logger)
except subprocess.CalledProcessError as e:
raise ParseError(f"Convert failed at {args}") from e
except Exception as e: # pragma: no cover
raise ParseError("Unknown error running convert") from e
def get_default_thumbnail() -> Path:
"""
Returns the path to a generic thumbnail
@@ -75,168 +127,46 @@ def get_default_thumbnail() -> Path:
return (Path(__file__).parent / "resources" / "document.webp").resolve()
_THUMBNAIL_MAX_WIDTH = 500
_THUMBNAIL_MAX_HEIGHT = 5000
# Used only when the page geometry cannot be read
_THUMBNAIL_FALLBACK_DPI = 150
# Applied before supersampling, so tiny pages are not enlarged
_THUMBNAIL_MAX_DPI = 300
# Rendering at a multiple and downsampling keeps text crisper
_THUMBNAIL_SUPERSAMPLE = 2
def rasterize_pdf_page_to_png(
in_path: Path,
out_path: Path,
*,
dpi: int,
logging_group=None,
) -> None:
"""
Rasterizes the first page of a PDF to a PNG with pdftoppm.
"""
# -singlefile drops the page number and -png appends ".png", so pass the
# path without its suffix
args = [
"pdftoppm",
"-f",
"1",
"-l",
"1",
"-r",
str(dpi),
"-png",
"-singlefile",
"-cropbox",
str(in_path),
str(out_path.with_suffix("")),
]
logger.debug("Execute: " + " ".join(args), extra={"group": logging_group})
try:
run_subprocess(args, logger=logger)
except subprocess.CalledProcessError as e:
raise ParseError(f"pdftoppm failed at {args}") from e
except Exception as e: # pragma: no cover
raise ParseError("Unknown error running pdftoppm") from e
def encode_thumbnail_webp(
png_path: Path,
out_path: Path,
*,
supersample: int = 1,
) -> None:
"""
Flattens alpha onto white, undoes supersampling, shrinks to fit and saves as WebP.
"""
from PIL import Image
try:
with Image.open(png_path) as im:
if im.mode in ("RGBA", "LA"):
flattened = Image.new("RGB", im.size, (255, 255, 255))
flattened.paste(im, mask=im.split()[-1])
else:
flattened = im.convert("RGB")
if supersample > 1:
flattened = flattened.resize(
(
max(1, round(flattened.width / supersample)),
max(1, round(flattened.height / supersample)),
),
Image.Resampling.LANCZOS,
)
flattened.thumbnail((_THUMBNAIL_MAX_WIDTH, _THUMBNAIL_MAX_HEIGHT))
flattened.save(out_path, format="WEBP")
except (OSError, Image.DecompressionBombError) as e:
raise ParseError(f"Unable to encode thumbnail from {png_path}") from e
def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> tuple[int, int]:
"""
Returns (dpi, supersample). Unknown geometry is not supersampled, since
the render size cannot be bounded.
"""
from paperless.parsers.utils import get_pdf_first_page_size_points
size = get_pdf_first_page_size_points(in_path)
if size is None:
logger.debug(
"Could not read PDF page size, using fallback DPI",
extra={"group": logging_group},
)
return _THUMBNAIL_FALLBACK_DPI, 1
width_pts, height_pts = size
dpi_for_width = _THUMBNAIL_MAX_WIDTH * 72 / width_pts
dpi_for_height = _THUMBNAIL_MAX_HEIGHT * 72 / height_pts
# Round up so the downsampled render is never a few pixels short of the
# target; the shrink-only clamp in encode_thumbnail_webp trims the excess.
dpi = max(
1,
math.ceil(min(_THUMBNAIL_MAX_DPI, dpi_for_width, dpi_for_height)),
)
# At the 1 DPI floor the page is already oversized, so do not supersample
return dpi, 1 if dpi == 1 else _THUMBNAIL_SUPERSAMPLE
def _render_pdf_thumbnail(
in_path: Path,
png_path: Path,
out_path: Path,
logging_group=None,
) -> None:
dpi, supersample = _compute_thumbnail_dpi(in_path, logging_group=logging_group)
rasterize_pdf_page_to_png(
in_path,
png_path,
dpi=dpi * supersample,
logging_group=logging_group,
)
encode_thumbnail_webp(png_path, out_path, supersample=supersample)
def _repair_pdf_with_qpdf(in_path: Path, out_path: Path) -> None:
# qpdf exits 3 after a repair; --warning-exit-0 keeps that from failing
try:
shutil.copy(in_path, out_path)
run_subprocess(
["qpdf", "--warning-exit-0", "--replace-input", str(out_path)],
logger=logger,
)
except (subprocess.CalledProcessError, OSError) as e:
raise ParseError(f"qpdf repair failed for {in_path}") from e
def make_thumbnail_from_pdf_qpdf_fallback(
in_path: Path,
temp_dir: Path,
logging_group=None,
) -> Path:
png_path = temp_dir / "page1_repaired.png"
out_path = temp_dir / "convert_qpdf.webp"
repaired_path = temp_dir / "repaired.pdf"
def make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group=None) -> Path:
out_path: Path = Path(temp_dir) / "convert_gs.webp"
# if convert fails, fall back to extracting
# the first PDF page as a PNG using Ghostscript
logger.warning(
"Thumbnail generation with pdftoppm failed, attempting qpdf repair and retry.",
"Thumbnail generation with ImageMagick failed, falling back "
"to ghostscript. Check your /etc/ImageMagick-x/policy.xml!",
extra={"group": logging_group},
)
# Ghostscript doesn't handle WebP outputs
gs_out_path: Path = Path(temp_dir) / "gs_out.png"
cmd = [settings.GS_BINARY, "-q", "-sDEVICE=pngalpha", "-o", gs_out_path, in_path]
try:
_repair_pdf_with_qpdf(in_path, repaired_path)
_render_pdf_thumbnail(repaired_path, png_path, out_path, logging_group)
try:
run_subprocess(cmd, logger=logger)
except subprocess.CalledProcessError as e:
raise ParseError(f"Thumbnail (gs) failed at {cmd}") from e
# then run convert on the output from gs to make WebP
run_convert(
density=300,
scale="500x5000>",
alpha="remove",
strip=True,
trim=False,
auto_orient=True,
input_file=gs_out_path,
output_file=out_path,
logging_group=logging_group,
)
return out_path
except ParseError as e:
logger.error(f"Unable to make thumbnail after qpdf repair: {e}")
logger.error(f"Unable to make thumbnail with Ghostscript: {e}")
# The caller might expect a generated thumbnail that can be moved,
# so we need to copy it before it gets moved.
# https://github.com/paperless-ngx/paperless-ngx/issues/3631
default_thumbnail_path = temp_dir / "document.webp"
default_thumbnail_path: Path = Path(temp_dir) / "document.webp"
copy_file_with_basic_stats(get_default_thumbnail(), default_thumbnail_path)
return default_thumbnail_path
@@ -245,18 +175,25 @@ def make_thumbnail_from_pdf(in_path: Path, temp_dir: Path, logging_group=None) -
"""
The thumbnail of a PDF is just a 500px wide image of the first page.
"""
png_path: Path = temp_dir / "page1.png"
out_path: Path = temp_dir / "convert.webp"
# Run convert to get a decent thumbnail
try:
_render_pdf_thumbnail(in_path, png_path, out_path, logging_group)
except ParseError as e:
logger.error(f"Unable to make thumbnail with pdftoppm: {e}")
out_path = make_thumbnail_from_pdf_qpdf_fallback(
in_path,
temp_dir,
logging_group,
run_convert(
density=300,
scale="500x5000>",
alpha="remove",
strip=True,
trim=False,
auto_orient=True,
use_cropbox=True,
input_file=f"{in_path}[0]",
output_file=str(out_path),
logging_group=logging_group,
)
except ParseError as e:
logger.error(f"Unable to make thumbnail with convert: {e}")
out_path = make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group)
return out_path
+194
View File
@@ -0,0 +1,194 @@
"""
Pure PDF page operations used by documents.bulk_edit.
This module deliberately knows nothing about Django, Celery or the documents
app: callers resolve documents, choose output paths and queue work. Every
function that writes a PDF removes unreferenced resources before saving.
pikepdf is always called as ``pikepdf.open(...)`` / ``pikepdf.new()`` (never
``from pikepdf import open``) so tests can patch those module attributes.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
from typing import NamedTuple
import pikepdf
if TYPE_CHECKING:
from collections.abc import Callable
from collections.abc import Iterable
from collections.abc import Mapping
from collections.abc import Sequence
from pathlib import Path
from types import TracebackType
class PageSpec(NamedTuple):
"""One page of an output PDF: a 1-indexed source page, optionally rotated."""
page: int
rotate: int = 0 # relative degrees, 0 leaves the page alone
def _require_positive(pages: Iterable[int]) -> None:
for page in pages:
if page < 1:
raise ValueError(f"Page numbers start at 1, got {page}")
def rotate_pdf(src: Path, dst: Path, degrees: int) -> None:
"""
Rotate every page relatively on the opened document, not a rebuild, so Info,
XMP and outlines are kept. ``src`` is not modified.
"""
with pikepdf.open(src) as pdf:
for page in pdf.pages:
page.rotate(degrees, relative=True)
pdf.remove_unreferenced_resources()
pdf.save(dst)
def remove_pages(src: Path, dst: Path, pages: Iterable[int]) -> None:
"""
Remove 1-indexed pages from the opened document, not a rebuild, so Info, XMP
and outlines are kept. ``src`` is not modified.
Duplicates are ignored. Pages are removed highest first so earlier removals
never shift the index of later ones.
"""
unique = sorted(set(pages))
_require_positive(unique)
with pikepdf.open(src) as pdf:
for page_num in reversed(unique):
del pdf.pages[page_num - 1]
pdf.remove_unreferenced_resources()
pdf.save(dst)
def build_pdfs(
src: Path,
outputs: Sequence[tuple[Sequence[PageSpec], Callable[[], Path]]],
) -> list[Path]:
"""
Build one new PDF per output from pages of ``src``, opening ``src`` once.
Each output is ``(page_specs, make_dst)``. Every page number is checked against
``src`` before any output is built, and ``make_dst`` is called after that
output's pages are copied and immediately before it is saved, so a bad page
number in any output never leaves a destination behind. Document-level data
(Info, XMP, outlines) is not carried over. Returns the written paths in output
order.
"""
for specs, _ in outputs:
_require_positive(spec.page for spec in specs)
written: list[Path] = []
with pikepdf.open(src) as source:
page_count = len(source.pages)
for specs, _ in outputs:
for spec in specs:
if spec.page > page_count:
raise IndexError(
f"Page {spec.page} is out of range, the PDF has "
f"{page_count} pages",
)
for specs, make_dst in outputs:
dst = pikepdf.new()
for spec in specs:
dst.pages.append(source.pages[spec.page - 1])
if spec.rotate:
dst.pages[-1].rotate(spec.rotate, relative=True)
dst.remove_unreferenced_resources()
path = make_dst()
dst.save(path)
dst.close()
written.append(path)
return written
def validate_page_operations(
operations: Sequence[Mapping[str, int]],
*,
single_output: bool,
) -> int:
"""
Validate ``edit_pdf`` style operations and return the output document count.
Each operation has ``page`` and optionally ``rotate`` and ``doc`` (the output
document index, default 0). The bounds rule is kept as it was: a ``doc`` index
must be below the number of operations.
"""
if not operations:
raise ValueError("Output document index is out of bounds")
max_idx = max(op.get("doc", 0) for op in operations)
if single_output and max_idx > 0:
raise ValueError("Multiple output documents specified")
if any(
op.get("doc", 0) < 0 or op.get("doc", 0) >= len(operations) for op in operations
):
raise ValueError("Output document index is out of bounds")
return max_idx + 1
def needs_decrypt(src: Path) -> bool:
"""
True if ``src`` is encrypted. A PDF that needs a password to open at all
counts as encrypted.
"""
try:
with pikepdf.open(src) as pdf:
return bool(pdf.is_encrypted)
except pikepdf.PasswordError:
return True
def decrypt_pdf(src: Path, make_dst: Callable[[], Path], password: str) -> Path:
"""
Write an unencrypted copy of ``src`` and return its path.
``make_dst`` is only called once the password has been accepted, so a wrong
password never leaves a destination behind.
"""
with pikepdf.open(src, password=password) as pdf:
pdf.remove_unreferenced_resources()
dst = make_dst()
pdf.save(dst)
return dst
class PdfMerger:
"""
Accumulates the pages of several PDFs into one new PDF.
``add`` raises if a source cannot be read; deciding whether to skip it is the
caller's policy. Use as a context manager so the merged PDF is closed.
"""
def __init__(self) -> None:
self._pdf = pikepdf.new()
self._version: str = self._pdf.pdf_version
def __enter__(self) -> PdfMerger:
return self
def __exit__(
self,
exc_type: type[BaseException] | None,
exc: BaseException | None,
tb: TracebackType | None,
) -> None:
self._pdf.close()
def add(self, path: Path) -> None:
with pikepdf.open(str(path)) as pdf:
self._version = max(self._version, pdf.pdf_version)
self._pdf.pages.extend(pdf.pages)
def save(self, dst: Path) -> None:
self._pdf.remove_unreferenced_resources()
self._pdf.save(dst, min_version=self._version)
+2
View File
@@ -2144,6 +2144,8 @@ class BulkEditSerializer(
raise serializers.ValidationError("pages must be a list")
if not all(isinstance(i, int) for i in parameters["pages"]):
raise serializers.ValidationError("pages must be a list of integers")
if any(i < 1 for i in parameters["pages"]):
raise serializers.ValidationError("pages must be positive integers")
def _validate_parameters_merge(self, parameters) -> None:
if "delete_originals" in parameters:
+30
View File
@@ -1843,6 +1843,36 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
m.assert_called_once()
self.assertEqual(m.call_args.kwargs["pages"], [[1], [2, 3, 4], [5]])
@mock.patch("documents.serialisers.bulk_edit.delete_pages")
def test_bulk_edit_delete_pages_rejects_pages_below_one(self, m) -> None:
"""
GIVEN:
- A legacy delete_pages bulk edit
WHEN:
- API to bulk edit is called with a page number below 1
THEN:
- API returns HTTP 400
- delete_pages is not called
"""
self.setup_mock(m, "delete_pages")
for pages in ([0], [-1], [1, 0]):
with self.subTest(pages=pages):
response = self.client.post(
"/api/documents/bulk_edit/",
json.dumps(
{
"documents": [self.doc2.id],
"method": "delete_pages",
"parameters": {"pages": pages},
},
),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn(b"pages must be positive integers", response.content)
m.assert_not_called()
@mock.patch("documents.views.bulk_edit.rotate")
def test_rotate_insufficient_permissions(self, m) -> None:
self.doc1.owner = User.objects.get(username="temp_admin")
+152 -231
View File
@@ -1,9 +1,9 @@
import shutil
from collections.abc import Callable
from datetime import date
from pathlib import Path
from unittest import mock
import pikepdf
from django.contrib.auth.models import Group
from django.contrib.auth.models import Permission
from django.contrib.auth.models import User
@@ -21,6 +21,7 @@ from documents.models import Document
from documents.models import DocumentType
from documents.models import StoragePath
from documents.models import Tag
from documents.pdf_ops import PageSpec
from documents.permissions import set_permissions_for_objects
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.permissions import grant_object
@@ -793,16 +794,14 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.img_doc.save()
@staticmethod
def mock_password_required_pdf(
mock_open: mock.Mock,
fake_pdf: mock.Mock,
) -> None:
password_context = mock.MagicMock()
password_context.__enter__.return_value = fake_pdf
mock_open.side_effect = [
pikepdf.PasswordError("password required"),
password_context,
]
def fake_decrypt(
src: Path,
make_dst: Callable[[], Path],
password: str,
) -> Path:
dst = make_dst()
dst.write_bytes(b"password removed")
return dst
@mock.patch("documents.tasks.consume_file.s")
def test_merge(self, mock_consume_file) -> None:
@@ -847,12 +846,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(result, "OK")
@mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.PdfMerger")
@mock.patch("documents.tasks.consume_file.s")
def test_merge_uses_latest_version_source_for_root_selection(
self,
mock_consume_file,
mock_open_pdf,
mock_merger,
) -> None:
version_file = self.dirs.scratch_dir / "sample2_version_merge.pdf"
shutil.copy(self.doc2.source_path, version_file)
@@ -863,16 +862,14 @@ class TestPDFActions(DirectoriesMixin, TestCase):
filename=version_file,
mime_type="application/pdf",
)
fake_pdf = mock.MagicMock()
fake_pdf.pdf_version = "1.7"
fake_pdf.pages = [mock.Mock()]
mock_open_pdf.return_value.__enter__.return_value = fake_pdf
merger = mock_merger.return_value.__enter__.return_value
merger.save.side_effect = lambda dst: shutil.copy(version.source_path, dst)
result = bulk_edit.merge([self.doc2.id])
self.assertEqual(result, "OK")
mock_open_pdf.assert_called_once_with(str(version.source_path))
mock_consume_file.assert_not_called()
merger.add.assert_called_once_with(version.source_path)
mock_consume_file.assert_called_once()
@mock.patch("documents.bulk_edit.delete.si")
@mock.patch("documents.tasks.consume_file.s")
@@ -1034,18 +1031,18 @@ class TestPDFActions(DirectoriesMixin, TestCase):
)
@mock.patch("documents.tasks.consume_file.delay")
@mock.patch("pikepdf.open")
def test_merge_with_errors(self, mock_open_pdf, mock_consume_file) -> None:
@mock.patch("documents.pdf_ops.PdfMerger.add")
def test_merge_with_errors(self, mock_add, mock_consume_file) -> None:
"""
GIVEN:
- Existing documents
WHEN:
- Merge action is called with 2 documents
- Error occurs when opening both files
- Error occurs when adding both files
THEN:
- Consume file should not be called
"""
mock_open_pdf.side_effect = Exception("Error opening PDF")
mock_add.side_effect = Exception("Error opening PDF")
doc_ids = [self.doc2.id, self.doc3.id]
with self.assertLogs("paperless.bulk_edit", level="ERROR") as cm:
@@ -1082,12 +1079,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(result, "OK")
@mock.patch("documents.bulk_edit.group")
@mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.build_pdfs")
@mock.patch("documents.tasks.consume_file.s")
def test_split_uses_latest_version_source_for_root_selection(
self,
mock_consume_file,
mock_open_pdf,
mock_build_pdfs,
mock_group,
) -> None:
version_file = self.dirs.scratch_dir / "sample2_version_split.pdf"
@@ -1099,17 +1096,15 @@ class TestPDFActions(DirectoriesMixin, TestCase):
filename=version_file,
mime_type="application/pdf",
)
fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock(), mock.Mock()]
mock_open_pdf.return_value.__enter__.return_value = fake_pdf
mock_build_pdfs.return_value = [version.source_path, version.source_path]
mock_group.return_value.delay.return_value = None
result = bulk_edit.split([self.doc2.id], [[1], [2]])
self.assertEqual(result, "OK")
mock_open_pdf.assert_called_once_with(version.source_path)
mock_consume_file.assert_not_called()
mock_group.return_value.delay.assert_not_called()
self.assertEqual(mock_build_pdfs.call_args.args[0], version.source_path)
self.assertEqual(mock_consume_file.call_count, 2)
mock_group.return_value.delay.assert_called_once()
@mock.patch("documents.bulk_edit.delete.si")
@mock.patch("documents.tasks.consume_file.s")
@@ -1197,18 +1192,18 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(self.doc2.archive_serial_number, 222)
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("pikepdf.Pdf.save")
def test_split_with_errors(self, mock_save_pdf, mock_consume_file) -> None:
@mock.patch("documents.pdf_ops.build_pdfs")
def test_split_with_errors(self, mock_build_pdfs, mock_consume_file) -> None:
"""
GIVEN:
- Existing documents
WHEN:
- Split action is called with 1 document and 2 page groups
- Error occurs when saving the files
- Error occurs when building the files
THEN:
- Consume file should not be called
"""
mock_save_pdf.side_effect = Exception("Error saving PDF")
mock_build_pdfs.side_effect = Exception("Error building PDFs")
doc_ids = [self.doc2.id]
pages = [[1, 2], [3]]
@@ -1243,10 +1238,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(result, "OK")
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("pikepdf.Pdf.save")
@mock.patch("documents.pdf_ops.rotate_pdf")
def test_rotate_with_error(
self,
mock_pdf_save,
mock_rotate_pdf,
mock_consume_delay,
) -> None:
"""
@@ -1254,11 +1249,11 @@ class TestPDFActions(DirectoriesMixin, TestCase):
- Existing documents
WHEN:
- Rotate action is called with 2 documents
- PikePDF raises an error
- Rotating the PDF raises an error
THEN:
- Rotate action should be called 0 times
"""
mock_pdf_save.side_effect = Exception("Error saving PDF")
mock_rotate_pdf.side_effect = Exception("Error rotating PDF")
doc_ids = [self.doc2.id, self.doc3.id]
with self.assertLogs("paperless.bulk_edit", level="ERROR") as cm:
@@ -1293,10 +1288,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.rotate_pdf")
def test_rotate_explicit_selection_uses_root_source_when_root_selected(
self,
mock_open,
mock_rotate_pdf,
mock_consume_delay,
mock_magic,
) -> None:
@@ -1305,9 +1300,6 @@ class TestPDFActions(DirectoriesMixin, TestCase):
title="B version 1",
root_document=self.doc2,
)
fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock()]
mock_open.return_value.__enter__.return_value = fake_pdf
result = bulk_edit.rotate(
[self.doc2.id],
@@ -1316,26 +1308,35 @@ class TestPDFActions(DirectoriesMixin, TestCase):
)
self.assertEqual(result, "OK")
mock_open.assert_called_once_with(self.doc2.source_path)
self.assertEqual(mock_rotate_pdf.call_args.args[0], self.doc2.source_path)
mock_consume_delay.assert_called_once()
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("pikepdf.Pdf.save")
@mock.patch("documents.pdf_ops.remove_pages")
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
def test_delete_pages(self, mock_magic, mock_pdf_save, mock_consume_delay) -> None:
def test_delete_pages(
self,
mock_magic,
mock_remove_pages,
mock_consume_delay,
) -> None:
"""
GIVEN:
- Existing documents
WHEN:
- Delete pages action is called with 1 document and 2 pages
THEN:
- Save should be called once
- The pages are removed from the document's source PDF
- A new version should be enqueued via consume_file
"""
doc_ids = [self.doc2.id]
pages = [1, 3]
result = bulk_edit.delete_pages(doc_ids, pages)
mock_pdf_save.assert_called_once()
mock_remove_pages.assert_called_once_with(
self.doc2.source_path,
mock.ANY,
[1, 3],
)
mock_consume_delay.assert_called_once()
task_kwargs = mock_consume_delay.call_args.kwargs["kwargs"]
self.assertEqual(task_kwargs["input_doc"].root_document_id, self.doc2.id)
@@ -1347,10 +1348,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.remove_pages")
def test_delete_pages_explicit_selection_uses_root_source_when_root_selected(
self,
mock_open,
mock_remove_pages,
mock_consume_delay,
mock_magic,
) -> None:
@@ -1359,9 +1360,6 @@ class TestPDFActions(DirectoriesMixin, TestCase):
title="B version 1",
root_document=self.doc2,
)
fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock(), mock.Mock()]
mock_open.return_value.__enter__.return_value = fake_pdf
result = bulk_edit.delete_pages(
[self.doc2.id],
@@ -1370,23 +1368,26 @@ class TestPDFActions(DirectoriesMixin, TestCase):
)
self.assertEqual(result, "OK")
mock_open.assert_called_once_with(self.doc2.source_path)
self.assertEqual(mock_remove_pages.call_args.args[0], self.doc2.source_path)
mock_consume_delay.assert_called_once()
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("pikepdf.Pdf.save")
def test_delete_pages_with_error(self, mock_pdf_save, mock_consume_delay) -> None:
@mock.patch("documents.pdf_ops.remove_pages")
def test_delete_pages_with_error(
self,
mock_remove_pages,
mock_consume_delay,
) -> None:
"""
GIVEN:
- Existing documents
WHEN:
- Delete pages action is called with 1 document and 2 pages
- PikePDF raises an error
- Removing the pages raises an error
THEN:
- Save should be called once
- No new version should be enqueued
"""
mock_pdf_save.side_effect = Exception("Error saving PDF")
mock_remove_pages.side_effect = Exception("Error removing pages")
doc_ids = [self.doc2.id]
pages = [1, 3]
@@ -1416,6 +1417,41 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(result, "OK")
mock_group.return_value.delay.assert_called_once()
@mock.patch("documents.bulk_edit.group")
@mock.patch("documents.pdf_ops.build_pdfs")
@mock.patch("documents.tasks.consume_file.s")
def test_edit_pdf_maps_operations_to_outputs(
self,
mock_consume_file: mock.Mock,
mock_build_pdfs: mock.Mock,
mock_group: mock.Mock,
) -> None:
"""
GIVEN:
- Existing document
WHEN:
- edit_pdf is called with operations interleaved across two outputs,
some of them rotated
THEN:
- Each output is built from its own operations, in operation order,
with the requested rotation
"""
mock_build_pdfs.return_value = [self.doc2.source_path, self.doc2.source_path]
mock_group.return_value.delay.return_value = None
operations = [
{"page": 3, "doc": 1},
{"page": 1, "doc": 0, "rotate": 90},
{"page": 2, "doc": 1, "rotate": 180},
]
bulk_edit.edit_pdf([self.doc2.id], operations)
outputs = mock_build_pdfs.call_args.args[1]
self.assertEqual(
[specs for specs, _ in outputs],
[[PageSpec(1, 90)], [PageSpec(3), PageSpec(2, 180)]],
)
@mock.patch("documents.bulk_edit.group")
@mock.patch("documents.tasks.consume_file.s")
def test_edit_pdf_with_user_override(self, mock_consume_file, mock_group) -> None:
@@ -1531,12 +1567,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("pikepdf.new")
@mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.build_pdfs")
def test_edit_pdf_explicit_selection_uses_root_source_when_root_selected(
self,
mock_open,
mock_new,
mock_build_pdfs,
mock_consume_delay,
mock_magic,
) -> None:
@@ -1545,12 +1579,7 @@ class TestPDFActions(DirectoriesMixin, TestCase):
title="B version 1",
root_document=self.doc2,
)
fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock()]
mock_open.return_value.__enter__.return_value = fake_pdf
output_pdf = mock.MagicMock()
output_pdf.pages = []
mock_new.return_value = output_pdf
mock_build_pdfs.return_value = [Path("edited.pdf")]
result = bulk_edit.edit_pdf(
[self.doc2.id],
@@ -1560,7 +1589,7 @@ class TestPDFActions(DirectoriesMixin, TestCase):
)
self.assertEqual(result, "OK")
mock_open.assert_called_once_with(self.doc2.source_path)
self.assertEqual(mock_build_pdfs.call_args.args[0], self.doc2.source_path)
mock_consume_delay.assert_called_once()
@mock.patch("documents.bulk_edit.group")
@@ -1586,31 +1615,6 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(result, "OK")
mock_group.return_value.delay.assert_called_once()
@mock.patch("documents.bulk_edit.group")
@mock.patch("documents.tasks.consume_file.s")
def test_edit_pdf_open_failure(
self,
mock_consume_file: mock.Mock,
mock_group: mock.Mock,
) -> None:
"""
GIVEN:
- Existing document
WHEN:
- edit_pdf fails to open PDF
THEN:
- Task group is not called
"""
doc_ids = [self.doc2.id]
operations = [
{"page": 9999}, # invalid page, forces error during PDF load
]
with self.assertLogs("paperless.bulk_edit", level="ERROR"):
with self.assertRaises(Exception):
bulk_edit.edit_pdf(doc_ids, operations)
mock_group.assert_not_called()
mock_consume_file.assert_not_called()
@mock.patch("documents.bulk_edit.group")
@mock.patch("documents.tasks.consume_file.s")
def test_edit_pdf_multiple_outputs_with_update_flag_errors(
@@ -1637,23 +1641,15 @@ class TestPDFActions(DirectoriesMixin, TestCase):
mock_group.assert_not_called()
mock_consume_file.assert_not_called()
@mock.patch("pikepdf.open")
def test_edit_pdf_rejects_invalid_operations(self, mock_open) -> None:
for operations in ([], [{"page": 1, "doc": 2**32}]):
with self.subTest(operations=operations):
with self.assertLogs("paperless.bulk_edit", level="ERROR"):
with self.assertRaisesRegex(ValueError, "index is out of bounds"):
bulk_edit.edit_pdf([self.doc2.id], operations)
mock_open.assert_not_called()
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file.delay")
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
@mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.decrypt_pdf")
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
def test_remove_password_update_document(
self,
mock_open,
mock_needs_decrypt,
mock_decrypt,
mock_mkdtemp,
mock_consume_delay,
mock_update_document,
@@ -1662,16 +1658,7 @@ class TestPDFActions(DirectoriesMixin, TestCase):
temp_dir = self.dirs.scratch_dir / "remove-password-update"
temp_dir.mkdir(parents=True, exist_ok=True)
mock_mkdtemp.return_value = str(temp_dir)
fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock(), mock.Mock(), mock.Mock()]
fake_pdf.is_encrypted = True
def save_side_effect(target_path):
Path(target_path).write_bytes(b"new pdf content")
fake_pdf.save.side_effect = save_side_effect
mock_open.return_value.__enter__.return_value = fake_pdf
mock_decrypt.side_effect = self.fake_decrypt
result = bulk_edit.remove_password(
[doc.id],
@@ -1680,14 +1667,8 @@ class TestPDFActions(DirectoriesMixin, TestCase):
)
self.assertEqual(result, "OK")
self.assertEqual(
mock_open.call_args_list,
[
mock.call(doc.source_path),
mock.call(doc.source_path, password="secret"),
],
)
fake_pdf.remove_unreferenced_resources.assert_called_once()
mock_needs_decrypt.assert_called_once_with(doc.source_path)
mock_decrypt.assert_called_once_with(doc.source_path, mock.ANY, "secret")
mock_update_document.assert_not_called()
mock_consume_delay.assert_called_once()
task_kwargs = mock_consume_delay.call_args.kwargs["kwargs"]
@@ -1700,40 +1681,15 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(task_kwargs["input_doc"].root_document_id, doc.id)
self.assertIsNotNone(task_kwargs["overrides"])
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
@mock.patch("pikepdf.open")
def test_remove_password_update_document_skips_unencrypted_pdf(
self,
mock_open,
mock_mkdtemp,
mock_consume_delay,
) -> None:
doc = self.doc1
fake_pdf = mock.MagicMock()
fake_pdf.is_encrypted = False
mock_open.return_value.__enter__.return_value = fake_pdf
result = bulk_edit.remove_password(
[doc.id],
password="secret",
update_document=True,
)
self.assertEqual(result, "OK")
mock_open.assert_called_once_with(doc.source_path)
fake_pdf.remove_unreferenced_resources.assert_not_called()
fake_pdf.save.assert_not_called()
mock_mkdtemp.assert_not_called()
mock_consume_delay.assert_not_called()
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file.delay")
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
@mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.decrypt_pdf")
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
def test_remove_password_update_document_uses_source_paths(
self,
mock_open,
mock_needs_decrypt,
mock_decrypt,
mock_mkdtemp,
mock_consume_delay,
mock_update_document,
@@ -1744,14 +1700,7 @@ class TestPDFActions(DirectoriesMixin, TestCase):
temp_dir = self.dirs.scratch_dir / "remove-password-source-file"
temp_dir.mkdir(parents=True, exist_ok=True)
mock_mkdtemp.return_value = str(temp_dir)
fake_pdf = mock.MagicMock()
self.mock_password_required_pdf(mock_open, fake_pdf)
def save_side_effect(target_path):
Path(target_path).write_bytes(b"new pdf content")
fake_pdf.save.side_effect = save_side_effect
mock_decrypt.side_effect = self.fake_decrypt
result = bulk_edit.remove_password(
[doc.id],
@@ -1761,22 +1710,19 @@ class TestPDFActions(DirectoriesMixin, TestCase):
)
self.assertEqual(result, "OK")
self.assertEqual(
mock_open.call_args_list,
[
mock.call(source_file),
mock.call(source_file, password="secret"),
],
)
mock_needs_decrypt.assert_called_once_with(source_file)
mock_decrypt.assert_called_once_with(source_file, mock.ANY, "secret")
mock_update_document.assert_not_called()
mock_consume_delay.assert_called_once()
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.decrypt_pdf")
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
def test_remove_password_explicit_selection_uses_root_source_when_root_selected(
self,
mock_open,
mock_needs_decrypt,
mock_decrypt,
mock_consume_delay,
mock_magic,
) -> None:
@@ -1785,8 +1731,7 @@ class TestPDFActions(DirectoriesMixin, TestCase):
title="A version 1",
root_document=self.doc1,
)
fake_pdf = mock.MagicMock()
self.mock_password_required_pdf(mock_open, fake_pdf)
mock_decrypt.return_value = Path("unprotected.pdf")
result = bulk_edit.remove_password(
[self.doc1.id],
@@ -1796,12 +1741,11 @@ class TestPDFActions(DirectoriesMixin, TestCase):
)
self.assertEqual(result, "OK")
self.assertEqual(
mock_open.call_args_list,
[
mock.call(self.doc1.source_path),
mock.call(self.doc1.source_path, password="secret"),
],
mock_needs_decrypt.assert_called_once_with(self.doc1.source_path)
mock_decrypt.assert_called_once_with(
self.doc1.source_path,
mock.ANY,
"secret",
)
mock_consume_delay.assert_called_once()
@@ -1809,10 +1753,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
@mock.patch("documents.bulk_edit.group")
@mock.patch("documents.tasks.consume_file.s")
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
@mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.decrypt_pdf")
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
def test_remove_password_creates_consumable_document(
self,
mock_open: mock.Mock,
mock_needs_decrypt: mock.Mock,
mock_decrypt: mock.Mock,
mock_mkdtemp: mock.Mock,
mock_consume_file: mock.Mock,
mock_group: mock.Mock,
@@ -1822,15 +1768,7 @@ class TestPDFActions(DirectoriesMixin, TestCase):
temp_dir = self.dirs.scratch_dir / "remove-password"
temp_dir.mkdir(parents=True, exist_ok=True)
mock_mkdtemp.return_value = str(temp_dir)
fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock(), mock.Mock()]
self.mock_password_required_pdf(mock_open, fake_pdf)
def save_side_effect(target_path: Path) -> None:
target_path.write_bytes(b"password removed")
fake_pdf.save.side_effect = save_side_effect
mock_decrypt.side_effect = self.fake_decrypt
mock_group.return_value.delay.return_value = None
user = User.objects.create(username="owner")
@@ -1845,13 +1783,7 @@ class TestPDFActions(DirectoriesMixin, TestCase):
)
self.assertEqual(result, "OK")
self.assertEqual(
mock_open.call_args_list,
[
mock.call(doc.source_path),
mock.call(doc.source_path, password="secret"),
],
)
mock_decrypt.assert_called_once_with(doc.source_path, mock.ANY, "secret")
mock_consume_file.assert_called_once()
call_kwargs = mock_consume_file.call_args.kwargs
consumable_document = call_kwargs["input_doc"]
@@ -1873,21 +1805,18 @@ class TestPDFActions(DirectoriesMixin, TestCase):
@mock.patch("documents.bulk_edit.chord")
@mock.patch("documents.bulk_edit.group")
@mock.patch("documents.tasks.consume_file.s")
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
@mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.decrypt_pdf")
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=False)
def test_remove_password_skips_unencrypted_pdf_without_queueing(
self,
mock_open: mock.Mock,
mock_mkdtemp: mock.Mock,
mock_needs_decrypt: mock.Mock,
mock_decrypt: mock.Mock,
mock_consume_file: mock.Mock,
mock_group: mock.Mock,
mock_chord: mock.Mock,
mock_delete: mock.Mock,
) -> None:
doc = self.doc2
fake_pdf = mock.MagicMock()
fake_pdf.is_encrypted = False
mock_open.return_value.__enter__.return_value = fake_pdf
result = bulk_edit.remove_password(
[doc.id],
@@ -1897,10 +1826,8 @@ class TestPDFActions(DirectoriesMixin, TestCase):
)
self.assertEqual(result, "OK")
mock_open.assert_called_once_with(doc.source_path)
fake_pdf.remove_unreferenced_resources.assert_not_called()
fake_pdf.save.assert_not_called()
mock_mkdtemp.assert_not_called()
mock_needs_decrypt.assert_called_once_with(doc.source_path)
mock_decrypt.assert_not_called()
mock_consume_file.assert_not_called()
mock_group.assert_not_called()
mock_chord.assert_not_called()
@@ -1911,10 +1838,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
@mock.patch("documents.bulk_edit.group")
@mock.patch("documents.tasks.consume_file.s")
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
@mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.decrypt_pdf")
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
def test_remove_password_deletes_original(
self,
mock_open: mock.Mock,
mock_needs_decrypt: mock.Mock,
mock_decrypt: mock.Mock,
mock_mkdtemp: mock.Mock,
mock_consume_file: mock.Mock,
mock_group: mock.Mock,
@@ -1925,15 +1854,7 @@ class TestPDFActions(DirectoriesMixin, TestCase):
temp_dir = self.dirs.scratch_dir / "remove-password-delete"
temp_dir.mkdir(parents=True, exist_ok=True)
mock_mkdtemp.return_value = str(temp_dir)
fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock(), mock.Mock()]
self.mock_password_required_pdf(mock_open, fake_pdf)
def save_side_effect(target_path: Path) -> None:
target_path.write_bytes(b"password removed")
fake_pdf.save.side_effect = save_side_effect
mock_decrypt.side_effect = self.fake_decrypt
mock_chord.return_value.delay.return_value = None
result = bulk_edit.remove_password(
@@ -1945,23 +1866,23 @@ class TestPDFActions(DirectoriesMixin, TestCase):
)
self.assertEqual(result, "OK")
self.assertEqual(
mock_open.call_args_list,
[
mock.call(doc.source_path),
mock.call(doc.source_path, password="secret"),
],
)
mock_decrypt.assert_called_once_with(doc.source_path, mock.ANY, "secret")
mock_consume_file.assert_called_once()
mock_group.assert_not_called()
mock_chord.assert_called_once()
mock_chord.return_value.delay.assert_called_once()
mock_delete.si.assert_called_once_with([doc.id])
@mock.patch("pikepdf.open")
def test_remove_password_open_failure(self, mock_open: mock.Mock) -> None:
mock_open.side_effect = RuntimeError("wrong password")
@mock.patch(
"documents.pdf_ops.decrypt_pdf",
side_effect=RuntimeError("wrong password"),
)
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
def test_remove_password_failure_raises_value_error(
self,
mock_needs_decrypt: mock.Mock,
mock_decrypt: mock.Mock,
) -> None:
with self.assertLogs("paperless.bulk_edit", level="ERROR") as cm:
with self.assertRaises(ValueError) as exc:
bulk_edit.remove_password([self.doc1.id], password="secret")
-382
View File
@@ -1,22 +1,11 @@
import subprocess
from collections.abc import Generator
from pathlib import Path
import pikepdf
import pytest
from PIL import Image
from pytest_django.fixtures import Settings
from pytest_mock import MockerFixture
from documents.parsers import ParseError
from documents.parsers import _compute_thumbnail_dpi
from documents.parsers import encode_thumbnail_webp
from documents.parsers import get_default_file_extension
from documents.parsers import get_default_thumbnail
from documents.parsers import get_supported_file_extensions
from documents.parsers import is_file_ext_supported
from documents.parsers import make_thumbnail_from_pdf
from documents.parsers import rasterize_pdf_page_to_png
from paperless.parsers.registry import get_parser_registry
from paperless.parsers.registry import reset_parser_registry
from paperless.parsers.tesseract import RasterisedDocumentParser
@@ -136,374 +125,3 @@ class TestParserAvailability:
assert is_file_ext_supported(".pdf")
assert not is_file_ext_supported(".hsdfh")
assert not is_file_ext_supported("")
class TestComputeThumbnailDpi:
@pytest.mark.parametrize(
("size", "expected"),
[
pytest.param((612.0, 792.0), (59, 2), id="letter-width-bound"),
pytest.param((792.0, 612.0), (46, 2), id="landscape-rounded-up"),
pytest.param((612.0, 100000.0), (4, 2), id="tall-strip-height-bound"),
pytest.param((200.0, 300.0), (180, 2), id="small-page-width-bound"),
pytest.param((72.0, 72.0), (300, 2), id="tiny-page-capped-at-300"),
pytest.param(
(1000000.0, 1000000.0),
(1, 1),
id="huge-page-minimum-one-unsupersampled",
),
pytest.param(None, (150, 1), id="unreadable-geometry-fallback"),
],
)
def test_dpi_from_page_size(
self,
mocker: MockerFixture,
tmp_path: Path,
size: tuple[float, float] | None,
expected: tuple[int, int],
) -> None:
"""
GIVEN:
- A first page of the given size, or unreadable geometry
WHEN:
- The thumbnail DPI is computed
THEN:
- The expected DPI and supersample factor are returned
"""
mocker.patch(
"paperless.parsers.utils.get_pdf_first_page_size_points",
return_value=size,
)
assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected
class TestMakeThumbnailFromPdf:
@pytest.fixture
def work_dir(self, tmp_path: Path) -> Path:
path = tmp_path / "work"
path.mkdir()
return path
@staticmethod
def _write_blank_pdf(path: Path, page_size: tuple[int, int] = (612, 792)) -> Path:
pdf = pikepdf.new()
pdf.add_blank_page(page_size=page_size)
pdf.save(path, object_stream_mode=pikepdf.ObjectStreamMode.disable)
return path
@pytest.mark.parametrize(
("size", "expected_dpi", "expected_supersample"),
[
pytest.param((612.0, 792.0), 118, 2, id="known-geometry-2x"),
pytest.param(None, 150, 1, id="unreadable-geometry-plain-fallback"),
],
)
def test_render_dpi_requested(
self,
mocker: MockerFixture,
tmp_path: Path,
work_dir: Path,
size: tuple[float, float] | None,
expected_dpi: int,
expected_supersample: int,
) -> None:
"""
GIVEN:
- Readable or unreadable page geometry
WHEN:
- A thumbnail is made
THEN:
- Rasterize and encode get the matching DPI and supersample factor
"""
mocker.patch(
"paperless.parsers.utils.get_pdf_first_page_size_points",
return_value=size,
)
rasterize = mocker.patch("documents.parsers.rasterize_pdf_page_to_png")
encode = mocker.patch("documents.parsers.encode_thumbnail_webp")
make_thumbnail_from_pdf(tmp_path / "in.pdf", work_dir)
assert rasterize.call_args.kwargs["dpi"] == expected_dpi
assert encode.call_args.kwargs["supersample"] == expected_supersample
@pytest.mark.parametrize(
("page_size", "expected_width"),
[
pytest.param((612, 792), 500, id="letter"),
pytest.param((792, 612), 500, id="landscape-letter"),
pytest.param((595, 842), 500, id="a4"),
pytest.param((200, 300), 500, id="small-page"),
pytest.param((72, 72), 300, id="tiny-page-capped"),
],
)
def test_thumbnail_width(
self,
tmp_path: Path,
work_dir: Path,
page_size: tuple[int, int],
expected_width: int,
) -> None:
"""
GIVEN:
- A PDF whose first page has the given size in points
WHEN:
- A thumbnail is made from it
THEN:
- The thumbnail has the expected width
"""
pdf_path = self._write_blank_pdf(tmp_path / "in.pdf", page_size)
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
assert thumb == work_dir / "convert.webp"
with Image.open(thumb) as im:
assert im.format == "WEBP"
assert im.width == expected_width
@classmethod
def _write_pdf_without_xref(cls, path: Path) -> Path:
"""
Cuts off the xref and trailer, which pdftoppm cannot recover from but qpdf can.
"""
cls._write_blank_pdf(path)
data = path.read_bytes()
path.write_bytes(data[: data.rindex(b"\nxref")])
return path
def test_qpdf_repair_produces_real_thumbnail(
self,
tmp_path: Path,
work_dir: Path,
) -> None:
"""
GIVEN:
- A PDF with its xref table and trailer cut off
WHEN:
- A thumbnail is made from it
THEN:
- The thumbnail is rendered from a qpdf repaired copy
- The original file is unchanged
"""
pdf_path = self._write_pdf_without_xref(tmp_path / "broken.pdf")
original_bytes = pdf_path.read_bytes()
with pytest.raises(ParseError):
rasterize_pdf_page_to_png(pdf_path, work_dir / "probe.png", dpi=50)
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
assert thumb == work_dir / "convert_qpdf.webp"
with Image.open(thumb) as im:
assert im.format == "WEBP"
assert im.width == 500
assert pdf_path.read_bytes() == original_bytes
@pytest.mark.parametrize(
"qpdf_error",
[
pytest.param(subprocess.CalledProcessError(2, "qpdf"), id="qpdf-fails"),
pytest.param(None, id="repaired-still-unrenderable"),
],
)
def test_double_failure_uses_default_thumbnail(
self,
mocker: MockerFixture,
tmp_path: Path,
work_dir: Path,
qpdf_error: subprocess.CalledProcessError | None,
) -> None:
"""
GIVEN:
- A PDF that cannot be rendered, even after qpdf repair
WHEN:
- A thumbnail is made from it
THEN:
- A copy of the default thumbnail is returned
"""
mocker.patch(
"documents.parsers.rasterize_pdf_page_to_png",
side_effect=ParseError("Does not compute."),
)
if qpdf_error is not None:
mocker.patch("documents.parsers.run_subprocess", side_effect=qpdf_error)
pdf_path = self._write_blank_pdf(tmp_path / "in.pdf")
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
assert thumb == work_dir / "document.webp"
assert thumb.read_bytes() == get_default_thumbnail().read_bytes()
class TestRasterizePdfPageToPng:
@staticmethod
def _write_pdf(
path: Path,
*,
crop_box: tuple[float, float, float, float] | None = None,
rotate: int | None = None,
) -> Path:
pdf = pikepdf.new()
pdf.add_blank_page(page_size=(144, 72))
pdf.add_blank_page(page_size=(300, 300))
page = pdf.pages[0]
if crop_box is not None:
page.obj.CropBox = pikepdf.Array(crop_box)
if rotate is not None:
page.obj.Rotate = rotate
pdf.save(path)
return path
@pytest.mark.parametrize(
("crop_box", "rotate", "expected_size"),
[
pytest.param(None, None, (144, 72), id="plain"),
pytest.param(None, 90, (72, 144), id="rotated-90"),
pytest.param((0, 0, 72, 36), None, (72, 36), id="crop-box"),
],
)
def test_renders_first_page_to_exact_path(
self,
tmp_path: Path,
crop_box: tuple[float, float, float, float] | None,
rotate: int | None,
expected_size: tuple[int, int],
) -> None:
"""
GIVEN:
- A two page PDF, first page optionally cropped or rotated
WHEN:
- The first page is rasterized at 72 DPI
THEN:
- Only out_path is written, sized to the first page's crop and rotation
"""
pdf_path = self._write_pdf(
tmp_path / "in.pdf",
crop_box=crop_box,
rotate=rotate,
)
out_dir = tmp_path / "out"
out_dir.mkdir()
out_path = out_dir / "page1.png"
rasterize_pdf_page_to_png(pdf_path, out_path, dpi=72)
assert list(out_dir.iterdir()) == [out_path]
with Image.open(out_path) as im:
assert im.format == "PNG"
assert im.size == expected_size
def test_failure_raises_parse_error(self, tmp_path: Path) -> None:
"""
GIVEN:
- A file that is not a PDF
WHEN:
- Rasterization is attempted
THEN:
- A ParseError is raised
"""
bad = tmp_path / "bad.pdf"
bad.write_bytes(b"not a pdf")
with pytest.raises(ParseError):
rasterize_pdf_page_to_png(bad, tmp_path / "page1.png", dpi=72)
class TestEncodeThumbnailWebp:
@pytest.mark.parametrize(
("mode", "color"),
[
pytest.param("RGBA", (0, 0, 0, 0), id="rgba"),
pytest.param("LA", (0, 0), id="la"),
],
)
def test_alpha_flattened_onto_white(
self,
tmp_path: Path,
mode: str,
color: tuple[int, ...],
) -> None:
"""
GIVEN:
- A fully transparent PNG with an alpha channel
WHEN:
- It is encoded as a thumbnail
THEN:
- The WebP output is RGB with the transparency flattened to white
"""
png_path = tmp_path / "in.png"
Image.new(mode, (20, 10), color).save(png_path)
out_path = tmp_path / "out.webp"
encode_thumbnail_webp(png_path, out_path)
with Image.open(out_path) as im:
assert im.format == "WEBP"
assert im.mode == "RGB"
assert im.size == (20, 10)
red, green, blue = im.getpixel((10, 5))
assert min(red, green, blue) >= 250
@pytest.mark.parametrize(
("in_size", "supersample", "expected_size"),
[
pytest.param((1000, 2000), 1, (500, 1000), id="too-wide-shrunk"),
pytest.param((100, 10000), 1, (50, 5000), id="too-tall-shrunk"),
pytest.param((100, 200), 1, (100, 200), id="small-not-enlarged"),
pytest.param((1000, 1400), 2, (500, 700), id="2x-halved"),
pytest.param((1001, 1401), 2, (500, 700), id="2x-odd-rounded"),
pytest.param((600, 800), 2, (300, 400), id="2x-small-not-enlarged"),
pytest.param((900, 1200), 2, (450, 600), id="2x-below-clamp"),
],
)
def test_size_clamped_and_downsampled(
self,
tmp_path: Path,
in_size: tuple[int, int],
supersample: int,
expected_size: tuple[int, int],
) -> None:
"""
GIVEN:
- A rendered image and its supersample factor
WHEN:
- It is encoded as a thumbnail with that factor
THEN:
- It is downsampled, fit within 500x5000 and never enlarged
"""
png_path = tmp_path / "in.png"
Image.new("RGB", in_size, (255, 255, 255)).save(png_path)
out_path = tmp_path / "out.webp"
encode_thumbnail_webp(png_path, out_path, supersample=supersample)
with Image.open(out_path) as im:
assert im.size == expected_size
@pytest.mark.parametrize(
"error",
[
pytest.param(OSError("broken image"), id="os-error"),
pytest.param(Image.DecompressionBombError("too large"), id="bomb"),
],
)
def test_decode_failure_raises_parse_error(
self,
tmp_path: Path,
mocker: MockerFixture,
error: Exception,
) -> None:
"""
GIVEN:
- Opening the rendered image fails
WHEN:
- It is encoded as a thumbnail
THEN:
- A ParseError is raised
"""
png_path = tmp_path / "in.png"
Image.new("RGB", (10, 10)).save(png_path)
mocker.patch("PIL.Image.open", side_effect=error)
with pytest.raises(ParseError):
encode_thumbnail_webp(png_path, tmp_path / "out.webp")
+520
View File
@@ -0,0 +1,520 @@
"""
Tests for documents.pdf_ops.
These use real PDFs from the sample directories. No database, Celery or mocks.
Pages are compared by a hash of their content stream, so page identity and order
are easy to assert.
"""
import hashlib
from collections.abc import Callable
from pathlib import Path
import pikepdf
import pytest
from documents import pdf_ops
from documents.pdf_ops import PageSpec
SRC_ROOT = Path(__file__).parents[2]
SAMPLES = Path(__file__).parent / "samples"
THREE_PAGES = SAMPLES / "documents" / "originals" / "0000002.pdf"
TWELVE_PAGES = SAMPLES / "barcodes" / "split-by-asn-2.pdf"
ENCRYPTED = SAMPLES / "password-is-test.pdf"
SIGNED = SRC_ROOT / "paperless" / "tests" / "samples" / "tesseract" / "signed.pdf"
def _page_fingerprint(page: pikepdf.Page) -> str:
contents = page.obj.get("/Contents")
assert contents is not None, "sample page has no /Contents"
streams = list(contents) if isinstance(contents, pikepdf.Array) else [contents]
return hashlib.sha256(b"".join(s.read_bytes() for s in streams)).hexdigest()
def fingerprints(path: Path) -> list[str]:
with pikepdf.open(path) as pdf:
return [_page_fingerprint(page) for page in pdf.pages]
def rotations(path: Path) -> list[int]:
with pikepdf.open(path) as pdf:
return [int(page.obj.get("/Rotate", 0)) for page in pdf.pages]
def docinfo_keys(path: Path) -> set[str]:
with pikepdf.open(path) as pdf:
return set(pdf.docinfo.keys())
def constant(path: Path) -> Callable[[], Path]:
return lambda: path
@pytest.fixture
def source_fingerprints() -> list[str]:
fps = fingerprints(THREE_PAGES)
assert len(set(fps)) == 3, "sample must have three distinct pages"
return fps
class TestRotatePdf:
def test_rotation_is_relative_and_applies_to_every_page(
self,
tmp_path: Path,
) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- It is rotated by 90 degrees, then the result is rotated by 90 again
THEN:
- Every page is rotated relative to its current rotation
- The page content itself is unchanged
"""
once = tmp_path / "once.pdf"
twice = tmp_path / "twice.pdf"
pdf_ops.rotate_pdf(THREE_PAGES, once, 90)
pdf_ops.rotate_pdf(once, twice, 90)
assert rotations(once) == [90, 90, 90]
assert rotations(twice) == [180, 180, 180]
assert fingerprints(twice) == fingerprints(THREE_PAGES)
def test_keeps_document_info(self, tmp_path: Path) -> None:
"""
GIVEN:
- A PDF with document info
WHEN:
- It is rotated
THEN:
- The document info is still present in the output
"""
dst = tmp_path / "out.pdf"
pdf_ops.rotate_pdf(THREE_PAGES, dst, 90)
assert "/Creator" in docinfo_keys(dst)
class TestRemovePages:
@pytest.mark.parametrize(
("pages", "kept"),
[
pytest.param([2], [0, 2], id="single"),
pytest.param([3, 1], [1], id="unordered"),
pytest.param([2, 2], [0, 2], id="duplicates-remove-once"),
pytest.param([], [0, 1, 2], id="empty-keeps-everything"),
pytest.param([1, 2, 3], [], id="every-page"),
],
)
def test_removes_only_the_requested_pages(
self,
tmp_path: Path,
source_fingerprints: list[str],
pages: list[int],
kept: list[int],
) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- Pages are removed, in any order and possibly repeated
THEN:
- Exactly the other pages remain, in their original order
"""
dst = tmp_path / "out.pdf"
pdf_ops.remove_pages(THREE_PAGES, dst, pages)
assert fingerprints(dst) == [source_fingerprints[i] for i in kept]
def test_keeps_document_info(self, tmp_path: Path) -> None:
"""
GIVEN:
- A PDF with document info
WHEN:
- A page is removed
THEN:
- The document info is still present in the output
"""
dst = tmp_path / "out.pdf"
pdf_ops.remove_pages(THREE_PAGES, dst, [1])
assert "/Creator" in docinfo_keys(dst)
@pytest.mark.parametrize(
"bad_page",
[
pytest.param(0, id="zero"),
pytest.param(-1, id="negative"),
],
)
def test_rejects_pages_below_one(self, tmp_path: Path, bad_page: int) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- Pages are removed and one of them is below 1
THEN:
- ValueError is raised
- No output file is written
"""
dst = tmp_path / "out.pdf"
with pytest.raises(ValueError, match="start at 1"):
pdf_ops.remove_pages(THREE_PAGES, dst, [1, bad_page])
assert not dst.exists()
class TestBuildPdfs:
def test_selects_and_orders_pages(
self,
tmp_path: Path,
source_fingerprints: list[str],
) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- One output is built from pages 3 then 1
THEN:
- The output has those pages in that order
- Its path is returned
"""
dst = tmp_path / "out.pdf"
written = pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(3), PageSpec(1)], constant(dst))],
)
assert written == [dst]
assert fingerprints(dst) == [source_fingerprints[2], source_fingerprints[0]]
def test_rotates_only_the_requested_pages(self, tmp_path: Path) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- One output is built with a different rotation on each page
THEN:
- Each output page has exactly the rotation requested for it
"""
dst = tmp_path / "out.pdf"
pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(1), PageSpec(2, 90), PageSpec(3, 180)], constant(dst))],
)
assert rotations(dst) == [0, 90, 180]
def test_writes_one_file_per_output_in_order(self, tmp_path: Path) -> None:
"""
GIVEN:
- A twelve page PDF
WHEN:
- Two outputs are built from different page ranges
THEN:
- Two files are written and returned in output order
- Each holds exactly its own pages
"""
first = tmp_path / "first.pdf"
second = tmp_path / "second.pdf"
source = fingerprints(TWELVE_PAGES)
written = pdf_ops.build_pdfs(
TWELVE_PAGES,
[
([PageSpec(p) for p in (1, 2, 3)], constant(first)),
([PageSpec(p) for p in range(4, 13)], constant(second)),
],
)
assert written == [first, second]
assert fingerprints(first) == source[:3]
assert fingerprints(second) == source[3:]
def test_empty_page_list_writes_a_zero_page_file(self, tmp_path: Path) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- An output with no pages is built
THEN:
- A PDF with zero pages is written
"""
dst = tmp_path / "out.pdf"
pdf_ops.build_pdfs(THREE_PAGES, [([], constant(dst))])
assert fingerprints(dst) == []
def test_destination_is_not_requested_when_a_page_is_out_of_range(
self,
tmp_path: Path,
) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- Several outputs are built
- A later output refers to a page past the end
THEN:
- IndexError is raised
- No destination was requested for any output, including earlier valid ones
"""
requested: list[Path] = []
def make_dst() -> Path:
requested.append(tmp_path / "out.pdf")
return requested[-1]
with pytest.raises(IndexError):
pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(1)], make_dst), ([PageSpec(99)], make_dst)],
)
assert requested == []
@pytest.mark.parametrize(
"bad_page",
[
pytest.param(0, id="zero"),
pytest.param(-1, id="negative"),
],
)
def test_rejects_pages_below_one_before_opening_anything(
self,
tmp_path: Path,
bad_page: int,
) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- Several outputs are built and a later one has a page below 1
THEN:
- ValueError is raised
- No destination was requested for any output
"""
requested: list[Path] = []
def make_dst() -> Path:
requested.append(tmp_path / "out.pdf")
return requested[-1]
with pytest.raises(ValueError, match="start at 1"):
pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(1)], make_dst), ([PageSpec(bad_page)], make_dst)],
)
assert requested == []
class TestValidatePageOperations:
def test_returns_the_output_count(self) -> None:
"""
GIVEN:
- Operations that all target the default output
WHEN:
- They are validated
THEN:
- One output document is reported
"""
operations = [{"page": 1}, {"page": 2}, {"page": 3}]
assert pdf_ops.validate_page_operations(operations, single_output=True) == 1
def test_gap_in_output_indices_counts_up_to_the_highest(self) -> None:
"""
GIVEN:
- Operations that target outputs 0 and 2 but never 1
WHEN:
- They are validated
THEN:
- Three output documents are reported
"""
operations = [
{"page": 1, "doc": 0},
{"page": 2, "doc": 2},
{"page": 3, "doc": 0},
]
count = pdf_ops.validate_page_operations(operations, single_output=False)
assert count == 3
def test_empty_operations_are_rejected(self) -> None:
"""
GIVEN:
- No operations
WHEN:
- They are validated
THEN:
- ValueError is raised
"""
with pytest.raises(ValueError, match="index is out of bounds"):
pdf_ops.validate_page_operations([], single_output=False)
def test_multiple_outputs_rejected_when_single_output_required(self) -> None:
"""
GIVEN:
- Operations that target two outputs
WHEN:
- They are validated with a single output required
THEN:
- ValueError is raised
"""
operations = [{"page": 1, "doc": 0}, {"page": 2, "doc": 1}]
with pytest.raises(ValueError, match="Multiple output documents"):
pdf_ops.validate_page_operations(operations, single_output=True)
@pytest.mark.parametrize(
"doc",
[
pytest.param(-1, id="negative"),
pytest.param(2, id="equal-to-operation-count"),
pytest.param(2**32, id="huge"),
],
)
def test_output_index_out_of_bounds(self, doc: int) -> None:
"""
GIVEN:
- Two operations, one with an output index that is out of bounds
WHEN:
- They are validated
THEN:
- ValueError is raised
"""
operations = [{"page": 1, "doc": 0}, {"page": 2, "doc": doc}]
with pytest.raises(ValueError, match="index is out of bounds"):
pdf_ops.validate_page_operations(operations, single_output=False)
class TestDecrypt:
@pytest.mark.parametrize(
("path", "expected"),
[
pytest.param(ENCRYPTED, True, id="password-required"),
pytest.param(SIGNED, True, id="opens-without-password-but-encrypted"),
pytest.param(THREE_PAGES, False, id="not-encrypted"),
],
)
def test_needs_decrypt(self, path: Path, *, expected: bool) -> None:
"""
GIVEN:
- A PDF that is encrypted, or encrypted but openable, or plain
WHEN:
- needs_decrypt is asked about it
THEN:
- Only the unencrypted PDF reports False
"""
assert pdf_ops.needs_decrypt(path) is expected
def test_decrypt_writes_an_unencrypted_copy(self, tmp_path: Path) -> None:
"""
GIVEN:
- A password protected PDF
WHEN:
- It is decrypted with the correct password
THEN:
- The path from make_dst is returned
- The written copy no longer needs decrypting
"""
dst = tmp_path / "out.pdf"
result = pdf_ops.decrypt_pdf(ENCRYPTED, constant(dst), "test")
assert result == dst
assert pdf_ops.needs_decrypt(dst) is False
def test_wrong_password_raises_and_never_requests_a_destination(
self,
tmp_path: Path,
) -> None:
"""
GIVEN:
- A password protected PDF
WHEN:
- It is decrypted with the wrong password
THEN:
- PasswordError is raised
- No destination was requested
"""
requested: list[Path] = []
def make_dst() -> Path:
requested.append(tmp_path / "out.pdf")
return requested[-1]
with pytest.raises(pikepdf.PasswordError):
pdf_ops.decrypt_pdf(ENCRYPTED, make_dst, "wrong")
assert requested == []
class TestPdfMerger:
def test_pages_are_appended_in_the_order_added(
self,
tmp_path: Path,
source_fingerprints: list[str],
) -> None:
"""
GIVEN:
- A reordered PDF and the original three page PDF
WHEN:
- Both are added to a merger in that order and saved
THEN:
- The output holds all pages in the order they were added
"""
reordered = tmp_path / "reordered.pdf"
merged = tmp_path / "merged.pdf"
pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(3), PageSpec(1)], constant(reordered))],
)
with pdf_ops.PdfMerger() as merger:
merger.add(reordered)
merger.add(THREE_PAGES)
merger.save(merged)
assert fingerprints(merged) == [
source_fingerprints[2],
source_fingerprints[0],
*source_fingerprints,
]
def test_output_version_is_at_least_the_highest_source_version(
self,
tmp_path: Path,
) -> None:
"""
GIVEN:
- Two PDFs with different PDF versions
WHEN:
- Both are added to a merger and saved
THEN:
- The output version is at least the highest source version
"""
merged = tmp_path / "merged.pdf"
with pikepdf.open(TWELVE_PAGES) as pdf:
source_versions = [pdf.pdf_version]
with pikepdf.open(THREE_PAGES) as pdf:
source_versions.append(pdf.pdf_version)
with pdf_ops.PdfMerger() as merger:
merger.add(TWELVE_PAGES)
merger.add(THREE_PAGES)
merger.save(merged)
with pikepdf.open(merged) as pdf:
assert pdf.pdf_version >= max(source_versions)
-46
View File
@@ -265,52 +265,6 @@ def get_page_count_for_pdf(
return None
def get_pdf_first_page_size_points(
path: Path,
log: logging.Logger | None = None,
) -> tuple[float, float] | None:
"""Return the first page's (width, height) in PDF points, post-rotation.
Uses the CropBox (MediaBox if absent), which must match pdftoppm's
``-cropbox`` or the computed DPI targets the wrong box.
Swaps width and height for 90/270 rotation. ``page.rotation`` resolves
inherited and negative ``/Rotate`` values, a raw lookup does not.
Parameters
----------
path:
Absolute path to the PDF file.
log:
Logger for warnings. Falls back to the module-level logger when omitted.
Returns
-------
tuple[float, float] | None
``(width_points, height_points)``, or ``None`` if the file cannot be
opened, has no pages, or the page box is degenerate.
"""
import pikepdf
_log = log or logger
try:
with pikepdf.Pdf.open(path) as pdf:
if len(pdf.pages) == 0:
return None
page = pdf.pages[0]
llx, lly, urx, ury = (float(v) for v in page.cropbox)
width = abs(urx - llx)
height = abs(ury - lly)
if width <= 0 or height <= 0:
return None
if page.rotation in (90, 270):
width, height = height, width
return width, height
except Exception as e:
_log.warning("Could not determine PDF page size for %s: %s", path, e)
return None
def extract_pdf_metadata(
document_path: Path,
log: logging.Logger | None = None,
+2
View File
@@ -986,6 +986,8 @@ GNUPG_HOME = os.getenv("HOME", "/tmp")
# Convert is part of the ImageMagick package
CONVERT_BINARY = os.getenv("PAPERLESS_CONVERT_BINARY", "convert")
CONVERT_TMPDIR = os.getenv("PAPERLESS_CONVERT_TMPDIR")
CONVERT_MEMORY_LIMIT = os.getenv("PAPERLESS_CONVERT_MEMORY_LIMIT")
GS_BINARY = os.getenv("PAPERLESS_GS_BINARY", "gs")
@@ -15,10 +15,9 @@ from typing import TYPE_CHECKING
import pytest
from ocrmypdf import SubprocessOutputError
from PIL import Image
from documents.parsers import ParseError
from documents.parsers import rasterize_pdf_page_to_png
from documents.parsers import run_convert
from paperless.models import ModeChoices
from paperless.parsers import ParserProtocol
from paperless.parsers.tesseract import RasterisedDocumentParser
@@ -281,71 +280,24 @@ class TestGetThumbnail:
)
assert thumb.is_file()
@pytest.mark.parametrize(
("filename", "expected_height"),
[
pytest.param("simple-digital.pdf", 647, id="portrait-letter"),
pytest.param("rotated.pdf", 386, id="landscape"),
],
)
def test_thumbnail_is_correct_format_and_size(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
filename: str,
expected_height: int,
) -> None:
"""
GIVEN:
- A portrait or landscape PDF
WHEN:
- A thumbnail is generated
THEN:
- A 500px wide WebP keeping the page's aspect ratio
"""
thumb = tesseract_parser.get_thumbnail(
tesseract_samples_dir / filename,
"application/pdf",
)
with Image.open(thumb) as im:
assert im.format == "WEBP"
assert im.width == 500
assert im.height == pytest.approx(expected_height, abs=2)
def test_thumbnail_fallback_on_pdftoppm_error(
def test_thumbnail_fallback_on_convert_error(
self,
mocker: MockerFixture,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
) -> None:
"""
GIVEN:
- Rasterizing the original PDF fails
WHEN:
- A thumbnail is generated
THEN:
- The PDF is repaired with qpdf and a real thumbnail is rendered
"""
original = tesseract_samples_dir / "simple-digital.pdf"
def _fail_on_original(in_path: Path, out_path: Path, **kwargs) -> None:
if in_path == original:
def _raise_on_pdf(input_file, output_file, **kwargs) -> None:
if ".pdf" in str(input_file):
raise ParseError("Does not compute.")
rasterize_pdf_page_to_png(in_path, out_path, **kwargs)
run_convert(input_file=input_file, output_file=output_file, **kwargs)
rasterize = mocker.patch(
"documents.parsers.rasterize_pdf_page_to_png",
side_effect=_fail_on_original,
mocker.patch("documents.parsers.run_convert", side_effect=_raise_on_pdf)
thumb = tesseract_parser.get_thumbnail(
tesseract_samples_dir / "simple-digital.pdf",
"application/pdf",
)
thumb = tesseract_parser.get_thumbnail(original, "application/pdf")
assert rasterize.call_count == 2
assert thumb.is_file()
assert thumb.name == "convert_qpdf.webp"
with Image.open(thumb) as im:
assert im.format == "WEBP"
assert im.width == 500
def test_thumbnail_encrypted_pdf(
self,
-145
View File
@@ -6,10 +6,8 @@ import codecs
from pathlib import Path
from typing import TYPE_CHECKING
import pikepdf
import pytest
from paperless.parsers.utils import get_pdf_first_page_size_points
from paperless.parsers.utils import is_tagged_pdf
from paperless.parsers.utils import pdf_born_digital_text
from paperless.parsers.utils import post_process_text
@@ -72,149 +70,6 @@ class TestIsTaggedPdf:
assert is_tagged_pdf(bad) is False
class TestGetPdfFirstPageSizePoints:
@staticmethod
def _write_pdf(
path: Path,
*,
media_box: tuple[float, float, float, float] = (0, 0, 600, 800),
crop_box: tuple[float, float, float, float] | None = None,
page_rotate: int | None = None,
inherited_rotate: int | None = None,
) -> Path:
pdf = pikepdf.new()
pdf.add_blank_page(page_size=(media_box[2], media_box[3]))
page = pdf.pages[0]
page.obj.MediaBox = pikepdf.Array(media_box)
if crop_box is not None:
page.obj.CropBox = pikepdf.Array(crop_box)
if page_rotate is not None:
page.obj.Rotate = page_rotate
if inherited_rotate is not None:
pdf.Root.Pages.Rotate = inherited_rotate
pdf.save(path)
return path
def test_letter_sample(self) -> None:
"""
GIVEN:
- A US Letter sample PDF with no CropBox and no rotation
WHEN:
- The first page size is requested
THEN:
- The MediaBox size in points is returned
"""
assert get_pdf_first_page_size_points(SAMPLES / "simple-digital.pdf") == (
612.0,
792.0,
)
@pytest.mark.parametrize(
("rotate", "expected"),
[
pytest.param(0, (600.0, 800.0), id="rotate-0"),
pytest.param(90, (800.0, 600.0), id="rotate-90"),
pytest.param(180, (600.0, 800.0), id="rotate-180"),
pytest.param(270, (800.0, 600.0), id="rotate-270"),
pytest.param(-90, (800.0, 600.0), id="rotate-negative-90"),
],
)
def test_page_rotation_swaps_dimensions(
self,
tmp_path: Path,
rotate: int,
expected: tuple[float, float],
) -> None:
"""
GIVEN:
- A portrait PDF page with /Rotate set directly on the page
WHEN:
- The first page size is requested
THEN:
- Width and height are swapped for quarter-turn rotations only
"""
pdf_path = self._write_pdf(tmp_path / "rotated.pdf", page_rotate=rotate)
assert get_pdf_first_page_size_points(pdf_path) == expected
def test_inherited_rotation_swaps_dimensions(self, tmp_path: Path) -> None:
"""
GIVEN:
- A page inheriting /Rotate 90 from the /Pages node
WHEN:
- The first page size is requested
THEN:
- Width and height are swapped
"""
pdf_path = self._write_pdf(tmp_path / "inherited.pdf", inherited_rotate=90)
assert get_pdf_first_page_size_points(pdf_path) == (800.0, 600.0)
def test_crop_box_preferred_over_media_box(self, tmp_path: Path) -> None:
"""
GIVEN:
- A PDF page with a CropBox smaller than its MediaBox
WHEN:
- The first page size is requested
THEN:
- The CropBox size is returned
"""
pdf_path = self._write_pdf(
tmp_path / "cropped.pdf",
crop_box=(50, 100, 350, 500),
)
assert get_pdf_first_page_size_points(pdf_path) == (300.0, 400.0)
def test_degenerate_box_returns_none(self, tmp_path: Path) -> None:
"""
GIVEN:
- A PDF page whose box has zero width
WHEN:
- The first page size is requested
THEN:
- None is returned
"""
pdf_path = self._write_pdf(
tmp_path / "degenerate.pdf",
media_box=(0, 0, 600, 800),
crop_box=(100, 0, 100, 800),
)
assert get_pdf_first_page_size_points(pdf_path) is None
def test_nonexistent_path_returns_none(self) -> None:
"""
GIVEN:
- A path that does not exist
WHEN:
- The first page size is requested
THEN:
- None is returned and nothing is raised
"""
assert get_pdf_first_page_size_points(Path("/nonexistent/file.pdf")) is None
def test_corrupt_pdf_returns_none(self, tmp_path: Path) -> None:
"""
GIVEN:
- A file that is not a PDF
WHEN:
- The first page size is requested
THEN:
- None is returned and nothing is raised
"""
bad = tmp_path / "bad.pdf"
bad.write_bytes(b"not a pdf")
assert get_pdf_first_page_size_points(bad) is None
def test_encrypted_pdf_returns_none(self) -> None:
"""
GIVEN:
- A password protected PDF
WHEN:
- The first page size is requested
THEN:
- None is returned and nothing is raised
"""
assert get_pdf_first_page_size_points(SAMPLES / "encrypted.pdf") is None
class TestPostProcessText:
@pytest.mark.parametrize(
("source", "expected"),