mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-10-08 00:57:14 +00:00
Compare commits
12
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2a8df9dca0 | ||
|
|
f1e85fff3e | ||
|
|
f549201b7c | ||
|
|
26efad77ac | ||
|
|
901dd4be93 | ||
|
|
8f73514fbb | ||
|
|
f30a65f440 | ||
|
|
2136659e2b | ||
|
|
26baf52a78 | ||
|
|
3143d1936a | ||
|
|
1a9b08394c | ||
|
|
ae3d8680fa |
No files matched your search
@@ -111,7 +111,7 @@ jobs:
|
||||
timeout-minutes: 12
|
||||
uses: $/.github/actions/apt-install
|
||||
with:
|
||||
packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils
|
||||
packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils qpdf
|
||||
- name: Configure ImageMagick
|
||||
run: |
|
||||
sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-7/policy.xml
|
||||
|
||||
@@ -38,6 +38,7 @@ src/documents/bulk_edit.py:0: error: Incompatible types in assignment (expressio
|
||||
src/documents/bulk_edit.py:0: error: Invalid index type "str" for "dict[FieldDataType, str]"; expected type "FieldDataType" [index]
|
||||
src/documents/bulk_edit.py:0: error: List comprehension has incompatible type List[tuple[int, Any]]; expected List[int] [misc]
|
||||
src/documents/bulk_edit.py:0: error: List comprehension has incompatible type List[tuple[int, None]]; expected List[int] [misc]
|
||||
src/documents/bulk_edit.py:0: error: Missing named argument "p" for "remove" of "PageList" [call-arg]
|
||||
src/documents/bulk_edit.py:0: error: Missing type arguments for generic type "dict" [type-arg]
|
||||
src/documents/bulk_edit.py:0: error: Missing type arguments for generic type "dict" [type-arg]
|
||||
src/documents/bulk_edit.py:0: error: Need type annotation for "to_create" (hint: "to_create: list[<type>] = ...") [var-annotated]
|
||||
|
||||
@@ -91,6 +91,13 @@
|
||||
"concise_description": "Argument `list[int]` is not assignable to parameter `args` with type `tuple[Any, ...] | None` in function `celery.app.task.Task.apply_async`",
|
||||
"severity": "error"
|
||||
},
|
||||
{
|
||||
"column": 33,
|
||||
"path": "src/documents/bulk_edit.py",
|
||||
"name": "missing-argument",
|
||||
"concise_description": "Missing argument `p` in function `pikepdf._core.PageList.remove`",
|
||||
"severity": "error"
|
||||
},
|
||||
{
|
||||
"column": 25,
|
||||
"path": "src/documents/caching.py",
|
||||
|
||||
+6
-18
@@ -1315,29 +1315,17 @@ valid crontab(5) expression describing when to run.
|
||||
|
||||
#### [`PAPERLESS_CONVERT_MEMORY_LIMIT=<num>`](#PAPERLESS_CONVERT_MEMORY_LIMIT) {#PAPERLESS_CONVERT_MEMORY_LIMIT}
|
||||
|
||||
: On smaller systems, or even in the case of Very Large Documents, the
|
||||
consumer may explode, complaining about how it's "unable to extend
|
||||
pixel cache". In such cases, try setting this to a reasonably low
|
||||
value, like 32. The default is to use whatever is necessary to do
|
||||
everything without writing to disk, and units are in megabytes.
|
||||
!!! warning
|
||||
|
||||
For more information on how to use this value, you should search the
|
||||
web for "MAGICK_MEMORY_LIMIT".
|
||||
|
||||
Defaults to 0, which disables the limit.
|
||||
Deprecated and has no effect, since PDF thumbnails no longer use
|
||||
ImageMagick. It will be removed in a future release.
|
||||
|
||||
#### [`PAPERLESS_CONVERT_TMPDIR=<path>`](#PAPERLESS_CONVERT_TMPDIR) {#PAPERLESS_CONVERT_TMPDIR}
|
||||
|
||||
: Similar to the memory limit, if you've got a small system and your
|
||||
OS mounts /tmp as tmpfs, you should set this to a path that's on a
|
||||
physical disk, like /home/your_user/tmp or something. ImageMagick
|
||||
will use this as scratch space when crunching through very large
|
||||
documents.
|
||||
!!! warning
|
||||
|
||||
For more information on how to use this value, you should search the
|
||||
web for "MAGICK_TMPDIR".
|
||||
|
||||
Default is none, which disables the temporary directory.
|
||||
Deprecated and has no effect, since PDF thumbnails no longer use
|
||||
ImageMagick. It will be removed in a future release.
|
||||
|
||||
#### [`PAPERLESS_APPS=<string>`](#PAPERLESS_APPS) {#PAPERLESS_APPS}
|
||||
|
||||
|
||||
+4
-7
@@ -177,12 +177,12 @@ to a positive number to enable polling and disable native filesystem notificatio
|
||||
- `pkg-config` for mysqlclient (python dependency)
|
||||
- `fonts-liberation` for generating thumbnails for plain text
|
||||
files
|
||||
- `imagemagick` >= 6 for PDF conversion
|
||||
- `imagemagick` >= 6 for image alpha handling
|
||||
- `gnupg` for decrypting GPG-encrypted email
|
||||
- `libpq-dev` for PostgreSQL
|
||||
- `libmagic-dev` for mime type detection
|
||||
- `mariadb-client` for MariaDB compile time
|
||||
- `poppler-utils` for barcode detection
|
||||
- `poppler-utils` for thumbnail generation and barcode detection
|
||||
|
||||
Use this list for your preferred package management:
|
||||
|
||||
@@ -416,11 +416,8 @@ to a positive number to enable polling and disable native filesystem notificatio
|
||||
You may need to change the path in the files. Example:
|
||||
`ExecStart=/opt/paperless/.local/bin/celery --app paperless worker --loglevel INFO`
|
||||
|
||||
12. Configure ImageMagick to allow processing of PDF documents and disable
|
||||
formats that Paperless-ngx does not use. Most distributions disable PDF
|
||||
processing by default, since PDF documents can contain malware. If you
|
||||
don't enable it, Paperless-ngx will fall back to Ghostscript for certain
|
||||
steps such as thumbnail generation.
|
||||
12. Harden ImageMagick by disabling formats that Paperless-ngx does not use.
|
||||
PDF processing is not needed and should stay disabled.
|
||||
|
||||
Configure the active ImageMagick policy file (commonly
|
||||
`/etc/ImageMagick-6/policy.xml` or `/etc/ImageMagick-7/policy.xml`) and
|
||||
|
||||
@@ -50,8 +50,6 @@ PAPERLESS_SECRET_KEY=change-me
|
||||
#PAPERLESS_OCR_ROTATE_PAGES=true
|
||||
#PAPERLESS_OCR_ROTATE_PAGES_THRESHOLD=12.0
|
||||
#PAPERLESS_OCR_USER_ARGS={}
|
||||
#PAPERLESS_CONVERT_MEMORY_LIMIT=0
|
||||
#PAPERLESS_CONVERT_TMPDIR=/var/tmp/paperless
|
||||
|
||||
# Software tweaks
|
||||
|
||||
|
||||
+219
-184
@@ -3,7 +3,6 @@ from __future__ import annotations
|
||||
import logging
|
||||
import tempfile
|
||||
import uuid
|
||||
from functools import partial
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
from typing import Literal
|
||||
@@ -18,7 +17,6 @@ from django.db.models import Max
|
||||
from django.db.models import Q
|
||||
from django.utils import timezone
|
||||
|
||||
from documents import pdf_ops
|
||||
from documents.data_models import ConsumableDocument
|
||||
from documents.data_models import DocumentMetadataOverrides
|
||||
from documents.data_models import DocumentSource
|
||||
@@ -118,11 +116,6 @@ def _resolve_root_and_source_doc(
|
||||
)
|
||||
|
||||
|
||||
def _scratch_path(name: str) -> Path:
|
||||
"""A path inside a fresh directory under SCRATCH_DIR."""
|
||||
return Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR)) / name
|
||||
|
||||
|
||||
def set_correspondent(
|
||||
doc_ids: list[int],
|
||||
correspondent: Correspondent,
|
||||
@@ -481,6 +474,8 @@ def rotate(
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
docs_by_root_id.setdefault(pair.root_doc.id, pair)
|
||||
|
||||
import pikepdf
|
||||
|
||||
for pair in docs_by_root_id.values():
|
||||
if pair.source_doc.mime_type != "application/pdf":
|
||||
logger.warning(
|
||||
@@ -493,7 +488,11 @@ def rotate(
|
||||
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
|
||||
/ f"{pair.root_doc.id}_rotated.pdf"
|
||||
)
|
||||
pdf_ops.rotate_pdf(pair.source_doc.source_path, filepath, degrees)
|
||||
with pikepdf.open(pair.source_doc.source_path) as pdf:
|
||||
for page in pdf.pages:
|
||||
page.rotate(degrees, relative=True)
|
||||
pdf.remove_unreferenced_resources()
|
||||
pdf.save(filepath)
|
||||
|
||||
# Preserve metadata/permissions via overrides; mark as new version
|
||||
overrides = DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
@@ -536,45 +535,48 @@ def merge(
|
||||
qs = Document.objects.select_related("root_document").filter(id__in=doc_ids)
|
||||
docs_by_id = {doc.id: doc for doc in qs}
|
||||
affected_docs: list[int] = []
|
||||
handoff_asn: int | None = None
|
||||
with pdf_ops.PdfMerger() as merger:
|
||||
# use doc_ids to preserve order
|
||||
for doc_id in doc_ids:
|
||||
doc = docs_by_id.get(doc_id)
|
||||
if doc is None:
|
||||
continue
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
try:
|
||||
# archive_path is None when there is no archive version
|
||||
archive_path = (
|
||||
pair.source_doc.archive_path
|
||||
if archive_fallback
|
||||
and pair.source_doc.mime_type != "application/pdf"
|
||||
else None
|
||||
)
|
||||
merger.add(
|
||||
archive_path
|
||||
if archive_path is not None
|
||||
else pair.source_doc.source_path,
|
||||
)
|
||||
affected_docs.append(doc.id)
|
||||
if handoff_asn is None and doc.archive_serial_number is not None:
|
||||
handoff_asn = doc.archive_serial_number
|
||||
except Exception as e:
|
||||
logger.exception(
|
||||
f"Error merging document {doc.id}, it will not be included in the merge: {e}",
|
||||
)
|
||||
if len(affected_docs) == 0:
|
||||
logger.warning("No documents were merged")
|
||||
return "OK"
|
||||
import pikepdf
|
||||
|
||||
filepath = (
|
||||
Path(
|
||||
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
|
||||
merged_pdf = pikepdf.new()
|
||||
version: str = merged_pdf.pdf_version
|
||||
handoff_asn: int | None = None
|
||||
# use doc_ids to preserve order
|
||||
for doc_id in doc_ids:
|
||||
doc = docs_by_id.get(doc_id)
|
||||
if doc is None:
|
||||
continue
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
try:
|
||||
doc_path = (
|
||||
pair.source_doc.archive_path
|
||||
if archive_fallback
|
||||
and pair.source_doc.mime_type != "application/pdf"
|
||||
and pair.source_doc.has_archive_version
|
||||
else pair.source_doc.source_path
|
||||
)
|
||||
/ f"{'_'.join([str(doc_id) for doc_id in affected_docs])[:100]}_merged.pdf"
|
||||
with pikepdf.open(str(doc_path)) as pdf:
|
||||
version = max(version, pdf.pdf_version)
|
||||
merged_pdf.pages.extend(pdf.pages)
|
||||
affected_docs.append(doc.id)
|
||||
if handoff_asn is None and doc.archive_serial_number is not None:
|
||||
handoff_asn = doc.archive_serial_number
|
||||
except Exception as e:
|
||||
logger.exception(
|
||||
f"Error merging document {doc.id}, it will not be included in the merge: {e}",
|
||||
)
|
||||
if len(affected_docs) == 0:
|
||||
logger.warning("No documents were merged")
|
||||
return "OK"
|
||||
|
||||
filepath = (
|
||||
Path(
|
||||
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
|
||||
)
|
||||
merger.save(filepath)
|
||||
/ f"{'_'.join([str(doc_id) for doc_id in affected_docs])[:100]}_merged.pdf"
|
||||
)
|
||||
merged_pdf.remove_unreferenced_resources()
|
||||
merged_pdf.save(filepath, min_version=version)
|
||||
merged_pdf.close()
|
||||
|
||||
if metadata_document_id:
|
||||
metadata_document = qs.get(id=metadata_document_id)
|
||||
@@ -750,60 +752,64 @@ def split(
|
||||
)
|
||||
doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
import pikepdf
|
||||
|
||||
consume_tasks = []
|
||||
|
||||
try:
|
||||
outputs = [
|
||||
(
|
||||
[pdf_ops.PageSpec(page) for page in split_doc],
|
||||
partial(_scratch_path, f"{doc.id}_{split_doc[0]}-{split_doc[-1]}.pdf"),
|
||||
)
|
||||
for split_doc in pages
|
||||
]
|
||||
filepaths = pdf_ops.build_pdfs(pair.source_doc.source_path, outputs)
|
||||
with pikepdf.open(pair.source_doc.source_path) as pdf:
|
||||
for idx, split_doc in enumerate(pages):
|
||||
dst: pikepdf.Pdf = pikepdf.new()
|
||||
for page in split_doc:
|
||||
dst.pages.append(pdf.pages[page - 1])
|
||||
filepath: Path = (
|
||||
Path(
|
||||
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
|
||||
)
|
||||
/ f"{doc.id}_{split_doc[0]}-{split_doc[-1]}.pdf"
|
||||
)
|
||||
dst.remove_unreferenced_resources()
|
||||
dst.save(filepath)
|
||||
dst.close()
|
||||
|
||||
for idx, (split_doc, filepath) in enumerate(
|
||||
zip(pages, filepaths, strict=True),
|
||||
):
|
||||
overrides: DocumentMetadataOverrides = (
|
||||
DocumentMetadataOverrides().from_document(doc)
|
||||
)
|
||||
overrides.title = f"{doc.title} (split {idx + 1})"
|
||||
if user is not None:
|
||||
overrides.owner_id = user.id
|
||||
if not delete_originals:
|
||||
overrides.skip_asn_if_exists = True
|
||||
logger.info(
|
||||
f"Adding split document with pages {split_doc} to the task queue.",
|
||||
)
|
||||
consume_tasks.append(
|
||||
consume_file.s(
|
||||
input_doc=ConsumableDocument(
|
||||
source=DocumentSource.ConsumeFolder,
|
||||
original_file=filepath,
|
||||
),
|
||||
overrides=overrides,
|
||||
).set(headers={"trigger_source": trigger_source}),
|
||||
)
|
||||
overrides: DocumentMetadataOverrides = (
|
||||
DocumentMetadataOverrides().from_document(doc)
|
||||
)
|
||||
overrides.title = f"{doc.title} (split {idx + 1})"
|
||||
if user is not None:
|
||||
overrides.owner_id = user.id
|
||||
if not delete_originals:
|
||||
overrides.skip_asn_if_exists = True
|
||||
logger.info(
|
||||
f"Adding split document with pages {split_doc} to the task queue.",
|
||||
)
|
||||
consume_tasks.append(
|
||||
consume_file.s(
|
||||
input_doc=ConsumableDocument(
|
||||
source=DocumentSource.ConsumeFolder,
|
||||
original_file=filepath,
|
||||
),
|
||||
overrides=overrides,
|
||||
).set(headers={"trigger_source": trigger_source}),
|
||||
)
|
||||
|
||||
if delete_originals:
|
||||
backup = release_archive_serial_numbers([doc.id])
|
||||
logger.info(
|
||||
"Queueing removal of original document after consumption of the split documents",
|
||||
)
|
||||
try:
|
||||
chord(
|
||||
header=consume_tasks,
|
||||
body=delete.si([doc.id]),
|
||||
).on_error(
|
||||
restore_archive_serial_numbers_task.s(backup),
|
||||
).apply_async()
|
||||
except Exception:
|
||||
restore_archive_serial_numbers(backup)
|
||||
raise
|
||||
else:
|
||||
group(consume_tasks).delay()
|
||||
if delete_originals:
|
||||
backup = release_archive_serial_numbers([doc.id])
|
||||
logger.info(
|
||||
"Queueing removal of original document after consumption of the split documents",
|
||||
)
|
||||
try:
|
||||
chord(
|
||||
header=consume_tasks,
|
||||
body=delete.si([doc.id]),
|
||||
).on_error(
|
||||
restore_archive_serial_numbers_task.s(backup),
|
||||
).apply_async()
|
||||
except Exception:
|
||||
restore_archive_serial_numbers(backup)
|
||||
raise
|
||||
else:
|
||||
group(consume_tasks).delay()
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Error splitting document {doc.id}: {e}")
|
||||
@@ -824,6 +830,8 @@ def delete_pages(
|
||||
)
|
||||
doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
pages = sorted(pages) # sort pages to avoid index issues
|
||||
import pikepdf
|
||||
|
||||
try:
|
||||
# Produce edited PDF to a temp file and create a new version
|
||||
@@ -831,7 +839,13 @@ def delete_pages(
|
||||
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
|
||||
/ f"{pair.root_doc.id}_pages_deleted.pdf"
|
||||
)
|
||||
pdf_ops.remove_pages(pair.source_doc.source_path, filepath, pages)
|
||||
with pikepdf.open(pair.source_doc.source_path) as pdf:
|
||||
offset = 1 # pages are 1-indexed
|
||||
for page_num in pages:
|
||||
pdf.pages.remove(pdf.pages[page_num - offset])
|
||||
offset += 1 # remove() changes the index of the pages
|
||||
pdf.remove_unreferenced_resources()
|
||||
pdf.save(filepath)
|
||||
|
||||
overrides = DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
if user is not None:
|
||||
@@ -880,28 +894,47 @@ def edit_pdf(
|
||||
)
|
||||
doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
import pikepdf
|
||||
|
||||
pdf_docs: list[pikepdf.Pdf] = []
|
||||
|
||||
try:
|
||||
output_count = pdf_ops.validate_page_operations(
|
||||
operations,
|
||||
single_output=update_document,
|
||||
)
|
||||
page_specs: list[list[pdf_ops.PageSpec]] = [[] for _ in range(output_count)]
|
||||
for op in operations:
|
||||
page_specs[op.get("doc", 0)].append(
|
||||
pdf_ops.PageSpec(op["page"], op.get("rotate", 0)),
|
||||
if not operations:
|
||||
raise ValueError("Output document index is out of bounds")
|
||||
|
||||
max_idx = max(op.get("doc", 0) for op in operations)
|
||||
if update_document and max_idx > 0:
|
||||
logger.error(
|
||||
"Update requested but multiple output documents specified",
|
||||
)
|
||||
raise ValueError("Multiple output documents specified")
|
||||
|
||||
if any(
|
||||
op.get("doc", 0) < 0 or op.get("doc", 0) >= len(operations)
|
||||
for op in operations
|
||||
):
|
||||
raise ValueError("Output document index is out of bounds")
|
||||
|
||||
with pikepdf.open(pair.source_doc.source_path) as src:
|
||||
# prepare output documents
|
||||
pdf_docs = [pikepdf.new() for _ in range(max_idx + 1)]
|
||||
|
||||
for op in operations:
|
||||
dst = pdf_docs[op.get("doc", 0)]
|
||||
page = src.pages[op["page"] - 1]
|
||||
dst.pages.append(page)
|
||||
if op.get("rotate"):
|
||||
dst.pages[-1].rotate(op["rotate"], relative=True)
|
||||
|
||||
if update_document:
|
||||
# Create a new version from the edited PDF rather than replacing in-place
|
||||
(filepath,) = pdf_ops.build_pdfs(
|
||||
pair.source_doc.source_path,
|
||||
[
|
||||
(
|
||||
page_specs[0],
|
||||
partial(_scratch_path, f"{pair.root_doc.id}_edited.pdf"),
|
||||
),
|
||||
],
|
||||
pdf = pdf_docs[0]
|
||||
pdf.remove_unreferenced_resources()
|
||||
filepath: Path = (
|
||||
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
|
||||
/ f"{pair.root_doc.id}_edited.pdf"
|
||||
)
|
||||
pdf.save(filepath)
|
||||
overrides = (
|
||||
DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
if include_metadata
|
||||
@@ -922,19 +955,6 @@ def edit_pdf(
|
||||
headers={"trigger_source": trigger_source},
|
||||
)
|
||||
else:
|
||||
version_filepaths = pdf_ops.build_pdfs(
|
||||
pair.source_doc.source_path,
|
||||
[
|
||||
(
|
||||
specs,
|
||||
partial(
|
||||
_scratch_path,
|
||||
f"{pair.root_doc.id}_edit_{idx}.pdf",
|
||||
),
|
||||
)
|
||||
for idx, specs in enumerate(page_specs, start=1)
|
||||
],
|
||||
)
|
||||
consume_tasks = []
|
||||
overrides = (
|
||||
DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
@@ -946,9 +966,15 @@ def edit_pdf(
|
||||
overrides.actor_id = user.id
|
||||
if not delete_original:
|
||||
overrides.skip_asn_if_exists = True
|
||||
if delete_original and output_count == 1:
|
||||
if delete_original and len(pdf_docs) == 1:
|
||||
overrides.asn = pair.root_doc.archive_serial_number
|
||||
for version_filepath in version_filepaths:
|
||||
for idx, pdf in enumerate(pdf_docs, start=1):
|
||||
version_filepath: Path = (
|
||||
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
|
||||
/ f"{pair.root_doc.id}_edit_{idx}.pdf"
|
||||
)
|
||||
pdf.remove_unreferenced_resources()
|
||||
pdf.save(version_filepath)
|
||||
consume_tasks.append(
|
||||
consume_file.s(
|
||||
input_doc=ConsumableDocument(
|
||||
@@ -998,6 +1024,8 @@ def remove_password(
|
||||
"""
|
||||
Remove password protection from PDF documents.
|
||||
"""
|
||||
import pikepdf
|
||||
|
||||
for doc_id in doc_ids:
|
||||
doc = Document.objects.select_related("root_document").get(id=doc_id)
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
@@ -1011,69 +1039,76 @@ def remove_password(
|
||||
doc.id,
|
||||
pair.source_doc.source_path,
|
||||
)
|
||||
if not pdf_ops.needs_decrypt(source_path):
|
||||
logger.info(
|
||||
"Skipping password removal for document %s because the "
|
||||
"source PDF is not encrypted",
|
||||
pair.root_doc.id,
|
||||
)
|
||||
continue
|
||||
try:
|
||||
with pikepdf.open(source_path) as pdf:
|
||||
if not pdf.is_encrypted:
|
||||
logger.info(
|
||||
"Skipping password removal for document %s because the "
|
||||
"source PDF is not encrypted",
|
||||
pair.root_doc.id,
|
||||
)
|
||||
continue
|
||||
except pikepdf.PasswordError:
|
||||
# Password-protected PDFs need the supplied password below.
|
||||
pass
|
||||
|
||||
filepath = pdf_ops.decrypt_pdf(
|
||||
source_path,
|
||||
partial(_scratch_path, f"{pair.root_doc.id}_unprotected.pdf"),
|
||||
password,
|
||||
)
|
||||
with pikepdf.open(source_path, password=password) as pdf:
|
||||
filepath: Path = (
|
||||
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
|
||||
/ f"{pair.root_doc.id}_unprotected.pdf"
|
||||
)
|
||||
pdf.remove_unreferenced_resources()
|
||||
pdf.save(filepath)
|
||||
|
||||
if update_document:
|
||||
# Create a new version rather than modifying the root/original in place.
|
||||
overrides = (
|
||||
DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
if include_metadata
|
||||
else DocumentMetadataOverrides()
|
||||
)
|
||||
if user is not None:
|
||||
overrides.owner_id = user.id
|
||||
overrides.actor_id = user.id
|
||||
consume_file.apply_async(
|
||||
kwargs={
|
||||
"input_doc": ConsumableDocument(
|
||||
source=DocumentSource.ConsumeFolder,
|
||||
original_file=filepath,
|
||||
root_document_id=pair.root_doc.id,
|
||||
),
|
||||
"overrides": overrides,
|
||||
},
|
||||
headers={"trigger_source": trigger_source},
|
||||
)
|
||||
else:
|
||||
consume_tasks = []
|
||||
overrides = (
|
||||
DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
if include_metadata
|
||||
else DocumentMetadataOverrides()
|
||||
)
|
||||
if user is not None:
|
||||
overrides.owner_id = user.id
|
||||
overrides.actor_id = user.id
|
||||
|
||||
consume_tasks.append(
|
||||
consume_file.s(
|
||||
input_doc=ConsumableDocument(
|
||||
source=DocumentSource.ConsumeFolder,
|
||||
original_file=filepath,
|
||||
),
|
||||
overrides=overrides,
|
||||
).set(headers={"trigger_source": trigger_source}),
|
||||
)
|
||||
|
||||
if delete_original:
|
||||
chord(
|
||||
header=consume_tasks,
|
||||
body=delete.si([doc.id]),
|
||||
).delay()
|
||||
if update_document:
|
||||
# Create a new version rather than modifying the root/original in place.
|
||||
overrides = (
|
||||
DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
if include_metadata
|
||||
else DocumentMetadataOverrides()
|
||||
)
|
||||
if user is not None:
|
||||
overrides.owner_id = user.id
|
||||
overrides.actor_id = user.id
|
||||
consume_file.apply_async(
|
||||
kwargs={
|
||||
"input_doc": ConsumableDocument(
|
||||
source=DocumentSource.ConsumeFolder,
|
||||
original_file=filepath,
|
||||
root_document_id=pair.root_doc.id,
|
||||
),
|
||||
"overrides": overrides,
|
||||
},
|
||||
headers={"trigger_source": trigger_source},
|
||||
)
|
||||
else:
|
||||
group(consume_tasks).delay()
|
||||
consume_tasks = []
|
||||
overrides = (
|
||||
DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
if include_metadata
|
||||
else DocumentMetadataOverrides()
|
||||
)
|
||||
if user is not None:
|
||||
overrides.owner_id = user.id
|
||||
overrides.actor_id = user.id
|
||||
|
||||
consume_tasks.append(
|
||||
consume_file.s(
|
||||
input_doc=ConsumableDocument(
|
||||
source=DocumentSource.ConsumeFolder,
|
||||
original_file=filepath,
|
||||
),
|
||||
overrides=overrides,
|
||||
).set(headers={"trigger_source": trigger_source}),
|
||||
)
|
||||
|
||||
if delete_original:
|
||||
chord(
|
||||
header=consume_tasks,
|
||||
body=delete.si([doc.id]),
|
||||
).delay()
|
||||
else:
|
||||
group(consume_tasks).delay()
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(
|
||||
|
||||
+161
-98
@@ -1,8 +1,8 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import math
|
||||
import mimetypes
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import tempfile
|
||||
@@ -68,58 +68,6 @@ def get_supported_file_extensions() -> set[str]:
|
||||
return extensions
|
||||
|
||||
|
||||
def run_convert(
|
||||
input_file,
|
||||
output_file,
|
||||
*,
|
||||
density=None,
|
||||
scale=None,
|
||||
alpha=None,
|
||||
strip=False,
|
||||
trim=False,
|
||||
type=None,
|
||||
depth=None,
|
||||
auto_orient=False,
|
||||
use_cropbox=False,
|
||||
extra=None,
|
||||
logging_group=None,
|
||||
) -> None:
|
||||
environment = os.environ.copy()
|
||||
if settings.CONVERT_MEMORY_LIMIT:
|
||||
# MAGICK_MEMORY_LIMIT sets the maximum amount of RAM the pixel cache can use.
|
||||
# MAGICK_MAP_LIMIT sets the maximum amount of memory-mapped I/O allowed.
|
||||
#
|
||||
# For large-format documents ImageMagick will hit the RAM limit and
|
||||
# immediately try to "map" the remaining data. If MAGICK_MAP_LIMIT isn't
|
||||
# also set, the process may trigger an OOM kill because the default
|
||||
# system/policy map limit is often too restrictive for these massive bitmaps.
|
||||
environment["MAGICK_MEMORY_LIMIT"] = settings.CONVERT_MEMORY_LIMIT
|
||||
environment["MAGICK_MAP_LIMIT"] = settings.CONVERT_MEMORY_LIMIT
|
||||
if settings.CONVERT_TMPDIR:
|
||||
environment["MAGICK_TMPDIR"] = settings.CONVERT_TMPDIR
|
||||
|
||||
args = [settings.CONVERT_BINARY]
|
||||
args += ["-density", str(density)] if density else []
|
||||
args += ["-scale", str(scale)] if scale else []
|
||||
args += ["-alpha", str(alpha)] if alpha else []
|
||||
args += ["-strip"] if strip else []
|
||||
args += ["-trim"] if trim else []
|
||||
args += ["-type", str(type)] if type else []
|
||||
args += ["-depth", str(depth)] if depth else []
|
||||
args += ["-auto-orient"] if auto_orient else []
|
||||
args += ["-define", "pdf:use-cropbox=true"] if use_cropbox else []
|
||||
args += [str(input_file), str(output_file)]
|
||||
|
||||
logger.debug("Execute: " + " ".join(args), extra={"group": logging_group})
|
||||
|
||||
try:
|
||||
run_subprocess(args, environment, logger)
|
||||
except subprocess.CalledProcessError as e:
|
||||
raise ParseError(f"Convert failed at {args}") from e
|
||||
except Exception as e: # pragma: no cover
|
||||
raise ParseError("Unknown error running convert") from e
|
||||
|
||||
|
||||
def get_default_thumbnail() -> Path:
|
||||
"""
|
||||
Returns the path to a generic thumbnail
|
||||
@@ -127,46 +75,168 @@ def get_default_thumbnail() -> Path:
|
||||
return (Path(__file__).parent / "resources" / "document.webp").resolve()
|
||||
|
||||
|
||||
def make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group=None) -> Path:
|
||||
out_path: Path = Path(temp_dir) / "convert_gs.webp"
|
||||
_THUMBNAIL_MAX_WIDTH = 500
|
||||
_THUMBNAIL_MAX_HEIGHT = 5000
|
||||
# Used only when the page geometry cannot be read
|
||||
_THUMBNAIL_FALLBACK_DPI = 150
|
||||
# Applied before supersampling, so tiny pages are not enlarged
|
||||
_THUMBNAIL_MAX_DPI = 300
|
||||
# Rendering at a multiple and downsampling keeps text crisper
|
||||
_THUMBNAIL_SUPERSAMPLE = 2
|
||||
|
||||
# if convert fails, fall back to extracting
|
||||
# the first PDF page as a PNG using Ghostscript
|
||||
logger.warning(
|
||||
"Thumbnail generation with ImageMagick failed, falling back "
|
||||
"to ghostscript. Check your /etc/ImageMagick-x/policy.xml!",
|
||||
extra={"group": logging_group},
|
||||
)
|
||||
# Ghostscript doesn't handle WebP outputs
|
||||
gs_out_path: Path = Path(temp_dir) / "gs_out.png"
|
||||
cmd = [settings.GS_BINARY, "-q", "-sDEVICE=pngalpha", "-o", gs_out_path, in_path]
|
||||
|
||||
def rasterize_pdf_page_to_png(
|
||||
in_path: Path,
|
||||
out_path: Path,
|
||||
*,
|
||||
dpi: int,
|
||||
logging_group=None,
|
||||
) -> None:
|
||||
"""
|
||||
Rasterizes the first page of a PDF to a PNG with pdftoppm.
|
||||
"""
|
||||
# -singlefile drops the page number and -png appends ".png", so pass the
|
||||
# path without its suffix
|
||||
args = [
|
||||
"pdftoppm",
|
||||
"-f",
|
||||
"1",
|
||||
"-l",
|
||||
"1",
|
||||
"-r",
|
||||
str(dpi),
|
||||
"-png",
|
||||
"-singlefile",
|
||||
"-cropbox",
|
||||
str(in_path),
|
||||
str(out_path.with_suffix("")),
|
||||
]
|
||||
|
||||
logger.debug("Execute: " + " ".join(args), extra={"group": logging_group})
|
||||
|
||||
try:
|
||||
try:
|
||||
run_subprocess(cmd, logger=logger)
|
||||
except subprocess.CalledProcessError as e:
|
||||
raise ParseError(f"Thumbnail (gs) failed at {cmd}") from e
|
||||
# then run convert on the output from gs to make WebP
|
||||
run_convert(
|
||||
density=300,
|
||||
scale="500x5000>",
|
||||
alpha="remove",
|
||||
strip=True,
|
||||
trim=False,
|
||||
auto_orient=True,
|
||||
input_file=gs_out_path,
|
||||
output_file=out_path,
|
||||
logging_group=logging_group,
|
||||
)
|
||||
run_subprocess(args, logger=logger)
|
||||
except subprocess.CalledProcessError as e:
|
||||
raise ParseError(f"pdftoppm failed at {args}") from e
|
||||
except Exception as e: # pragma: no cover
|
||||
raise ParseError("Unknown error running pdftoppm") from e
|
||||
|
||||
|
||||
def encode_thumbnail_webp(
|
||||
png_path: Path,
|
||||
out_path: Path,
|
||||
*,
|
||||
supersample: int = 1,
|
||||
) -> None:
|
||||
"""
|
||||
Flattens alpha onto white, undoes supersampling, shrinks to fit and saves as WebP.
|
||||
"""
|
||||
from PIL import Image
|
||||
|
||||
try:
|
||||
with Image.open(png_path) as im:
|
||||
if im.mode in ("RGBA", "LA"):
|
||||
flattened = Image.new("RGB", im.size, (255, 255, 255))
|
||||
flattened.paste(im, mask=im.split()[-1])
|
||||
else:
|
||||
flattened = im.convert("RGB")
|
||||
|
||||
if supersample > 1:
|
||||
flattened = flattened.resize(
|
||||
(
|
||||
max(1, round(flattened.width / supersample)),
|
||||
max(1, round(flattened.height / supersample)),
|
||||
),
|
||||
Image.Resampling.LANCZOS,
|
||||
)
|
||||
|
||||
flattened.thumbnail((_THUMBNAIL_MAX_WIDTH, _THUMBNAIL_MAX_HEIGHT))
|
||||
flattened.save(out_path, format="WEBP")
|
||||
except (OSError, Image.DecompressionBombError) as e:
|
||||
raise ParseError(f"Unable to encode thumbnail from {png_path}") from e
|
||||
|
||||
|
||||
def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> tuple[int, int]:
|
||||
"""
|
||||
Returns (dpi, supersample). Unknown geometry is not supersampled, since
|
||||
the render size cannot be bounded.
|
||||
"""
|
||||
from paperless.parsers.utils import get_pdf_first_page_size_points
|
||||
|
||||
size = get_pdf_first_page_size_points(in_path)
|
||||
if size is None:
|
||||
logger.debug(
|
||||
"Could not read PDF page size, using fallback DPI",
|
||||
extra={"group": logging_group},
|
||||
)
|
||||
return _THUMBNAIL_FALLBACK_DPI, 1
|
||||
|
||||
width_pts, height_pts = size
|
||||
dpi_for_width = _THUMBNAIL_MAX_WIDTH * 72 / width_pts
|
||||
dpi_for_height = _THUMBNAIL_MAX_HEIGHT * 72 / height_pts
|
||||
# Round up so the downsampled render is never a few pixels short of the
|
||||
# target; the shrink-only clamp in encode_thumbnail_webp trims the excess.
|
||||
dpi = max(
|
||||
1,
|
||||
math.ceil(min(_THUMBNAIL_MAX_DPI, dpi_for_width, dpi_for_height)),
|
||||
)
|
||||
# At the 1 DPI floor the page is already oversized, so do not supersample
|
||||
return dpi, 1 if dpi == 1 else _THUMBNAIL_SUPERSAMPLE
|
||||
|
||||
|
||||
def _render_pdf_thumbnail(
|
||||
in_path: Path,
|
||||
png_path: Path,
|
||||
out_path: Path,
|
||||
logging_group=None,
|
||||
) -> None:
|
||||
dpi, supersample = _compute_thumbnail_dpi(in_path, logging_group=logging_group)
|
||||
rasterize_pdf_page_to_png(
|
||||
in_path,
|
||||
png_path,
|
||||
dpi=dpi * supersample,
|
||||
logging_group=logging_group,
|
||||
)
|
||||
encode_thumbnail_webp(png_path, out_path, supersample=supersample)
|
||||
|
||||
|
||||
def _repair_pdf_with_qpdf(in_path: Path, out_path: Path) -> None:
|
||||
# qpdf exits 3 after a repair; --warning-exit-0 keeps that from failing
|
||||
try:
|
||||
shutil.copy(in_path, out_path)
|
||||
run_subprocess(
|
||||
["qpdf", "--warning-exit-0", "--replace-input", str(out_path)],
|
||||
logger=logger,
|
||||
)
|
||||
except (subprocess.CalledProcessError, OSError) as e:
|
||||
raise ParseError(f"qpdf repair failed for {in_path}") from e
|
||||
|
||||
|
||||
def make_thumbnail_from_pdf_qpdf_fallback(
|
||||
in_path: Path,
|
||||
temp_dir: Path,
|
||||
logging_group=None,
|
||||
) -> Path:
|
||||
png_path = temp_dir / "page1_repaired.png"
|
||||
out_path = temp_dir / "convert_qpdf.webp"
|
||||
repaired_path = temp_dir / "repaired.pdf"
|
||||
|
||||
logger.warning(
|
||||
"Thumbnail generation with pdftoppm failed, attempting qpdf repair and retry.",
|
||||
extra={"group": logging_group},
|
||||
)
|
||||
|
||||
try:
|
||||
_repair_pdf_with_qpdf(in_path, repaired_path)
|
||||
_render_pdf_thumbnail(repaired_path, png_path, out_path, logging_group)
|
||||
return out_path
|
||||
|
||||
except ParseError as e:
|
||||
logger.error(f"Unable to make thumbnail with Ghostscript: {e}")
|
||||
logger.error(f"Unable to make thumbnail after qpdf repair: {e}")
|
||||
# The caller might expect a generated thumbnail that can be moved,
|
||||
# so we need to copy it before it gets moved.
|
||||
# https://github.com/paperless-ngx/paperless-ngx/issues/3631
|
||||
default_thumbnail_path: Path = Path(temp_dir) / "document.webp"
|
||||
default_thumbnail_path = temp_dir / "document.webp"
|
||||
copy_file_with_basic_stats(get_default_thumbnail(), default_thumbnail_path)
|
||||
return default_thumbnail_path
|
||||
|
||||
@@ -175,25 +245,18 @@ def make_thumbnail_from_pdf(in_path: Path, temp_dir: Path, logging_group=None) -
|
||||
"""
|
||||
The thumbnail of a PDF is just a 500px wide image of the first page.
|
||||
"""
|
||||
png_path: Path = temp_dir / "page1.png"
|
||||
out_path: Path = temp_dir / "convert.webp"
|
||||
|
||||
# Run convert to get a decent thumbnail
|
||||
try:
|
||||
run_convert(
|
||||
density=300,
|
||||
scale="500x5000>",
|
||||
alpha="remove",
|
||||
strip=True,
|
||||
trim=False,
|
||||
auto_orient=True,
|
||||
use_cropbox=True,
|
||||
input_file=f"{in_path}[0]",
|
||||
output_file=str(out_path),
|
||||
logging_group=logging_group,
|
||||
)
|
||||
_render_pdf_thumbnail(in_path, png_path, out_path, logging_group)
|
||||
except ParseError as e:
|
||||
logger.error(f"Unable to make thumbnail with convert: {e}")
|
||||
out_path = make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group)
|
||||
logger.error(f"Unable to make thumbnail with pdftoppm: {e}")
|
||||
out_path = make_thumbnail_from_pdf_qpdf_fallback(
|
||||
in_path,
|
||||
temp_dir,
|
||||
logging_group,
|
||||
)
|
||||
|
||||
return out_path
|
||||
|
||||
|
||||
@@ -1,194 +0,0 @@
|
||||
"""
|
||||
Pure PDF page operations used by documents.bulk_edit.
|
||||
|
||||
This module deliberately knows nothing about Django, Celery or the documents
|
||||
app: callers resolve documents, choose output paths and queue work. Every
|
||||
function that writes a PDF removes unreferenced resources before saving.
|
||||
|
||||
pikepdf is always called as ``pikepdf.open(...)`` / ``pikepdf.new()`` (never
|
||||
``from pikepdf import open``) so tests can patch those module attributes.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
from typing import NamedTuple
|
||||
|
||||
import pikepdf
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Callable
|
||||
from collections.abc import Iterable
|
||||
from collections.abc import Mapping
|
||||
from collections.abc import Sequence
|
||||
from pathlib import Path
|
||||
from types import TracebackType
|
||||
|
||||
|
||||
class PageSpec(NamedTuple):
|
||||
"""One page of an output PDF: a 1-indexed source page, optionally rotated."""
|
||||
|
||||
page: int
|
||||
rotate: int = 0 # relative degrees, 0 leaves the page alone
|
||||
|
||||
|
||||
def _require_positive(pages: Iterable[int]) -> None:
|
||||
for page in pages:
|
||||
if page < 1:
|
||||
raise ValueError(f"Page numbers start at 1, got {page}")
|
||||
|
||||
|
||||
def rotate_pdf(src: Path, dst: Path, degrees: int) -> None:
|
||||
"""
|
||||
Rotate every page relatively on the opened document, not a rebuild, so Info,
|
||||
XMP and outlines are kept. ``src`` is not modified.
|
||||
"""
|
||||
with pikepdf.open(src) as pdf:
|
||||
for page in pdf.pages:
|
||||
page.rotate(degrees, relative=True)
|
||||
pdf.remove_unreferenced_resources()
|
||||
pdf.save(dst)
|
||||
|
||||
|
||||
def remove_pages(src: Path, dst: Path, pages: Iterable[int]) -> None:
|
||||
"""
|
||||
Remove 1-indexed pages from the opened document, not a rebuild, so Info, XMP
|
||||
and outlines are kept. ``src`` is not modified.
|
||||
|
||||
Duplicates are ignored. Pages are removed highest first so earlier removals
|
||||
never shift the index of later ones.
|
||||
"""
|
||||
unique = sorted(set(pages))
|
||||
_require_positive(unique)
|
||||
with pikepdf.open(src) as pdf:
|
||||
for page_num in reversed(unique):
|
||||
del pdf.pages[page_num - 1]
|
||||
pdf.remove_unreferenced_resources()
|
||||
pdf.save(dst)
|
||||
|
||||
|
||||
def build_pdfs(
|
||||
src: Path,
|
||||
outputs: Sequence[tuple[Sequence[PageSpec], Callable[[], Path]]],
|
||||
) -> list[Path]:
|
||||
"""
|
||||
Build one new PDF per output from pages of ``src``, opening ``src`` once.
|
||||
|
||||
Each output is ``(page_specs, make_dst)``. Every page number is checked against
|
||||
``src`` before any output is built, and ``make_dst`` is called after that
|
||||
output's pages are copied and immediately before it is saved, so a bad page
|
||||
number in any output never leaves a destination behind. Document-level data
|
||||
(Info, XMP, outlines) is not carried over. Returns the written paths in output
|
||||
order.
|
||||
"""
|
||||
for specs, _ in outputs:
|
||||
_require_positive(spec.page for spec in specs)
|
||||
|
||||
written: list[Path] = []
|
||||
with pikepdf.open(src) as source:
|
||||
page_count = len(source.pages)
|
||||
for specs, _ in outputs:
|
||||
for spec in specs:
|
||||
if spec.page > page_count:
|
||||
raise IndexError(
|
||||
f"Page {spec.page} is out of range, the PDF has "
|
||||
f"{page_count} pages",
|
||||
)
|
||||
for specs, make_dst in outputs:
|
||||
dst = pikepdf.new()
|
||||
for spec in specs:
|
||||
dst.pages.append(source.pages[spec.page - 1])
|
||||
if spec.rotate:
|
||||
dst.pages[-1].rotate(spec.rotate, relative=True)
|
||||
dst.remove_unreferenced_resources()
|
||||
path = make_dst()
|
||||
dst.save(path)
|
||||
dst.close()
|
||||
written.append(path)
|
||||
return written
|
||||
|
||||
|
||||
def validate_page_operations(
|
||||
operations: Sequence[Mapping[str, int]],
|
||||
*,
|
||||
single_output: bool,
|
||||
) -> int:
|
||||
"""
|
||||
Validate ``edit_pdf`` style operations and return the output document count.
|
||||
|
||||
Each operation has ``page`` and optionally ``rotate`` and ``doc`` (the output
|
||||
document index, default 0). The bounds rule is kept as it was: a ``doc`` index
|
||||
must be below the number of operations.
|
||||
"""
|
||||
if not operations:
|
||||
raise ValueError("Output document index is out of bounds")
|
||||
|
||||
max_idx = max(op.get("doc", 0) for op in operations)
|
||||
if single_output and max_idx > 0:
|
||||
raise ValueError("Multiple output documents specified")
|
||||
|
||||
if any(
|
||||
op.get("doc", 0) < 0 or op.get("doc", 0) >= len(operations) for op in operations
|
||||
):
|
||||
raise ValueError("Output document index is out of bounds")
|
||||
|
||||
return max_idx + 1
|
||||
|
||||
|
||||
def needs_decrypt(src: Path) -> bool:
|
||||
"""
|
||||
True if ``src`` is encrypted. A PDF that needs a password to open at all
|
||||
counts as encrypted.
|
||||
"""
|
||||
try:
|
||||
with pikepdf.open(src) as pdf:
|
||||
return bool(pdf.is_encrypted)
|
||||
except pikepdf.PasswordError:
|
||||
return True
|
||||
|
||||
|
||||
def decrypt_pdf(src: Path, make_dst: Callable[[], Path], password: str) -> Path:
|
||||
"""
|
||||
Write an unencrypted copy of ``src`` and return its path.
|
||||
|
||||
``make_dst`` is only called once the password has been accepted, so a wrong
|
||||
password never leaves a destination behind.
|
||||
"""
|
||||
with pikepdf.open(src, password=password) as pdf:
|
||||
pdf.remove_unreferenced_resources()
|
||||
dst = make_dst()
|
||||
pdf.save(dst)
|
||||
return dst
|
||||
|
||||
|
||||
class PdfMerger:
|
||||
"""
|
||||
Accumulates the pages of several PDFs into one new PDF.
|
||||
|
||||
``add`` raises if a source cannot be read; deciding whether to skip it is the
|
||||
caller's policy. Use as a context manager so the merged PDF is closed.
|
||||
"""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._pdf = pikepdf.new()
|
||||
self._version: str = self._pdf.pdf_version
|
||||
|
||||
def __enter__(self) -> PdfMerger:
|
||||
return self
|
||||
|
||||
def __exit__(
|
||||
self,
|
||||
exc_type: type[BaseException] | None,
|
||||
exc: BaseException | None,
|
||||
tb: TracebackType | None,
|
||||
) -> None:
|
||||
self._pdf.close()
|
||||
|
||||
def add(self, path: Path) -> None:
|
||||
with pikepdf.open(str(path)) as pdf:
|
||||
self._version = max(self._version, pdf.pdf_version)
|
||||
self._pdf.pages.extend(pdf.pages)
|
||||
|
||||
def save(self, dst: Path) -> None:
|
||||
self._pdf.remove_unreferenced_resources()
|
||||
self._pdf.save(dst, min_version=self._version)
|
||||
@@ -2144,8 +2144,6 @@ class BulkEditSerializer(
|
||||
raise serializers.ValidationError("pages must be a list")
|
||||
if not all(isinstance(i, int) for i in parameters["pages"]):
|
||||
raise serializers.ValidationError("pages must be a list of integers")
|
||||
if any(i < 1 for i in parameters["pages"]):
|
||||
raise serializers.ValidationError("pages must be positive integers")
|
||||
|
||||
def _validate_parameters_merge(self, parameters) -> None:
|
||||
if "delete_originals" in parameters:
|
||||
|
||||
@@ -1843,36 +1843,6 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
|
||||
m.assert_called_once()
|
||||
self.assertEqual(m.call_args.kwargs["pages"], [[1], [2, 3, 4], [5]])
|
||||
|
||||
@mock.patch("documents.serialisers.bulk_edit.delete_pages")
|
||||
def test_bulk_edit_delete_pages_rejects_pages_below_one(self, m) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A legacy delete_pages bulk edit
|
||||
WHEN:
|
||||
- API to bulk edit is called with a page number below 1
|
||||
THEN:
|
||||
- API returns HTTP 400
|
||||
- delete_pages is not called
|
||||
"""
|
||||
self.setup_mock(m, "delete_pages")
|
||||
|
||||
for pages in ([0], [-1], [1, 0]):
|
||||
with self.subTest(pages=pages):
|
||||
response = self.client.post(
|
||||
"/api/documents/bulk_edit/",
|
||||
json.dumps(
|
||||
{
|
||||
"documents": [self.doc2.id],
|
||||
"method": "delete_pages",
|
||||
"parameters": {"pages": pages},
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
self.assertIn(b"pages must be positive integers", response.content)
|
||||
m.assert_not_called()
|
||||
|
||||
@mock.patch("documents.views.bulk_edit.rotate")
|
||||
def test_rotate_insufficient_permissions(self, m) -> None:
|
||||
self.doc1.owner = User.objects.get(username="temp_admin")
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
import shutil
|
||||
from collections.abc import Callable
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
import pikepdf
|
||||
from django.contrib.auth.models import Group
|
||||
from django.contrib.auth.models import Permission
|
||||
from django.contrib.auth.models import User
|
||||
@@ -21,7 +21,6 @@ from documents.models import Document
|
||||
from documents.models import DocumentType
|
||||
from documents.models import StoragePath
|
||||
from documents.models import Tag
|
||||
from documents.pdf_ops import PageSpec
|
||||
from documents.permissions import set_permissions_for_objects
|
||||
from paperless_testing.dirs import DirectoriesMixin
|
||||
from paperless_testing.permissions import grant_object
|
||||
@@ -794,14 +793,16 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
self.img_doc.save()
|
||||
|
||||
@staticmethod
|
||||
def fake_decrypt(
|
||||
src: Path,
|
||||
make_dst: Callable[[], Path],
|
||||
password: str,
|
||||
) -> Path:
|
||||
dst = make_dst()
|
||||
dst.write_bytes(b"password removed")
|
||||
return dst
|
||||
def mock_password_required_pdf(
|
||||
mock_open: mock.Mock,
|
||||
fake_pdf: mock.Mock,
|
||||
) -> None:
|
||||
password_context = mock.MagicMock()
|
||||
password_context.__enter__.return_value = fake_pdf
|
||||
mock_open.side_effect = [
|
||||
pikepdf.PasswordError("password required"),
|
||||
password_context,
|
||||
]
|
||||
|
||||
@mock.patch("documents.tasks.consume_file.s")
|
||||
def test_merge(self, mock_consume_file) -> None:
|
||||
@@ -846,12 +847,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
|
||||
@mock.patch("documents.pdf_ops.PdfMerger")
|
||||
@mock.patch("pikepdf.open")
|
||||
@mock.patch("documents.tasks.consume_file.s")
|
||||
def test_merge_uses_latest_version_source_for_root_selection(
|
||||
self,
|
||||
mock_consume_file,
|
||||
mock_merger,
|
||||
mock_open_pdf,
|
||||
) -> None:
|
||||
version_file = self.dirs.scratch_dir / "sample2_version_merge.pdf"
|
||||
shutil.copy(self.doc2.source_path, version_file)
|
||||
@@ -862,14 +863,16 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
filename=version_file,
|
||||
mime_type="application/pdf",
|
||||
)
|
||||
merger = mock_merger.return_value.__enter__.return_value
|
||||
merger.save.side_effect = lambda dst: shutil.copy(version.source_path, dst)
|
||||
fake_pdf = mock.MagicMock()
|
||||
fake_pdf.pdf_version = "1.7"
|
||||
fake_pdf.pages = [mock.Mock()]
|
||||
mock_open_pdf.return_value.__enter__.return_value = fake_pdf
|
||||
|
||||
result = bulk_edit.merge([self.doc2.id])
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
merger.add.assert_called_once_with(version.source_path)
|
||||
mock_consume_file.assert_called_once()
|
||||
mock_open_pdf.assert_called_once_with(str(version.source_path))
|
||||
mock_consume_file.assert_not_called()
|
||||
|
||||
@mock.patch("documents.bulk_edit.delete.si")
|
||||
@mock.patch("documents.tasks.consume_file.s")
|
||||
@@ -1031,18 +1034,18 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
)
|
||||
|
||||
@mock.patch("documents.tasks.consume_file.delay")
|
||||
@mock.patch("documents.pdf_ops.PdfMerger.add")
|
||||
def test_merge_with_errors(self, mock_add, mock_consume_file) -> None:
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_merge_with_errors(self, mock_open_pdf, mock_consume_file) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Existing documents
|
||||
WHEN:
|
||||
- Merge action is called with 2 documents
|
||||
- Error occurs when adding both files
|
||||
- Error occurs when opening both files
|
||||
THEN:
|
||||
- Consume file should not be called
|
||||
"""
|
||||
mock_add.side_effect = Exception("Error opening PDF")
|
||||
mock_open_pdf.side_effect = Exception("Error opening PDF")
|
||||
doc_ids = [self.doc2.id, self.doc3.id]
|
||||
|
||||
with self.assertLogs("paperless.bulk_edit", level="ERROR") as cm:
|
||||
@@ -1079,12 +1082,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
self.assertEqual(result, "OK")
|
||||
|
||||
@mock.patch("documents.bulk_edit.group")
|
||||
@mock.patch("documents.pdf_ops.build_pdfs")
|
||||
@mock.patch("pikepdf.open")
|
||||
@mock.patch("documents.tasks.consume_file.s")
|
||||
def test_split_uses_latest_version_source_for_root_selection(
|
||||
self,
|
||||
mock_consume_file,
|
||||
mock_build_pdfs,
|
||||
mock_open_pdf,
|
||||
mock_group,
|
||||
) -> None:
|
||||
version_file = self.dirs.scratch_dir / "sample2_version_split.pdf"
|
||||
@@ -1096,15 +1099,17 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
filename=version_file,
|
||||
mime_type="application/pdf",
|
||||
)
|
||||
mock_build_pdfs.return_value = [version.source_path, version.source_path]
|
||||
fake_pdf = mock.MagicMock()
|
||||
fake_pdf.pages = [mock.Mock(), mock.Mock()]
|
||||
mock_open_pdf.return_value.__enter__.return_value = fake_pdf
|
||||
mock_group.return_value.delay.return_value = None
|
||||
|
||||
result = bulk_edit.split([self.doc2.id], [[1], [2]])
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
self.assertEqual(mock_build_pdfs.call_args.args[0], version.source_path)
|
||||
self.assertEqual(mock_consume_file.call_count, 2)
|
||||
mock_group.return_value.delay.assert_called_once()
|
||||
mock_open_pdf.assert_called_once_with(version.source_path)
|
||||
mock_consume_file.assert_not_called()
|
||||
mock_group.return_value.delay.assert_not_called()
|
||||
|
||||
@mock.patch("documents.bulk_edit.delete.si")
|
||||
@mock.patch("documents.tasks.consume_file.s")
|
||||
@@ -1192,18 +1197,18 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
self.assertEqual(self.doc2.archive_serial_number, 222)
|
||||
|
||||
@mock.patch("documents.tasks.consume_file.apply_async")
|
||||
@mock.patch("documents.pdf_ops.build_pdfs")
|
||||
def test_split_with_errors(self, mock_build_pdfs, mock_consume_file) -> None:
|
||||
@mock.patch("pikepdf.Pdf.save")
|
||||
def test_split_with_errors(self, mock_save_pdf, mock_consume_file) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Existing documents
|
||||
WHEN:
|
||||
- Split action is called with 1 document and 2 page groups
|
||||
- Error occurs when building the files
|
||||
- Error occurs when saving the files
|
||||
THEN:
|
||||
- Consume file should not be called
|
||||
"""
|
||||
mock_build_pdfs.side_effect = Exception("Error building PDFs")
|
||||
mock_save_pdf.side_effect = Exception("Error saving PDF")
|
||||
doc_ids = [self.doc2.id]
|
||||
pages = [[1, 2], [3]]
|
||||
|
||||
@@ -1238,10 +1243,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
self.assertEqual(result, "OK")
|
||||
|
||||
@mock.patch("documents.tasks.consume_file.apply_async")
|
||||
@mock.patch("documents.pdf_ops.rotate_pdf")
|
||||
@mock.patch("pikepdf.Pdf.save")
|
||||
def test_rotate_with_error(
|
||||
self,
|
||||
mock_rotate_pdf,
|
||||
mock_pdf_save,
|
||||
mock_consume_delay,
|
||||
) -> None:
|
||||
"""
|
||||
@@ -1249,11 +1254,11 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
- Existing documents
|
||||
WHEN:
|
||||
- Rotate action is called with 2 documents
|
||||
- Rotating the PDF raises an error
|
||||
- PikePDF raises an error
|
||||
THEN:
|
||||
- Rotate action should be called 0 times
|
||||
"""
|
||||
mock_rotate_pdf.side_effect = Exception("Error rotating PDF")
|
||||
mock_pdf_save.side_effect = Exception("Error saving PDF")
|
||||
doc_ids = [self.doc2.id, self.doc3.id]
|
||||
|
||||
with self.assertLogs("paperless.bulk_edit", level="ERROR") as cm:
|
||||
@@ -1288,10 +1293,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
|
||||
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
|
||||
@mock.patch("documents.tasks.consume_file.apply_async")
|
||||
@mock.patch("documents.pdf_ops.rotate_pdf")
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_rotate_explicit_selection_uses_root_source_when_root_selected(
|
||||
self,
|
||||
mock_rotate_pdf,
|
||||
mock_open,
|
||||
mock_consume_delay,
|
||||
mock_magic,
|
||||
) -> None:
|
||||
@@ -1300,6 +1305,9 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
title="B version 1",
|
||||
root_document=self.doc2,
|
||||
)
|
||||
fake_pdf = mock.MagicMock()
|
||||
fake_pdf.pages = [mock.Mock()]
|
||||
mock_open.return_value.__enter__.return_value = fake_pdf
|
||||
|
||||
result = bulk_edit.rotate(
|
||||
[self.doc2.id],
|
||||
@@ -1308,35 +1316,26 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
)
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
self.assertEqual(mock_rotate_pdf.call_args.args[0], self.doc2.source_path)
|
||||
mock_open.assert_called_once_with(self.doc2.source_path)
|
||||
mock_consume_delay.assert_called_once()
|
||||
|
||||
@mock.patch("documents.tasks.consume_file.apply_async")
|
||||
@mock.patch("documents.pdf_ops.remove_pages")
|
||||
@mock.patch("pikepdf.Pdf.save")
|
||||
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
|
||||
def test_delete_pages(
|
||||
self,
|
||||
mock_magic,
|
||||
mock_remove_pages,
|
||||
mock_consume_delay,
|
||||
) -> None:
|
||||
def test_delete_pages(self, mock_magic, mock_pdf_save, mock_consume_delay) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Existing documents
|
||||
WHEN:
|
||||
- Delete pages action is called with 1 document and 2 pages
|
||||
THEN:
|
||||
- The pages are removed from the document's source PDF
|
||||
- Save should be called once
|
||||
- A new version should be enqueued via consume_file
|
||||
"""
|
||||
doc_ids = [self.doc2.id]
|
||||
pages = [1, 3]
|
||||
result = bulk_edit.delete_pages(doc_ids, pages)
|
||||
mock_remove_pages.assert_called_once_with(
|
||||
self.doc2.source_path,
|
||||
mock.ANY,
|
||||
[1, 3],
|
||||
)
|
||||
mock_pdf_save.assert_called_once()
|
||||
mock_consume_delay.assert_called_once()
|
||||
task_kwargs = mock_consume_delay.call_args.kwargs["kwargs"]
|
||||
self.assertEqual(task_kwargs["input_doc"].root_document_id, self.doc2.id)
|
||||
@@ -1348,10 +1347,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
|
||||
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
|
||||
@mock.patch("documents.tasks.consume_file.apply_async")
|
||||
@mock.patch("documents.pdf_ops.remove_pages")
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_delete_pages_explicit_selection_uses_root_source_when_root_selected(
|
||||
self,
|
||||
mock_remove_pages,
|
||||
mock_open,
|
||||
mock_consume_delay,
|
||||
mock_magic,
|
||||
) -> None:
|
||||
@@ -1360,6 +1359,9 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
title="B version 1",
|
||||
root_document=self.doc2,
|
||||
)
|
||||
fake_pdf = mock.MagicMock()
|
||||
fake_pdf.pages = [mock.Mock(), mock.Mock()]
|
||||
mock_open.return_value.__enter__.return_value = fake_pdf
|
||||
|
||||
result = bulk_edit.delete_pages(
|
||||
[self.doc2.id],
|
||||
@@ -1368,26 +1370,23 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
)
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
self.assertEqual(mock_remove_pages.call_args.args[0], self.doc2.source_path)
|
||||
mock_open.assert_called_once_with(self.doc2.source_path)
|
||||
mock_consume_delay.assert_called_once()
|
||||
|
||||
@mock.patch("documents.tasks.consume_file.apply_async")
|
||||
@mock.patch("documents.pdf_ops.remove_pages")
|
||||
def test_delete_pages_with_error(
|
||||
self,
|
||||
mock_remove_pages,
|
||||
mock_consume_delay,
|
||||
) -> None:
|
||||
@mock.patch("pikepdf.Pdf.save")
|
||||
def test_delete_pages_with_error(self, mock_pdf_save, mock_consume_delay) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Existing documents
|
||||
WHEN:
|
||||
- Delete pages action is called with 1 document and 2 pages
|
||||
- Removing the pages raises an error
|
||||
- PikePDF raises an error
|
||||
THEN:
|
||||
- Save should be called once
|
||||
- No new version should be enqueued
|
||||
"""
|
||||
mock_remove_pages.side_effect = Exception("Error removing pages")
|
||||
mock_pdf_save.side_effect = Exception("Error saving PDF")
|
||||
doc_ids = [self.doc2.id]
|
||||
pages = [1, 3]
|
||||
|
||||
@@ -1417,41 +1416,6 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
self.assertEqual(result, "OK")
|
||||
mock_group.return_value.delay.assert_called_once()
|
||||
|
||||
@mock.patch("documents.bulk_edit.group")
|
||||
@mock.patch("documents.pdf_ops.build_pdfs")
|
||||
@mock.patch("documents.tasks.consume_file.s")
|
||||
def test_edit_pdf_maps_operations_to_outputs(
|
||||
self,
|
||||
mock_consume_file: mock.Mock,
|
||||
mock_build_pdfs: mock.Mock,
|
||||
mock_group: mock.Mock,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Existing document
|
||||
WHEN:
|
||||
- edit_pdf is called with operations interleaved across two outputs,
|
||||
some of them rotated
|
||||
THEN:
|
||||
- Each output is built from its own operations, in operation order,
|
||||
with the requested rotation
|
||||
"""
|
||||
mock_build_pdfs.return_value = [self.doc2.source_path, self.doc2.source_path]
|
||||
mock_group.return_value.delay.return_value = None
|
||||
operations = [
|
||||
{"page": 3, "doc": 1},
|
||||
{"page": 1, "doc": 0, "rotate": 90},
|
||||
{"page": 2, "doc": 1, "rotate": 180},
|
||||
]
|
||||
|
||||
bulk_edit.edit_pdf([self.doc2.id], operations)
|
||||
|
||||
outputs = mock_build_pdfs.call_args.args[1]
|
||||
self.assertEqual(
|
||||
[specs for specs, _ in outputs],
|
||||
[[PageSpec(1, 90)], [PageSpec(3), PageSpec(2, 180)]],
|
||||
)
|
||||
|
||||
@mock.patch("documents.bulk_edit.group")
|
||||
@mock.patch("documents.tasks.consume_file.s")
|
||||
def test_edit_pdf_with_user_override(self, mock_consume_file, mock_group) -> None:
|
||||
@@ -1567,10 +1531,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
|
||||
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
|
||||
@mock.patch("documents.tasks.consume_file.apply_async")
|
||||
@mock.patch("documents.pdf_ops.build_pdfs")
|
||||
@mock.patch("pikepdf.new")
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_edit_pdf_explicit_selection_uses_root_source_when_root_selected(
|
||||
self,
|
||||
mock_build_pdfs,
|
||||
mock_open,
|
||||
mock_new,
|
||||
mock_consume_delay,
|
||||
mock_magic,
|
||||
) -> None:
|
||||
@@ -1579,7 +1545,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
title="B version 1",
|
||||
root_document=self.doc2,
|
||||
)
|
||||
mock_build_pdfs.return_value = [Path("edited.pdf")]
|
||||
fake_pdf = mock.MagicMock()
|
||||
fake_pdf.pages = [mock.Mock()]
|
||||
mock_open.return_value.__enter__.return_value = fake_pdf
|
||||
output_pdf = mock.MagicMock()
|
||||
output_pdf.pages = []
|
||||
mock_new.return_value = output_pdf
|
||||
|
||||
result = bulk_edit.edit_pdf(
|
||||
[self.doc2.id],
|
||||
@@ -1589,7 +1560,7 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
)
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
self.assertEqual(mock_build_pdfs.call_args.args[0], self.doc2.source_path)
|
||||
mock_open.assert_called_once_with(self.doc2.source_path)
|
||||
mock_consume_delay.assert_called_once()
|
||||
|
||||
@mock.patch("documents.bulk_edit.group")
|
||||
@@ -1615,6 +1586,31 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
self.assertEqual(result, "OK")
|
||||
mock_group.return_value.delay.assert_called_once()
|
||||
|
||||
@mock.patch("documents.bulk_edit.group")
|
||||
@mock.patch("documents.tasks.consume_file.s")
|
||||
def test_edit_pdf_open_failure(
|
||||
self,
|
||||
mock_consume_file: mock.Mock,
|
||||
mock_group: mock.Mock,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Existing document
|
||||
WHEN:
|
||||
- edit_pdf fails to open PDF
|
||||
THEN:
|
||||
- Task group is not called
|
||||
"""
|
||||
doc_ids = [self.doc2.id]
|
||||
operations = [
|
||||
{"page": 9999}, # invalid page, forces error during PDF load
|
||||
]
|
||||
with self.assertLogs("paperless.bulk_edit", level="ERROR"):
|
||||
with self.assertRaises(Exception):
|
||||
bulk_edit.edit_pdf(doc_ids, operations)
|
||||
mock_group.assert_not_called()
|
||||
mock_consume_file.assert_not_called()
|
||||
|
||||
@mock.patch("documents.bulk_edit.group")
|
||||
@mock.patch("documents.tasks.consume_file.s")
|
||||
def test_edit_pdf_multiple_outputs_with_update_flag_errors(
|
||||
@@ -1641,15 +1637,23 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
mock_group.assert_not_called()
|
||||
mock_consume_file.assert_not_called()
|
||||
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_edit_pdf_rejects_invalid_operations(self, mock_open) -> None:
|
||||
for operations in ([], [{"page": 1, "doc": 2**32}]):
|
||||
with self.subTest(operations=operations):
|
||||
with self.assertLogs("paperless.bulk_edit", level="ERROR"):
|
||||
with self.assertRaisesRegex(ValueError, "index is out of bounds"):
|
||||
bulk_edit.edit_pdf([self.doc2.id], operations)
|
||||
|
||||
mock_open.assert_not_called()
|
||||
|
||||
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file.delay")
|
||||
@mock.patch("documents.tasks.consume_file.apply_async")
|
||||
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
|
||||
@mock.patch("documents.pdf_ops.decrypt_pdf")
|
||||
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_remove_password_update_document(
|
||||
self,
|
||||
mock_needs_decrypt,
|
||||
mock_decrypt,
|
||||
mock_open,
|
||||
mock_mkdtemp,
|
||||
mock_consume_delay,
|
||||
mock_update_document,
|
||||
@@ -1658,7 +1662,16 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
temp_dir = self.dirs.scratch_dir / "remove-password-update"
|
||||
temp_dir.mkdir(parents=True, exist_ok=True)
|
||||
mock_mkdtemp.return_value = str(temp_dir)
|
||||
mock_decrypt.side_effect = self.fake_decrypt
|
||||
|
||||
fake_pdf = mock.MagicMock()
|
||||
fake_pdf.pages = [mock.Mock(), mock.Mock(), mock.Mock()]
|
||||
fake_pdf.is_encrypted = True
|
||||
|
||||
def save_side_effect(target_path):
|
||||
Path(target_path).write_bytes(b"new pdf content")
|
||||
|
||||
fake_pdf.save.side_effect = save_side_effect
|
||||
mock_open.return_value.__enter__.return_value = fake_pdf
|
||||
|
||||
result = bulk_edit.remove_password(
|
||||
[doc.id],
|
||||
@@ -1667,8 +1680,14 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
)
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
mock_needs_decrypt.assert_called_once_with(doc.source_path)
|
||||
mock_decrypt.assert_called_once_with(doc.source_path, mock.ANY, "secret")
|
||||
self.assertEqual(
|
||||
mock_open.call_args_list,
|
||||
[
|
||||
mock.call(doc.source_path),
|
||||
mock.call(doc.source_path, password="secret"),
|
||||
],
|
||||
)
|
||||
fake_pdf.remove_unreferenced_resources.assert_called_once()
|
||||
mock_update_document.assert_not_called()
|
||||
mock_consume_delay.assert_called_once()
|
||||
task_kwargs = mock_consume_delay.call_args.kwargs["kwargs"]
|
||||
@@ -1681,15 +1700,40 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
self.assertEqual(task_kwargs["input_doc"].root_document_id, doc.id)
|
||||
self.assertIsNotNone(task_kwargs["overrides"])
|
||||
|
||||
@mock.patch("documents.tasks.consume_file.apply_async")
|
||||
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_remove_password_update_document_skips_unencrypted_pdf(
|
||||
self,
|
||||
mock_open,
|
||||
mock_mkdtemp,
|
||||
mock_consume_delay,
|
||||
) -> None:
|
||||
doc = self.doc1
|
||||
fake_pdf = mock.MagicMock()
|
||||
fake_pdf.is_encrypted = False
|
||||
mock_open.return_value.__enter__.return_value = fake_pdf
|
||||
|
||||
result = bulk_edit.remove_password(
|
||||
[doc.id],
|
||||
password="secret",
|
||||
update_document=True,
|
||||
)
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
mock_open.assert_called_once_with(doc.source_path)
|
||||
fake_pdf.remove_unreferenced_resources.assert_not_called()
|
||||
fake_pdf.save.assert_not_called()
|
||||
mock_mkdtemp.assert_not_called()
|
||||
mock_consume_delay.assert_not_called()
|
||||
|
||||
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file.delay")
|
||||
@mock.patch("documents.tasks.consume_file.apply_async")
|
||||
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
|
||||
@mock.patch("documents.pdf_ops.decrypt_pdf")
|
||||
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_remove_password_update_document_uses_source_paths(
|
||||
self,
|
||||
mock_needs_decrypt,
|
||||
mock_decrypt,
|
||||
mock_open,
|
||||
mock_mkdtemp,
|
||||
mock_consume_delay,
|
||||
mock_update_document,
|
||||
@@ -1700,7 +1744,14 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
temp_dir = self.dirs.scratch_dir / "remove-password-source-file"
|
||||
temp_dir.mkdir(parents=True, exist_ok=True)
|
||||
mock_mkdtemp.return_value = str(temp_dir)
|
||||
mock_decrypt.side_effect = self.fake_decrypt
|
||||
|
||||
fake_pdf = mock.MagicMock()
|
||||
self.mock_password_required_pdf(mock_open, fake_pdf)
|
||||
|
||||
def save_side_effect(target_path):
|
||||
Path(target_path).write_bytes(b"new pdf content")
|
||||
|
||||
fake_pdf.save.side_effect = save_side_effect
|
||||
|
||||
result = bulk_edit.remove_password(
|
||||
[doc.id],
|
||||
@@ -1710,19 +1761,22 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
)
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
mock_needs_decrypt.assert_called_once_with(source_file)
|
||||
mock_decrypt.assert_called_once_with(source_file, mock.ANY, "secret")
|
||||
self.assertEqual(
|
||||
mock_open.call_args_list,
|
||||
[
|
||||
mock.call(source_file),
|
||||
mock.call(source_file, password="secret"),
|
||||
],
|
||||
)
|
||||
mock_update_document.assert_not_called()
|
||||
mock_consume_delay.assert_called_once()
|
||||
|
||||
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
|
||||
@mock.patch("documents.tasks.consume_file.apply_async")
|
||||
@mock.patch("documents.pdf_ops.decrypt_pdf")
|
||||
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_remove_password_explicit_selection_uses_root_source_when_root_selected(
|
||||
self,
|
||||
mock_needs_decrypt,
|
||||
mock_decrypt,
|
||||
mock_open,
|
||||
mock_consume_delay,
|
||||
mock_magic,
|
||||
) -> None:
|
||||
@@ -1731,7 +1785,8 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
title="A version 1",
|
||||
root_document=self.doc1,
|
||||
)
|
||||
mock_decrypt.return_value = Path("unprotected.pdf")
|
||||
fake_pdf = mock.MagicMock()
|
||||
self.mock_password_required_pdf(mock_open, fake_pdf)
|
||||
|
||||
result = bulk_edit.remove_password(
|
||||
[self.doc1.id],
|
||||
@@ -1741,11 +1796,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
)
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
mock_needs_decrypt.assert_called_once_with(self.doc1.source_path)
|
||||
mock_decrypt.assert_called_once_with(
|
||||
self.doc1.source_path,
|
||||
mock.ANY,
|
||||
"secret",
|
||||
self.assertEqual(
|
||||
mock_open.call_args_list,
|
||||
[
|
||||
mock.call(self.doc1.source_path),
|
||||
mock.call(self.doc1.source_path, password="secret"),
|
||||
],
|
||||
)
|
||||
mock_consume_delay.assert_called_once()
|
||||
|
||||
@@ -1753,12 +1809,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
@mock.patch("documents.bulk_edit.group")
|
||||
@mock.patch("documents.tasks.consume_file.s")
|
||||
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
|
||||
@mock.patch("documents.pdf_ops.decrypt_pdf")
|
||||
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_remove_password_creates_consumable_document(
|
||||
self,
|
||||
mock_needs_decrypt: mock.Mock,
|
||||
mock_decrypt: mock.Mock,
|
||||
mock_open: mock.Mock,
|
||||
mock_mkdtemp: mock.Mock,
|
||||
mock_consume_file: mock.Mock,
|
||||
mock_group: mock.Mock,
|
||||
@@ -1768,7 +1822,15 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
temp_dir = self.dirs.scratch_dir / "remove-password"
|
||||
temp_dir.mkdir(parents=True, exist_ok=True)
|
||||
mock_mkdtemp.return_value = str(temp_dir)
|
||||
mock_decrypt.side_effect = self.fake_decrypt
|
||||
|
||||
fake_pdf = mock.MagicMock()
|
||||
fake_pdf.pages = [mock.Mock(), mock.Mock()]
|
||||
self.mock_password_required_pdf(mock_open, fake_pdf)
|
||||
|
||||
def save_side_effect(target_path: Path) -> None:
|
||||
target_path.write_bytes(b"password removed")
|
||||
|
||||
fake_pdf.save.side_effect = save_side_effect
|
||||
mock_group.return_value.delay.return_value = None
|
||||
|
||||
user = User.objects.create(username="owner")
|
||||
@@ -1783,7 +1845,13 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
)
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
mock_decrypt.assert_called_once_with(doc.source_path, mock.ANY, "secret")
|
||||
self.assertEqual(
|
||||
mock_open.call_args_list,
|
||||
[
|
||||
mock.call(doc.source_path),
|
||||
mock.call(doc.source_path, password="secret"),
|
||||
],
|
||||
)
|
||||
mock_consume_file.assert_called_once()
|
||||
call_kwargs = mock_consume_file.call_args.kwargs
|
||||
consumable_document = call_kwargs["input_doc"]
|
||||
@@ -1805,18 +1873,21 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
@mock.patch("documents.bulk_edit.chord")
|
||||
@mock.patch("documents.bulk_edit.group")
|
||||
@mock.patch("documents.tasks.consume_file.s")
|
||||
@mock.patch("documents.pdf_ops.decrypt_pdf")
|
||||
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=False)
|
||||
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_remove_password_skips_unencrypted_pdf_without_queueing(
|
||||
self,
|
||||
mock_needs_decrypt: mock.Mock,
|
||||
mock_decrypt: mock.Mock,
|
||||
mock_open: mock.Mock,
|
||||
mock_mkdtemp: mock.Mock,
|
||||
mock_consume_file: mock.Mock,
|
||||
mock_group: mock.Mock,
|
||||
mock_chord: mock.Mock,
|
||||
mock_delete: mock.Mock,
|
||||
) -> None:
|
||||
doc = self.doc2
|
||||
fake_pdf = mock.MagicMock()
|
||||
fake_pdf.is_encrypted = False
|
||||
mock_open.return_value.__enter__.return_value = fake_pdf
|
||||
|
||||
result = bulk_edit.remove_password(
|
||||
[doc.id],
|
||||
@@ -1826,8 +1897,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
)
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
mock_needs_decrypt.assert_called_once_with(doc.source_path)
|
||||
mock_decrypt.assert_not_called()
|
||||
mock_open.assert_called_once_with(doc.source_path)
|
||||
fake_pdf.remove_unreferenced_resources.assert_not_called()
|
||||
fake_pdf.save.assert_not_called()
|
||||
mock_mkdtemp.assert_not_called()
|
||||
mock_consume_file.assert_not_called()
|
||||
mock_group.assert_not_called()
|
||||
mock_chord.assert_not_called()
|
||||
@@ -1838,12 +1911,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
@mock.patch("documents.bulk_edit.group")
|
||||
@mock.patch("documents.tasks.consume_file.s")
|
||||
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
|
||||
@mock.patch("documents.pdf_ops.decrypt_pdf")
|
||||
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_remove_password_deletes_original(
|
||||
self,
|
||||
mock_needs_decrypt: mock.Mock,
|
||||
mock_decrypt: mock.Mock,
|
||||
mock_open: mock.Mock,
|
||||
mock_mkdtemp: mock.Mock,
|
||||
mock_consume_file: mock.Mock,
|
||||
mock_group: mock.Mock,
|
||||
@@ -1854,7 +1925,15 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
temp_dir = self.dirs.scratch_dir / "remove-password-delete"
|
||||
temp_dir.mkdir(parents=True, exist_ok=True)
|
||||
mock_mkdtemp.return_value = str(temp_dir)
|
||||
mock_decrypt.side_effect = self.fake_decrypt
|
||||
|
||||
fake_pdf = mock.MagicMock()
|
||||
fake_pdf.pages = [mock.Mock(), mock.Mock()]
|
||||
self.mock_password_required_pdf(mock_open, fake_pdf)
|
||||
|
||||
def save_side_effect(target_path: Path) -> None:
|
||||
target_path.write_bytes(b"password removed")
|
||||
|
||||
fake_pdf.save.side_effect = save_side_effect
|
||||
mock_chord.return_value.delay.return_value = None
|
||||
|
||||
result = bulk_edit.remove_password(
|
||||
@@ -1866,23 +1945,23 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
)
|
||||
|
||||
self.assertEqual(result, "OK")
|
||||
mock_decrypt.assert_called_once_with(doc.source_path, mock.ANY, "secret")
|
||||
self.assertEqual(
|
||||
mock_open.call_args_list,
|
||||
[
|
||||
mock.call(doc.source_path),
|
||||
mock.call(doc.source_path, password="secret"),
|
||||
],
|
||||
)
|
||||
mock_consume_file.assert_called_once()
|
||||
mock_group.assert_not_called()
|
||||
mock_chord.assert_called_once()
|
||||
mock_chord.return_value.delay.assert_called_once()
|
||||
mock_delete.si.assert_called_once_with([doc.id])
|
||||
|
||||
@mock.patch(
|
||||
"documents.pdf_ops.decrypt_pdf",
|
||||
side_effect=RuntimeError("wrong password"),
|
||||
)
|
||||
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
|
||||
def test_remove_password_failure_raises_value_error(
|
||||
self,
|
||||
mock_needs_decrypt: mock.Mock,
|
||||
mock_decrypt: mock.Mock,
|
||||
) -> None:
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_remove_password_open_failure(self, mock_open: mock.Mock) -> None:
|
||||
mock_open.side_effect = RuntimeError("wrong password")
|
||||
|
||||
with self.assertLogs("paperless.bulk_edit", level="ERROR") as cm:
|
||||
with self.assertRaises(ValueError) as exc:
|
||||
bulk_edit.remove_password([self.doc1.id], password="secret")
|
||||
|
||||
@@ -1,11 +1,22 @@
|
||||
import subprocess
|
||||
from collections.abc import Generator
|
||||
from pathlib import Path
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
from PIL import Image
|
||||
from pytest_django.fixtures import Settings
|
||||
from pytest_mock import MockerFixture
|
||||
|
||||
from documents.parsers import ParseError
|
||||
from documents.parsers import _compute_thumbnail_dpi
|
||||
from documents.parsers import encode_thumbnail_webp
|
||||
from documents.parsers import get_default_file_extension
|
||||
from documents.parsers import get_default_thumbnail
|
||||
from documents.parsers import get_supported_file_extensions
|
||||
from documents.parsers import is_file_ext_supported
|
||||
from documents.parsers import make_thumbnail_from_pdf
|
||||
from documents.parsers import rasterize_pdf_page_to_png
|
||||
from paperless.parsers.registry import get_parser_registry
|
||||
from paperless.parsers.registry import reset_parser_registry
|
||||
from paperless.parsers.tesseract import RasterisedDocumentParser
|
||||
@@ -125,3 +136,374 @@ class TestParserAvailability:
|
||||
assert is_file_ext_supported(".pdf")
|
||||
assert not is_file_ext_supported(".hsdfh")
|
||||
assert not is_file_ext_supported("")
|
||||
|
||||
|
||||
class TestComputeThumbnailDpi:
|
||||
@pytest.mark.parametrize(
|
||||
("size", "expected"),
|
||||
[
|
||||
pytest.param((612.0, 792.0), (59, 2), id="letter-width-bound"),
|
||||
pytest.param((792.0, 612.0), (46, 2), id="landscape-rounded-up"),
|
||||
pytest.param((612.0, 100000.0), (4, 2), id="tall-strip-height-bound"),
|
||||
pytest.param((200.0, 300.0), (180, 2), id="small-page-width-bound"),
|
||||
pytest.param((72.0, 72.0), (300, 2), id="tiny-page-capped-at-300"),
|
||||
pytest.param(
|
||||
(1000000.0, 1000000.0),
|
||||
(1, 1),
|
||||
id="huge-page-minimum-one-unsupersampled",
|
||||
),
|
||||
pytest.param(None, (150, 1), id="unreadable-geometry-fallback"),
|
||||
],
|
||||
)
|
||||
def test_dpi_from_page_size(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tmp_path: Path,
|
||||
size: tuple[float, float] | None,
|
||||
expected: tuple[int, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A first page of the given size, or unreadable geometry
|
||||
WHEN:
|
||||
- The thumbnail DPI is computed
|
||||
THEN:
|
||||
- The expected DPI and supersample factor are returned
|
||||
"""
|
||||
mocker.patch(
|
||||
"paperless.parsers.utils.get_pdf_first_page_size_points",
|
||||
return_value=size,
|
||||
)
|
||||
assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected
|
||||
|
||||
|
||||
class TestMakeThumbnailFromPdf:
|
||||
@pytest.fixture
|
||||
def work_dir(self, tmp_path: Path) -> Path:
|
||||
path = tmp_path / "work"
|
||||
path.mkdir()
|
||||
return path
|
||||
|
||||
@staticmethod
|
||||
def _write_blank_pdf(path: Path, page_size: tuple[int, int] = (612, 792)) -> Path:
|
||||
pdf = pikepdf.new()
|
||||
pdf.add_blank_page(page_size=page_size)
|
||||
pdf.save(path, object_stream_mode=pikepdf.ObjectStreamMode.disable)
|
||||
return path
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("size", "expected_dpi", "expected_supersample"),
|
||||
[
|
||||
pytest.param((612.0, 792.0), 118, 2, id="known-geometry-2x"),
|
||||
pytest.param(None, 150, 1, id="unreadable-geometry-plain-fallback"),
|
||||
],
|
||||
)
|
||||
def test_render_dpi_requested(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tmp_path: Path,
|
||||
work_dir: Path,
|
||||
size: tuple[float, float] | None,
|
||||
expected_dpi: int,
|
||||
expected_supersample: int,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Readable or unreadable page geometry
|
||||
WHEN:
|
||||
- A thumbnail is made
|
||||
THEN:
|
||||
- Rasterize and encode get the matching DPI and supersample factor
|
||||
"""
|
||||
mocker.patch(
|
||||
"paperless.parsers.utils.get_pdf_first_page_size_points",
|
||||
return_value=size,
|
||||
)
|
||||
rasterize = mocker.patch("documents.parsers.rasterize_pdf_page_to_png")
|
||||
encode = mocker.patch("documents.parsers.encode_thumbnail_webp")
|
||||
|
||||
make_thumbnail_from_pdf(tmp_path / "in.pdf", work_dir)
|
||||
|
||||
assert rasterize.call_args.kwargs["dpi"] == expected_dpi
|
||||
assert encode.call_args.kwargs["supersample"] == expected_supersample
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("page_size", "expected_width"),
|
||||
[
|
||||
pytest.param((612, 792), 500, id="letter"),
|
||||
pytest.param((792, 612), 500, id="landscape-letter"),
|
||||
pytest.param((595, 842), 500, id="a4"),
|
||||
pytest.param((200, 300), 500, id="small-page"),
|
||||
pytest.param((72, 72), 300, id="tiny-page-capped"),
|
||||
],
|
||||
)
|
||||
def test_thumbnail_width(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
work_dir: Path,
|
||||
page_size: tuple[int, int],
|
||||
expected_width: int,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF whose first page has the given size in points
|
||||
WHEN:
|
||||
- A thumbnail is made from it
|
||||
THEN:
|
||||
- The thumbnail has the expected width
|
||||
"""
|
||||
pdf_path = self._write_blank_pdf(tmp_path / "in.pdf", page_size)
|
||||
|
||||
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
|
||||
|
||||
assert thumb == work_dir / "convert.webp"
|
||||
with Image.open(thumb) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.width == expected_width
|
||||
|
||||
@classmethod
|
||||
def _write_pdf_without_xref(cls, path: Path) -> Path:
|
||||
"""
|
||||
Cuts off the xref and trailer, which pdftoppm cannot recover from but qpdf can.
|
||||
"""
|
||||
cls._write_blank_pdf(path)
|
||||
data = path.read_bytes()
|
||||
path.write_bytes(data[: data.rindex(b"\nxref")])
|
||||
return path
|
||||
|
||||
def test_qpdf_repair_produces_real_thumbnail(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
work_dir: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF with its xref table and trailer cut off
|
||||
WHEN:
|
||||
- A thumbnail is made from it
|
||||
THEN:
|
||||
- The thumbnail is rendered from a qpdf repaired copy
|
||||
- The original file is unchanged
|
||||
"""
|
||||
pdf_path = self._write_pdf_without_xref(tmp_path / "broken.pdf")
|
||||
original_bytes = pdf_path.read_bytes()
|
||||
|
||||
with pytest.raises(ParseError):
|
||||
rasterize_pdf_page_to_png(pdf_path, work_dir / "probe.png", dpi=50)
|
||||
|
||||
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
|
||||
|
||||
assert thumb == work_dir / "convert_qpdf.webp"
|
||||
with Image.open(thumb) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.width == 500
|
||||
assert pdf_path.read_bytes() == original_bytes
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"qpdf_error",
|
||||
[
|
||||
pytest.param(subprocess.CalledProcessError(2, "qpdf"), id="qpdf-fails"),
|
||||
pytest.param(None, id="repaired-still-unrenderable"),
|
||||
],
|
||||
)
|
||||
def test_double_failure_uses_default_thumbnail(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tmp_path: Path,
|
||||
work_dir: Path,
|
||||
qpdf_error: subprocess.CalledProcessError | None,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF that cannot be rendered, even after qpdf repair
|
||||
WHEN:
|
||||
- A thumbnail is made from it
|
||||
THEN:
|
||||
- A copy of the default thumbnail is returned
|
||||
"""
|
||||
mocker.patch(
|
||||
"documents.parsers.rasterize_pdf_page_to_png",
|
||||
side_effect=ParseError("Does not compute."),
|
||||
)
|
||||
if qpdf_error is not None:
|
||||
mocker.patch("documents.parsers.run_subprocess", side_effect=qpdf_error)
|
||||
pdf_path = self._write_blank_pdf(tmp_path / "in.pdf")
|
||||
|
||||
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
|
||||
|
||||
assert thumb == work_dir / "document.webp"
|
||||
assert thumb.read_bytes() == get_default_thumbnail().read_bytes()
|
||||
|
||||
|
||||
class TestRasterizePdfPageToPng:
|
||||
@staticmethod
|
||||
def _write_pdf(
|
||||
path: Path,
|
||||
*,
|
||||
crop_box: tuple[float, float, float, float] | None = None,
|
||||
rotate: int | None = None,
|
||||
) -> Path:
|
||||
pdf = pikepdf.new()
|
||||
pdf.add_blank_page(page_size=(144, 72))
|
||||
pdf.add_blank_page(page_size=(300, 300))
|
||||
page = pdf.pages[0]
|
||||
if crop_box is not None:
|
||||
page.obj.CropBox = pikepdf.Array(crop_box)
|
||||
if rotate is not None:
|
||||
page.obj.Rotate = rotate
|
||||
pdf.save(path)
|
||||
return path
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("crop_box", "rotate", "expected_size"),
|
||||
[
|
||||
pytest.param(None, None, (144, 72), id="plain"),
|
||||
pytest.param(None, 90, (72, 144), id="rotated-90"),
|
||||
pytest.param((0, 0, 72, 36), None, (72, 36), id="crop-box"),
|
||||
],
|
||||
)
|
||||
def test_renders_first_page_to_exact_path(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
crop_box: tuple[float, float, float, float] | None,
|
||||
rotate: int | None,
|
||||
expected_size: tuple[int, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A two page PDF, first page optionally cropped or rotated
|
||||
WHEN:
|
||||
- The first page is rasterized at 72 DPI
|
||||
THEN:
|
||||
- Only out_path is written, sized to the first page's crop and rotation
|
||||
"""
|
||||
pdf_path = self._write_pdf(
|
||||
tmp_path / "in.pdf",
|
||||
crop_box=crop_box,
|
||||
rotate=rotate,
|
||||
)
|
||||
out_dir = tmp_path / "out"
|
||||
out_dir.mkdir()
|
||||
out_path = out_dir / "page1.png"
|
||||
|
||||
rasterize_pdf_page_to_png(pdf_path, out_path, dpi=72)
|
||||
|
||||
assert list(out_dir.iterdir()) == [out_path]
|
||||
with Image.open(out_path) as im:
|
||||
assert im.format == "PNG"
|
||||
assert im.size == expected_size
|
||||
|
||||
def test_failure_raises_parse_error(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A file that is not a PDF
|
||||
WHEN:
|
||||
- Rasterization is attempted
|
||||
THEN:
|
||||
- A ParseError is raised
|
||||
"""
|
||||
bad = tmp_path / "bad.pdf"
|
||||
bad.write_bytes(b"not a pdf")
|
||||
|
||||
with pytest.raises(ParseError):
|
||||
rasterize_pdf_page_to_png(bad, tmp_path / "page1.png", dpi=72)
|
||||
|
||||
|
||||
class TestEncodeThumbnailWebp:
|
||||
@pytest.mark.parametrize(
|
||||
("mode", "color"),
|
||||
[
|
||||
pytest.param("RGBA", (0, 0, 0, 0), id="rgba"),
|
||||
pytest.param("LA", (0, 0), id="la"),
|
||||
],
|
||||
)
|
||||
def test_alpha_flattened_onto_white(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
mode: str,
|
||||
color: tuple[int, ...],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A fully transparent PNG with an alpha channel
|
||||
WHEN:
|
||||
- It is encoded as a thumbnail
|
||||
THEN:
|
||||
- The WebP output is RGB with the transparency flattened to white
|
||||
"""
|
||||
png_path = tmp_path / "in.png"
|
||||
Image.new(mode, (20, 10), color).save(png_path)
|
||||
out_path = tmp_path / "out.webp"
|
||||
|
||||
encode_thumbnail_webp(png_path, out_path)
|
||||
|
||||
with Image.open(out_path) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.mode == "RGB"
|
||||
assert im.size == (20, 10)
|
||||
red, green, blue = im.getpixel((10, 5))
|
||||
assert min(red, green, blue) >= 250
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("in_size", "supersample", "expected_size"),
|
||||
[
|
||||
pytest.param((1000, 2000), 1, (500, 1000), id="too-wide-shrunk"),
|
||||
pytest.param((100, 10000), 1, (50, 5000), id="too-tall-shrunk"),
|
||||
pytest.param((100, 200), 1, (100, 200), id="small-not-enlarged"),
|
||||
pytest.param((1000, 1400), 2, (500, 700), id="2x-halved"),
|
||||
pytest.param((1001, 1401), 2, (500, 700), id="2x-odd-rounded"),
|
||||
pytest.param((600, 800), 2, (300, 400), id="2x-small-not-enlarged"),
|
||||
pytest.param((900, 1200), 2, (450, 600), id="2x-below-clamp"),
|
||||
],
|
||||
)
|
||||
def test_size_clamped_and_downsampled(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
in_size: tuple[int, int],
|
||||
supersample: int,
|
||||
expected_size: tuple[int, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A rendered image and its supersample factor
|
||||
WHEN:
|
||||
- It is encoded as a thumbnail with that factor
|
||||
THEN:
|
||||
- It is downsampled, fit within 500x5000 and never enlarged
|
||||
"""
|
||||
png_path = tmp_path / "in.png"
|
||||
Image.new("RGB", in_size, (255, 255, 255)).save(png_path)
|
||||
out_path = tmp_path / "out.webp"
|
||||
|
||||
encode_thumbnail_webp(png_path, out_path, supersample=supersample)
|
||||
|
||||
with Image.open(out_path) as im:
|
||||
assert im.size == expected_size
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"error",
|
||||
[
|
||||
pytest.param(OSError("broken image"), id="os-error"),
|
||||
pytest.param(Image.DecompressionBombError("too large"), id="bomb"),
|
||||
],
|
||||
)
|
||||
def test_decode_failure_raises_parse_error(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
mocker: MockerFixture,
|
||||
error: Exception,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Opening the rendered image fails
|
||||
WHEN:
|
||||
- It is encoded as a thumbnail
|
||||
THEN:
|
||||
- A ParseError is raised
|
||||
"""
|
||||
png_path = tmp_path / "in.png"
|
||||
Image.new("RGB", (10, 10)).save(png_path)
|
||||
mocker.patch("PIL.Image.open", side_effect=error)
|
||||
|
||||
with pytest.raises(ParseError):
|
||||
encode_thumbnail_webp(png_path, tmp_path / "out.webp")
|
||||
@@ -1,520 +0,0 @@
|
||||
"""
|
||||
Tests for documents.pdf_ops.
|
||||
|
||||
These use real PDFs from the sample directories. No database, Celery or mocks.
|
||||
Pages are compared by a hash of their content stream, so page identity and order
|
||||
are easy to assert.
|
||||
"""
|
||||
|
||||
import hashlib
|
||||
from collections.abc import Callable
|
||||
from pathlib import Path
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
|
||||
from documents import pdf_ops
|
||||
from documents.pdf_ops import PageSpec
|
||||
|
||||
SRC_ROOT = Path(__file__).parents[2]
|
||||
SAMPLES = Path(__file__).parent / "samples"
|
||||
THREE_PAGES = SAMPLES / "documents" / "originals" / "0000002.pdf"
|
||||
TWELVE_PAGES = SAMPLES / "barcodes" / "split-by-asn-2.pdf"
|
||||
ENCRYPTED = SAMPLES / "password-is-test.pdf"
|
||||
SIGNED = SRC_ROOT / "paperless" / "tests" / "samples" / "tesseract" / "signed.pdf"
|
||||
|
||||
|
||||
def _page_fingerprint(page: pikepdf.Page) -> str:
|
||||
contents = page.obj.get("/Contents")
|
||||
assert contents is not None, "sample page has no /Contents"
|
||||
streams = list(contents) if isinstance(contents, pikepdf.Array) else [contents]
|
||||
return hashlib.sha256(b"".join(s.read_bytes() for s in streams)).hexdigest()
|
||||
|
||||
|
||||
def fingerprints(path: Path) -> list[str]:
|
||||
with pikepdf.open(path) as pdf:
|
||||
return [_page_fingerprint(page) for page in pdf.pages]
|
||||
|
||||
|
||||
def rotations(path: Path) -> list[int]:
|
||||
with pikepdf.open(path) as pdf:
|
||||
return [int(page.obj.get("/Rotate", 0)) for page in pdf.pages]
|
||||
|
||||
|
||||
def docinfo_keys(path: Path) -> set[str]:
|
||||
with pikepdf.open(path) as pdf:
|
||||
return set(pdf.docinfo.keys())
|
||||
|
||||
|
||||
def constant(path: Path) -> Callable[[], Path]:
|
||||
return lambda: path
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def source_fingerprints() -> list[str]:
|
||||
fps = fingerprints(THREE_PAGES)
|
||||
assert len(set(fps)) == 3, "sample must have three distinct pages"
|
||||
return fps
|
||||
|
||||
|
||||
class TestRotatePdf:
|
||||
def test_rotation_is_relative_and_applies_to_every_page(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A three page PDF
|
||||
WHEN:
|
||||
- It is rotated by 90 degrees, then the result is rotated by 90 again
|
||||
THEN:
|
||||
- Every page is rotated relative to its current rotation
|
||||
- The page content itself is unchanged
|
||||
"""
|
||||
once = tmp_path / "once.pdf"
|
||||
twice = tmp_path / "twice.pdf"
|
||||
|
||||
pdf_ops.rotate_pdf(THREE_PAGES, once, 90)
|
||||
pdf_ops.rotate_pdf(once, twice, 90)
|
||||
|
||||
assert rotations(once) == [90, 90, 90]
|
||||
assert rotations(twice) == [180, 180, 180]
|
||||
assert fingerprints(twice) == fingerprints(THREE_PAGES)
|
||||
|
||||
def test_keeps_document_info(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF with document info
|
||||
WHEN:
|
||||
- It is rotated
|
||||
THEN:
|
||||
- The document info is still present in the output
|
||||
"""
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.rotate_pdf(THREE_PAGES, dst, 90)
|
||||
|
||||
assert "/Creator" in docinfo_keys(dst)
|
||||
|
||||
|
||||
class TestRemovePages:
|
||||
@pytest.mark.parametrize(
|
||||
("pages", "kept"),
|
||||
[
|
||||
pytest.param([2], [0, 2], id="single"),
|
||||
pytest.param([3, 1], [1], id="unordered"),
|
||||
pytest.param([2, 2], [0, 2], id="duplicates-remove-once"),
|
||||
pytest.param([], [0, 1, 2], id="empty-keeps-everything"),
|
||||
pytest.param([1, 2, 3], [], id="every-page"),
|
||||
],
|
||||
)
|
||||
def test_removes_only_the_requested_pages(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
source_fingerprints: list[str],
|
||||
pages: list[int],
|
||||
kept: list[int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A three page PDF
|
||||
WHEN:
|
||||
- Pages are removed, in any order and possibly repeated
|
||||
THEN:
|
||||
- Exactly the other pages remain, in their original order
|
||||
"""
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.remove_pages(THREE_PAGES, dst, pages)
|
||||
|
||||
assert fingerprints(dst) == [source_fingerprints[i] for i in kept]
|
||||
|
||||
def test_keeps_document_info(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF with document info
|
||||
WHEN:
|
||||
- A page is removed
|
||||
THEN:
|
||||
- The document info is still present in the output
|
||||
"""
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.remove_pages(THREE_PAGES, dst, [1])
|
||||
|
||||
assert "/Creator" in docinfo_keys(dst)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"bad_page",
|
||||
[
|
||||
pytest.param(0, id="zero"),
|
||||
pytest.param(-1, id="negative"),
|
||||
],
|
||||
)
|
||||
def test_rejects_pages_below_one(self, tmp_path: Path, bad_page: int) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A three page PDF
|
||||
WHEN:
|
||||
- Pages are removed and one of them is below 1
|
||||
THEN:
|
||||
- ValueError is raised
|
||||
- No output file is written
|
||||
"""
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
with pytest.raises(ValueError, match="start at 1"):
|
||||
pdf_ops.remove_pages(THREE_PAGES, dst, [1, bad_page])
|
||||
|
||||
assert not dst.exists()
|
||||
|
||||
|
||||
class TestBuildPdfs:
|
||||
def test_selects_and_orders_pages(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
source_fingerprints: list[str],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A three page PDF
|
||||
WHEN:
|
||||
- One output is built from pages 3 then 1
|
||||
THEN:
|
||||
- The output has those pages in that order
|
||||
- Its path is returned
|
||||
"""
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
written = pdf_ops.build_pdfs(
|
||||
THREE_PAGES,
|
||||
[([PageSpec(3), PageSpec(1)], constant(dst))],
|
||||
)
|
||||
|
||||
assert written == [dst]
|
||||
assert fingerprints(dst) == [source_fingerprints[2], source_fingerprints[0]]
|
||||
|
||||
def test_rotates_only_the_requested_pages(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A three page PDF
|
||||
WHEN:
|
||||
- One output is built with a different rotation on each page
|
||||
THEN:
|
||||
- Each output page has exactly the rotation requested for it
|
||||
"""
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.build_pdfs(
|
||||
THREE_PAGES,
|
||||
[([PageSpec(1), PageSpec(2, 90), PageSpec(3, 180)], constant(dst))],
|
||||
)
|
||||
|
||||
assert rotations(dst) == [0, 90, 180]
|
||||
|
||||
def test_writes_one_file_per_output_in_order(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A twelve page PDF
|
||||
WHEN:
|
||||
- Two outputs are built from different page ranges
|
||||
THEN:
|
||||
- Two files are written and returned in output order
|
||||
- Each holds exactly its own pages
|
||||
"""
|
||||
first = tmp_path / "first.pdf"
|
||||
second = tmp_path / "second.pdf"
|
||||
source = fingerprints(TWELVE_PAGES)
|
||||
|
||||
written = pdf_ops.build_pdfs(
|
||||
TWELVE_PAGES,
|
||||
[
|
||||
([PageSpec(p) for p in (1, 2, 3)], constant(first)),
|
||||
([PageSpec(p) for p in range(4, 13)], constant(second)),
|
||||
],
|
||||
)
|
||||
|
||||
assert written == [first, second]
|
||||
assert fingerprints(first) == source[:3]
|
||||
assert fingerprints(second) == source[3:]
|
||||
|
||||
def test_empty_page_list_writes_a_zero_page_file(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A three page PDF
|
||||
WHEN:
|
||||
- An output with no pages is built
|
||||
THEN:
|
||||
- A PDF with zero pages is written
|
||||
"""
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.build_pdfs(THREE_PAGES, [([], constant(dst))])
|
||||
|
||||
assert fingerprints(dst) == []
|
||||
|
||||
def test_destination_is_not_requested_when_a_page_is_out_of_range(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A three page PDF
|
||||
WHEN:
|
||||
- Several outputs are built
|
||||
- A later output refers to a page past the end
|
||||
THEN:
|
||||
- IndexError is raised
|
||||
- No destination was requested for any output, including earlier valid ones
|
||||
"""
|
||||
requested: list[Path] = []
|
||||
|
||||
def make_dst() -> Path:
|
||||
requested.append(tmp_path / "out.pdf")
|
||||
return requested[-1]
|
||||
|
||||
with pytest.raises(IndexError):
|
||||
pdf_ops.build_pdfs(
|
||||
THREE_PAGES,
|
||||
[([PageSpec(1)], make_dst), ([PageSpec(99)], make_dst)],
|
||||
)
|
||||
|
||||
assert requested == []
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"bad_page",
|
||||
[
|
||||
pytest.param(0, id="zero"),
|
||||
pytest.param(-1, id="negative"),
|
||||
],
|
||||
)
|
||||
def test_rejects_pages_below_one_before_opening_anything(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
bad_page: int,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A three page PDF
|
||||
WHEN:
|
||||
- Several outputs are built and a later one has a page below 1
|
||||
THEN:
|
||||
- ValueError is raised
|
||||
- No destination was requested for any output
|
||||
"""
|
||||
requested: list[Path] = []
|
||||
|
||||
def make_dst() -> Path:
|
||||
requested.append(tmp_path / "out.pdf")
|
||||
return requested[-1]
|
||||
|
||||
with pytest.raises(ValueError, match="start at 1"):
|
||||
pdf_ops.build_pdfs(
|
||||
THREE_PAGES,
|
||||
[([PageSpec(1)], make_dst), ([PageSpec(bad_page)], make_dst)],
|
||||
)
|
||||
|
||||
assert requested == []
|
||||
|
||||
|
||||
class TestValidatePageOperations:
|
||||
def test_returns_the_output_count(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Operations that all target the default output
|
||||
WHEN:
|
||||
- They are validated
|
||||
THEN:
|
||||
- One output document is reported
|
||||
"""
|
||||
operations = [{"page": 1}, {"page": 2}, {"page": 3}]
|
||||
|
||||
assert pdf_ops.validate_page_operations(operations, single_output=True) == 1
|
||||
|
||||
def test_gap_in_output_indices_counts_up_to_the_highest(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Operations that target outputs 0 and 2 but never 1
|
||||
WHEN:
|
||||
- They are validated
|
||||
THEN:
|
||||
- Three output documents are reported
|
||||
"""
|
||||
operations = [
|
||||
{"page": 1, "doc": 0},
|
||||
{"page": 2, "doc": 2},
|
||||
{"page": 3, "doc": 0},
|
||||
]
|
||||
|
||||
count = pdf_ops.validate_page_operations(operations, single_output=False)
|
||||
|
||||
assert count == 3
|
||||
|
||||
def test_empty_operations_are_rejected(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- No operations
|
||||
WHEN:
|
||||
- They are validated
|
||||
THEN:
|
||||
- ValueError is raised
|
||||
"""
|
||||
with pytest.raises(ValueError, match="index is out of bounds"):
|
||||
pdf_ops.validate_page_operations([], single_output=False)
|
||||
|
||||
def test_multiple_outputs_rejected_when_single_output_required(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Operations that target two outputs
|
||||
WHEN:
|
||||
- They are validated with a single output required
|
||||
THEN:
|
||||
- ValueError is raised
|
||||
"""
|
||||
operations = [{"page": 1, "doc": 0}, {"page": 2, "doc": 1}]
|
||||
|
||||
with pytest.raises(ValueError, match="Multiple output documents"):
|
||||
pdf_ops.validate_page_operations(operations, single_output=True)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"doc",
|
||||
[
|
||||
pytest.param(-1, id="negative"),
|
||||
pytest.param(2, id="equal-to-operation-count"),
|
||||
pytest.param(2**32, id="huge"),
|
||||
],
|
||||
)
|
||||
def test_output_index_out_of_bounds(self, doc: int) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two operations, one with an output index that is out of bounds
|
||||
WHEN:
|
||||
- They are validated
|
||||
THEN:
|
||||
- ValueError is raised
|
||||
"""
|
||||
operations = [{"page": 1, "doc": 0}, {"page": 2, "doc": doc}]
|
||||
|
||||
with pytest.raises(ValueError, match="index is out of bounds"):
|
||||
pdf_ops.validate_page_operations(operations, single_output=False)
|
||||
|
||||
|
||||
class TestDecrypt:
|
||||
@pytest.mark.parametrize(
|
||||
("path", "expected"),
|
||||
[
|
||||
pytest.param(ENCRYPTED, True, id="password-required"),
|
||||
pytest.param(SIGNED, True, id="opens-without-password-but-encrypted"),
|
||||
pytest.param(THREE_PAGES, False, id="not-encrypted"),
|
||||
],
|
||||
)
|
||||
def test_needs_decrypt(self, path: Path, *, expected: bool) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF that is encrypted, or encrypted but openable, or plain
|
||||
WHEN:
|
||||
- needs_decrypt is asked about it
|
||||
THEN:
|
||||
- Only the unencrypted PDF reports False
|
||||
"""
|
||||
assert pdf_ops.needs_decrypt(path) is expected
|
||||
|
||||
def test_decrypt_writes_an_unencrypted_copy(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A password protected PDF
|
||||
WHEN:
|
||||
- It is decrypted with the correct password
|
||||
THEN:
|
||||
- The path from make_dst is returned
|
||||
- The written copy no longer needs decrypting
|
||||
"""
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
result = pdf_ops.decrypt_pdf(ENCRYPTED, constant(dst), "test")
|
||||
|
||||
assert result == dst
|
||||
assert pdf_ops.needs_decrypt(dst) is False
|
||||
|
||||
def test_wrong_password_raises_and_never_requests_a_destination(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A password protected PDF
|
||||
WHEN:
|
||||
- It is decrypted with the wrong password
|
||||
THEN:
|
||||
- PasswordError is raised
|
||||
- No destination was requested
|
||||
"""
|
||||
requested: list[Path] = []
|
||||
|
||||
def make_dst() -> Path:
|
||||
requested.append(tmp_path / "out.pdf")
|
||||
return requested[-1]
|
||||
|
||||
with pytest.raises(pikepdf.PasswordError):
|
||||
pdf_ops.decrypt_pdf(ENCRYPTED, make_dst, "wrong")
|
||||
|
||||
assert requested == []
|
||||
|
||||
|
||||
class TestPdfMerger:
|
||||
def test_pages_are_appended_in_the_order_added(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
source_fingerprints: list[str],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A reordered PDF and the original three page PDF
|
||||
WHEN:
|
||||
- Both are added to a merger in that order and saved
|
||||
THEN:
|
||||
- The output holds all pages in the order they were added
|
||||
"""
|
||||
reordered = tmp_path / "reordered.pdf"
|
||||
merged = tmp_path / "merged.pdf"
|
||||
pdf_ops.build_pdfs(
|
||||
THREE_PAGES,
|
||||
[([PageSpec(3), PageSpec(1)], constant(reordered))],
|
||||
)
|
||||
|
||||
with pdf_ops.PdfMerger() as merger:
|
||||
merger.add(reordered)
|
||||
merger.add(THREE_PAGES)
|
||||
merger.save(merged)
|
||||
|
||||
assert fingerprints(merged) == [
|
||||
source_fingerprints[2],
|
||||
source_fingerprints[0],
|
||||
*source_fingerprints,
|
||||
]
|
||||
|
||||
def test_output_version_is_at_least_the_highest_source_version(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two PDFs with different PDF versions
|
||||
WHEN:
|
||||
- Both are added to a merger and saved
|
||||
THEN:
|
||||
- The output version is at least the highest source version
|
||||
"""
|
||||
merged = tmp_path / "merged.pdf"
|
||||
with pikepdf.open(TWELVE_PAGES) as pdf:
|
||||
source_versions = [pdf.pdf_version]
|
||||
with pikepdf.open(THREE_PAGES) as pdf:
|
||||
source_versions.append(pdf.pdf_version)
|
||||
|
||||
with pdf_ops.PdfMerger() as merger:
|
||||
merger.add(TWELVE_PAGES)
|
||||
merger.add(THREE_PAGES)
|
||||
merger.save(merged)
|
||||
|
||||
with pikepdf.open(merged) as pdf:
|
||||
assert pdf.pdf_version >= max(source_versions)
|
||||
@@ -265,6 +265,52 @@ def get_page_count_for_pdf(
|
||||
return None
|
||||
|
||||
|
||||
def get_pdf_first_page_size_points(
|
||||
path: Path,
|
||||
log: logging.Logger | None = None,
|
||||
) -> tuple[float, float] | None:
|
||||
"""Return the first page's (width, height) in PDF points, post-rotation.
|
||||
|
||||
Uses the CropBox (MediaBox if absent), which must match pdftoppm's
|
||||
``-cropbox`` or the computed DPI targets the wrong box.
|
||||
|
||||
Swaps width and height for 90/270 rotation. ``page.rotation`` resolves
|
||||
inherited and negative ``/Rotate`` values, a raw lookup does not.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
path:
|
||||
Absolute path to the PDF file.
|
||||
log:
|
||||
Logger for warnings. Falls back to the module-level logger when omitted.
|
||||
|
||||
Returns
|
||||
-------
|
||||
tuple[float, float] | None
|
||||
``(width_points, height_points)``, or ``None`` if the file cannot be
|
||||
opened, has no pages, or the page box is degenerate.
|
||||
"""
|
||||
import pikepdf
|
||||
|
||||
_log = log or logger
|
||||
try:
|
||||
with pikepdf.Pdf.open(path) as pdf:
|
||||
if len(pdf.pages) == 0:
|
||||
return None
|
||||
page = pdf.pages[0]
|
||||
llx, lly, urx, ury = (float(v) for v in page.cropbox)
|
||||
width = abs(urx - llx)
|
||||
height = abs(ury - lly)
|
||||
if width <= 0 or height <= 0:
|
||||
return None
|
||||
if page.rotation in (90, 270):
|
||||
width, height = height, width
|
||||
return width, height
|
||||
except Exception as e:
|
||||
_log.warning("Could not determine PDF page size for %s: %s", path, e)
|
||||
return None
|
||||
|
||||
|
||||
def extract_pdf_metadata(
|
||||
document_path: Path,
|
||||
log: logging.Logger | None = None,
|
||||
|
||||
@@ -986,8 +986,6 @@ GNUPG_HOME = os.getenv("HOME", "/tmp")
|
||||
|
||||
# Convert is part of the ImageMagick package
|
||||
CONVERT_BINARY = os.getenv("PAPERLESS_CONVERT_BINARY", "convert")
|
||||
CONVERT_TMPDIR = os.getenv("PAPERLESS_CONVERT_TMPDIR")
|
||||
CONVERT_MEMORY_LIMIT = os.getenv("PAPERLESS_CONVERT_MEMORY_LIMIT")
|
||||
|
||||
GS_BINARY = os.getenv("PAPERLESS_GS_BINARY", "gs")
|
||||
|
||||
|
||||
@@ -15,9 +15,10 @@ from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
from ocrmypdf import SubprocessOutputError
|
||||
from PIL import Image
|
||||
|
||||
from documents.parsers import ParseError
|
||||
from documents.parsers import run_convert
|
||||
from documents.parsers import rasterize_pdf_page_to_png
|
||||
from paperless.models import ModeChoices
|
||||
from paperless.parsers import ParserProtocol
|
||||
from paperless.parsers.tesseract import RasterisedDocumentParser
|
||||
@@ -280,24 +281,71 @@ class TestGetThumbnail:
|
||||
)
|
||||
assert thumb.is_file()
|
||||
|
||||
def test_thumbnail_fallback_on_convert_error(
|
||||
@pytest.mark.parametrize(
|
||||
("filename", "expected_height"),
|
||||
[
|
||||
pytest.param("simple-digital.pdf", 647, id="portrait-letter"),
|
||||
pytest.param("rotated.pdf", 386, id="landscape"),
|
||||
],
|
||||
)
|
||||
def test_thumbnail_is_correct_format_and_size(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
tesseract_samples_dir: Path,
|
||||
filename: str,
|
||||
expected_height: int,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A portrait or landscape PDF
|
||||
WHEN:
|
||||
- A thumbnail is generated
|
||||
THEN:
|
||||
- A 500px wide WebP keeping the page's aspect ratio
|
||||
"""
|
||||
thumb = tesseract_parser.get_thumbnail(
|
||||
tesseract_samples_dir / filename,
|
||||
"application/pdf",
|
||||
)
|
||||
with Image.open(thumb) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.width == 500
|
||||
assert im.height == pytest.approx(expected_height, abs=2)
|
||||
|
||||
def test_thumbnail_fallback_on_pdftoppm_error(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
def _raise_on_pdf(input_file, output_file, **kwargs) -> None:
|
||||
if ".pdf" in str(input_file):
|
||||
"""
|
||||
GIVEN:
|
||||
- Rasterizing the original PDF fails
|
||||
WHEN:
|
||||
- A thumbnail is generated
|
||||
THEN:
|
||||
- The PDF is repaired with qpdf and a real thumbnail is rendered
|
||||
"""
|
||||
original = tesseract_samples_dir / "simple-digital.pdf"
|
||||
|
||||
def _fail_on_original(in_path: Path, out_path: Path, **kwargs) -> None:
|
||||
if in_path == original:
|
||||
raise ParseError("Does not compute.")
|
||||
run_convert(input_file=input_file, output_file=output_file, **kwargs)
|
||||
rasterize_pdf_page_to_png(in_path, out_path, **kwargs)
|
||||
|
||||
mocker.patch("documents.parsers.run_convert", side_effect=_raise_on_pdf)
|
||||
|
||||
thumb = tesseract_parser.get_thumbnail(
|
||||
tesseract_samples_dir / "simple-digital.pdf",
|
||||
"application/pdf",
|
||||
rasterize = mocker.patch(
|
||||
"documents.parsers.rasterize_pdf_page_to_png",
|
||||
side_effect=_fail_on_original,
|
||||
)
|
||||
|
||||
thumb = tesseract_parser.get_thumbnail(original, "application/pdf")
|
||||
|
||||
assert rasterize.call_count == 2
|
||||
assert thumb.is_file()
|
||||
assert thumb.name == "convert_qpdf.webp"
|
||||
with Image.open(thumb) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.width == 500
|
||||
|
||||
def test_thumbnail_encrypted_pdf(
|
||||
self,
|
||||
|
||||
@@ -6,8 +6,10 @@ import codecs
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
|
||||
from paperless.parsers.utils import get_pdf_first_page_size_points
|
||||
from paperless.parsers.utils import is_tagged_pdf
|
||||
from paperless.parsers.utils import pdf_born_digital_text
|
||||
from paperless.parsers.utils import post_process_text
|
||||
@@ -70,6 +72,149 @@ class TestIsTaggedPdf:
|
||||
assert is_tagged_pdf(bad) is False
|
||||
|
||||
|
||||
class TestGetPdfFirstPageSizePoints:
|
||||
@staticmethod
|
||||
def _write_pdf(
|
||||
path: Path,
|
||||
*,
|
||||
media_box: tuple[float, float, float, float] = (0, 0, 600, 800),
|
||||
crop_box: tuple[float, float, float, float] | None = None,
|
||||
page_rotate: int | None = None,
|
||||
inherited_rotate: int | None = None,
|
||||
) -> Path:
|
||||
pdf = pikepdf.new()
|
||||
pdf.add_blank_page(page_size=(media_box[2], media_box[3]))
|
||||
page = pdf.pages[0]
|
||||
page.obj.MediaBox = pikepdf.Array(media_box)
|
||||
if crop_box is not None:
|
||||
page.obj.CropBox = pikepdf.Array(crop_box)
|
||||
if page_rotate is not None:
|
||||
page.obj.Rotate = page_rotate
|
||||
if inherited_rotate is not None:
|
||||
pdf.Root.Pages.Rotate = inherited_rotate
|
||||
pdf.save(path)
|
||||
return path
|
||||
|
||||
def test_letter_sample(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A US Letter sample PDF with no CropBox and no rotation
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- The MediaBox size in points is returned
|
||||
"""
|
||||
assert get_pdf_first_page_size_points(SAMPLES / "simple-digital.pdf") == (
|
||||
612.0,
|
||||
792.0,
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("rotate", "expected"),
|
||||
[
|
||||
pytest.param(0, (600.0, 800.0), id="rotate-0"),
|
||||
pytest.param(90, (800.0, 600.0), id="rotate-90"),
|
||||
pytest.param(180, (600.0, 800.0), id="rotate-180"),
|
||||
pytest.param(270, (800.0, 600.0), id="rotate-270"),
|
||||
pytest.param(-90, (800.0, 600.0), id="rotate-negative-90"),
|
||||
],
|
||||
)
|
||||
def test_page_rotation_swaps_dimensions(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
rotate: int,
|
||||
expected: tuple[float, float],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A portrait PDF page with /Rotate set directly on the page
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- Width and height are swapped for quarter-turn rotations only
|
||||
"""
|
||||
pdf_path = self._write_pdf(tmp_path / "rotated.pdf", page_rotate=rotate)
|
||||
assert get_pdf_first_page_size_points(pdf_path) == expected
|
||||
|
||||
def test_inherited_rotation_swaps_dimensions(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A page inheriting /Rotate 90 from the /Pages node
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- Width and height are swapped
|
||||
"""
|
||||
pdf_path = self._write_pdf(tmp_path / "inherited.pdf", inherited_rotate=90)
|
||||
assert get_pdf_first_page_size_points(pdf_path) == (800.0, 600.0)
|
||||
|
||||
def test_crop_box_preferred_over_media_box(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF page with a CropBox smaller than its MediaBox
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- The CropBox size is returned
|
||||
"""
|
||||
pdf_path = self._write_pdf(
|
||||
tmp_path / "cropped.pdf",
|
||||
crop_box=(50, 100, 350, 500),
|
||||
)
|
||||
assert get_pdf_first_page_size_points(pdf_path) == (300.0, 400.0)
|
||||
|
||||
def test_degenerate_box_returns_none(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF page whose box has zero width
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- None is returned
|
||||
"""
|
||||
pdf_path = self._write_pdf(
|
||||
tmp_path / "degenerate.pdf",
|
||||
media_box=(0, 0, 600, 800),
|
||||
crop_box=(100, 0, 100, 800),
|
||||
)
|
||||
assert get_pdf_first_page_size_points(pdf_path) is None
|
||||
|
||||
def test_nonexistent_path_returns_none(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A path that does not exist
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- None is returned and nothing is raised
|
||||
"""
|
||||
assert get_pdf_first_page_size_points(Path("/nonexistent/file.pdf")) is None
|
||||
|
||||
def test_corrupt_pdf_returns_none(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A file that is not a PDF
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- None is returned and nothing is raised
|
||||
"""
|
||||
bad = tmp_path / "bad.pdf"
|
||||
bad.write_bytes(b"not a pdf")
|
||||
assert get_pdf_first_page_size_points(bad) is None
|
||||
|
||||
def test_encrypted_pdf_returns_none(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A password protected PDF
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- None is returned and nothing is raised
|
||||
"""
|
||||
assert get_pdf_first_page_size_points(SAMPLES / "encrypted.pdf") is None
|
||||
|
||||
|
||||
class TestPostProcessText:
|
||||
@pytest.mark.parametrize(
|
||||
("source", "expected"),
|
||||
|
||||
Reference in new issue
Block a user