Compare commits

..
Author SHA1 Message Date
Trenton HandClaude Sonnet 5.5 6d61214bba Fix: satisfy pyrefly in test_pdf_ops and clarify pdf_ops docstrings
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-10-03 15:03:31 -07:00
Trenton HandClaude Sonnet 5.5 dcfe389909 Chore: drop stale type-check baseline entries for bulk_edit
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-10-03 14:59:01 -07:00
Trenton HandClaude Sonnet 5.5 0beb0a1d0b Fix: reject delete_pages page numbers below 1
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-10-03 14:58:36 -07:00
Trenton HandClaude Sonnet 5.5 bfaea71a83 Refactor: use pdf_ops for PDF page work in bulk_edit
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-10-03 14:54:15 -07:00
Trenton HandClaude Sonnet 5.5 0b7cecb6bb Refactor: add pure pdf_ops module with real-PDF tests
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-10-03 14:47:53 -07:00
49 changed files with 1117 additions and 575 deletions

No files matched your search

-1
View File
@@ -38,7 +38,6 @@ src/documents/bulk_edit.py:0: error: Incompatible types in assignment (expressio
src/documents/bulk_edit.py:0: error: Invalid index type "str" for "dict[FieldDataType, str]"; expected type "FieldDataType" [index]
src/documents/bulk_edit.py:0: error: List comprehension has incompatible type List[tuple[int, Any]]; expected List[int] [misc]
src/documents/bulk_edit.py:0: error: List comprehension has incompatible type List[tuple[int, None]]; expected List[int] [misc]
src/documents/bulk_edit.py:0: error: Missing named argument "p" for "remove" of "PageList" [call-arg]
src/documents/bulk_edit.py:0: error: Missing type arguments for generic type "dict" [type-arg]
src/documents/bulk_edit.py:0: error: Missing type arguments for generic type "dict" [type-arg]
src/documents/bulk_edit.py:0: error: Need type annotation for "to_create" (hint: "to_create: list[<type>] = ...") [var-annotated]
-7
View File
@@ -91,13 +91,6 @@
"concise_description": "Argument `list[int]` is not assignable to parameter `args` with type `tuple[Any, ...] | None` in function `celery.app.task.Task.apply_async`",
"severity": "error"
},
{
"column": 33,
"path": "src/documents/bulk_edit.py",
"name": "missing-argument",
"concise_description": "Missing argument `p` in function `pikepdf._core.PageList.remove`",
"severity": "error"
},
{
"column": 25,
"path": "src/documents/caching.py",
+2 -2
View File
@@ -256,8 +256,8 @@ isort.force-single-line = true
[tool.codespell]
ignore-words-list = "criterias,afterall,valeu,ureue,equest,ure,assertIn,Oktober,commitish,NIN,nin,reprot"
skip = """\
src-ui/src/locale/*,src-ui/pnpm-lock.yaml,src-ui/e2e/*,src/paperless/tests/samples/mail/*,src/documents/tests/samples\
/*,src/paperless_testing/sample_files/*,*.po,*.json\
src-ui/src/locale/*,src-ui/pnpm-lock.yaml,src-ui/e2e/*,src/paperless_mail/tests/samples/*,src/paperless/tests/samples\
/mail/*,src/documents/tests/samples/*,*.po,*.json\
"""
write-changes = true
-38
View File
@@ -11,8 +11,6 @@ from typing import TYPE_CHECKING
import pytest
from paperless_testing import samples
if TYPE_CHECKING:
from collections.abc import Generator
from pathlib import Path
@@ -151,39 +149,3 @@ def fake_progress_manager(
monkeypatch.setattr("documents.tasks.ProgressManager", FakeProgressManager)
return FakeProgressManager
@pytest.fixture(scope="session")
def shared_samples_dir() -> Path:
"""Directory of sample files used by more than one app's tests."""
return samples.SHARED_SAMPLES_DIR
@pytest.fixture(scope="session")
def simple_digital_pdf_file() -> Path:
"""One-page PDF with a text layer."""
return samples.SIMPLE_DIGITAL_PDF
@pytest.fixture(scope="session")
def multi_page_digital_pdf_file() -> Path:
"""Three-page PDF with a text layer."""
return samples.MULTI_PAGE_DIGITAL_PDF
@pytest.fixture(scope="session")
def with_form_pdf_file() -> Path:
"""PDF containing a fillable form."""
return samples.WITH_FORM_PDF
@pytest.fixture(scope="session")
def multi_page_images_pdf_file() -> Path:
"""Multi-page PDF of scanned images, no text layer."""
return samples.MULTI_PAGE_IMAGES_PDF
@pytest.fixture(scope="session")
def thumbnail_webp_file() -> Path:
"""Small WebP thumbnail."""
return samples.THUMBNAIL_WEBP
+185 -219
View File
@@ -3,6 +3,7 @@ from __future__ import annotations
import logging
import tempfile
import uuid
from functools import partial
from pathlib import Path
from typing import TYPE_CHECKING
from typing import Literal
@@ -17,6 +18,7 @@ from django.db.models import Max
from django.db.models import Q
from django.utils import timezone
from documents import pdf_ops
from documents.data_models import ConsumableDocument
from documents.data_models import DocumentMetadataOverrides
from documents.data_models import DocumentSource
@@ -116,6 +118,11 @@ def _resolve_root_and_source_doc(
)
def _scratch_path(name: str) -> Path:
"""A path inside a fresh directory under SCRATCH_DIR."""
return Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR)) / name
def set_correspondent(
doc_ids: list[int],
correspondent: Correspondent,
@@ -474,8 +481,6 @@ def rotate(
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
docs_by_root_id.setdefault(pair.root_doc.id, pair)
import pikepdf
for pair in docs_by_root_id.values():
if pair.source_doc.mime_type != "application/pdf":
logger.warning(
@@ -488,11 +493,7 @@ def rotate(
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_rotated.pdf"
)
with pikepdf.open(pair.source_doc.source_path) as pdf:
for page in pdf.pages:
page.rotate(degrees, relative=True)
pdf.remove_unreferenced_resources()
pdf.save(filepath)
pdf_ops.rotate_pdf(pair.source_doc.source_path, filepath, degrees)
# Preserve metadata/permissions via overrides; mark as new version
overrides = DocumentMetadataOverrides().from_document(pair.root_doc)
@@ -535,48 +536,45 @@ def merge(
qs = Document.objects.select_related("root_document").filter(id__in=doc_ids)
docs_by_id = {doc.id: doc for doc in qs}
affected_docs: list[int] = []
import pikepdf
merged_pdf = pikepdf.new()
version: str = merged_pdf.pdf_version
handoff_asn: int | None = None
# use doc_ids to preserve order
for doc_id in doc_ids:
doc = docs_by_id.get(doc_id)
if doc is None:
continue
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
try:
doc_path = (
pair.source_doc.archive_path
if archive_fallback
and pair.source_doc.mime_type != "application/pdf"
and pair.source_doc.has_archive_version
else pair.source_doc.source_path
)
with pikepdf.open(str(doc_path)) as pdf:
version = max(version, pdf.pdf_version)
merged_pdf.pages.extend(pdf.pages)
affected_docs.append(doc.id)
if handoff_asn is None and doc.archive_serial_number is not None:
handoff_asn = doc.archive_serial_number
except Exception as e:
logger.exception(
f"Error merging document {doc.id}, it will not be included in the merge: {e}",
)
if len(affected_docs) == 0:
logger.warning("No documents were merged")
return "OK"
with pdf_ops.PdfMerger() as merger:
# use doc_ids to preserve order
for doc_id in doc_ids:
doc = docs_by_id.get(doc_id)
if doc is None:
continue
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
try:
# archive_path is None when there is no archive version
archive_path = (
pair.source_doc.archive_path
if archive_fallback
and pair.source_doc.mime_type != "application/pdf"
else None
)
merger.add(
archive_path
if archive_path is not None
else pair.source_doc.source_path,
)
affected_docs.append(doc.id)
if handoff_asn is None and doc.archive_serial_number is not None:
handoff_asn = doc.archive_serial_number
except Exception as e:
logger.exception(
f"Error merging document {doc.id}, it will not be included in the merge: {e}",
)
if len(affected_docs) == 0:
logger.warning("No documents were merged")
return "OK"
filepath = (
Path(
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
filepath = (
Path(
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
)
/ f"{'_'.join([str(doc_id) for doc_id in affected_docs])[:100]}_merged.pdf"
)
/ f"{'_'.join([str(doc_id) for doc_id in affected_docs])[:100]}_merged.pdf"
)
merged_pdf.remove_unreferenced_resources()
merged_pdf.save(filepath, min_version=version)
merged_pdf.close()
merger.save(filepath)
if metadata_document_id:
metadata_document = qs.get(id=metadata_document_id)
@@ -752,64 +750,60 @@ def split(
)
doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
import pikepdf
consume_tasks = []
try:
with pikepdf.open(pair.source_doc.source_path) as pdf:
for idx, split_doc in enumerate(pages):
dst: pikepdf.Pdf = pikepdf.new()
for page in split_doc:
dst.pages.append(pdf.pages[page - 1])
filepath: Path = (
Path(
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
)
/ f"{doc.id}_{split_doc[0]}-{split_doc[-1]}.pdf"
)
dst.remove_unreferenced_resources()
dst.save(filepath)
dst.close()
outputs = [
(
[pdf_ops.PageSpec(page) for page in split_doc],
partial(_scratch_path, f"{doc.id}_{split_doc[0]}-{split_doc[-1]}.pdf"),
)
for split_doc in pages
]
filepaths = pdf_ops.build_pdfs(pair.source_doc.source_path, outputs)
overrides: DocumentMetadataOverrides = (
DocumentMetadataOverrides().from_document(doc)
)
overrides.title = f"{doc.title} (split {idx + 1})"
if user is not None:
overrides.owner_id = user.id
if not delete_originals:
overrides.skip_asn_if_exists = True
logger.info(
f"Adding split document with pages {split_doc} to the task queue.",
)
consume_tasks.append(
consume_file.s(
input_doc=ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
),
overrides=overrides,
).set(headers={"trigger_source": trigger_source}),
)
for idx, (split_doc, filepath) in enumerate(
zip(pages, filepaths, strict=True),
):
overrides: DocumentMetadataOverrides = (
DocumentMetadataOverrides().from_document(doc)
)
overrides.title = f"{doc.title} (split {idx + 1})"
if user is not None:
overrides.owner_id = user.id
if not delete_originals:
overrides.skip_asn_if_exists = True
logger.info(
f"Adding split document with pages {split_doc} to the task queue.",
)
consume_tasks.append(
consume_file.s(
input_doc=ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
),
overrides=overrides,
).set(headers={"trigger_source": trigger_source}),
)
if delete_originals:
backup = release_archive_serial_numbers([doc.id])
logger.info(
"Queueing removal of original document after consumption of the split documents",
)
try:
chord(
header=consume_tasks,
body=delete.si([doc.id]),
).on_error(
restore_archive_serial_numbers_task.s(backup),
).apply_async()
except Exception:
restore_archive_serial_numbers(backup)
raise
else:
group(consume_tasks).delay()
if delete_originals:
backup = release_archive_serial_numbers([doc.id])
logger.info(
"Queueing removal of original document after consumption of the split documents",
)
try:
chord(
header=consume_tasks,
body=delete.si([doc.id]),
).on_error(
restore_archive_serial_numbers_task.s(backup),
).apply_async()
except Exception:
restore_archive_serial_numbers(backup)
raise
else:
group(consume_tasks).delay()
except Exception as e:
logger.exception(f"Error splitting document {doc.id}: {e}")
@@ -830,8 +824,7 @@ def delete_pages(
)
doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
pages = sorted(pages) # sort pages to avoid index issues
import pikepdf
pages = sorted(set(pages))
try:
# Produce edited PDF to a temp file and create a new version
@@ -839,13 +832,7 @@ def delete_pages(
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_pages_deleted.pdf"
)
with pikepdf.open(pair.source_doc.source_path) as pdf:
offset = 1 # pages are 1-indexed
for page_num in pages:
pdf.pages.remove(pdf.pages[page_num - offset])
offset += 1 # remove() changes the index of the pages
pdf.remove_unreferenced_resources()
pdf.save(filepath)
pdf_ops.remove_pages(pair.source_doc.source_path, filepath, pages)
overrides = DocumentMetadataOverrides().from_document(pair.root_doc)
if user is not None:
@@ -894,47 +881,28 @@ def edit_pdf(
)
doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
import pikepdf
pdf_docs: list[pikepdf.Pdf] = []
try:
if not operations:
raise ValueError("Output document index is out of bounds")
max_idx = max(op.get("doc", 0) for op in operations)
if update_document and max_idx > 0:
logger.error(
"Update requested but multiple output documents specified",
output_count = pdf_ops.validate_page_operations(
operations,
single_output=update_document,
)
page_specs: list[list[pdf_ops.PageSpec]] = [[] for _ in range(output_count)]
for op in operations:
page_specs[op.get("doc", 0)].append(
pdf_ops.PageSpec(op["page"], op.get("rotate", 0)),
)
raise ValueError("Multiple output documents specified")
if any(
op.get("doc", 0) < 0 or op.get("doc", 0) >= len(operations)
for op in operations
):
raise ValueError("Output document index is out of bounds")
with pikepdf.open(pair.source_doc.source_path) as src:
# prepare output documents
pdf_docs = [pikepdf.new() for _ in range(max_idx + 1)]
for op in operations:
dst = pdf_docs[op.get("doc", 0)]
page = src.pages[op["page"] - 1]
dst.pages.append(page)
if op.get("rotate"):
dst.pages[-1].rotate(op["rotate"], relative=True)
if update_document:
# Create a new version from the edited PDF rather than replacing in-place
pdf = pdf_docs[0]
pdf.remove_unreferenced_resources()
filepath: Path = (
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_edited.pdf"
(filepath,) = pdf_ops.build_pdfs(
pair.source_doc.source_path,
[
(
page_specs[0],
partial(_scratch_path, f"{pair.root_doc.id}_edited.pdf"),
),
],
)
pdf.save(filepath)
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata
@@ -955,6 +923,19 @@ def edit_pdf(
headers={"trigger_source": trigger_source},
)
else:
version_filepaths = pdf_ops.build_pdfs(
pair.source_doc.source_path,
[
(
specs,
partial(
_scratch_path,
f"{pair.root_doc.id}_edit_{idx}.pdf",
),
)
for idx, specs in enumerate(page_specs, start=1)
],
)
consume_tasks = []
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
@@ -966,15 +947,9 @@ def edit_pdf(
overrides.actor_id = user.id
if not delete_original:
overrides.skip_asn_if_exists = True
if delete_original and len(pdf_docs) == 1:
if delete_original and output_count == 1:
overrides.asn = pair.root_doc.archive_serial_number
for idx, pdf in enumerate(pdf_docs, start=1):
version_filepath: Path = (
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_edit_{idx}.pdf"
)
pdf.remove_unreferenced_resources()
pdf.save(version_filepath)
for version_filepath in version_filepaths:
consume_tasks.append(
consume_file.s(
input_doc=ConsumableDocument(
@@ -1024,8 +999,6 @@ def remove_password(
"""
Remove password protection from PDF documents.
"""
import pikepdf
for doc_id in doc_ids:
doc = Document.objects.select_related("root_document").get(id=doc_id)
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
@@ -1039,76 +1012,69 @@ def remove_password(
doc.id,
pair.source_doc.source_path,
)
try:
with pikepdf.open(source_path) as pdf:
if not pdf.is_encrypted:
logger.info(
"Skipping password removal for document %s because the "
"source PDF is not encrypted",
pair.root_doc.id,
)
continue
except pikepdf.PasswordError:
# Password-protected PDFs need the supplied password below.
pass
with pikepdf.open(source_path, password=password) as pdf:
filepath: Path = (
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_unprotected.pdf"
if not pdf_ops.needs_decrypt(source_path):
logger.info(
"Skipping password removal for document %s because the "
"source PDF is not encrypted",
pair.root_doc.id,
)
pdf.remove_unreferenced_resources()
pdf.save(filepath)
continue
if update_document:
# Create a new version rather than modifying the root/original in place.
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata
else DocumentMetadataOverrides()
)
if user is not None:
overrides.owner_id = user.id
overrides.actor_id = user.id
consume_file.apply_async(
kwargs={
"input_doc": ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
root_document_id=pair.root_doc.id,
),
"overrides": overrides,
},
headers={"trigger_source": trigger_source},
)
filepath = pdf_ops.decrypt_pdf(
source_path,
partial(_scratch_path, f"{pair.root_doc.id}_unprotected.pdf"),
password,
)
if update_document:
# Create a new version rather than modifying the root/original in place.
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata
else DocumentMetadataOverrides()
)
if user is not None:
overrides.owner_id = user.id
overrides.actor_id = user.id
consume_file.apply_async(
kwargs={
"input_doc": ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
root_document_id=pair.root_doc.id,
),
"overrides": overrides,
},
headers={"trigger_source": trigger_source},
)
else:
consume_tasks = []
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata
else DocumentMetadataOverrides()
)
if user is not None:
overrides.owner_id = user.id
overrides.actor_id = user.id
consume_tasks.append(
consume_file.s(
input_doc=ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
),
overrides=overrides,
).set(headers={"trigger_source": trigger_source}),
)
if delete_original:
chord(
header=consume_tasks,
body=delete.si([doc.id]),
).delay()
else:
consume_tasks = []
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata
else DocumentMetadataOverrides()
)
if user is not None:
overrides.owner_id = user.id
overrides.actor_id = user.id
consume_tasks.append(
consume_file.s(
input_doc=ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
),
overrides=overrides,
).set(headers={"trigger_source": trigger_source}),
)
if delete_original:
chord(
header=consume_tasks,
body=delete.si([doc.id]),
).delay()
else:
group(consume_tasks).delay()
group(consume_tasks).delay()
except Exception as e:
logger.exception(
+184
View File
@@ -0,0 +1,184 @@
"""
Pure PDF page operations used by documents.bulk_edit.
This module deliberately knows nothing about Django, Celery or the documents
app: callers resolve documents, choose output paths and queue work. Every
function that writes a PDF removes unreferenced resources before saving.
pikepdf is always called as ``pikepdf.open(...)`` / ``pikepdf.new()`` (never
``from pikepdf import open``) so tests can patch those module attributes.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
from typing import NamedTuple
import pikepdf
if TYPE_CHECKING:
from collections.abc import Callable
from collections.abc import Iterable
from collections.abc import Mapping
from collections.abc import Sequence
from pathlib import Path
from types import TracebackType
class PageSpec(NamedTuple):
"""One page of an output PDF: a 1-indexed source page, optionally rotated."""
page: int
rotate: int = 0 # relative degrees, 0 leaves the page alone
def _require_positive(pages: Iterable[int]) -> None:
for page in pages:
if page < 1:
raise ValueError(f"Page numbers start at 1, got {page}")
def rotate_pdf(src: Path, dst: Path, degrees: int) -> None:
"""
Rotate every page relatively on the opened document, not a rebuild, so Info,
XMP and outlines are kept. ``src`` is not modified.
"""
with pikepdf.open(src) as pdf:
for page in pdf.pages:
page.rotate(degrees, relative=True)
pdf.remove_unreferenced_resources()
pdf.save(dst)
def remove_pages(src: Path, dst: Path, pages: Iterable[int]) -> None:
"""
Remove 1-indexed pages from the opened document, not a rebuild, so Info, XMP
and outlines are kept. ``src`` is not modified.
Duplicates are ignored. Pages are removed highest first so earlier removals
never shift the index of later ones.
"""
unique = sorted(set(pages))
_require_positive(unique)
with pikepdf.open(src) as pdf:
for page_num in reversed(unique):
del pdf.pages[page_num - 1]
pdf.remove_unreferenced_resources()
pdf.save(dst)
def build_pdfs(
src: Path,
outputs: Sequence[tuple[Sequence[PageSpec], Callable[[], Path]]],
) -> list[Path]:
"""
Build one new PDF per output from pages of ``src``, opening ``src`` once.
Each output is ``(page_specs, make_dst)``. ``make_dst`` is called after that
output's pages are copied and immediately before it is saved, so a bad page
number never leaves a destination behind. Document-level data (Info, XMP,
outlines) is not carried over. Returns the written paths in output order.
"""
for specs, _ in outputs:
_require_positive(spec.page for spec in specs)
written: list[Path] = []
with pikepdf.open(src) as source:
for specs, make_dst in outputs:
dst = pikepdf.new()
for spec in specs:
dst.pages.append(source.pages[spec.page - 1])
if spec.rotate:
dst.pages[-1].rotate(spec.rotate, relative=True)
dst.remove_unreferenced_resources()
path = make_dst()
dst.save(path)
dst.close()
written.append(path)
return written
def validate_page_operations(
operations: Sequence[Mapping[str, int]],
*,
single_output: bool,
) -> int:
"""
Validate ``edit_pdf`` style operations and return the output document count.
Each operation has ``page`` and optionally ``rotate`` and ``doc`` (the output
document index, default 0). The bounds rule is kept as it was: a ``doc`` index
must be below the number of operations.
"""
if not operations:
raise ValueError("Output document index is out of bounds")
max_idx = max(op.get("doc", 0) for op in operations)
if single_output and max_idx > 0:
raise ValueError("Multiple output documents specified")
if any(
op.get("doc", 0) < 0 or op.get("doc", 0) >= len(operations) for op in operations
):
raise ValueError("Output document index is out of bounds")
return max_idx + 1
def needs_decrypt(src: Path) -> bool:
"""
True if ``src`` is encrypted. A PDF that needs a password to open at all
counts as encrypted.
"""
try:
with pikepdf.open(src) as pdf:
return bool(pdf.is_encrypted)
except pikepdf.PasswordError:
return True
def decrypt_pdf(src: Path, make_dst: Callable[[], Path], password: str) -> Path:
"""
Write an unencrypted copy of ``src`` and return its path.
``make_dst`` is only called once the password has been accepted, so a wrong
password never leaves a destination behind.
"""
with pikepdf.open(src, password=password) as pdf:
pdf.remove_unreferenced_resources()
dst = make_dst()
pdf.save(dst)
return dst
class PdfMerger:
"""
Accumulates the pages of several PDFs into one new PDF.
``add`` raises if a source cannot be read; deciding whether to skip it is the
caller's policy. Use as a context manager so the merged PDF is closed.
"""
def __init__(self) -> None:
self._pdf = pikepdf.new()
self._version: str = self._pdf.pdf_version
def __enter__(self) -> PdfMerger:
return self
def __exit__(
self,
exc_type: type[BaseException] | None,
exc: BaseException | None,
tb: TracebackType | None,
) -> None:
self._pdf.close()
def add(self, path: Path) -> None:
with pikepdf.open(str(path)) as pdf:
self._version = max(self._version, pdf.pdf_version)
self._pdf.pages.extend(pdf.pages)
def save(self, dst: Path) -> None:
self._pdf.remove_unreferenced_resources()
self._pdf.save(dst, min_version=self._version)
+2
View File
@@ -2137,6 +2137,8 @@ class BulkEditSerializer(
raise serializers.ValidationError("pages must be a list")
if not all(isinstance(i, int) for i in parameters["pages"]):
raise serializers.ValidationError("pages must be a list of integers")
if any(i < 1 for i in parameters["pages"]):
raise serializers.ValidationError("pages must be positive integers")
def _validate_parameters_merge(self, parameters) -> None:
if "delete_originals" in parameters:
+10 -6
View File
@@ -12,25 +12,29 @@ if TYPE_CHECKING:
from paperless_testing.dirs import PaperlessDirs
@pytest.fixture(scope="session")
def document_samples_dir() -> Path:
"""Path to the shared test sample documents."""
return Path(__file__).parent / "samples" / "documents"
@pytest.fixture()
def sample_doc(
paperless_dirs: "PaperlessDirs",
simple_digital_pdf_file: Path,
multi_page_images_pdf_file: Path,
thumbnail_webp_file: Path,
document_samples_dir: Path,
) -> "Document":
"""Create a document with valid files and matching checksums."""
with filelock.FileLock(paperless_dirs.media_lock):
shutil.copy(
simple_digital_pdf_file,
document_samples_dir / "originals" / "0000001.pdf",
paperless_dirs.originals_dir / "0000001.pdf",
)
shutil.copy(
multi_page_images_pdf_file,
document_samples_dir / "archive" / "0000001.pdf",
paperless_dirs.archive_dir / "0000001.pdf",
)
shutil.copy(
thumbnail_webp_file,
document_samples_dir / "thumbnails" / "0000001.webp",
paperless_dirs.thumbnail_dir / "0000001.webp",
)
Binary file not shown.

After

Width:  |  Height:  |  Size: 32 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 2.6 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 2.6 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 2.6 KiB

Binary file not shown.
+1
View File
@@ -0,0 +1 @@
This is a test file.
@@ -53,7 +53,7 @@ class TestBulkDownload(DirectoriesMixin, SampleDirMixin, APITestCase):
archive_checksum="D",
)
shutil.copy(self.SIMPLE_PDF, self.doc2.source_path)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", self.doc2.source_path)
shutil.copy(self.SAMPLE_DIR / "simple.png", self.doc2b.source_path)
shutil.copy(self.SAMPLE_DIR / "simple.jpg", self.doc3.source_path)
shutil.copy(self.SAMPLE_DIR / "test_with_bom.pdf", self.doc3.archive_path)
+30
View File
@@ -1843,6 +1843,36 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
m.assert_called_once()
self.assertEqual(m.call_args.kwargs["pages"], [[1], [2, 3, 4], [5]])
@mock.patch("documents.serialisers.bulk_edit.delete_pages")
def test_bulk_edit_delete_pages_rejects_pages_below_one(self, m) -> None:
"""
GIVEN:
- A legacy delete_pages bulk edit
WHEN:
- API to bulk edit is called with a page number below 1
THEN:
- API returns HTTP 400
- delete_pages is not called
"""
self.setup_mock(m, "delete_pages")
for pages in ([0], [-1], [1, 0]):
with self.subTest(pages=pages):
response = self.client.post(
"/api/documents/bulk_edit/",
json.dumps(
{
"documents": [self.doc2.id],
"method": "delete_pages",
"parameters": {"pages": pages},
},
),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn(b"pages must be positive integers", response.content)
m.assert_not_called()
@mock.patch("documents.views.bulk_edit.rotate")
def test_rotate_insufficient_permissions(self, m) -> None:
self.doc1.owner = User.objects.get(username="temp_admin")
+33 -49
View File
@@ -56,8 +56,6 @@ from paperless_testing.http import read_streaming_response
from paperless_testing.permissions import grant_all_global
from paperless_testing.permissions import grant_global
from paperless_testing.permissions import grant_object
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
from paperless_testing.samples import THUMBNAIL_WEBP
class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
@@ -1854,11 +1852,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with SimpleUploadedFile(
"simple.pdf",
SIMPLE_DIGITAL_PDF.read_bytes(),
content_type="application/pdf",
) as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f},
@@ -1886,7 +1880,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
payload = SimpleUploadedFile(
"../../outside.pdf",
SIMPLE_DIGITAL_PDF.read_bytes(),
(Path(__file__).parent / "samples" / "simple.pdf").read_bytes(),
content_type="application/pdf",
)
@@ -1915,7 +1909,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
pdf_bytes = SIMPLE_DIGITAL_PDF.read_bytes()
pdf_bytes = (Path(__file__).parent / "samples" / "simple.pdf").read_bytes()
boundary = "paperless-boundary"
payload = (
(
@@ -1991,7 +1985,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
def test_upload_insufficient_permissions(self) -> None:
self.client.force_authenticate(user=UserFactory(username="testuser2"))
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f},
@@ -2004,11 +1998,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with SimpleUploadedFile(
"simple.pdf",
SIMPLE_DIGITAL_PDF.read_bytes(),
content_type="application/pdf",
) as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{
@@ -2041,7 +2031,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"documenst": f},
@@ -2067,7 +2057,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "title": "my custom title"},
@@ -2087,7 +2077,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
)
c = Correspondent.objects.create(name="test-corres")
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "correspondent": c.id},
@@ -2106,7 +2096,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "correspondent": 3456},
@@ -2121,7 +2111,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
)
dt = DocumentType.objects.create(name="invoice")
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "document_type": dt.id},
@@ -2140,7 +2130,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "document_type": 34578},
@@ -2155,7 +2145,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
)
sp = StoragePath.objects.create(name="invoices")
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "storage_path": sp.id},
@@ -2174,7 +2164,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "storage_path": 34578},
@@ -2190,7 +2180,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
t1 = Tag.objects.create(name="tag1")
t2 = Tag.objects.create(name="tag2")
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "tags": [t2.id, t1.id]},
@@ -2211,7 +2201,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
t1 = Tag.objects.create(name="tag1")
t2 = Tag.objects.create(name="tag2")
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "tags": [t2.id, t1.id, 734563]},
@@ -2235,7 +2225,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
0,
tzinfo=zoneinfo.ZoneInfo("America/Los_Angeles"),
)
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "created": created},
@@ -2251,11 +2241,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with SimpleUploadedFile(
"simple.pdf",
SIMPLE_DIGITAL_PDF.read_bytes(),
content_type="application/pdf",
) as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "archive_serial_number": 500},
@@ -2282,11 +2268,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
data_type=CustomField.FieldDataType.STRING,
)
with SimpleUploadedFile(
"simple.pdf",
SIMPLE_DIGITAL_PDF.read_bytes(),
content_type="application/pdf",
) as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{
@@ -2341,7 +2323,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
w1.actions.add(action1)
w1.save()
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{
@@ -2382,11 +2364,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
data_type=CustomField.FieldDataType.INT,
)
with SimpleUploadedFile(
"simple.pdf",
SIMPLE_DIGITAL_PDF.read_bytes(),
content_type="application/pdf",
) as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{
@@ -2435,7 +2413,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
]
for payload in error_payloads:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
data = {"document": f, **payload}
response = self.client.post(
"/api/documents/post_document/",
@@ -2493,7 +2471,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with SIMPLE_DIGITAL_PDF.open("rb") as f:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "from_webui": True},
@@ -2532,8 +2510,14 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
archive_filename="archive.pdf",
)
source_file: Path = THUMBNAIL_WEBP
archive_file: Path = SIMPLE_DIGITAL_PDF
source_file: Path = (
Path(__file__).parent
/ "samples"
/ "documents"
/ "thumbnails"
/ "0000001.webp"
)
archive_file: Path = Path(__file__).parent / "samples" / "simple.pdf"
shutil.copy(source_file, doc.source_path)
shutil.copy(archive_file, doc.archive_path)
@@ -2566,7 +2550,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
mime_type="application/pdf",
)
shutil.copy(SIMPLE_DIGITAL_PDF, doc.source_path)
shutil.copy(Path(__file__).parent / "samples" / "simple.pdf", doc.source_path)
response = self.client.get(f"/api/documents/{doc.pk}/metadata/")
self.assertEqual(response.status_code, status.HTTP_200_OK)
@@ -4304,8 +4288,8 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
filename="test2.pdf",
)
archive_file = SIMPLE_DIGITAL_PDF
source_file = SIMPLE_DIGITAL_PDF
archive_file = Path(__file__).parent / "samples" / "simple.pdf"
source_file = Path(__file__).parent / "samples" / "simple.pdf"
shutil.copy(archive_file, doc.archive_path)
shutil.copy(source_file, doc2.source_path)
+5 -8
View File
@@ -12,9 +12,6 @@ from documents.tests.utils import SampleDirMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.factories import UserFactory
from paperless_testing.permissions import grant_global
from paperless_testing.samples import MULTI_PAGE_DIGITAL_PDF
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
from paperless_testing.samples import WITH_FORM_PDF
class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
@@ -45,15 +42,15 @@ class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
# Copy sample files to document paths (using different files to distinguish versions)
shutil.copy(
SIMPLE_DIGITAL_PDF,
self.SAMPLE_DIR / "documents" / "originals" / "0000001.pdf",
self.doc1.archive_path,
)
shutil.copy(
MULTI_PAGE_DIGITAL_PDF,
self.SAMPLE_DIR / "documents" / "originals" / "0000002.pdf",
self.doc1.source_path,
)
shutil.copy(
WITH_FORM_PDF,
self.SAMPLE_DIR / "documents" / "originals" / "0000003.pdf",
self.doc2.source_path,
)
@@ -380,7 +377,7 @@ class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
checksum="3",
filename="test3.pdf",
)
shutil.copy(self.SIMPLE_PDF, doc3.source_path)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", doc3.source_path)
doc4 = Document.objects.create(
title="test1",
@@ -389,7 +386,7 @@ class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
checksum="4",
filename="test4.pdf",
)
shutil.copy(self.SIMPLE_PDF, doc4.source_path)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", doc4.source_path)
response = self.client.post(
self.ENDPOINT,
+2 -2
View File
@@ -150,7 +150,7 @@ class TestBarcode(
- No barcodes detected
- No pages to split on
"""
test_file = self.SIMPLE_PDF
test_file = self.SAMPLE_DIR / "simple.pdf"
with self.get_reader(test_file) as reader:
reader.detect()
separator_page_numbers = reader.get_separation_pages()
@@ -442,7 +442,7 @@ class TestBarcode(
THEN:
- Nothing happens
"""
test_file = self.SIMPLE_PDF
test_file = self.SAMPLE_DIR / "simple.pdf"
with self.get_reader(test_file) as reader:
try:
+30 -9
View File
@@ -24,9 +24,6 @@ from documents.models import Tag
from documents.permissions import set_permissions_for_objects
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.permissions import grant_object
from paperless_testing.samples import MULTI_PAGE_DIGITAL_PDF
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
from paperless_testing.samples import WITH_FORM_PDF
class TestBulkEdit(DirectoriesMixin, TestCase):
@@ -704,27 +701,47 @@ class TestPDFActions(DirectoriesMixin, TestCase):
super().setUp()
sample1 = self.dirs.scratch_dir / "sample.pdf"
shutil.copy(
SIMPLE_DIGITAL_PDF,
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
sample1,
)
sample1_archive = self.dirs.archive_dir / "sample_archive.pdf"
shutil.copy(
SIMPLE_DIGITAL_PDF,
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
sample1_archive,
)
sample2 = self.dirs.scratch_dir / "sample2.pdf"
shutil.copy(
MULTI_PAGE_DIGITAL_PDF,
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000002.pdf",
sample2,
)
sample2_archive = self.dirs.archive_dir / "sample2_archive.pdf"
shutil.copy(
MULTI_PAGE_DIGITAL_PDF,
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000002.pdf",
sample2_archive,
)
sample3 = self.dirs.scratch_dir / "sample3.pdf"
shutil.copy(
WITH_FORM_PDF,
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000003.pdf",
sample3,
)
self.doc1 = Document.objects.create(
@@ -758,7 +775,11 @@ class TestPDFActions(DirectoriesMixin, TestCase):
)
img_doc_archive = self.dirs.archive_dir / "sample_image.pdf"
shutil.copy(
SIMPLE_DIGITAL_PDF,
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
img_doc_archive,
)
self.img_doc = Document.objects.create(
+28 -9
View File
@@ -36,9 +36,6 @@ from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.factories import UserFactory
from paperless_testing.fakes.progress import FakeProgressManager
from paperless_testing.samples import MULTI_PAGE_DIGITAL_PDF
from paperless_testing.samples import MULTI_PAGE_IMAGES_PDF
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
class _BaseNewStyleParser:
@@ -115,7 +112,9 @@ class _BaseNewStyleParser:
class DummyParser(_BaseNewStyleParser):
_ARCHIVE_SRC = MULTI_PAGE_IMAGES_PDF
_ARCHIVE_SRC = (
Path(__file__).parent / "samples" / "documents" / "archive" / "0000001.pdf"
)
def parse(self, document_path, mime_type, *, produce_archive: bool = True) -> None:
self._text = "The Text"
@@ -201,19 +200,33 @@ class TestConsumer(
self.addCleanup(patcher.stop)
def get_test_file(self):
src = SIMPLE_DIGITAL_PDF
src = (
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf"
)
dst = self.dirs.scratch_dir / "sample.pdf"
shutil.copy(src, dst)
return dst
def get_test_file2(self):
src = MULTI_PAGE_DIGITAL_PDF
src = (
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000002.pdf"
)
dst = self.dirs.scratch_dir / "sample2.pdf"
shutil.copy(src, dst)
return dst
def get_test_archive_file(self):
src = MULTI_PAGE_IMAGES_PDF
src = (
Path(__file__).parent / "samples" / "documents" / "archive" / "0000001.pdf"
)
dst = self.dirs.scratch_dir / "sample_archive.pdf"
shutil.copy(src, dst)
return dst
@@ -1049,7 +1062,7 @@ class TestConsumer(
@mock.patch("documents.consumer.get_parser_registry")
def test_similar_filenames(self, m) -> None:
shutil.copy(
SIMPLE_DIGITAL_PDF,
Path(__file__).parent / "samples" / "simple.pdf",
settings.CONSUMPTION_DIR / "simple.pdf",
)
shutil.copy(
@@ -1590,7 +1603,13 @@ class TestConsumerRemoteOCR(
self.addCleanup(patcher.stop)
def _consume(self, *, overrides: DocumentMetadataOverrides | None = None) -> bool:
src = SIMPLE_DIGITAL_PDF
src = (
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf"
)
dst = self.dirs.scratch_dir / "sample.pdf"
shutil.copy(src, dst)
+2 -5
View File
@@ -37,15 +37,12 @@ class TestDoubleSided(
self.double_sided_dir.mkdir()
self.staging_file = self.dirs.scratch_dir / STAGING_FILE_NAME
def _sample(self, name: str) -> Path:
return self.SIMPLE_PDF if name == "simple.pdf" else self.SAMPLE_DIR / name
def consume_file(self, srcname, dstname: str | Path = "foo.pdf"):
"""
Starts the consume process and also ensures the
destination file does not exist afterwards
"""
src = self._sample(srcname)
src = self.SAMPLE_DIR / srcname
dst = self.double_sided_dir / dstname
dst.parent.mkdir(parents=True, exist_ok=True)
shutil.copy(src, dst)
@@ -60,7 +57,7 @@ class TestDoubleSided(
return msg
def create_staging_file(self, src="double-sided-odd.pdf", datetime=None) -> None:
shutil.copy(self._sample(src), self.staging_file)
shutil.copy(self.SAMPLE_DIR / src, self.staging_file)
if datetime is None:
datetime = dt.datetime.now()
os.utime(str(self.staging_file), (datetime.timestamp(),) * 2)
+1 -2
View File
@@ -22,9 +22,8 @@ from documents.models import Document
from documents.tasks import update_document_content_maybe_archive_file
from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
sample_file: Path = SIMPLE_DIGITAL_PDF
sample_file: Path = Path(__file__).parent / "samples" / "simple.pdf"
@pytest.mark.management
+76 -20
View File
@@ -51,7 +51,6 @@ from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.dirs import paperless_environment
from paperless_testing.permissions import grant_object
from paperless_testing.samples import install_document_samples
@pytest.mark.management
@@ -197,7 +196,10 @@ class TestExportImport(
def test_exporter(self, *, use_filename_format=False) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
num_permission_objects = Permission.objects.count()
@@ -301,7 +303,10 @@ class TestExportImport(
def test_exporter_with_filename_format(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
with override_settings(
FILENAME_FORMAT="{created_year}/{correspondent}/{title}",
@@ -310,7 +315,10 @@ class TestExportImport(
def test_exporter_includes_share_links_and_bundles(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
share_link = ShareLink.objects.create(
slug="share-link-slug",
@@ -409,7 +417,10 @@ class TestExportImport(
def test_update_export_changed_time(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
self._do_export()
self.assertIsFile(self.target / "manifest.json")
@@ -445,7 +456,10 @@ class TestExportImport(
def test_update_export_changed_checksum(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
self._do_export()
@@ -472,7 +486,10 @@ class TestExportImport(
def test_update_export_deleted_document(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
manifest = self._do_export()
@@ -504,7 +521,10 @@ class TestExportImport(
@override_settings(FILENAME_FORMAT="{title}/{correspondent}")
def test_update_export_changed_location(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
self._do_export(use_filename_format=True)
self.assertIsFile(self.target / "wow1" / "c.pdf")
@@ -546,7 +566,10 @@ class TestExportImport(
- Zipfile contains exported files
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
args = ["document_exporter", self.target, "--zip"]
@@ -575,7 +598,10 @@ class TestExportImport(
- Zipfile contains exported files
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
args = ["document_exporter", self.target, "--zip", "--use-filename-format"]
@@ -611,7 +637,10 @@ class TestExportImport(
- The existing file and directory in target are removed
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
# Create stuff in target directory
existing_file = self.target / "test.txt"
@@ -707,7 +736,10 @@ class TestExportImport(
- Documents can be imported again
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
manifest = self._do_export()
has_archive = False
@@ -750,7 +782,10 @@ class TestExportImport(
- Documents can be imported again
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
manifest = self._do_export()
has_thumbnail = False
@@ -795,7 +830,10 @@ class TestExportImport(
- Documents can be imported again
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
manifest = self._do_export(split_manifest=True)
has_document = False
@@ -828,7 +866,10 @@ class TestExportImport(
- Documents can be imported again
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
self._do_export(use_folder_prefix=True)
@@ -855,7 +896,10 @@ class TestExportImport(
- Documents can be imported again
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
self._do_export(use_folder_prefix=True, split_manifest=True)
@@ -882,7 +926,10 @@ class TestExportImport(
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
num_content_type_objects = ContentType.objects.count()
num_permission_objects = Permission.objects.count()
@@ -919,7 +966,10 @@ class TestExportImport(
def test_exporter_with_auditlog_disabled(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
with override_settings(
AUDIT_LOG_ENABLED=False,
@@ -939,7 +989,10 @@ class TestExportImport(
survive the round-trip with deleted_at preserved
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
install_document_samples(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
# d1 has self.note and self.cfi1 attached via setUp
self.d1.delete()
@@ -983,7 +1036,10 @@ class TestExportImport(
"""
shutil.rmtree(self.dirs.media_dir / "documents")
install_document_samples(self.dirs.media_dir / "documents")
shutil.copytree(
self.SAMPLE_DIR / "documents",
self.dirs.media_dir / "documents",
)
_ = self._do_export(data_only=True)
@@ -11,7 +11,6 @@ from documents.models import Document
from documents.parsers import get_default_thumbnail
from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
@pytest.mark.management
@@ -25,7 +24,7 @@ class TestMakeThumbnails(DirectoriesMixin, FileSystemAssertsMixin, TestCase):
filename="test.pdf",
)
shutil.copy(
SIMPLE_DIGITAL_PDF,
Path(__file__).parent / "samples" / "simple.pdf",
self.d1.source_path,
)
@@ -37,7 +36,7 @@ class TestMakeThumbnails(DirectoriesMixin, FileSystemAssertsMixin, TestCase):
filename="test2.pdf",
)
shutil.copy(
SIMPLE_DIGITAL_PDF,
Path(__file__).parent / "samples" / "simple.pdf",
self.d2.source_path,
)
+386
View File
@@ -0,0 +1,386 @@
"""
Tests for documents.pdf_ops.
These use real PDFs from the sample directories. No database, Celery or mocks.
Pages are compared by a hash of their content stream, so page identity and order
are easy to assert.
"""
import ast
import hashlib
from collections.abc import Callable
from pathlib import Path
import pikepdf
import pytest
from documents import pdf_ops
from documents.pdf_ops import PageSpec
SRC_ROOT = Path(__file__).parents[2]
SAMPLES = Path(__file__).parent / "samples"
THREE_PAGES = SAMPLES / "documents" / "originals" / "0000002.pdf"
TWELVE_PAGES = SAMPLES / "barcodes" / "split-by-asn-2.pdf"
ENCRYPTED = SAMPLES / "password-is-test.pdf"
SIGNED = SRC_ROOT / "paperless" / "tests" / "samples" / "tesseract" / "signed.pdf"
def _page_fingerprint(page: pikepdf.Page) -> str:
contents = page.obj.get("/Contents")
assert contents is not None, "sample page has no /Contents"
streams = list(contents) if isinstance(contents, pikepdf.Array) else [contents]
return hashlib.sha1(b"".join(s.read_bytes() for s in streams)).hexdigest()
def fingerprints(path: Path) -> list[str]:
with pikepdf.open(path) as pdf:
return [_page_fingerprint(page) for page in pdf.pages]
def rotations(path: Path) -> list[int]:
with pikepdf.open(path) as pdf:
return [int(page.obj.get("/Rotate", 0)) for page in pdf.pages]
def docinfo_keys(path: Path) -> set[str]:
with pikepdf.open(path) as pdf:
return set(pdf.docinfo.keys())
def constant(path: Path) -> Callable[[], Path]:
return lambda: path
@pytest.fixture
def source_fingerprints() -> list[str]:
fps = fingerprints(THREE_PAGES)
assert len(set(fps)) == 3, "sample must have three distinct pages"
return fps
class TestRotatePdf:
def test_rotation_is_relative_and_applies_to_every_page(self, tmp_path: Path):
once = tmp_path / "once.pdf"
twice = tmp_path / "twice.pdf"
pdf_ops.rotate_pdf(THREE_PAGES, once, 90)
pdf_ops.rotate_pdf(once, twice, 90)
assert rotations(once) == [90, 90, 90]
assert rotations(twice) == [180, 180, 180]
assert fingerprints(twice) == fingerprints(THREE_PAGES)
def test_keeps_document_info(self, tmp_path: Path):
dst = tmp_path / "out.pdf"
pdf_ops.rotate_pdf(THREE_PAGES, dst, 90)
assert "/Creator" in docinfo_keys(dst)
class TestRemovePages:
def test_removes_selected_pages(
self,
tmp_path: Path,
source_fingerprints: list[str],
):
dst = tmp_path / "out.pdf"
pdf_ops.remove_pages(THREE_PAGES, dst, [2])
assert fingerprints(dst) == [source_fingerprints[0], source_fingerprints[2]]
def test_duplicate_page_numbers_remove_the_page_once(
self,
tmp_path: Path,
source_fingerprints: list[str],
):
dst = tmp_path / "out.pdf"
pdf_ops.remove_pages(THREE_PAGES, dst, [2, 2])
assert fingerprints(dst) == [source_fingerprints[0], source_fingerprints[2]]
def test_unordered_pages(self, tmp_path: Path, source_fingerprints: list[str]):
dst = tmp_path / "out.pdf"
pdf_ops.remove_pages(THREE_PAGES, dst, [3, 1])
assert fingerprints(dst) == [source_fingerprints[1]]
def test_empty_list_keeps_every_page(
self,
tmp_path: Path,
source_fingerprints: list[str],
):
dst = tmp_path / "out.pdf"
pdf_ops.remove_pages(THREE_PAGES, dst, [])
assert fingerprints(dst) == source_fingerprints
def test_removing_every_page_writes_an_empty_pdf(self, tmp_path: Path):
dst = tmp_path / "out.pdf"
pdf_ops.remove_pages(THREE_PAGES, dst, [1, 2, 3])
assert fingerprints(dst) == []
def test_keeps_document_info(self, tmp_path: Path):
dst = tmp_path / "out.pdf"
pdf_ops.remove_pages(THREE_PAGES, dst, [1])
assert "/Creator" in docinfo_keys(dst)
@pytest.mark.parametrize("bad_page", [0, -1])
def test_rejects_pages_below_one(self, tmp_path: Path, bad_page: int):
dst = tmp_path / "out.pdf"
with pytest.raises(ValueError, match="start at 1"):
pdf_ops.remove_pages(THREE_PAGES, dst, [1, bad_page])
assert not dst.exists()
def test_page_past_the_end_raises_and_writes_nothing(self, tmp_path: Path):
dst = tmp_path / "out.pdf"
with pytest.raises(IndexError):
pdf_ops.remove_pages(THREE_PAGES, dst, [99])
assert not dst.exists()
class TestBuildPdfs:
def test_selects_and_orders_pages(
self,
tmp_path: Path,
source_fingerprints: list[str],
):
dst = tmp_path / "out.pdf"
written = pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(3), PageSpec(1)], constant(dst))],
)
assert written == [dst]
assert fingerprints(dst) == [source_fingerprints[2], source_fingerprints[0]]
def test_rotates_only_the_requested_pages(self, tmp_path: Path):
dst = tmp_path / "out.pdf"
pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(1), PageSpec(2, 90), PageSpec(3, 180)], constant(dst))],
)
assert rotations(dst) == [0, 90, 180]
def test_writes_one_file_per_output_in_order(self, tmp_path: Path):
first = tmp_path / "first.pdf"
second = tmp_path / "second.pdf"
source = fingerprints(TWELVE_PAGES)
written = pdf_ops.build_pdfs(
TWELVE_PAGES,
[
([PageSpec(p) for p in (1, 2, 3)], constant(first)),
([PageSpec(p) for p in range(4, 13)], constant(second)),
],
)
assert written == [first, second]
assert fingerprints(first) == source[:3]
assert fingerprints(second) == source[3:]
def test_empty_page_list_writes_a_zero_page_file(self, tmp_path: Path):
dst = tmp_path / "out.pdf"
pdf_ops.build_pdfs(THREE_PAGES, [([], constant(dst))])
assert fingerprints(dst) == []
def test_does_not_carry_over_document_info(self, tmp_path: Path):
dst = tmp_path / "out.pdf"
assert "/Creator" in docinfo_keys(THREE_PAGES)
pdf_ops.build_pdfs(THREE_PAGES, [([PageSpec(1)], constant(dst))])
assert "/Creator" not in docinfo_keys(dst)
def test_destination_is_not_requested_when_a_page_is_out_of_range(
self,
tmp_path: Path,
):
requested: list[Path] = []
def make_dst() -> Path:
requested.append(tmp_path / "out.pdf")
return requested[-1]
with pytest.raises(IndexError):
pdf_ops.build_pdfs(THREE_PAGES, [([PageSpec(99)], make_dst)])
assert requested == []
@pytest.mark.parametrize("bad_page", [0, -1])
def test_rejects_pages_below_one_before_opening_anything(
self,
tmp_path: Path,
bad_page: int,
):
requested: list[Path] = []
def make_dst() -> Path:
requested.append(tmp_path / "out.pdf")
return requested[-1]
with pytest.raises(ValueError, match="start at 1"):
pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(1)], make_dst), ([PageSpec(bad_page)], make_dst)],
)
assert requested == []
class TestValidatePageOperations:
def test_returns_the_output_count(self):
operations = [{"page": 1}, {"page": 2}, {"page": 3}]
assert pdf_ops.validate_page_operations(operations, single_output=True) == 1
def test_gap_in_output_indices_counts_up_to_the_highest(self):
operations = [
{"page": 1, "doc": 0},
{"page": 2, "doc": 2},
{"page": 3, "doc": 0},
]
count = pdf_ops.validate_page_operations(operations, single_output=False)
assert count == 3
def test_empty_operations_are_rejected(self):
with pytest.raises(ValueError, match="index is out of bounds"):
pdf_ops.validate_page_operations([], single_output=False)
def test_multiple_outputs_rejected_when_single_output_required(self):
operations = [{"page": 1, "doc": 0}, {"page": 2, "doc": 1}]
with pytest.raises(ValueError, match="Multiple output documents"):
pdf_ops.validate_page_operations(operations, single_output=True)
@pytest.mark.parametrize("doc", [-1, 2, 2**32])
def test_output_index_out_of_bounds(self, doc: int):
operations = [{"page": 1, "doc": 0}, {"page": 2, "doc": doc}]
with pytest.raises(ValueError, match="index is out of bounds"):
pdf_ops.validate_page_operations(operations, single_output=False)
class TestDecrypt:
def test_needs_decrypt(self):
assert pdf_ops.needs_decrypt(ENCRYPTED) is True
assert pdf_ops.needs_decrypt(THREE_PAGES) is False
def test_pdf_that_opens_without_a_password_but_is_flagged_encrypted(self):
assert pdf_ops.needs_decrypt(SIGNED) is True
def test_decrypt_writes_an_unencrypted_copy(self, tmp_path: Path):
dst = tmp_path / "out.pdf"
result = pdf_ops.decrypt_pdf(ENCRYPTED, constant(dst), "test")
assert result == dst
assert pdf_ops.needs_decrypt(dst) is False
def test_wrong_password_raises_and_never_requests_a_destination(
self,
tmp_path: Path,
):
requested: list[Path] = []
def make_dst() -> Path:
requested.append(tmp_path / "out.pdf")
return requested[-1]
with pytest.raises(pikepdf.PasswordError):
pdf_ops.decrypt_pdf(ENCRYPTED, make_dst, "wrong")
assert requested == []
class TestPdfMerger:
def test_pages_are_appended_in_the_order_added(
self,
tmp_path: Path,
source_fingerprints: list[str],
):
reordered = tmp_path / "reordered.pdf"
merged = tmp_path / "merged.pdf"
pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(3), PageSpec(1)], constant(reordered))],
)
with pdf_ops.PdfMerger() as merger:
merger.add(reordered)
merger.add(THREE_PAGES)
merger.save(merged)
assert fingerprints(merged) == [
source_fingerprints[2],
source_fingerprints[0],
*source_fingerprints,
]
def test_output_version_is_at_least_the_highest_source_version(
self,
tmp_path: Path,
):
merged = tmp_path / "merged.pdf"
with pikepdf.open(TWELVE_PAGES) as pdf:
source_versions = [pdf.pdf_version]
with pikepdf.open(THREE_PAGES) as pdf:
source_versions.append(pdf.pdf_version)
with pdf_ops.PdfMerger() as merger:
merger.add(TWELVE_PAGES)
merger.add(THREE_PAGES)
merger.save(merged)
with pikepdf.open(merged) as pdf:
assert pdf.pdf_version >= max(source_versions)
def test_unreadable_source_raises_so_the_caller_can_skip_it(
self,
tmp_path: Path,
):
garbage = tmp_path / "garbage.pdf"
garbage.write_bytes(b"not a pdf")
with pdf_ops.PdfMerger() as merger:
with pytest.raises(pikepdf.PdfError):
merger.add(garbage)
def test_pdf_ops_imports_only_the_standard_library_and_pikepdf():
tree = ast.parse(Path(pdf_ops.__file__).read_text())
imported: set[str] = set()
for node in ast.walk(tree):
if isinstance(node, ast.Import):
imported.update(alias.name.split(".")[0] for alias in node.names)
elif isinstance(node, ast.ImportFrom) and node.level == 0 and node.module:
imported.add(node.module.split(".")[0])
coupled = imported & {
"django",
"celery",
"documents",
"paperless",
"paperless_mail",
"paperless_ai",
}
assert not coupled
-61
View File
@@ -1,61 +0,0 @@
import hashlib
from collections import defaultdict
from pathlib import Path
from paperless_testing.samples import SHARED_SAMPLES_DIR
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
from paperless_testing.samples import install_document_samples
EXPECTED_MEDIA_FILES = [
"originals/0000001.pdf",
"originals/0000002.pdf",
"originals/0000003.pdf",
"originals/0000004.pdf",
"originals/0000005.pdf",
"originals/0000006.pdf",
"archive/0000001.pdf",
"thumbnails/0000001.webp",
"thumbnails/0000002.webp",
"thumbnails/0000003.webp",
"thumbnails/0000004.webp",
]
def _sha(path: Path) -> str:
return hashlib.sha256(path.read_bytes()).hexdigest()
def test_install_recreates_every_historical_opaque_name(tmp_path: Path) -> None:
install_document_samples(tmp_path)
for relpath in EXPECTED_MEDIA_FILES:
assert (tmp_path / relpath).is_file(), relpath
def test_install_uses_the_shared_bytes(tmp_path: Path) -> None:
install_document_samples(tmp_path)
assert _sha(tmp_path / "originals" / "0000001.pdf") == _sha(SIMPLE_DIGITAL_PDF)
APP_SAMPLE_TREES = [
Path(__file__).parent / "samples",
Path(__file__).parents[2] / "paperless" / "tests" / "samples",
SHARED_SAMPLES_DIR,
]
def test_no_two_sample_files_have_identical_bytes() -> None:
by_hash: dict[str, list[Path]] = defaultdict(list)
for tree in APP_SAMPLE_TREES:
assert tree.is_dir(), f"Sample tree does not exist: {tree}"
for path in tree.rglob("*"):
# Skip empty placeholders, which would all hash identically
if path.is_file() and path.stat().st_size > 0:
by_hash[_sha(path)].append(path)
assert by_hash, "No sample files were found to compare"
duplicates = {h: ps for h, ps in by_hash.items() if len(ps) > 1}
lines = [" " + ", ".join(str(p) for p in ps) for ps in duplicates.values()]
message = (
"Duplicate sample files, move one to paperless_testing/sample_files:\n"
+ "\n".join(lines)
)
assert not duplicates, message
+15 -4
View File
@@ -20,7 +20,6 @@ from documents.sanity_checker import SanityCheckMessages
from documents.tests.helpers import dummy_preprocess
from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
@pytest.mark.django_db
@@ -229,12 +228,20 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
"""
sample1 = self.dirs.scratch_dir / "sample.pdf"
shutil.copy(
SIMPLE_DIGITAL_PDF,
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
sample1,
)
sample1_archive = self.dirs.archive_dir / "sample_archive.pdf"
shutil.copy(
SIMPLE_DIGITAL_PDF,
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
sample1_archive,
)
doc = Document.objects.create(
@@ -262,7 +269,11 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
"""
sample1 = self.dirs.scratch_dir / "sample.pdf"
shutil.copy(
SIMPLE_DIGITAL_PDF,
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
sample1,
)
doc = Document.objects.create(
+27 -27
View File
@@ -178,7 +178,7 @@ class TestWorkflows(
self.assertEqual(action.__str__(), "WorkflowAction 1")
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -290,7 +290,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -409,7 +409,7 @@ class TestWorkflows(
w2.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -478,7 +478,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -531,7 +531,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -608,7 +608,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -687,7 +687,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -765,7 +765,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -876,7 +876,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -928,7 +928,7 @@ class TestWorkflows(
generated = generate_unique_filename(doc)
destination = (settings.ORIGINALS_DIR / generated).resolve()
create_source_path_directory(destination)
shutil.copy(self.SIMPLE_PDF, destination)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", destination)
Document.objects.filter(pk=doc.pk).update(filename=generated.as_posix())
doc.refresh_from_db()
@@ -2023,7 +2023,7 @@ class TestWorkflows(
superuser = UserFactory(username="superuser", superuser=True)
self.client.force_authenticate(user=superuser)
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
tasks.consume_file(
@@ -3009,7 +3009,7 @@ class TestWorkflows(
generated = generate_unique_filename(doc)
destination = (settings.ORIGINALS_DIR / generated).resolve()
create_source_path_directory(destination)
shutil.copy(self.SIMPLE_PDF, destination)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", destination)
Document.objects.filter(pk=doc.pk).update(filename=generated.as_posix())
doc.refresh_from_db()
doc.tags.set([self.t1, self.t2])
@@ -3073,7 +3073,7 @@ class TestWorkflows(
generated = generate_unique_filename(doc)
destination = (settings.ORIGINALS_DIR / generated).resolve()
create_source_path_directory(destination)
shutil.copy(self.SIMPLE_PDF, destination)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", destination)
Document.objects.filter(pk=doc.pk).update(filename=generated.as_posix())
doc.refresh_from_db()
doc.tags.set([self.t1])
@@ -3222,7 +3222,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -3346,7 +3346,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -3496,7 +3496,7 @@ class TestWorkflows(
workflow.actions.set([assignment_action, email_action])
temp_working_copy = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "working-copy.pdf",
)
@@ -3596,7 +3596,7 @@ class TestWorkflows(
# move the file
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -3704,7 +3704,7 @@ class TestWorkflows(
generated = generate_unique_filename(doc)
destination = (settings.ORIGINALS_DIR / generated).resolve()
create_source_path_directory(destination)
shutil.copy(self.SIMPLE_PDF, destination)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", destination)
Document.objects.filter(pk=doc.pk).update(filename=generated.as_posix())
run_workflows(WorkflowTrigger.WorkflowTriggerType.DOCUMENT_UPDATED, doc)
@@ -3915,7 +3915,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -4036,7 +4036,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -4387,7 +4387,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -4835,7 +4835,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
@@ -5027,7 +5027,7 @@ class TestWorkflows(
# Create a test file to be consumed
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
test_file_path = Path(test_file)
@@ -5091,7 +5091,7 @@ class TestWorkflows(
# Create a test file to be consumed
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple2.pdf",
)
test_file_path = Path(test_file)
@@ -5529,7 +5529,7 @@ class TestDateWorkflowLocalization(
)
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
tmp_path / "simple.pdf",
)
@@ -5596,7 +5596,7 @@ class TestRemoteOCRWorkflowAction(DirectoriesMixin, SampleDirMixin, APITestCase)
self._make_workflow(WorkflowTrigger.WorkflowTriggerType.CONSUMPTION)
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
overrides = DocumentMetadataOverrides()
@@ -5814,7 +5814,7 @@ class TestApplyAISuggestionsWorkflowAction(
)
test_file = shutil.copy(
self.SIMPLE_PDF,
self.SAMPLE_DIR / "simple.pdf",
self.dirs.scratch_dir / "simple.pdf",
)
-3
View File
@@ -11,7 +11,6 @@ from documents.data_models import ConsumableDocument
from documents.data_models import DocumentMetadataOverrides
from documents.data_models import DocumentSource
from paperless_testing.fakes.progress import FakeProgressManager
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
class ConsumeTaskMixin:
@@ -54,8 +53,6 @@ class SampleDirMixin:
BARCODE_SAMPLE_DIR = SAMPLE_DIR / "barcodes"
SIMPLE_PDF = SIMPLE_DIGITAL_PDF
class GetConsumerMixin:
@contextmanager
+24
View File
@@ -432,6 +432,30 @@ def tesseract_samples_dir(parser_samples_dir: Path) -> Path:
return parser_samples_dir / "tesseract"
@pytest.fixture(scope="session")
def multi_page_images_pdf_file(tesseract_samples_dir: Path) -> Path:
"""Path to a multi-page PDF with images.
Returns
-------
Path
Absolute path to ``tesseract/multi-page-images.pdf``.
"""
return tesseract_samples_dir / "multi-page-images.pdf"
@pytest.fixture(scope="session")
def simple_digital_pdf_file(tesseract_samples_dir: Path) -> Path:
"""Path to a simple digital PDF sample file.
Returns
-------
Path
Absolute path to ``tesseract/simple-digital.pdf``.
"""
return tesseract_samples_dir / "simple-digital.pdf"
@pytest.fixture(scope="session")
def simple_no_dpi_png_file(tesseract_samples_dir: Path) -> Path:
"""Path to a simple PNG without DPI information.
@@ -160,11 +160,11 @@ class TestGetPageCount:
def test_single_page_pdf(
self,
tesseract_parser: RasterisedDocumentParser,
simple_digital_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
assert (
tesseract_parser.get_page_count(
simple_digital_pdf_file,
tesseract_samples_dir / "simple-digital.pdf",
"application/pdf",
)
== 1
@@ -187,7 +187,7 @@ class TestGetPageCount:
self,
mocker: MockerFixture,
tesseract_parser: RasterisedDocumentParser,
simple_digital_pdf_file: Path,
tesseract_samples_dir: Path,
caplog,
) -> None:
"""
@@ -203,7 +203,7 @@ class TestGetPageCount:
with caplog.at_level(logging.WARNING):
page_count = tesseract_parser.get_page_count(
simple_digital_pdf_file,
tesseract_samples_dir / "simple-digital.pdf",
"application/pdf",
)
assert page_count is None
@@ -272,10 +272,10 @@ class TestGetThumbnail:
def test_thumbnail_is_file(
self,
tesseract_parser: RasterisedDocumentParser,
simple_digital_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
thumb = tesseract_parser.get_thumbnail(
simple_digital_pdf_file,
tesseract_samples_dir / "simple-digital.pdf",
"application/pdf",
)
assert thumb.is_file()
@@ -284,7 +284,7 @@ class TestGetThumbnail:
self,
mocker: MockerFixture,
tesseract_parser: RasterisedDocumentParser,
simple_digital_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
def _raise_on_pdf(input_file, output_file, **kwargs) -> None:
if ".pdf" in str(input_file):
@@ -294,7 +294,7 @@ class TestGetThumbnail:
mocker.patch("documents.parsers.run_convert", side_effect=_raise_on_pdf)
thumb = tesseract_parser.get_thumbnail(
simple_digital_pdf_file,
tesseract_samples_dir / "simple-digital.pdf",
"application/pdf",
)
assert thumb.is_file()
@@ -320,11 +320,11 @@ class TestExtractText:
def test_extract_text_from_digital_pdf(
self,
tesseract_parser: RasterisedDocumentParser,
simple_digital_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
text = tesseract_parser.extract_text(
None,
simple_digital_pdf_file,
tesseract_samples_dir / "simple-digital.pdf",
)
assert text is not None
assert "This is a test document." in text.strip()
@@ -339,7 +339,7 @@ class TestParsePdf:
def test_simple_digital_creates_archive(
self,
tesseract_parser: RasterisedDocumentParser,
multi_page_digital_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
"""
GIVEN:
@@ -353,7 +353,7 @@ class TestParsePdf:
- Text is extracted
"""
tesseract_parser.parse(
multi_page_digital_pdf_file,
tesseract_samples_dir / "multi-page-digital.pdf",
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -368,10 +368,10 @@ class TestParsePdf:
def test_with_form_default(
self,
tesseract_parser: RasterisedDocumentParser,
with_form_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
tesseract_parser.parse(
with_form_pdf_file,
tesseract_samples_dir / "with-form.pdf",
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -384,11 +384,11 @@ class TestParsePdf:
def test_with_form_redo_no_archive_when_not_requested(
self,
tesseract_parser: RasterisedDocumentParser,
with_form_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
tesseract_parser.settings.mode = ModeChoices.REDO
tesseract_parser.parse(
with_form_pdf_file,
tesseract_samples_dir / "with-form.pdf",
"application/pdf",
produce_archive=False,
)
@@ -401,11 +401,11 @@ class TestParsePdf:
def test_with_form_force(
self,
tesseract_parser: RasterisedDocumentParser,
with_form_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
tesseract_parser.settings.mode = ModeChoices.FORCE
tesseract_parser.parse(
with_form_pdf_file,
tesseract_samples_dir / "with-form.pdf",
"application/pdf",
)
assert_ordered_substrings(
@@ -446,7 +446,7 @@ class TestParsePdf:
self,
mocker: MockerFixture,
tesseract_parser: RasterisedDocumentParser,
simple_digital_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
mocker.patch(
"ocrmypdf.ocr",
@@ -454,7 +454,7 @@ class TestParsePdf:
)
with pytest.raises(ParseError):
tesseract_parser.parse(
simple_digital_pdf_file,
tesseract_samples_dir / "simple-digital.pdf",
"application/pdf",
)
@@ -530,10 +530,10 @@ class TestParseMultiPage:
def test_multi_page_digital(
self,
tesseract_parser: RasterisedDocumentParser,
multi_page_digital_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
tesseract_parser.parse(
multi_page_digital_pdf_file,
tesseract_samples_dir / "multi-page-digital.pdf",
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -557,12 +557,12 @@ class TestParseMultiPage:
self,
mode: str,
tesseract_parser: RasterisedDocumentParser,
multi_page_digital_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
tesseract_parser.settings.pages = 2
tesseract_parser.settings.mode = mode
tesseract_parser.parse(
multi_page_digital_pdf_file,
tesseract_samples_dir / "multi-page-digital.pdf",
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -576,11 +576,11 @@ class TestParseMultiPage:
def test_multi_page_images_skip(
self,
tesseract_parser: RasterisedDocumentParser,
multi_page_images_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
tesseract_parser.settings.mode = ModeChoices.AUTO
tesseract_parser.parse(
multi_page_images_pdf_file,
tesseract_samples_dir / "multi-page-images.pdf",
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -594,7 +594,7 @@ class TestParseMultiPage:
def test_multi_page_images_redo_pages_2(
self,
tesseract_parser: RasterisedDocumentParser,
multi_page_images_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
"""
GIVEN:
@@ -609,7 +609,7 @@ class TestParseMultiPage:
tesseract_parser.settings.pages = 2
tesseract_parser.settings.mode = ModeChoices.REDO
tesseract_parser.parse(
multi_page_images_pdf_file,
tesseract_samples_dir / "multi-page-images.pdf",
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -622,7 +622,7 @@ class TestParseMultiPage:
def test_multi_page_images_force_page_1(
self,
tesseract_parser: RasterisedDocumentParser,
multi_page_images_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
"""
GIVEN:
@@ -637,7 +637,7 @@ class TestParseMultiPage:
tesseract_parser.settings.pages = 1
tesseract_parser.settings.mode = ModeChoices.FORCE
tesseract_parser.parse(
multi_page_images_pdf_file,
tesseract_samples_dir / "multi-page-images.pdf",
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -733,7 +733,7 @@ class TestSkipArchive:
def test_skip_noarchive_with_text_layer(
self,
tesseract_parser: RasterisedDocumentParser,
multi_page_digital_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
"""
GIVEN:
@@ -747,7 +747,7 @@ class TestSkipArchive:
"""
tesseract_parser.settings.mode = ModeChoices.AUTO
tesseract_parser.parse(
multi_page_digital_pdf_file,
tesseract_samples_dir / "multi-page-digital.pdf",
"application/pdf",
produce_archive=False,
)
@@ -762,7 +762,7 @@ class TestSkipArchive:
def test_skip_noarchive_image_only_creates_archive(
self,
tesseract_parser: RasterisedDocumentParser,
multi_page_images_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
"""
GIVEN:
@@ -775,7 +775,7 @@ class TestSkipArchive:
"""
tesseract_parser.settings.mode = ModeChoices.AUTO
tesseract_parser.parse(
multi_page_images_pdf_file,
tesseract_samples_dir / "multi-page-images.pdf",
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -787,29 +787,29 @@ class TestSkipArchive:
)
@pytest.mark.parametrize(
("produce_archive", "sample_fixture", "expect_archive"),
("produce_archive", "filename", "expect_archive"),
[
pytest.param(
True,
"multi_page_digital_pdf_file",
"multi-page-digital.pdf",
True,
id="produce-archive-with-text",
),
pytest.param(
True,
"multi_page_images_pdf_file",
"multi-page-images.pdf",
True,
id="produce-archive-no-text",
),
pytest.param(
False,
"multi_page_digital_pdf_file",
"multi-page-digital.pdf",
False,
id="no-archive-with-text-layer",
),
pytest.param(
False,
"multi_page_images_pdf_file",
"multi-page-images.pdf",
False,
id="no-archive-no-text-layer",
),
@@ -818,10 +818,10 @@ class TestSkipArchive:
def test_produce_archive_flag(
self,
produce_archive: bool, # noqa: FBT001
sample_fixture: str,
filename: str,
expect_archive: bool, # noqa: FBT001
tesseract_parser: RasterisedDocumentParser,
request: pytest.FixtureRequest,
tesseract_samples_dir: Path,
) -> None:
"""
GIVEN:
@@ -834,9 +834,8 @@ class TestSkipArchive:
- Text is always extracted
"""
tesseract_parser.settings.mode = ModeChoices.AUTO
sample = request.getfixturevalue(sample_fixture)
tesseract_parser.parse(
sample,
tesseract_samples_dir / filename,
"application/pdf",
produce_archive=produce_archive,
)
@@ -853,7 +852,7 @@ class TestSkipArchive:
self,
mocker: MockerFixture,
tesseract_parser: RasterisedDocumentParser,
simple_digital_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
"""
GIVEN:
@@ -869,7 +868,7 @@ class TestSkipArchive:
tesseract_parser.settings.mode = ModeChoices.AUTO
mock_ocr = mocker.patch("ocrmypdf.ocr")
tesseract_parser.parse(
simple_digital_pdf_file,
tesseract_samples_dir / "simple-digital.pdf",
"application/pdf",
produce_archive=False,
)
@@ -908,7 +907,7 @@ class TestSkipArchive:
def test_tagged_pdf_produces_pdfa_archive_without_ocr(
self,
tesseract_parser: RasterisedDocumentParser,
simple_digital_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
"""
GIVEN:
@@ -923,7 +922,7 @@ class TestSkipArchive:
"""
tesseract_parser.settings.mode = ModeChoices.AUTO
tesseract_parser.parse(
simple_digital_pdf_file,
tesseract_samples_dir / "simple-digital.pdf",
"application/pdf",
produce_archive=True,
)
@@ -0,0 +1,19 @@
<html>
<head>
<meta http-equiv="content-type" content="text/html; charset=UTF-8">
</head>
<body>
<p>Some Text</p>
<p>
<img src="cid:part1.pNdUSz0s.D3NqVtPg@example.de" alt="Has to be rewritten to work..">
<img src="http://localhost:8080/assets/logo_full_white.svg" alt="This image should not be shown.">
</p>
<p>and an embedded image.<br>
</p>
<p id="changeme">Paragraph unchanged.</p>
<scRipt>
document.getElementById("changeme").innerHTML = "Paragraph changed via Java Script.";
</script>
</body>
</html>
Binary file not shown.
Binary file not shown.

After

Width:  |  Height:  |  Size: 2.8 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 6.9 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 32 KiB

Binary file not shown.
+6 -4
View File
@@ -16,6 +16,8 @@ from paperless.parsers.utils import read_file_handle_unicode_errors
if TYPE_CHECKING:
from pytest_mock import MockerFixture
SAMPLES = Path(__file__).parent / "samples" / "tesseract"
class TestReadFileHandleUnicodeErrors:
def test_plain_utf8(self, tmp_path: Path) -> None:
@@ -53,11 +55,11 @@ class TestReadFileHandleUnicodeErrors:
class TestIsTaggedPdf:
def test_tagged_pdf_returns_true(self, simple_digital_pdf_file: Path) -> None:
assert is_tagged_pdf(simple_digital_pdf_file) is True
def test_tagged_pdf_returns_true(self) -> None:
assert is_tagged_pdf(SAMPLES / "simple-digital.pdf") is True
def test_untagged_pdf_returns_false(self, multi_page_images_pdf_file: Path) -> None:
assert is_tagged_pdf(multi_page_images_pdf_file) is False
def test_untagged_pdf_returns_false(self) -> None:
assert is_tagged_pdf(SAMPLES / "multi-page-images.pdf") is False
def test_nonexistent_path_returns_false(self) -> None:
assert is_tagged_pdf(Path("/nonexistent/file.pdf")) is False
-48
View File
@@ -1,48 +0,0 @@
"""Test sample files shared across tests.
A sample used by a single app stays in that app's ``tests/samples`` tree.
Samples needed by more than one app, or stored under more than one name, live
here exactly once.
"""
import shutil
from pathlib import Path
SHARED_SAMPLES_DIR = (Path(__file__).parent / "sample_files").resolve()
SIMPLE_DIGITAL_PDF = SHARED_SAMPLES_DIR / "simple-digital.pdf"
MULTI_PAGE_DIGITAL_PDF = SHARED_SAMPLES_DIR / "multi-page-digital.pdf"
WITH_FORM_PDF = SHARED_SAMPLES_DIR / "with-form.pdf"
MULTI_PAGE_IMAGES_PDF = SHARED_SAMPLES_DIR / "multi-page-images.pdf"
THUMBNAIL_WEBP = SHARED_SAMPLES_DIR / "thumbnail.webp"
# The fake media tree under documents/tests/samples/documents keeps only the
# files unique to it. The files that duplicate a shared sample are laid back
# down under their historical opaque names (DB rows and exporter manifests
# refer to them) by install_document_samples().
_DOCUMENT_SAMPLES_DIR = (
Path(__file__).parent.parent / "documents" / "tests" / "samples" / "documents"
).resolve()
_SHARED_AS_MEDIA = (
("originals/0000001.pdf", SIMPLE_DIGITAL_PDF),
("originals/0000002.pdf", MULTI_PAGE_DIGITAL_PDF),
("originals/0000003.pdf", WITH_FORM_PDF),
("archive/0000001.pdf", MULTI_PAGE_IMAGES_PDF),
("thumbnails/0000001.webp", THUMBNAIL_WEBP),
("thumbnails/0000002.webp", THUMBNAIL_WEBP),
("thumbnails/0000003.webp", THUMBNAIL_WEBP),
("thumbnails/0000004.webp", THUMBNAIL_WEBP),
)
def install_document_samples(dest: Path) -> None:
"""Populate ``dest`` with the full fake media tree.
Replaces ``shutil.copytree(<samples>/documents, dest)`` in tests.
"""
shutil.copytree(_DOCUMENT_SAMPLES_DIR, dest, dirs_exist_ok=True)
for relpath, source in _SHARED_AS_MEDIA:
target = dest / relpath
target.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(source, target)