Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6d61214bba | ||
|
|
dcfe389909 | ||
|
|
0beb0a1d0b | ||
|
|
bfaea71a83 | ||
|
|
0b7cecb6bb |
No files matched your search
@@ -38,7 +38,6 @@ src/documents/bulk_edit.py:0: error: Incompatible types in assignment (expressio
|
||||
src/documents/bulk_edit.py:0: error: Invalid index type "str" for "dict[FieldDataType, str]"; expected type "FieldDataType" [index]
|
||||
src/documents/bulk_edit.py:0: error: List comprehension has incompatible type List[tuple[int, Any]]; expected List[int] [misc]
|
||||
src/documents/bulk_edit.py:0: error: List comprehension has incompatible type List[tuple[int, None]]; expected List[int] [misc]
|
||||
src/documents/bulk_edit.py:0: error: Missing named argument "p" for "remove" of "PageList" [call-arg]
|
||||
src/documents/bulk_edit.py:0: error: Missing type arguments for generic type "dict" [type-arg]
|
||||
src/documents/bulk_edit.py:0: error: Missing type arguments for generic type "dict" [type-arg]
|
||||
src/documents/bulk_edit.py:0: error: Need type annotation for "to_create" (hint: "to_create: list[<type>] = ...") [var-annotated]
|
||||
|
||||
@@ -91,13 +91,6 @@
|
||||
"concise_description": "Argument `list[int]` is not assignable to parameter `args` with type `tuple[Any, ...] | None` in function `celery.app.task.Task.apply_async`",
|
||||
"severity": "error"
|
||||
},
|
||||
{
|
||||
"column": 33,
|
||||
"path": "src/documents/bulk_edit.py",
|
||||
"name": "missing-argument",
|
||||
"concise_description": "Missing argument `p` in function `pikepdf._core.PageList.remove`",
|
||||
"severity": "error"
|
||||
},
|
||||
{
|
||||
"column": 25,
|
||||
"path": "src/documents/caching.py",
|
||||
|
||||
@@ -256,8 +256,8 @@ isort.force-single-line = true
|
||||
[tool.codespell]
|
||||
ignore-words-list = "criterias,afterall,valeu,ureue,equest,ure,assertIn,Oktober,commitish,NIN,nin,reprot"
|
||||
skip = """\
|
||||
src-ui/src/locale/*,src-ui/pnpm-lock.yaml,src-ui/e2e/*,src/paperless/tests/samples/mail/*,src/documents/tests/samples\
|
||||
/*,src/paperless_testing/sample_files/*,*.po,*.json\
|
||||
src-ui/src/locale/*,src-ui/pnpm-lock.yaml,src-ui/e2e/*,src/paperless_mail/tests/samples/*,src/paperless/tests/samples\
|
||||
/mail/*,src/documents/tests/samples/*,*.po,*.json\
|
||||
"""
|
||||
write-changes = true
|
||||
|
||||
|
||||
@@ -11,8 +11,6 @@ from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from paperless_testing import samples
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Generator
|
||||
from pathlib import Path
|
||||
@@ -151,39 +149,3 @@ def fake_progress_manager(
|
||||
|
||||
monkeypatch.setattr("documents.tasks.ProgressManager", FakeProgressManager)
|
||||
return FakeProgressManager
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def shared_samples_dir() -> Path:
|
||||
"""Directory of sample files used by more than one app's tests."""
|
||||
return samples.SHARED_SAMPLES_DIR
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def simple_digital_pdf_file() -> Path:
|
||||
"""One-page PDF with a text layer."""
|
||||
return samples.SIMPLE_DIGITAL_PDF
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def multi_page_digital_pdf_file() -> Path:
|
||||
"""Three-page PDF with a text layer."""
|
||||
return samples.MULTI_PAGE_DIGITAL_PDF
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def with_form_pdf_file() -> Path:
|
||||
"""PDF containing a fillable form."""
|
||||
return samples.WITH_FORM_PDF
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def multi_page_images_pdf_file() -> Path:
|
||||
"""Multi-page PDF of scanned images, no text layer."""
|
||||
return samples.MULTI_PAGE_IMAGES_PDF
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def thumbnail_webp_file() -> Path:
|
||||
"""Small WebP thumbnail."""
|
||||
return samples.THUMBNAIL_WEBP
|
||||
@@ -3,6 +3,7 @@ from __future__ import annotations
|
||||
import logging
|
||||
import tempfile
|
||||
import uuid
|
||||
from functools import partial
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
from typing import Literal
|
||||
@@ -17,6 +18,7 @@ from django.db.models import Max
|
||||
from django.db.models import Q
|
||||
from django.utils import timezone
|
||||
|
||||
from documents import pdf_ops
|
||||
from documents.data_models import ConsumableDocument
|
||||
from documents.data_models import DocumentMetadataOverrides
|
||||
from documents.data_models import DocumentSource
|
||||
@@ -116,6 +118,11 @@ def _resolve_root_and_source_doc(
|
||||
)
|
||||
|
||||
|
||||
def _scratch_path(name: str) -> Path:
|
||||
"""A path inside a fresh directory under SCRATCH_DIR."""
|
||||
return Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR)) / name
|
||||
|
||||
|
||||
def set_correspondent(
|
||||
doc_ids: list[int],
|
||||
correspondent: Correspondent,
|
||||
@@ -474,8 +481,6 @@ def rotate(
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
docs_by_root_id.setdefault(pair.root_doc.id, pair)
|
||||
|
||||
import pikepdf
|
||||
|
||||
for pair in docs_by_root_id.values():
|
||||
if pair.source_doc.mime_type != "application/pdf":
|
||||
logger.warning(
|
||||
@@ -488,11 +493,7 @@ def rotate(
|
||||
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
|
||||
/ f"{pair.root_doc.id}_rotated.pdf"
|
||||
)
|
||||
with pikepdf.open(pair.source_doc.source_path) as pdf:
|
||||
for page in pdf.pages:
|
||||
page.rotate(degrees, relative=True)
|
||||
pdf.remove_unreferenced_resources()
|
||||
pdf.save(filepath)
|
||||
pdf_ops.rotate_pdf(pair.source_doc.source_path, filepath, degrees)
|
||||
|
||||
# Preserve metadata/permissions via overrides; mark as new version
|
||||
overrides = DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
@@ -535,48 +536,45 @@ def merge(
|
||||
qs = Document.objects.select_related("root_document").filter(id__in=doc_ids)
|
||||
docs_by_id = {doc.id: doc for doc in qs}
|
||||
affected_docs: list[int] = []
|
||||
import pikepdf
|
||||
|
||||
merged_pdf = pikepdf.new()
|
||||
version: str = merged_pdf.pdf_version
|
||||
handoff_asn: int | None = None
|
||||
# use doc_ids to preserve order
|
||||
for doc_id in doc_ids:
|
||||
doc = docs_by_id.get(doc_id)
|
||||
if doc is None:
|
||||
continue
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
try:
|
||||
doc_path = (
|
||||
pair.source_doc.archive_path
|
||||
if archive_fallback
|
||||
and pair.source_doc.mime_type != "application/pdf"
|
||||
and pair.source_doc.has_archive_version
|
||||
else pair.source_doc.source_path
|
||||
)
|
||||
with pikepdf.open(str(doc_path)) as pdf:
|
||||
version = max(version, pdf.pdf_version)
|
||||
merged_pdf.pages.extend(pdf.pages)
|
||||
affected_docs.append(doc.id)
|
||||
if handoff_asn is None and doc.archive_serial_number is not None:
|
||||
handoff_asn = doc.archive_serial_number
|
||||
except Exception as e:
|
||||
logger.exception(
|
||||
f"Error merging document {doc.id}, it will not be included in the merge: {e}",
|
||||
)
|
||||
if len(affected_docs) == 0:
|
||||
logger.warning("No documents were merged")
|
||||
return "OK"
|
||||
with pdf_ops.PdfMerger() as merger:
|
||||
# use doc_ids to preserve order
|
||||
for doc_id in doc_ids:
|
||||
doc = docs_by_id.get(doc_id)
|
||||
if doc is None:
|
||||
continue
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
try:
|
||||
# archive_path is None when there is no archive version
|
||||
archive_path = (
|
||||
pair.source_doc.archive_path
|
||||
if archive_fallback
|
||||
and pair.source_doc.mime_type != "application/pdf"
|
||||
else None
|
||||
)
|
||||
merger.add(
|
||||
archive_path
|
||||
if archive_path is not None
|
||||
else pair.source_doc.source_path,
|
||||
)
|
||||
affected_docs.append(doc.id)
|
||||
if handoff_asn is None and doc.archive_serial_number is not None:
|
||||
handoff_asn = doc.archive_serial_number
|
||||
except Exception as e:
|
||||
logger.exception(
|
||||
f"Error merging document {doc.id}, it will not be included in the merge: {e}",
|
||||
)
|
||||
if len(affected_docs) == 0:
|
||||
logger.warning("No documents were merged")
|
||||
return "OK"
|
||||
|
||||
filepath = (
|
||||
Path(
|
||||
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
|
||||
filepath = (
|
||||
Path(
|
||||
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
|
||||
)
|
||||
/ f"{'_'.join([str(doc_id) for doc_id in affected_docs])[:100]}_merged.pdf"
|
||||
)
|
||||
/ f"{'_'.join([str(doc_id) for doc_id in affected_docs])[:100]}_merged.pdf"
|
||||
)
|
||||
merged_pdf.remove_unreferenced_resources()
|
||||
merged_pdf.save(filepath, min_version=version)
|
||||
merged_pdf.close()
|
||||
merger.save(filepath)
|
||||
|
||||
if metadata_document_id:
|
||||
metadata_document = qs.get(id=metadata_document_id)
|
||||
@@ -752,64 +750,60 @@ def split(
|
||||
)
|
||||
doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
import pikepdf
|
||||
|
||||
consume_tasks = []
|
||||
|
||||
try:
|
||||
with pikepdf.open(pair.source_doc.source_path) as pdf:
|
||||
for idx, split_doc in enumerate(pages):
|
||||
dst: pikepdf.Pdf = pikepdf.new()
|
||||
for page in split_doc:
|
||||
dst.pages.append(pdf.pages[page - 1])
|
||||
filepath: Path = (
|
||||
Path(
|
||||
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
|
||||
)
|
||||
/ f"{doc.id}_{split_doc[0]}-{split_doc[-1]}.pdf"
|
||||
)
|
||||
dst.remove_unreferenced_resources()
|
||||
dst.save(filepath)
|
||||
dst.close()
|
||||
outputs = [
|
||||
(
|
||||
[pdf_ops.PageSpec(page) for page in split_doc],
|
||||
partial(_scratch_path, f"{doc.id}_{split_doc[0]}-{split_doc[-1]}.pdf"),
|
||||
)
|
||||
for split_doc in pages
|
||||
]
|
||||
filepaths = pdf_ops.build_pdfs(pair.source_doc.source_path, outputs)
|
||||
|
||||
overrides: DocumentMetadataOverrides = (
|
||||
DocumentMetadataOverrides().from_document(doc)
|
||||
)
|
||||
overrides.title = f"{doc.title} (split {idx + 1})"
|
||||
if user is not None:
|
||||
overrides.owner_id = user.id
|
||||
if not delete_originals:
|
||||
overrides.skip_asn_if_exists = True
|
||||
logger.info(
|
||||
f"Adding split document with pages {split_doc} to the task queue.",
|
||||
)
|
||||
consume_tasks.append(
|
||||
consume_file.s(
|
||||
input_doc=ConsumableDocument(
|
||||
source=DocumentSource.ConsumeFolder,
|
||||
original_file=filepath,
|
||||
),
|
||||
overrides=overrides,
|
||||
).set(headers={"trigger_source": trigger_source}),
|
||||
)
|
||||
for idx, (split_doc, filepath) in enumerate(
|
||||
zip(pages, filepaths, strict=True),
|
||||
):
|
||||
overrides: DocumentMetadataOverrides = (
|
||||
DocumentMetadataOverrides().from_document(doc)
|
||||
)
|
||||
overrides.title = f"{doc.title} (split {idx + 1})"
|
||||
if user is not None:
|
||||
overrides.owner_id = user.id
|
||||
if not delete_originals:
|
||||
overrides.skip_asn_if_exists = True
|
||||
logger.info(
|
||||
f"Adding split document with pages {split_doc} to the task queue.",
|
||||
)
|
||||
consume_tasks.append(
|
||||
consume_file.s(
|
||||
input_doc=ConsumableDocument(
|
||||
source=DocumentSource.ConsumeFolder,
|
||||
original_file=filepath,
|
||||
),
|
||||
overrides=overrides,
|
||||
).set(headers={"trigger_source": trigger_source}),
|
||||
)
|
||||
|
||||
if delete_originals:
|
||||
backup = release_archive_serial_numbers([doc.id])
|
||||
logger.info(
|
||||
"Queueing removal of original document after consumption of the split documents",
|
||||
)
|
||||
try:
|
||||
chord(
|
||||
header=consume_tasks,
|
||||
body=delete.si([doc.id]),
|
||||
).on_error(
|
||||
restore_archive_serial_numbers_task.s(backup),
|
||||
).apply_async()
|
||||
except Exception:
|
||||
restore_archive_serial_numbers(backup)
|
||||
raise
|
||||
else:
|
||||
group(consume_tasks).delay()
|
||||
if delete_originals:
|
||||
backup = release_archive_serial_numbers([doc.id])
|
||||
logger.info(
|
||||
"Queueing removal of original document after consumption of the split documents",
|
||||
)
|
||||
try:
|
||||
chord(
|
||||
header=consume_tasks,
|
||||
body=delete.si([doc.id]),
|
||||
).on_error(
|
||||
restore_archive_serial_numbers_task.s(backup),
|
||||
).apply_async()
|
||||
except Exception:
|
||||
restore_archive_serial_numbers(backup)
|
||||
raise
|
||||
else:
|
||||
group(consume_tasks).delay()
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Error splitting document {doc.id}: {e}")
|
||||
@@ -830,8 +824,7 @@ def delete_pages(
|
||||
)
|
||||
doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
pages = sorted(pages) # sort pages to avoid index issues
|
||||
import pikepdf
|
||||
pages = sorted(set(pages))
|
||||
|
||||
try:
|
||||
# Produce edited PDF to a temp file and create a new version
|
||||
@@ -839,13 +832,7 @@ def delete_pages(
|
||||
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
|
||||
/ f"{pair.root_doc.id}_pages_deleted.pdf"
|
||||
)
|
||||
with pikepdf.open(pair.source_doc.source_path) as pdf:
|
||||
offset = 1 # pages are 1-indexed
|
||||
for page_num in pages:
|
||||
pdf.pages.remove(pdf.pages[page_num - offset])
|
||||
offset += 1 # remove() changes the index of the pages
|
||||
pdf.remove_unreferenced_resources()
|
||||
pdf.save(filepath)
|
||||
pdf_ops.remove_pages(pair.source_doc.source_path, filepath, pages)
|
||||
|
||||
overrides = DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
if user is not None:
|
||||
@@ -894,47 +881,28 @@ def edit_pdf(
|
||||
)
|
||||
doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
import pikepdf
|
||||
|
||||
pdf_docs: list[pikepdf.Pdf] = []
|
||||
|
||||
try:
|
||||
if not operations:
|
||||
raise ValueError("Output document index is out of bounds")
|
||||
|
||||
max_idx = max(op.get("doc", 0) for op in operations)
|
||||
if update_document and max_idx > 0:
|
||||
logger.error(
|
||||
"Update requested but multiple output documents specified",
|
||||
output_count = pdf_ops.validate_page_operations(
|
||||
operations,
|
||||
single_output=update_document,
|
||||
)
|
||||
page_specs: list[list[pdf_ops.PageSpec]] = [[] for _ in range(output_count)]
|
||||
for op in operations:
|
||||
page_specs[op.get("doc", 0)].append(
|
||||
pdf_ops.PageSpec(op["page"], op.get("rotate", 0)),
|
||||
)
|
||||
raise ValueError("Multiple output documents specified")
|
||||
|
||||
if any(
|
||||
op.get("doc", 0) < 0 or op.get("doc", 0) >= len(operations)
|
||||
for op in operations
|
||||
):
|
||||
raise ValueError("Output document index is out of bounds")
|
||||
|
||||
with pikepdf.open(pair.source_doc.source_path) as src:
|
||||
# prepare output documents
|
||||
pdf_docs = [pikepdf.new() for _ in range(max_idx + 1)]
|
||||
|
||||
for op in operations:
|
||||
dst = pdf_docs[op.get("doc", 0)]
|
||||
page = src.pages[op["page"] - 1]
|
||||
dst.pages.append(page)
|
||||
if op.get("rotate"):
|
||||
dst.pages[-1].rotate(op["rotate"], relative=True)
|
||||
|
||||
if update_document:
|
||||
# Create a new version from the edited PDF rather than replacing in-place
|
||||
pdf = pdf_docs[0]
|
||||
pdf.remove_unreferenced_resources()
|
||||
filepath: Path = (
|
||||
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
|
||||
/ f"{pair.root_doc.id}_edited.pdf"
|
||||
(filepath,) = pdf_ops.build_pdfs(
|
||||
pair.source_doc.source_path,
|
||||
[
|
||||
(
|
||||
page_specs[0],
|
||||
partial(_scratch_path, f"{pair.root_doc.id}_edited.pdf"),
|
||||
),
|
||||
],
|
||||
)
|
||||
pdf.save(filepath)
|
||||
overrides = (
|
||||
DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
if include_metadata
|
||||
@@ -955,6 +923,19 @@ def edit_pdf(
|
||||
headers={"trigger_source": trigger_source},
|
||||
)
|
||||
else:
|
||||
version_filepaths = pdf_ops.build_pdfs(
|
||||
pair.source_doc.source_path,
|
||||
[
|
||||
(
|
||||
specs,
|
||||
partial(
|
||||
_scratch_path,
|
||||
f"{pair.root_doc.id}_edit_{idx}.pdf",
|
||||
),
|
||||
)
|
||||
for idx, specs in enumerate(page_specs, start=1)
|
||||
],
|
||||
)
|
||||
consume_tasks = []
|
||||
overrides = (
|
||||
DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
@@ -966,15 +947,9 @@ def edit_pdf(
|
||||
overrides.actor_id = user.id
|
||||
if not delete_original:
|
||||
overrides.skip_asn_if_exists = True
|
||||
if delete_original and len(pdf_docs) == 1:
|
||||
if delete_original and output_count == 1:
|
||||
overrides.asn = pair.root_doc.archive_serial_number
|
||||
for idx, pdf in enumerate(pdf_docs, start=1):
|
||||
version_filepath: Path = (
|
||||
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
|
||||
/ f"{pair.root_doc.id}_edit_{idx}.pdf"
|
||||
)
|
||||
pdf.remove_unreferenced_resources()
|
||||
pdf.save(version_filepath)
|
||||
for version_filepath in version_filepaths:
|
||||
consume_tasks.append(
|
||||
consume_file.s(
|
||||
input_doc=ConsumableDocument(
|
||||
@@ -1024,8 +999,6 @@ def remove_password(
|
||||
"""
|
||||
Remove password protection from PDF documents.
|
||||
"""
|
||||
import pikepdf
|
||||
|
||||
for doc_id in doc_ids:
|
||||
doc = Document.objects.select_related("root_document").get(id=doc_id)
|
||||
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
|
||||
@@ -1039,76 +1012,69 @@ def remove_password(
|
||||
doc.id,
|
||||
pair.source_doc.source_path,
|
||||
)
|
||||
try:
|
||||
with pikepdf.open(source_path) as pdf:
|
||||
if not pdf.is_encrypted:
|
||||
logger.info(
|
||||
"Skipping password removal for document %s because the "
|
||||
"source PDF is not encrypted",
|
||||
pair.root_doc.id,
|
||||
)
|
||||
continue
|
||||
except pikepdf.PasswordError:
|
||||
# Password-protected PDFs need the supplied password below.
|
||||
pass
|
||||
|
||||
with pikepdf.open(source_path, password=password) as pdf:
|
||||
filepath: Path = (
|
||||
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
|
||||
/ f"{pair.root_doc.id}_unprotected.pdf"
|
||||
if not pdf_ops.needs_decrypt(source_path):
|
||||
logger.info(
|
||||
"Skipping password removal for document %s because the "
|
||||
"source PDF is not encrypted",
|
||||
pair.root_doc.id,
|
||||
)
|
||||
pdf.remove_unreferenced_resources()
|
||||
pdf.save(filepath)
|
||||
continue
|
||||
|
||||
if update_document:
|
||||
# Create a new version rather than modifying the root/original in place.
|
||||
overrides = (
|
||||
DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
if include_metadata
|
||||
else DocumentMetadataOverrides()
|
||||
)
|
||||
if user is not None:
|
||||
overrides.owner_id = user.id
|
||||
overrides.actor_id = user.id
|
||||
consume_file.apply_async(
|
||||
kwargs={
|
||||
"input_doc": ConsumableDocument(
|
||||
source=DocumentSource.ConsumeFolder,
|
||||
original_file=filepath,
|
||||
root_document_id=pair.root_doc.id,
|
||||
),
|
||||
"overrides": overrides,
|
||||
},
|
||||
headers={"trigger_source": trigger_source},
|
||||
)
|
||||
filepath = pdf_ops.decrypt_pdf(
|
||||
source_path,
|
||||
partial(_scratch_path, f"{pair.root_doc.id}_unprotected.pdf"),
|
||||
password,
|
||||
)
|
||||
|
||||
if update_document:
|
||||
# Create a new version rather than modifying the root/original in place.
|
||||
overrides = (
|
||||
DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
if include_metadata
|
||||
else DocumentMetadataOverrides()
|
||||
)
|
||||
if user is not None:
|
||||
overrides.owner_id = user.id
|
||||
overrides.actor_id = user.id
|
||||
consume_file.apply_async(
|
||||
kwargs={
|
||||
"input_doc": ConsumableDocument(
|
||||
source=DocumentSource.ConsumeFolder,
|
||||
original_file=filepath,
|
||||
root_document_id=pair.root_doc.id,
|
||||
),
|
||||
"overrides": overrides,
|
||||
},
|
||||
headers={"trigger_source": trigger_source},
|
||||
)
|
||||
else:
|
||||
consume_tasks = []
|
||||
overrides = (
|
||||
DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
if include_metadata
|
||||
else DocumentMetadataOverrides()
|
||||
)
|
||||
if user is not None:
|
||||
overrides.owner_id = user.id
|
||||
overrides.actor_id = user.id
|
||||
|
||||
consume_tasks.append(
|
||||
consume_file.s(
|
||||
input_doc=ConsumableDocument(
|
||||
source=DocumentSource.ConsumeFolder,
|
||||
original_file=filepath,
|
||||
),
|
||||
overrides=overrides,
|
||||
).set(headers={"trigger_source": trigger_source}),
|
||||
)
|
||||
|
||||
if delete_original:
|
||||
chord(
|
||||
header=consume_tasks,
|
||||
body=delete.si([doc.id]),
|
||||
).delay()
|
||||
else:
|
||||
consume_tasks = []
|
||||
overrides = (
|
||||
DocumentMetadataOverrides().from_document(pair.root_doc)
|
||||
if include_metadata
|
||||
else DocumentMetadataOverrides()
|
||||
)
|
||||
if user is not None:
|
||||
overrides.owner_id = user.id
|
||||
overrides.actor_id = user.id
|
||||
|
||||
consume_tasks.append(
|
||||
consume_file.s(
|
||||
input_doc=ConsumableDocument(
|
||||
source=DocumentSource.ConsumeFolder,
|
||||
original_file=filepath,
|
||||
),
|
||||
overrides=overrides,
|
||||
).set(headers={"trigger_source": trigger_source}),
|
||||
)
|
||||
|
||||
if delete_original:
|
||||
chord(
|
||||
header=consume_tasks,
|
||||
body=delete.si([doc.id]),
|
||||
).delay()
|
||||
else:
|
||||
group(consume_tasks).delay()
|
||||
group(consume_tasks).delay()
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(
|
||||
|
||||
@@ -0,0 +1,184 @@
|
||||
"""
|
||||
Pure PDF page operations used by documents.bulk_edit.
|
||||
|
||||
This module deliberately knows nothing about Django, Celery or the documents
|
||||
app: callers resolve documents, choose output paths and queue work. Every
|
||||
function that writes a PDF removes unreferenced resources before saving.
|
||||
|
||||
pikepdf is always called as ``pikepdf.open(...)`` / ``pikepdf.new()`` (never
|
||||
``from pikepdf import open``) so tests can patch those module attributes.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
from typing import NamedTuple
|
||||
|
||||
import pikepdf
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Callable
|
||||
from collections.abc import Iterable
|
||||
from collections.abc import Mapping
|
||||
from collections.abc import Sequence
|
||||
from pathlib import Path
|
||||
from types import TracebackType
|
||||
|
||||
|
||||
class PageSpec(NamedTuple):
|
||||
"""One page of an output PDF: a 1-indexed source page, optionally rotated."""
|
||||
|
||||
page: int
|
||||
rotate: int = 0 # relative degrees, 0 leaves the page alone
|
||||
|
||||
|
||||
def _require_positive(pages: Iterable[int]) -> None:
|
||||
for page in pages:
|
||||
if page < 1:
|
||||
raise ValueError(f"Page numbers start at 1, got {page}")
|
||||
|
||||
|
||||
def rotate_pdf(src: Path, dst: Path, degrees: int) -> None:
|
||||
"""
|
||||
Rotate every page relatively on the opened document, not a rebuild, so Info,
|
||||
XMP and outlines are kept. ``src`` is not modified.
|
||||
"""
|
||||
with pikepdf.open(src) as pdf:
|
||||
for page in pdf.pages:
|
||||
page.rotate(degrees, relative=True)
|
||||
pdf.remove_unreferenced_resources()
|
||||
pdf.save(dst)
|
||||
|
||||
|
||||
def remove_pages(src: Path, dst: Path, pages: Iterable[int]) -> None:
|
||||
"""
|
||||
Remove 1-indexed pages from the opened document, not a rebuild, so Info, XMP
|
||||
and outlines are kept. ``src`` is not modified.
|
||||
|
||||
Duplicates are ignored. Pages are removed highest first so earlier removals
|
||||
never shift the index of later ones.
|
||||
"""
|
||||
unique = sorted(set(pages))
|
||||
_require_positive(unique)
|
||||
with pikepdf.open(src) as pdf:
|
||||
for page_num in reversed(unique):
|
||||
del pdf.pages[page_num - 1]
|
||||
pdf.remove_unreferenced_resources()
|
||||
pdf.save(dst)
|
||||
|
||||
|
||||
def build_pdfs(
|
||||
src: Path,
|
||||
outputs: Sequence[tuple[Sequence[PageSpec], Callable[[], Path]]],
|
||||
) -> list[Path]:
|
||||
"""
|
||||
Build one new PDF per output from pages of ``src``, opening ``src`` once.
|
||||
|
||||
Each output is ``(page_specs, make_dst)``. ``make_dst`` is called after that
|
||||
output's pages are copied and immediately before it is saved, so a bad page
|
||||
number never leaves a destination behind. Document-level data (Info, XMP,
|
||||
outlines) is not carried over. Returns the written paths in output order.
|
||||
"""
|
||||
for specs, _ in outputs:
|
||||
_require_positive(spec.page for spec in specs)
|
||||
|
||||
written: list[Path] = []
|
||||
with pikepdf.open(src) as source:
|
||||
for specs, make_dst in outputs:
|
||||
dst = pikepdf.new()
|
||||
for spec in specs:
|
||||
dst.pages.append(source.pages[spec.page - 1])
|
||||
if spec.rotate:
|
||||
dst.pages[-1].rotate(spec.rotate, relative=True)
|
||||
dst.remove_unreferenced_resources()
|
||||
path = make_dst()
|
||||
dst.save(path)
|
||||
dst.close()
|
||||
written.append(path)
|
||||
return written
|
||||
|
||||
|
||||
def validate_page_operations(
|
||||
operations: Sequence[Mapping[str, int]],
|
||||
*,
|
||||
single_output: bool,
|
||||
) -> int:
|
||||
"""
|
||||
Validate ``edit_pdf`` style operations and return the output document count.
|
||||
|
||||
Each operation has ``page`` and optionally ``rotate`` and ``doc`` (the output
|
||||
document index, default 0). The bounds rule is kept as it was: a ``doc`` index
|
||||
must be below the number of operations.
|
||||
"""
|
||||
if not operations:
|
||||
raise ValueError("Output document index is out of bounds")
|
||||
|
||||
max_idx = max(op.get("doc", 0) for op in operations)
|
||||
if single_output and max_idx > 0:
|
||||
raise ValueError("Multiple output documents specified")
|
||||
|
||||
if any(
|
||||
op.get("doc", 0) < 0 or op.get("doc", 0) >= len(operations) for op in operations
|
||||
):
|
||||
raise ValueError("Output document index is out of bounds")
|
||||
|
||||
return max_idx + 1
|
||||
|
||||
|
||||
def needs_decrypt(src: Path) -> bool:
|
||||
"""
|
||||
True if ``src`` is encrypted. A PDF that needs a password to open at all
|
||||
counts as encrypted.
|
||||
"""
|
||||
try:
|
||||
with pikepdf.open(src) as pdf:
|
||||
return bool(pdf.is_encrypted)
|
||||
except pikepdf.PasswordError:
|
||||
return True
|
||||
|
||||
|
||||
def decrypt_pdf(src: Path, make_dst: Callable[[], Path], password: str) -> Path:
|
||||
"""
|
||||
Write an unencrypted copy of ``src`` and return its path.
|
||||
|
||||
``make_dst`` is only called once the password has been accepted, so a wrong
|
||||
password never leaves a destination behind.
|
||||
"""
|
||||
with pikepdf.open(src, password=password) as pdf:
|
||||
pdf.remove_unreferenced_resources()
|
||||
dst = make_dst()
|
||||
pdf.save(dst)
|
||||
return dst
|
||||
|
||||
|
||||
class PdfMerger:
|
||||
"""
|
||||
Accumulates the pages of several PDFs into one new PDF.
|
||||
|
||||
``add`` raises if a source cannot be read; deciding whether to skip it is the
|
||||
caller's policy. Use as a context manager so the merged PDF is closed.
|
||||
"""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._pdf = pikepdf.new()
|
||||
self._version: str = self._pdf.pdf_version
|
||||
|
||||
def __enter__(self) -> PdfMerger:
|
||||
return self
|
||||
|
||||
def __exit__(
|
||||
self,
|
||||
exc_type: type[BaseException] | None,
|
||||
exc: BaseException | None,
|
||||
tb: TracebackType | None,
|
||||
) -> None:
|
||||
self._pdf.close()
|
||||
|
||||
def add(self, path: Path) -> None:
|
||||
with pikepdf.open(str(path)) as pdf:
|
||||
self._version = max(self._version, pdf.pdf_version)
|
||||
self._pdf.pages.extend(pdf.pages)
|
||||
|
||||
def save(self, dst: Path) -> None:
|
||||
self._pdf.remove_unreferenced_resources()
|
||||
self._pdf.save(dst, min_version=self._version)
|
||||
@@ -2137,6 +2137,8 @@ class BulkEditSerializer(
|
||||
raise serializers.ValidationError("pages must be a list")
|
||||
if not all(isinstance(i, int) for i in parameters["pages"]):
|
||||
raise serializers.ValidationError("pages must be a list of integers")
|
||||
if any(i < 1 for i in parameters["pages"]):
|
||||
raise serializers.ValidationError("pages must be positive integers")
|
||||
|
||||
def _validate_parameters_merge(self, parameters) -> None:
|
||||
if "delete_originals" in parameters:
|
||||
|
||||
@@ -12,25 +12,29 @@ if TYPE_CHECKING:
|
||||
from paperless_testing.dirs import PaperlessDirs
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def document_samples_dir() -> Path:
|
||||
"""Path to the shared test sample documents."""
|
||||
return Path(__file__).parent / "samples" / "documents"
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def sample_doc(
|
||||
paperless_dirs: "PaperlessDirs",
|
||||
simple_digital_pdf_file: Path,
|
||||
multi_page_images_pdf_file: Path,
|
||||
thumbnail_webp_file: Path,
|
||||
document_samples_dir: Path,
|
||||
) -> "Document":
|
||||
"""Create a document with valid files and matching checksums."""
|
||||
with filelock.FileLock(paperless_dirs.media_lock):
|
||||
shutil.copy(
|
||||
simple_digital_pdf_file,
|
||||
document_samples_dir / "originals" / "0000001.pdf",
|
||||
paperless_dirs.originals_dir / "0000001.pdf",
|
||||
)
|
||||
shutil.copy(
|
||||
multi_page_images_pdf_file,
|
||||
document_samples_dir / "archive" / "0000001.pdf",
|
||||
paperless_dirs.archive_dir / "0000001.pdf",
|
||||
)
|
||||
shutil.copy(
|
||||
thumbnail_webp_file,
|
||||
document_samples_dir / "thumbnails" / "0000001.webp",
|
||||
paperless_dirs.thumbnail_dir / "0000001.webp",
|
||||
)
|
||||
|
||||
|
||||
|
After Width: | Height: | Size: 32 KiB |
|
After Width: | Height: | Size: 2.6 KiB |
|
After Width: | Height: | Size: 2.6 KiB |
|
After Width: | Height: | Size: 2.6 KiB |
@@ -0,0 +1 @@
|
||||
This is a test file.
|
||||
@@ -53,7 +53,7 @@ class TestBulkDownload(DirectoriesMixin, SampleDirMixin, APITestCase):
|
||||
archive_checksum="D",
|
||||
)
|
||||
|
||||
shutil.copy(self.SIMPLE_PDF, self.doc2.source_path)
|
||||
shutil.copy(self.SAMPLE_DIR / "simple.pdf", self.doc2.source_path)
|
||||
shutil.copy(self.SAMPLE_DIR / "simple.png", self.doc2b.source_path)
|
||||
shutil.copy(self.SAMPLE_DIR / "simple.jpg", self.doc3.source_path)
|
||||
shutil.copy(self.SAMPLE_DIR / "test_with_bom.pdf", self.doc3.archive_path)
|
||||
|
||||
@@ -1843,6 +1843,36 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
|
||||
m.assert_called_once()
|
||||
self.assertEqual(m.call_args.kwargs["pages"], [[1], [2, 3, 4], [5]])
|
||||
|
||||
@mock.patch("documents.serialisers.bulk_edit.delete_pages")
|
||||
def test_bulk_edit_delete_pages_rejects_pages_below_one(self, m) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A legacy delete_pages bulk edit
|
||||
WHEN:
|
||||
- API to bulk edit is called with a page number below 1
|
||||
THEN:
|
||||
- API returns HTTP 400
|
||||
- delete_pages is not called
|
||||
"""
|
||||
self.setup_mock(m, "delete_pages")
|
||||
|
||||
for pages in ([0], [-1], [1, 0]):
|
||||
with self.subTest(pages=pages):
|
||||
response = self.client.post(
|
||||
"/api/documents/bulk_edit/",
|
||||
json.dumps(
|
||||
{
|
||||
"documents": [self.doc2.id],
|
||||
"method": "delete_pages",
|
||||
"parameters": {"pages": pages},
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
self.assertIn(b"pages must be positive integers", response.content)
|
||||
m.assert_not_called()
|
||||
|
||||
@mock.patch("documents.views.bulk_edit.rotate")
|
||||
def test_rotate_insufficient_permissions(self, m) -> None:
|
||||
self.doc1.owner = User.objects.get(username="temp_admin")
|
||||
|
||||
@@ -56,8 +56,6 @@ from paperless_testing.http import read_streaming_response
|
||||
from paperless_testing.permissions import grant_all_global
|
||||
from paperless_testing.permissions import grant_global
|
||||
from paperless_testing.permissions import grant_object
|
||||
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
|
||||
from paperless_testing.samples import THUMBNAIL_WEBP
|
||||
|
||||
|
||||
class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
@@ -1854,11 +1852,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
id=str(uuid.uuid4()),
|
||||
)
|
||||
|
||||
with SimpleUploadedFile(
|
||||
"simple.pdf",
|
||||
SIMPLE_DIGITAL_PDF.read_bytes(),
|
||||
content_type="application/pdf",
|
||||
) as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f},
|
||||
@@ -1886,7 +1880,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
|
||||
payload = SimpleUploadedFile(
|
||||
"../../outside.pdf",
|
||||
SIMPLE_DIGITAL_PDF.read_bytes(),
|
||||
(Path(__file__).parent / "samples" / "simple.pdf").read_bytes(),
|
||||
content_type="application/pdf",
|
||||
)
|
||||
|
||||
@@ -1915,7 +1909,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
id=str(uuid.uuid4()),
|
||||
)
|
||||
|
||||
pdf_bytes = SIMPLE_DIGITAL_PDF.read_bytes()
|
||||
pdf_bytes = (Path(__file__).parent / "samples" / "simple.pdf").read_bytes()
|
||||
boundary = "paperless-boundary"
|
||||
payload = (
|
||||
(
|
||||
@@ -1991,7 +1985,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
def test_upload_insufficient_permissions(self) -> None:
|
||||
self.client.force_authenticate(user=UserFactory(username="testuser2"))
|
||||
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f},
|
||||
@@ -2004,11 +1998,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
id=str(uuid.uuid4()),
|
||||
)
|
||||
|
||||
with SimpleUploadedFile(
|
||||
"simple.pdf",
|
||||
SIMPLE_DIGITAL_PDF.read_bytes(),
|
||||
content_type="application/pdf",
|
||||
) as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{
|
||||
@@ -2041,7 +2031,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
id=str(uuid.uuid4()),
|
||||
)
|
||||
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"documenst": f},
|
||||
@@ -2067,7 +2057,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
id=str(uuid.uuid4()),
|
||||
)
|
||||
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f, "title": "my custom title"},
|
||||
@@ -2087,7 +2077,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
)
|
||||
|
||||
c = Correspondent.objects.create(name="test-corres")
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f, "correspondent": c.id},
|
||||
@@ -2106,7 +2096,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
id=str(uuid.uuid4()),
|
||||
)
|
||||
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f, "correspondent": 3456},
|
||||
@@ -2121,7 +2111,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
)
|
||||
|
||||
dt = DocumentType.objects.create(name="invoice")
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f, "document_type": dt.id},
|
||||
@@ -2140,7 +2130,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
id=str(uuid.uuid4()),
|
||||
)
|
||||
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f, "document_type": 34578},
|
||||
@@ -2155,7 +2145,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
)
|
||||
|
||||
sp = StoragePath.objects.create(name="invoices")
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f, "storage_path": sp.id},
|
||||
@@ -2174,7 +2164,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
id=str(uuid.uuid4()),
|
||||
)
|
||||
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f, "storage_path": 34578},
|
||||
@@ -2190,7 +2180,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
|
||||
t1 = Tag.objects.create(name="tag1")
|
||||
t2 = Tag.objects.create(name="tag2")
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f, "tags": [t2.id, t1.id]},
|
||||
@@ -2211,7 +2201,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
|
||||
t1 = Tag.objects.create(name="tag1")
|
||||
t2 = Tag.objects.create(name="tag2")
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f, "tags": [t2.id, t1.id, 734563]},
|
||||
@@ -2235,7 +2225,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
0,
|
||||
tzinfo=zoneinfo.ZoneInfo("America/Los_Angeles"),
|
||||
)
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f, "created": created},
|
||||
@@ -2251,11 +2241,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
id=str(uuid.uuid4()),
|
||||
)
|
||||
|
||||
with SimpleUploadedFile(
|
||||
"simple.pdf",
|
||||
SIMPLE_DIGITAL_PDF.read_bytes(),
|
||||
content_type="application/pdf",
|
||||
) as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f, "archive_serial_number": 500},
|
||||
@@ -2282,11 +2268,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
|
||||
with SimpleUploadedFile(
|
||||
"simple.pdf",
|
||||
SIMPLE_DIGITAL_PDF.read_bytes(),
|
||||
content_type="application/pdf",
|
||||
) as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{
|
||||
@@ -2341,7 +2323,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
w1.actions.add(action1)
|
||||
w1.save()
|
||||
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{
|
||||
@@ -2382,11 +2364,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
data_type=CustomField.FieldDataType.INT,
|
||||
)
|
||||
|
||||
with SimpleUploadedFile(
|
||||
"simple.pdf",
|
||||
SIMPLE_DIGITAL_PDF.read_bytes(),
|
||||
content_type="application/pdf",
|
||||
) as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{
|
||||
@@ -2435,7 +2413,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
]
|
||||
|
||||
for payload in error_payloads:
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
data = {"document": f, **payload}
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
@@ -2493,7 +2471,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
id=str(uuid.uuid4()),
|
||||
)
|
||||
|
||||
with SIMPLE_DIGITAL_PDF.open("rb") as f:
|
||||
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
|
||||
response = self.client.post(
|
||||
"/api/documents/post_document/",
|
||||
{"document": f, "from_webui": True},
|
||||
@@ -2532,8 +2510,14 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
archive_filename="archive.pdf",
|
||||
)
|
||||
|
||||
source_file: Path = THUMBNAIL_WEBP
|
||||
archive_file: Path = SIMPLE_DIGITAL_PDF
|
||||
source_file: Path = (
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "thumbnails"
|
||||
/ "0000001.webp"
|
||||
)
|
||||
archive_file: Path = Path(__file__).parent / "samples" / "simple.pdf"
|
||||
|
||||
shutil.copy(source_file, doc.source_path)
|
||||
shutil.copy(archive_file, doc.archive_path)
|
||||
@@ -2566,7 +2550,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
mime_type="application/pdf",
|
||||
)
|
||||
|
||||
shutil.copy(SIMPLE_DIGITAL_PDF, doc.source_path)
|
||||
shutil.copy(Path(__file__).parent / "samples" / "simple.pdf", doc.source_path)
|
||||
|
||||
response = self.client.get(f"/api/documents/{doc.pk}/metadata/")
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
@@ -4304,8 +4288,8 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
filename="test2.pdf",
|
||||
)
|
||||
|
||||
archive_file = SIMPLE_DIGITAL_PDF
|
||||
source_file = SIMPLE_DIGITAL_PDF
|
||||
archive_file = Path(__file__).parent / "samples" / "simple.pdf"
|
||||
source_file = Path(__file__).parent / "samples" / "simple.pdf"
|
||||
|
||||
shutil.copy(archive_file, doc.archive_path)
|
||||
shutil.copy(source_file, doc2.source_path)
|
||||
|
||||
@@ -12,9 +12,6 @@ from documents.tests.utils import SampleDirMixin
|
||||
from paperless_testing.dirs import DirectoriesMixin
|
||||
from paperless_testing.factories import UserFactory
|
||||
from paperless_testing.permissions import grant_global
|
||||
from paperless_testing.samples import MULTI_PAGE_DIGITAL_PDF
|
||||
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
|
||||
from paperless_testing.samples import WITH_FORM_PDF
|
||||
|
||||
|
||||
class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
|
||||
@@ -45,15 +42,15 @@ class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
|
||||
|
||||
# Copy sample files to document paths (using different files to distinguish versions)
|
||||
shutil.copy(
|
||||
SIMPLE_DIGITAL_PDF,
|
||||
self.SAMPLE_DIR / "documents" / "originals" / "0000001.pdf",
|
||||
self.doc1.archive_path,
|
||||
)
|
||||
shutil.copy(
|
||||
MULTI_PAGE_DIGITAL_PDF,
|
||||
self.SAMPLE_DIR / "documents" / "originals" / "0000002.pdf",
|
||||
self.doc1.source_path,
|
||||
)
|
||||
shutil.copy(
|
||||
WITH_FORM_PDF,
|
||||
self.SAMPLE_DIR / "documents" / "originals" / "0000003.pdf",
|
||||
self.doc2.source_path,
|
||||
)
|
||||
|
||||
@@ -380,7 +377,7 @@ class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
|
||||
checksum="3",
|
||||
filename="test3.pdf",
|
||||
)
|
||||
shutil.copy(self.SIMPLE_PDF, doc3.source_path)
|
||||
shutil.copy(self.SAMPLE_DIR / "simple.pdf", doc3.source_path)
|
||||
|
||||
doc4 = Document.objects.create(
|
||||
title="test1",
|
||||
@@ -389,7 +386,7 @@ class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
|
||||
checksum="4",
|
||||
filename="test4.pdf",
|
||||
)
|
||||
shutil.copy(self.SIMPLE_PDF, doc4.source_path)
|
||||
shutil.copy(self.SAMPLE_DIR / "simple.pdf", doc4.source_path)
|
||||
|
||||
response = self.client.post(
|
||||
self.ENDPOINT,
|
||||
|
||||
@@ -150,7 +150,7 @@ class TestBarcode(
|
||||
- No barcodes detected
|
||||
- No pages to split on
|
||||
"""
|
||||
test_file = self.SIMPLE_PDF
|
||||
test_file = self.SAMPLE_DIR / "simple.pdf"
|
||||
with self.get_reader(test_file) as reader:
|
||||
reader.detect()
|
||||
separator_page_numbers = reader.get_separation_pages()
|
||||
@@ -442,7 +442,7 @@ class TestBarcode(
|
||||
THEN:
|
||||
- Nothing happens
|
||||
"""
|
||||
test_file = self.SIMPLE_PDF
|
||||
test_file = self.SAMPLE_DIR / "simple.pdf"
|
||||
|
||||
with self.get_reader(test_file) as reader:
|
||||
try:
|
||||
|
||||
@@ -24,9 +24,6 @@ from documents.models import Tag
|
||||
from documents.permissions import set_permissions_for_objects
|
||||
from paperless_testing.dirs import DirectoriesMixin
|
||||
from paperless_testing.permissions import grant_object
|
||||
from paperless_testing.samples import MULTI_PAGE_DIGITAL_PDF
|
||||
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
|
||||
from paperless_testing.samples import WITH_FORM_PDF
|
||||
|
||||
|
||||
class TestBulkEdit(DirectoriesMixin, TestCase):
|
||||
@@ -704,27 +701,47 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
super().setUp()
|
||||
sample1 = self.dirs.scratch_dir / "sample.pdf"
|
||||
shutil.copy(
|
||||
SIMPLE_DIGITAL_PDF,
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000001.pdf",
|
||||
sample1,
|
||||
)
|
||||
sample1_archive = self.dirs.archive_dir / "sample_archive.pdf"
|
||||
shutil.copy(
|
||||
SIMPLE_DIGITAL_PDF,
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000001.pdf",
|
||||
sample1_archive,
|
||||
)
|
||||
sample2 = self.dirs.scratch_dir / "sample2.pdf"
|
||||
shutil.copy(
|
||||
MULTI_PAGE_DIGITAL_PDF,
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000002.pdf",
|
||||
sample2,
|
||||
)
|
||||
sample2_archive = self.dirs.archive_dir / "sample2_archive.pdf"
|
||||
shutil.copy(
|
||||
MULTI_PAGE_DIGITAL_PDF,
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000002.pdf",
|
||||
sample2_archive,
|
||||
)
|
||||
sample3 = self.dirs.scratch_dir / "sample3.pdf"
|
||||
shutil.copy(
|
||||
WITH_FORM_PDF,
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000003.pdf",
|
||||
sample3,
|
||||
)
|
||||
self.doc1 = Document.objects.create(
|
||||
@@ -758,7 +775,11 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
)
|
||||
img_doc_archive = self.dirs.archive_dir / "sample_image.pdf"
|
||||
shutil.copy(
|
||||
SIMPLE_DIGITAL_PDF,
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000001.pdf",
|
||||
img_doc_archive,
|
||||
)
|
||||
self.img_doc = Document.objects.create(
|
||||
|
||||
@@ -36,9 +36,6 @@ from paperless_testing.assertions import FileSystemAssertsMixin
|
||||
from paperless_testing.dirs import DirectoriesMixin
|
||||
from paperless_testing.factories import UserFactory
|
||||
from paperless_testing.fakes.progress import FakeProgressManager
|
||||
from paperless_testing.samples import MULTI_PAGE_DIGITAL_PDF
|
||||
from paperless_testing.samples import MULTI_PAGE_IMAGES_PDF
|
||||
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
|
||||
|
||||
|
||||
class _BaseNewStyleParser:
|
||||
@@ -115,7 +112,9 @@ class _BaseNewStyleParser:
|
||||
|
||||
|
||||
class DummyParser(_BaseNewStyleParser):
|
||||
_ARCHIVE_SRC = MULTI_PAGE_IMAGES_PDF
|
||||
_ARCHIVE_SRC = (
|
||||
Path(__file__).parent / "samples" / "documents" / "archive" / "0000001.pdf"
|
||||
)
|
||||
|
||||
def parse(self, document_path, mime_type, *, produce_archive: bool = True) -> None:
|
||||
self._text = "The Text"
|
||||
@@ -201,19 +200,33 @@ class TestConsumer(
|
||||
self.addCleanup(patcher.stop)
|
||||
|
||||
def get_test_file(self):
|
||||
src = SIMPLE_DIGITAL_PDF
|
||||
src = (
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000001.pdf"
|
||||
)
|
||||
dst = self.dirs.scratch_dir / "sample.pdf"
|
||||
shutil.copy(src, dst)
|
||||
return dst
|
||||
|
||||
def get_test_file2(self):
|
||||
src = MULTI_PAGE_DIGITAL_PDF
|
||||
src = (
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000002.pdf"
|
||||
)
|
||||
dst = self.dirs.scratch_dir / "sample2.pdf"
|
||||
shutil.copy(src, dst)
|
||||
return dst
|
||||
|
||||
def get_test_archive_file(self):
|
||||
src = MULTI_PAGE_IMAGES_PDF
|
||||
src = (
|
||||
Path(__file__).parent / "samples" / "documents" / "archive" / "0000001.pdf"
|
||||
)
|
||||
dst = self.dirs.scratch_dir / "sample_archive.pdf"
|
||||
shutil.copy(src, dst)
|
||||
return dst
|
||||
@@ -1049,7 +1062,7 @@ class TestConsumer(
|
||||
@mock.patch("documents.consumer.get_parser_registry")
|
||||
def test_similar_filenames(self, m) -> None:
|
||||
shutil.copy(
|
||||
SIMPLE_DIGITAL_PDF,
|
||||
Path(__file__).parent / "samples" / "simple.pdf",
|
||||
settings.CONSUMPTION_DIR / "simple.pdf",
|
||||
)
|
||||
shutil.copy(
|
||||
@@ -1590,7 +1603,13 @@ class TestConsumerRemoteOCR(
|
||||
self.addCleanup(patcher.stop)
|
||||
|
||||
def _consume(self, *, overrides: DocumentMetadataOverrides | None = None) -> bool:
|
||||
src = SIMPLE_DIGITAL_PDF
|
||||
src = (
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000001.pdf"
|
||||
)
|
||||
dst = self.dirs.scratch_dir / "sample.pdf"
|
||||
shutil.copy(src, dst)
|
||||
|
||||
|
||||
@@ -37,15 +37,12 @@ class TestDoubleSided(
|
||||
self.double_sided_dir.mkdir()
|
||||
self.staging_file = self.dirs.scratch_dir / STAGING_FILE_NAME
|
||||
|
||||
def _sample(self, name: str) -> Path:
|
||||
return self.SIMPLE_PDF if name == "simple.pdf" else self.SAMPLE_DIR / name
|
||||
|
||||
def consume_file(self, srcname, dstname: str | Path = "foo.pdf"):
|
||||
"""
|
||||
Starts the consume process and also ensures the
|
||||
destination file does not exist afterwards
|
||||
"""
|
||||
src = self._sample(srcname)
|
||||
src = self.SAMPLE_DIR / srcname
|
||||
dst = self.double_sided_dir / dstname
|
||||
dst.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copy(src, dst)
|
||||
@@ -60,7 +57,7 @@ class TestDoubleSided(
|
||||
return msg
|
||||
|
||||
def create_staging_file(self, src="double-sided-odd.pdf", datetime=None) -> None:
|
||||
shutil.copy(self._sample(src), self.staging_file)
|
||||
shutil.copy(self.SAMPLE_DIR / src, self.staging_file)
|
||||
if datetime is None:
|
||||
datetime = dt.datetime.now()
|
||||
os.utime(str(self.staging_file), (datetime.timestamp(),) * 2)
|
||||
|
||||
@@ -22,9 +22,8 @@ from documents.models import Document
|
||||
from documents.tasks import update_document_content_maybe_archive_file
|
||||
from paperless_testing.assertions import FileSystemAssertsMixin
|
||||
from paperless_testing.dirs import DirectoriesMixin
|
||||
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
|
||||
|
||||
sample_file: Path = SIMPLE_DIGITAL_PDF
|
||||
sample_file: Path = Path(__file__).parent / "samples" / "simple.pdf"
|
||||
|
||||
|
||||
@pytest.mark.management
|
||||
|
||||
@@ -51,7 +51,6 @@ from paperless_testing.assertions import FileSystemAssertsMixin
|
||||
from paperless_testing.dirs import DirectoriesMixin
|
||||
from paperless_testing.dirs import paperless_environment
|
||||
from paperless_testing.permissions import grant_object
|
||||
from paperless_testing.samples import install_document_samples
|
||||
|
||||
|
||||
@pytest.mark.management
|
||||
@@ -197,7 +196,10 @@ class TestExportImport(
|
||||
|
||||
def test_exporter(self, *, use_filename_format=False) -> None:
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
num_permission_objects = Permission.objects.count()
|
||||
|
||||
@@ -301,7 +303,10 @@ class TestExportImport(
|
||||
|
||||
def test_exporter_with_filename_format(self) -> None:
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
with override_settings(
|
||||
FILENAME_FORMAT="{created_year}/{correspondent}/{title}",
|
||||
@@ -310,7 +315,10 @@ class TestExportImport(
|
||||
|
||||
def test_exporter_includes_share_links_and_bundles(self) -> None:
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
share_link = ShareLink.objects.create(
|
||||
slug="share-link-slug",
|
||||
@@ -409,7 +417,10 @@ class TestExportImport(
|
||||
|
||||
def test_update_export_changed_time(self) -> None:
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
self._do_export()
|
||||
self.assertIsFile(self.target / "manifest.json")
|
||||
@@ -445,7 +456,10 @@ class TestExportImport(
|
||||
|
||||
def test_update_export_changed_checksum(self) -> None:
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
self._do_export()
|
||||
|
||||
@@ -472,7 +486,10 @@ class TestExportImport(
|
||||
|
||||
def test_update_export_deleted_document(self) -> None:
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
manifest = self._do_export()
|
||||
|
||||
@@ -504,7 +521,10 @@ class TestExportImport(
|
||||
@override_settings(FILENAME_FORMAT="{title}/{correspondent}")
|
||||
def test_update_export_changed_location(self) -> None:
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
self._do_export(use_filename_format=True)
|
||||
self.assertIsFile(self.target / "wow1" / "c.pdf")
|
||||
@@ -546,7 +566,10 @@ class TestExportImport(
|
||||
- Zipfile contains exported files
|
||||
"""
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
args = ["document_exporter", self.target, "--zip"]
|
||||
|
||||
@@ -575,7 +598,10 @@ class TestExportImport(
|
||||
- Zipfile contains exported files
|
||||
"""
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
args = ["document_exporter", self.target, "--zip", "--use-filename-format"]
|
||||
|
||||
@@ -611,7 +637,10 @@ class TestExportImport(
|
||||
- The existing file and directory in target are removed
|
||||
"""
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
# Create stuff in target directory
|
||||
existing_file = self.target / "test.txt"
|
||||
@@ -707,7 +736,10 @@ class TestExportImport(
|
||||
- Documents can be imported again
|
||||
"""
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
manifest = self._do_export()
|
||||
has_archive = False
|
||||
@@ -750,7 +782,10 @@ class TestExportImport(
|
||||
- Documents can be imported again
|
||||
"""
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
manifest = self._do_export()
|
||||
has_thumbnail = False
|
||||
@@ -795,7 +830,10 @@ class TestExportImport(
|
||||
- Documents can be imported again
|
||||
"""
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
manifest = self._do_export(split_manifest=True)
|
||||
has_document = False
|
||||
@@ -828,7 +866,10 @@ class TestExportImport(
|
||||
- Documents can be imported again
|
||||
"""
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
self._do_export(use_folder_prefix=True)
|
||||
|
||||
@@ -855,7 +896,10 @@ class TestExportImport(
|
||||
- Documents can be imported again
|
||||
"""
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
self._do_export(use_folder_prefix=True, split_manifest=True)
|
||||
|
||||
@@ -882,7 +926,10 @@ class TestExportImport(
|
||||
"""
|
||||
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
num_content_type_objects = ContentType.objects.count()
|
||||
num_permission_objects = Permission.objects.count()
|
||||
@@ -919,7 +966,10 @@ class TestExportImport(
|
||||
|
||||
def test_exporter_with_auditlog_disabled(self) -> None:
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
with override_settings(
|
||||
AUDIT_LOG_ENABLED=False,
|
||||
@@ -939,7 +989,10 @@ class TestExportImport(
|
||||
survive the round-trip with deleted_at preserved
|
||||
"""
|
||||
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
|
||||
install_document_samples(Path(self.dirs.media_dir) / "documents")
|
||||
shutil.copytree(
|
||||
Path(__file__).parent / "samples" / "documents",
|
||||
Path(self.dirs.media_dir) / "documents",
|
||||
)
|
||||
|
||||
# d1 has self.note and self.cfi1 attached via setUp
|
||||
self.d1.delete()
|
||||
@@ -983,7 +1036,10 @@ class TestExportImport(
|
||||
"""
|
||||
|
||||
shutil.rmtree(self.dirs.media_dir / "documents")
|
||||
install_document_samples(self.dirs.media_dir / "documents")
|
||||
shutil.copytree(
|
||||
self.SAMPLE_DIR / "documents",
|
||||
self.dirs.media_dir / "documents",
|
||||
)
|
||||
|
||||
_ = self._do_export(data_only=True)
|
||||
|
||||
|
||||
@@ -11,7 +11,6 @@ from documents.models import Document
|
||||
from documents.parsers import get_default_thumbnail
|
||||
from paperless_testing.assertions import FileSystemAssertsMixin
|
||||
from paperless_testing.dirs import DirectoriesMixin
|
||||
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
|
||||
|
||||
|
||||
@pytest.mark.management
|
||||
@@ -25,7 +24,7 @@ class TestMakeThumbnails(DirectoriesMixin, FileSystemAssertsMixin, TestCase):
|
||||
filename="test.pdf",
|
||||
)
|
||||
shutil.copy(
|
||||
SIMPLE_DIGITAL_PDF,
|
||||
Path(__file__).parent / "samples" / "simple.pdf",
|
||||
self.d1.source_path,
|
||||
)
|
||||
|
||||
@@ -37,7 +36,7 @@ class TestMakeThumbnails(DirectoriesMixin, FileSystemAssertsMixin, TestCase):
|
||||
filename="test2.pdf",
|
||||
)
|
||||
shutil.copy(
|
||||
SIMPLE_DIGITAL_PDF,
|
||||
Path(__file__).parent / "samples" / "simple.pdf",
|
||||
self.d2.source_path,
|
||||
)
|
||||
|
||||
|
||||
@@ -0,0 +1,386 @@
|
||||
"""
|
||||
Tests for documents.pdf_ops.
|
||||
|
||||
These use real PDFs from the sample directories. No database, Celery or mocks.
|
||||
Pages are compared by a hash of their content stream, so page identity and order
|
||||
are easy to assert.
|
||||
"""
|
||||
|
||||
import ast
|
||||
import hashlib
|
||||
from collections.abc import Callable
|
||||
from pathlib import Path
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
|
||||
from documents import pdf_ops
|
||||
from documents.pdf_ops import PageSpec
|
||||
|
||||
SRC_ROOT = Path(__file__).parents[2]
|
||||
SAMPLES = Path(__file__).parent / "samples"
|
||||
THREE_PAGES = SAMPLES / "documents" / "originals" / "0000002.pdf"
|
||||
TWELVE_PAGES = SAMPLES / "barcodes" / "split-by-asn-2.pdf"
|
||||
ENCRYPTED = SAMPLES / "password-is-test.pdf"
|
||||
SIGNED = SRC_ROOT / "paperless" / "tests" / "samples" / "tesseract" / "signed.pdf"
|
||||
|
||||
|
||||
def _page_fingerprint(page: pikepdf.Page) -> str:
|
||||
contents = page.obj.get("/Contents")
|
||||
assert contents is not None, "sample page has no /Contents"
|
||||
streams = list(contents) if isinstance(contents, pikepdf.Array) else [contents]
|
||||
return hashlib.sha1(b"".join(s.read_bytes() for s in streams)).hexdigest()
|
||||
|
||||
|
||||
def fingerprints(path: Path) -> list[str]:
|
||||
with pikepdf.open(path) as pdf:
|
||||
return [_page_fingerprint(page) for page in pdf.pages]
|
||||
|
||||
|
||||
def rotations(path: Path) -> list[int]:
|
||||
with pikepdf.open(path) as pdf:
|
||||
return [int(page.obj.get("/Rotate", 0)) for page in pdf.pages]
|
||||
|
||||
|
||||
def docinfo_keys(path: Path) -> set[str]:
|
||||
with pikepdf.open(path) as pdf:
|
||||
return set(pdf.docinfo.keys())
|
||||
|
||||
|
||||
def constant(path: Path) -> Callable[[], Path]:
|
||||
return lambda: path
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def source_fingerprints() -> list[str]:
|
||||
fps = fingerprints(THREE_PAGES)
|
||||
assert len(set(fps)) == 3, "sample must have three distinct pages"
|
||||
return fps
|
||||
|
||||
|
||||
class TestRotatePdf:
|
||||
def test_rotation_is_relative_and_applies_to_every_page(self, tmp_path: Path):
|
||||
once = tmp_path / "once.pdf"
|
||||
twice = tmp_path / "twice.pdf"
|
||||
|
||||
pdf_ops.rotate_pdf(THREE_PAGES, once, 90)
|
||||
pdf_ops.rotate_pdf(once, twice, 90)
|
||||
|
||||
assert rotations(once) == [90, 90, 90]
|
||||
assert rotations(twice) == [180, 180, 180]
|
||||
assert fingerprints(twice) == fingerprints(THREE_PAGES)
|
||||
|
||||
def test_keeps_document_info(self, tmp_path: Path):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.rotate_pdf(THREE_PAGES, dst, 90)
|
||||
|
||||
assert "/Creator" in docinfo_keys(dst)
|
||||
|
||||
|
||||
class TestRemovePages:
|
||||
def test_removes_selected_pages(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
source_fingerprints: list[str],
|
||||
):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.remove_pages(THREE_PAGES, dst, [2])
|
||||
|
||||
assert fingerprints(dst) == [source_fingerprints[0], source_fingerprints[2]]
|
||||
|
||||
def test_duplicate_page_numbers_remove_the_page_once(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
source_fingerprints: list[str],
|
||||
):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.remove_pages(THREE_PAGES, dst, [2, 2])
|
||||
|
||||
assert fingerprints(dst) == [source_fingerprints[0], source_fingerprints[2]]
|
||||
|
||||
def test_unordered_pages(self, tmp_path: Path, source_fingerprints: list[str]):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.remove_pages(THREE_PAGES, dst, [3, 1])
|
||||
|
||||
assert fingerprints(dst) == [source_fingerprints[1]]
|
||||
|
||||
def test_empty_list_keeps_every_page(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
source_fingerprints: list[str],
|
||||
):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.remove_pages(THREE_PAGES, dst, [])
|
||||
|
||||
assert fingerprints(dst) == source_fingerprints
|
||||
|
||||
def test_removing_every_page_writes_an_empty_pdf(self, tmp_path: Path):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.remove_pages(THREE_PAGES, dst, [1, 2, 3])
|
||||
|
||||
assert fingerprints(dst) == []
|
||||
|
||||
def test_keeps_document_info(self, tmp_path: Path):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.remove_pages(THREE_PAGES, dst, [1])
|
||||
|
||||
assert "/Creator" in docinfo_keys(dst)
|
||||
|
||||
@pytest.mark.parametrize("bad_page", [0, -1])
|
||||
def test_rejects_pages_below_one(self, tmp_path: Path, bad_page: int):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
with pytest.raises(ValueError, match="start at 1"):
|
||||
pdf_ops.remove_pages(THREE_PAGES, dst, [1, bad_page])
|
||||
|
||||
assert not dst.exists()
|
||||
|
||||
def test_page_past_the_end_raises_and_writes_nothing(self, tmp_path: Path):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
with pytest.raises(IndexError):
|
||||
pdf_ops.remove_pages(THREE_PAGES, dst, [99])
|
||||
|
||||
assert not dst.exists()
|
||||
|
||||
|
||||
class TestBuildPdfs:
|
||||
def test_selects_and_orders_pages(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
source_fingerprints: list[str],
|
||||
):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
written = pdf_ops.build_pdfs(
|
||||
THREE_PAGES,
|
||||
[([PageSpec(3), PageSpec(1)], constant(dst))],
|
||||
)
|
||||
|
||||
assert written == [dst]
|
||||
assert fingerprints(dst) == [source_fingerprints[2], source_fingerprints[0]]
|
||||
|
||||
def test_rotates_only_the_requested_pages(self, tmp_path: Path):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.build_pdfs(
|
||||
THREE_PAGES,
|
||||
[([PageSpec(1), PageSpec(2, 90), PageSpec(3, 180)], constant(dst))],
|
||||
)
|
||||
|
||||
assert rotations(dst) == [0, 90, 180]
|
||||
|
||||
def test_writes_one_file_per_output_in_order(self, tmp_path: Path):
|
||||
first = tmp_path / "first.pdf"
|
||||
second = tmp_path / "second.pdf"
|
||||
source = fingerprints(TWELVE_PAGES)
|
||||
|
||||
written = pdf_ops.build_pdfs(
|
||||
TWELVE_PAGES,
|
||||
[
|
||||
([PageSpec(p) for p in (1, 2, 3)], constant(first)),
|
||||
([PageSpec(p) for p in range(4, 13)], constant(second)),
|
||||
],
|
||||
)
|
||||
|
||||
assert written == [first, second]
|
||||
assert fingerprints(first) == source[:3]
|
||||
assert fingerprints(second) == source[3:]
|
||||
|
||||
def test_empty_page_list_writes_a_zero_page_file(self, tmp_path: Path):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
pdf_ops.build_pdfs(THREE_PAGES, [([], constant(dst))])
|
||||
|
||||
assert fingerprints(dst) == []
|
||||
|
||||
def test_does_not_carry_over_document_info(self, tmp_path: Path):
|
||||
dst = tmp_path / "out.pdf"
|
||||
assert "/Creator" in docinfo_keys(THREE_PAGES)
|
||||
|
||||
pdf_ops.build_pdfs(THREE_PAGES, [([PageSpec(1)], constant(dst))])
|
||||
|
||||
assert "/Creator" not in docinfo_keys(dst)
|
||||
|
||||
def test_destination_is_not_requested_when_a_page_is_out_of_range(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
):
|
||||
requested: list[Path] = []
|
||||
|
||||
def make_dst() -> Path:
|
||||
requested.append(tmp_path / "out.pdf")
|
||||
return requested[-1]
|
||||
|
||||
with pytest.raises(IndexError):
|
||||
pdf_ops.build_pdfs(THREE_PAGES, [([PageSpec(99)], make_dst)])
|
||||
|
||||
assert requested == []
|
||||
|
||||
@pytest.mark.parametrize("bad_page", [0, -1])
|
||||
def test_rejects_pages_below_one_before_opening_anything(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
bad_page: int,
|
||||
):
|
||||
requested: list[Path] = []
|
||||
|
||||
def make_dst() -> Path:
|
||||
requested.append(tmp_path / "out.pdf")
|
||||
return requested[-1]
|
||||
|
||||
with pytest.raises(ValueError, match="start at 1"):
|
||||
pdf_ops.build_pdfs(
|
||||
THREE_PAGES,
|
||||
[([PageSpec(1)], make_dst), ([PageSpec(bad_page)], make_dst)],
|
||||
)
|
||||
|
||||
assert requested == []
|
||||
|
||||
|
||||
class TestValidatePageOperations:
|
||||
def test_returns_the_output_count(self):
|
||||
operations = [{"page": 1}, {"page": 2}, {"page": 3}]
|
||||
|
||||
assert pdf_ops.validate_page_operations(operations, single_output=True) == 1
|
||||
|
||||
def test_gap_in_output_indices_counts_up_to_the_highest(self):
|
||||
operations = [
|
||||
{"page": 1, "doc": 0},
|
||||
{"page": 2, "doc": 2},
|
||||
{"page": 3, "doc": 0},
|
||||
]
|
||||
|
||||
count = pdf_ops.validate_page_operations(operations, single_output=False)
|
||||
|
||||
assert count == 3
|
||||
|
||||
def test_empty_operations_are_rejected(self):
|
||||
with pytest.raises(ValueError, match="index is out of bounds"):
|
||||
pdf_ops.validate_page_operations([], single_output=False)
|
||||
|
||||
def test_multiple_outputs_rejected_when_single_output_required(self):
|
||||
operations = [{"page": 1, "doc": 0}, {"page": 2, "doc": 1}]
|
||||
|
||||
with pytest.raises(ValueError, match="Multiple output documents"):
|
||||
pdf_ops.validate_page_operations(operations, single_output=True)
|
||||
|
||||
@pytest.mark.parametrize("doc", [-1, 2, 2**32])
|
||||
def test_output_index_out_of_bounds(self, doc: int):
|
||||
operations = [{"page": 1, "doc": 0}, {"page": 2, "doc": doc}]
|
||||
|
||||
with pytest.raises(ValueError, match="index is out of bounds"):
|
||||
pdf_ops.validate_page_operations(operations, single_output=False)
|
||||
|
||||
|
||||
class TestDecrypt:
|
||||
def test_needs_decrypt(self):
|
||||
assert pdf_ops.needs_decrypt(ENCRYPTED) is True
|
||||
assert pdf_ops.needs_decrypt(THREE_PAGES) is False
|
||||
|
||||
def test_pdf_that_opens_without_a_password_but_is_flagged_encrypted(self):
|
||||
assert pdf_ops.needs_decrypt(SIGNED) is True
|
||||
|
||||
def test_decrypt_writes_an_unencrypted_copy(self, tmp_path: Path):
|
||||
dst = tmp_path / "out.pdf"
|
||||
|
||||
result = pdf_ops.decrypt_pdf(ENCRYPTED, constant(dst), "test")
|
||||
|
||||
assert result == dst
|
||||
assert pdf_ops.needs_decrypt(dst) is False
|
||||
|
||||
def test_wrong_password_raises_and_never_requests_a_destination(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
):
|
||||
requested: list[Path] = []
|
||||
|
||||
def make_dst() -> Path:
|
||||
requested.append(tmp_path / "out.pdf")
|
||||
return requested[-1]
|
||||
|
||||
with pytest.raises(pikepdf.PasswordError):
|
||||
pdf_ops.decrypt_pdf(ENCRYPTED, make_dst, "wrong")
|
||||
|
||||
assert requested == []
|
||||
|
||||
|
||||
class TestPdfMerger:
|
||||
def test_pages_are_appended_in_the_order_added(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
source_fingerprints: list[str],
|
||||
):
|
||||
reordered = tmp_path / "reordered.pdf"
|
||||
merged = tmp_path / "merged.pdf"
|
||||
pdf_ops.build_pdfs(
|
||||
THREE_PAGES,
|
||||
[([PageSpec(3), PageSpec(1)], constant(reordered))],
|
||||
)
|
||||
|
||||
with pdf_ops.PdfMerger() as merger:
|
||||
merger.add(reordered)
|
||||
merger.add(THREE_PAGES)
|
||||
merger.save(merged)
|
||||
|
||||
assert fingerprints(merged) == [
|
||||
source_fingerprints[2],
|
||||
source_fingerprints[0],
|
||||
*source_fingerprints,
|
||||
]
|
||||
|
||||
def test_output_version_is_at_least_the_highest_source_version(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
):
|
||||
merged = tmp_path / "merged.pdf"
|
||||
with pikepdf.open(TWELVE_PAGES) as pdf:
|
||||
source_versions = [pdf.pdf_version]
|
||||
with pikepdf.open(THREE_PAGES) as pdf:
|
||||
source_versions.append(pdf.pdf_version)
|
||||
|
||||
with pdf_ops.PdfMerger() as merger:
|
||||
merger.add(TWELVE_PAGES)
|
||||
merger.add(THREE_PAGES)
|
||||
merger.save(merged)
|
||||
|
||||
with pikepdf.open(merged) as pdf:
|
||||
assert pdf.pdf_version >= max(source_versions)
|
||||
|
||||
def test_unreadable_source_raises_so_the_caller_can_skip_it(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
):
|
||||
garbage = tmp_path / "garbage.pdf"
|
||||
garbage.write_bytes(b"not a pdf")
|
||||
|
||||
with pdf_ops.PdfMerger() as merger:
|
||||
with pytest.raises(pikepdf.PdfError):
|
||||
merger.add(garbage)
|
||||
|
||||
|
||||
def test_pdf_ops_imports_only_the_standard_library_and_pikepdf():
|
||||
tree = ast.parse(Path(pdf_ops.__file__).read_text())
|
||||
imported: set[str] = set()
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, ast.Import):
|
||||
imported.update(alias.name.split(".")[0] for alias in node.names)
|
||||
elif isinstance(node, ast.ImportFrom) and node.level == 0 and node.module:
|
||||
imported.add(node.module.split(".")[0])
|
||||
|
||||
coupled = imported & {
|
||||
"django",
|
||||
"celery",
|
||||
"documents",
|
||||
"paperless",
|
||||
"paperless_mail",
|
||||
"paperless_ai",
|
||||
}
|
||||
assert not coupled
|
||||
@@ -1,61 +0,0 @@
|
||||
import hashlib
|
||||
from collections import defaultdict
|
||||
from pathlib import Path
|
||||
|
||||
from paperless_testing.samples import SHARED_SAMPLES_DIR
|
||||
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
|
||||
from paperless_testing.samples import install_document_samples
|
||||
|
||||
EXPECTED_MEDIA_FILES = [
|
||||
"originals/0000001.pdf",
|
||||
"originals/0000002.pdf",
|
||||
"originals/0000003.pdf",
|
||||
"originals/0000004.pdf",
|
||||
"originals/0000005.pdf",
|
||||
"originals/0000006.pdf",
|
||||
"archive/0000001.pdf",
|
||||
"thumbnails/0000001.webp",
|
||||
"thumbnails/0000002.webp",
|
||||
"thumbnails/0000003.webp",
|
||||
"thumbnails/0000004.webp",
|
||||
]
|
||||
|
||||
|
||||
def _sha(path: Path) -> str:
|
||||
return hashlib.sha256(path.read_bytes()).hexdigest()
|
||||
|
||||
|
||||
def test_install_recreates_every_historical_opaque_name(tmp_path: Path) -> None:
|
||||
install_document_samples(tmp_path)
|
||||
for relpath in EXPECTED_MEDIA_FILES:
|
||||
assert (tmp_path / relpath).is_file(), relpath
|
||||
|
||||
|
||||
def test_install_uses_the_shared_bytes(tmp_path: Path) -> None:
|
||||
install_document_samples(tmp_path)
|
||||
assert _sha(tmp_path / "originals" / "0000001.pdf") == _sha(SIMPLE_DIGITAL_PDF)
|
||||
|
||||
|
||||
APP_SAMPLE_TREES = [
|
||||
Path(__file__).parent / "samples",
|
||||
Path(__file__).parents[2] / "paperless" / "tests" / "samples",
|
||||
SHARED_SAMPLES_DIR,
|
||||
]
|
||||
|
||||
|
||||
def test_no_two_sample_files_have_identical_bytes() -> None:
|
||||
by_hash: dict[str, list[Path]] = defaultdict(list)
|
||||
for tree in APP_SAMPLE_TREES:
|
||||
assert tree.is_dir(), f"Sample tree does not exist: {tree}"
|
||||
for path in tree.rglob("*"):
|
||||
# Skip empty placeholders, which would all hash identically
|
||||
if path.is_file() and path.stat().st_size > 0:
|
||||
by_hash[_sha(path)].append(path)
|
||||
assert by_hash, "No sample files were found to compare"
|
||||
duplicates = {h: ps for h, ps in by_hash.items() if len(ps) > 1}
|
||||
lines = [" " + ", ".join(str(p) for p in ps) for ps in duplicates.values()]
|
||||
message = (
|
||||
"Duplicate sample files, move one to paperless_testing/sample_files:\n"
|
||||
+ "\n".join(lines)
|
||||
)
|
||||
assert not duplicates, message
|
||||
@@ -20,7 +20,6 @@ from documents.sanity_checker import SanityCheckMessages
|
||||
from documents.tests.helpers import dummy_preprocess
|
||||
from paperless_testing.assertions import FileSystemAssertsMixin
|
||||
from paperless_testing.dirs import DirectoriesMixin
|
||||
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
|
||||
|
||||
|
||||
@pytest.mark.django_db
|
||||
@@ -229,12 +228,20 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
|
||||
"""
|
||||
sample1 = self.dirs.scratch_dir / "sample.pdf"
|
||||
shutil.copy(
|
||||
SIMPLE_DIGITAL_PDF,
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000001.pdf",
|
||||
sample1,
|
||||
)
|
||||
sample1_archive = self.dirs.archive_dir / "sample_archive.pdf"
|
||||
shutil.copy(
|
||||
SIMPLE_DIGITAL_PDF,
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000001.pdf",
|
||||
sample1_archive,
|
||||
)
|
||||
doc = Document.objects.create(
|
||||
@@ -262,7 +269,11 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
|
||||
"""
|
||||
sample1 = self.dirs.scratch_dir / "sample.pdf"
|
||||
shutil.copy(
|
||||
SIMPLE_DIGITAL_PDF,
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000001.pdf",
|
||||
sample1,
|
||||
)
|
||||
doc = Document.objects.create(
|
||||
|
||||
@@ -178,7 +178,7 @@ class TestWorkflows(
|
||||
self.assertEqual(action.__str__(), "WorkflowAction 1")
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -290,7 +290,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -409,7 +409,7 @@ class TestWorkflows(
|
||||
w2.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -478,7 +478,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -531,7 +531,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -608,7 +608,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -687,7 +687,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -765,7 +765,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -876,7 +876,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -928,7 +928,7 @@ class TestWorkflows(
|
||||
generated = generate_unique_filename(doc)
|
||||
destination = (settings.ORIGINALS_DIR / generated).resolve()
|
||||
create_source_path_directory(destination)
|
||||
shutil.copy(self.SIMPLE_PDF, destination)
|
||||
shutil.copy(self.SAMPLE_DIR / "simple.pdf", destination)
|
||||
Document.objects.filter(pk=doc.pk).update(filename=generated.as_posix())
|
||||
doc.refresh_from_db()
|
||||
|
||||
@@ -2023,7 +2023,7 @@ class TestWorkflows(
|
||||
superuser = UserFactory(username="superuser", superuser=True)
|
||||
self.client.force_authenticate(user=superuser)
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
tasks.consume_file(
|
||||
@@ -3009,7 +3009,7 @@ class TestWorkflows(
|
||||
generated = generate_unique_filename(doc)
|
||||
destination = (settings.ORIGINALS_DIR / generated).resolve()
|
||||
create_source_path_directory(destination)
|
||||
shutil.copy(self.SIMPLE_PDF, destination)
|
||||
shutil.copy(self.SAMPLE_DIR / "simple.pdf", destination)
|
||||
Document.objects.filter(pk=doc.pk).update(filename=generated.as_posix())
|
||||
doc.refresh_from_db()
|
||||
doc.tags.set([self.t1, self.t2])
|
||||
@@ -3073,7 +3073,7 @@ class TestWorkflows(
|
||||
generated = generate_unique_filename(doc)
|
||||
destination = (settings.ORIGINALS_DIR / generated).resolve()
|
||||
create_source_path_directory(destination)
|
||||
shutil.copy(self.SIMPLE_PDF, destination)
|
||||
shutil.copy(self.SAMPLE_DIR / "simple.pdf", destination)
|
||||
Document.objects.filter(pk=doc.pk).update(filename=generated.as_posix())
|
||||
doc.refresh_from_db()
|
||||
doc.tags.set([self.t1])
|
||||
@@ -3222,7 +3222,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -3346,7 +3346,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -3496,7 +3496,7 @@ class TestWorkflows(
|
||||
workflow.actions.set([assignment_action, email_action])
|
||||
|
||||
temp_working_copy = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "working-copy.pdf",
|
||||
)
|
||||
|
||||
@@ -3596,7 +3596,7 @@ class TestWorkflows(
|
||||
|
||||
# move the file
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -3704,7 +3704,7 @@ class TestWorkflows(
|
||||
generated = generate_unique_filename(doc)
|
||||
destination = (settings.ORIGINALS_DIR / generated).resolve()
|
||||
create_source_path_directory(destination)
|
||||
shutil.copy(self.SIMPLE_PDF, destination)
|
||||
shutil.copy(self.SAMPLE_DIR / "simple.pdf", destination)
|
||||
Document.objects.filter(pk=doc.pk).update(filename=generated.as_posix())
|
||||
|
||||
run_workflows(WorkflowTrigger.WorkflowTriggerType.DOCUMENT_UPDATED, doc)
|
||||
@@ -3915,7 +3915,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -4036,7 +4036,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -4387,7 +4387,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -4835,7 +4835,7 @@ class TestWorkflows(
|
||||
w.save()
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -5027,7 +5027,7 @@ class TestWorkflows(
|
||||
|
||||
# Create a test file to be consumed
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
test_file_path = Path(test_file)
|
||||
@@ -5091,7 +5091,7 @@ class TestWorkflows(
|
||||
|
||||
# Create a test file to be consumed
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple2.pdf",
|
||||
)
|
||||
test_file_path = Path(test_file)
|
||||
@@ -5529,7 +5529,7 @@ class TestDateWorkflowLocalization(
|
||||
)
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
tmp_path / "simple.pdf",
|
||||
)
|
||||
|
||||
@@ -5596,7 +5596,7 @@ class TestRemoteOCRWorkflowAction(DirectoriesMixin, SampleDirMixin, APITestCase)
|
||||
self._make_workflow(WorkflowTrigger.WorkflowTriggerType.CONSUMPTION)
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
overrides = DocumentMetadataOverrides()
|
||||
@@ -5814,7 +5814,7 @@ class TestApplyAISuggestionsWorkflowAction(
|
||||
)
|
||||
|
||||
test_file = shutil.copy(
|
||||
self.SIMPLE_PDF,
|
||||
self.SAMPLE_DIR / "simple.pdf",
|
||||
self.dirs.scratch_dir / "simple.pdf",
|
||||
)
|
||||
|
||||
|
||||
@@ -11,7 +11,6 @@ from documents.data_models import ConsumableDocument
|
||||
from documents.data_models import DocumentMetadataOverrides
|
||||
from documents.data_models import DocumentSource
|
||||
from paperless_testing.fakes.progress import FakeProgressManager
|
||||
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
|
||||
|
||||
|
||||
class ConsumeTaskMixin:
|
||||
@@ -54,8 +53,6 @@ class SampleDirMixin:
|
||||
|
||||
BARCODE_SAMPLE_DIR = SAMPLE_DIR / "barcodes"
|
||||
|
||||
SIMPLE_PDF = SIMPLE_DIGITAL_PDF
|
||||
|
||||
|
||||
class GetConsumerMixin:
|
||||
@contextmanager
|
||||
|
||||
@@ -432,6 +432,30 @@ def tesseract_samples_dir(parser_samples_dir: Path) -> Path:
|
||||
return parser_samples_dir / "tesseract"
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def multi_page_images_pdf_file(tesseract_samples_dir: Path) -> Path:
|
||||
"""Path to a multi-page PDF with images.
|
||||
|
||||
Returns
|
||||
-------
|
||||
Path
|
||||
Absolute path to ``tesseract/multi-page-images.pdf``.
|
||||
"""
|
||||
return tesseract_samples_dir / "multi-page-images.pdf"
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def simple_digital_pdf_file(tesseract_samples_dir: Path) -> Path:
|
||||
"""Path to a simple digital PDF sample file.
|
||||
|
||||
Returns
|
||||
-------
|
||||
Path
|
||||
Absolute path to ``tesseract/simple-digital.pdf``.
|
||||
"""
|
||||
return tesseract_samples_dir / "simple-digital.pdf"
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def simple_no_dpi_png_file(tesseract_samples_dir: Path) -> Path:
|
||||
"""Path to a simple PNG without DPI information.
|
||||
|
||||
@@ -160,11 +160,11 @@ class TestGetPageCount:
|
||||
def test_single_page_pdf(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
simple_digital_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
assert (
|
||||
tesseract_parser.get_page_count(
|
||||
simple_digital_pdf_file,
|
||||
tesseract_samples_dir / "simple-digital.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
== 1
|
||||
@@ -187,7 +187,7 @@ class TestGetPageCount:
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
simple_digital_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
caplog,
|
||||
) -> None:
|
||||
"""
|
||||
@@ -203,7 +203,7 @@ class TestGetPageCount:
|
||||
|
||||
with caplog.at_level(logging.WARNING):
|
||||
page_count = tesseract_parser.get_page_count(
|
||||
simple_digital_pdf_file,
|
||||
tesseract_samples_dir / "simple-digital.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
assert page_count is None
|
||||
@@ -272,10 +272,10 @@ class TestGetThumbnail:
|
||||
def test_thumbnail_is_file(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
simple_digital_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
thumb = tesseract_parser.get_thumbnail(
|
||||
simple_digital_pdf_file,
|
||||
tesseract_samples_dir / "simple-digital.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
assert thumb.is_file()
|
||||
@@ -284,7 +284,7 @@ class TestGetThumbnail:
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
simple_digital_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
def _raise_on_pdf(input_file, output_file, **kwargs) -> None:
|
||||
if ".pdf" in str(input_file):
|
||||
@@ -294,7 +294,7 @@ class TestGetThumbnail:
|
||||
mocker.patch("documents.parsers.run_convert", side_effect=_raise_on_pdf)
|
||||
|
||||
thumb = tesseract_parser.get_thumbnail(
|
||||
simple_digital_pdf_file,
|
||||
tesseract_samples_dir / "simple-digital.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
assert thumb.is_file()
|
||||
@@ -320,11 +320,11 @@ class TestExtractText:
|
||||
def test_extract_text_from_digital_pdf(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
simple_digital_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
text = tesseract_parser.extract_text(
|
||||
None,
|
||||
simple_digital_pdf_file,
|
||||
tesseract_samples_dir / "simple-digital.pdf",
|
||||
)
|
||||
assert text is not None
|
||||
assert "This is a test document." in text.strip()
|
||||
@@ -339,7 +339,7 @@ class TestParsePdf:
|
||||
def test_simple_digital_creates_archive(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
multi_page_digital_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
@@ -353,7 +353,7 @@ class TestParsePdf:
|
||||
- Text is extracted
|
||||
"""
|
||||
tesseract_parser.parse(
|
||||
multi_page_digital_pdf_file,
|
||||
tesseract_samples_dir / "multi-page-digital.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
assert tesseract_parser.archive_path is not None
|
||||
@@ -368,10 +368,10 @@ class TestParsePdf:
|
||||
def test_with_form_default(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
with_form_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
tesseract_parser.parse(
|
||||
with_form_pdf_file,
|
||||
tesseract_samples_dir / "with-form.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
assert tesseract_parser.archive_path is not None
|
||||
@@ -384,11 +384,11 @@ class TestParsePdf:
|
||||
def test_with_form_redo_no_archive_when_not_requested(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
with_form_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
tesseract_parser.settings.mode = ModeChoices.REDO
|
||||
tesseract_parser.parse(
|
||||
with_form_pdf_file,
|
||||
tesseract_samples_dir / "with-form.pdf",
|
||||
"application/pdf",
|
||||
produce_archive=False,
|
||||
)
|
||||
@@ -401,11 +401,11 @@ class TestParsePdf:
|
||||
def test_with_form_force(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
with_form_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
tesseract_parser.settings.mode = ModeChoices.FORCE
|
||||
tesseract_parser.parse(
|
||||
with_form_pdf_file,
|
||||
tesseract_samples_dir / "with-form.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
assert_ordered_substrings(
|
||||
@@ -446,7 +446,7 @@ class TestParsePdf:
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
simple_digital_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
mocker.patch(
|
||||
"ocrmypdf.ocr",
|
||||
@@ -454,7 +454,7 @@ class TestParsePdf:
|
||||
)
|
||||
with pytest.raises(ParseError):
|
||||
tesseract_parser.parse(
|
||||
simple_digital_pdf_file,
|
||||
tesseract_samples_dir / "simple-digital.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
|
||||
@@ -530,10 +530,10 @@ class TestParseMultiPage:
|
||||
def test_multi_page_digital(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
multi_page_digital_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
tesseract_parser.parse(
|
||||
multi_page_digital_pdf_file,
|
||||
tesseract_samples_dir / "multi-page-digital.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
assert tesseract_parser.archive_path is not None
|
||||
@@ -557,12 +557,12 @@ class TestParseMultiPage:
|
||||
self,
|
||||
mode: str,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
multi_page_digital_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
tesseract_parser.settings.pages = 2
|
||||
tesseract_parser.settings.mode = mode
|
||||
tesseract_parser.parse(
|
||||
multi_page_digital_pdf_file,
|
||||
tesseract_samples_dir / "multi-page-digital.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
assert tesseract_parser.archive_path is not None
|
||||
@@ -576,11 +576,11 @@ class TestParseMultiPage:
|
||||
def test_multi_page_images_skip(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
multi_page_images_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
tesseract_parser.settings.mode = ModeChoices.AUTO
|
||||
tesseract_parser.parse(
|
||||
multi_page_images_pdf_file,
|
||||
tesseract_samples_dir / "multi-page-images.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
assert tesseract_parser.archive_path is not None
|
||||
@@ -594,7 +594,7 @@ class TestParseMultiPage:
|
||||
def test_multi_page_images_redo_pages_2(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
multi_page_images_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
@@ -609,7 +609,7 @@ class TestParseMultiPage:
|
||||
tesseract_parser.settings.pages = 2
|
||||
tesseract_parser.settings.mode = ModeChoices.REDO
|
||||
tesseract_parser.parse(
|
||||
multi_page_images_pdf_file,
|
||||
tesseract_samples_dir / "multi-page-images.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
assert tesseract_parser.archive_path is not None
|
||||
@@ -622,7 +622,7 @@ class TestParseMultiPage:
|
||||
def test_multi_page_images_force_page_1(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
multi_page_images_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
@@ -637,7 +637,7 @@ class TestParseMultiPage:
|
||||
tesseract_parser.settings.pages = 1
|
||||
tesseract_parser.settings.mode = ModeChoices.FORCE
|
||||
tesseract_parser.parse(
|
||||
multi_page_images_pdf_file,
|
||||
tesseract_samples_dir / "multi-page-images.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
assert tesseract_parser.archive_path is not None
|
||||
@@ -733,7 +733,7 @@ class TestSkipArchive:
|
||||
def test_skip_noarchive_with_text_layer(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
multi_page_digital_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
@@ -747,7 +747,7 @@ class TestSkipArchive:
|
||||
"""
|
||||
tesseract_parser.settings.mode = ModeChoices.AUTO
|
||||
tesseract_parser.parse(
|
||||
multi_page_digital_pdf_file,
|
||||
tesseract_samples_dir / "multi-page-digital.pdf",
|
||||
"application/pdf",
|
||||
produce_archive=False,
|
||||
)
|
||||
@@ -762,7 +762,7 @@ class TestSkipArchive:
|
||||
def test_skip_noarchive_image_only_creates_archive(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
multi_page_images_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
@@ -775,7 +775,7 @@ class TestSkipArchive:
|
||||
"""
|
||||
tesseract_parser.settings.mode = ModeChoices.AUTO
|
||||
tesseract_parser.parse(
|
||||
multi_page_images_pdf_file,
|
||||
tesseract_samples_dir / "multi-page-images.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
assert tesseract_parser.archive_path is not None
|
||||
@@ -787,29 +787,29 @@ class TestSkipArchive:
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("produce_archive", "sample_fixture", "expect_archive"),
|
||||
("produce_archive", "filename", "expect_archive"),
|
||||
[
|
||||
pytest.param(
|
||||
True,
|
||||
"multi_page_digital_pdf_file",
|
||||
"multi-page-digital.pdf",
|
||||
True,
|
||||
id="produce-archive-with-text",
|
||||
),
|
||||
pytest.param(
|
||||
True,
|
||||
"multi_page_images_pdf_file",
|
||||
"multi-page-images.pdf",
|
||||
True,
|
||||
id="produce-archive-no-text",
|
||||
),
|
||||
pytest.param(
|
||||
False,
|
||||
"multi_page_digital_pdf_file",
|
||||
"multi-page-digital.pdf",
|
||||
False,
|
||||
id="no-archive-with-text-layer",
|
||||
),
|
||||
pytest.param(
|
||||
False,
|
||||
"multi_page_images_pdf_file",
|
||||
"multi-page-images.pdf",
|
||||
False,
|
||||
id="no-archive-no-text-layer",
|
||||
),
|
||||
@@ -818,10 +818,10 @@ class TestSkipArchive:
|
||||
def test_produce_archive_flag(
|
||||
self,
|
||||
produce_archive: bool, # noqa: FBT001
|
||||
sample_fixture: str,
|
||||
filename: str,
|
||||
expect_archive: bool, # noqa: FBT001
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
request: pytest.FixtureRequest,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
@@ -834,9 +834,8 @@ class TestSkipArchive:
|
||||
- Text is always extracted
|
||||
"""
|
||||
tesseract_parser.settings.mode = ModeChoices.AUTO
|
||||
sample = request.getfixturevalue(sample_fixture)
|
||||
tesseract_parser.parse(
|
||||
sample,
|
||||
tesseract_samples_dir / filename,
|
||||
"application/pdf",
|
||||
produce_archive=produce_archive,
|
||||
)
|
||||
@@ -853,7 +852,7 @@ class TestSkipArchive:
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
simple_digital_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
@@ -869,7 +868,7 @@ class TestSkipArchive:
|
||||
tesseract_parser.settings.mode = ModeChoices.AUTO
|
||||
mock_ocr = mocker.patch("ocrmypdf.ocr")
|
||||
tesseract_parser.parse(
|
||||
simple_digital_pdf_file,
|
||||
tesseract_samples_dir / "simple-digital.pdf",
|
||||
"application/pdf",
|
||||
produce_archive=False,
|
||||
)
|
||||
@@ -908,7 +907,7 @@ class TestSkipArchive:
|
||||
def test_tagged_pdf_produces_pdfa_archive_without_ocr(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
simple_digital_pdf_file: Path,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
@@ -923,7 +922,7 @@ class TestSkipArchive:
|
||||
"""
|
||||
tesseract_parser.settings.mode = ModeChoices.AUTO
|
||||
tesseract_parser.parse(
|
||||
simple_digital_pdf_file,
|
||||
tesseract_samples_dir / "simple-digital.pdf",
|
||||
"application/pdf",
|
||||
produce_archive=True,
|
||||
)
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
<html>
|
||||
<head>
|
||||
<meta http-equiv="content-type" content="text/html; charset=UTF-8">
|
||||
</head>
|
||||
<body>
|
||||
<p>Some Text</p>
|
||||
<p>
|
||||
<img src="cid:part1.pNdUSz0s.D3NqVtPg@example.de" alt="Has to be rewritten to work..">
|
||||
<img src="http://localhost:8080/assets/logo_full_white.svg" alt="This image should not be shown.">
|
||||
</p>
|
||||
|
||||
<p>and an embedded image.<br>
|
||||
</p>
|
||||
<p id="changeme">Paragraph unchanged.</p>
|
||||
<scRipt>
|
||||
document.getElementById("changeme").innerHTML = "Paragraph changed via Java Script.";
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
|
After Width: | Height: | Size: 2.8 KiB |
|
After Width: | Height: | Size: 6.9 KiB |
|
After Width: | Height: | Size: 32 KiB |
@@ -16,6 +16,8 @@ from paperless.parsers.utils import read_file_handle_unicode_errors
|
||||
if TYPE_CHECKING:
|
||||
from pytest_mock import MockerFixture
|
||||
|
||||
SAMPLES = Path(__file__).parent / "samples" / "tesseract"
|
||||
|
||||
|
||||
class TestReadFileHandleUnicodeErrors:
|
||||
def test_plain_utf8(self, tmp_path: Path) -> None:
|
||||
@@ -53,11 +55,11 @@ class TestReadFileHandleUnicodeErrors:
|
||||
|
||||
|
||||
class TestIsTaggedPdf:
|
||||
def test_tagged_pdf_returns_true(self, simple_digital_pdf_file: Path) -> None:
|
||||
assert is_tagged_pdf(simple_digital_pdf_file) is True
|
||||
def test_tagged_pdf_returns_true(self) -> None:
|
||||
assert is_tagged_pdf(SAMPLES / "simple-digital.pdf") is True
|
||||
|
||||
def test_untagged_pdf_returns_false(self, multi_page_images_pdf_file: Path) -> None:
|
||||
assert is_tagged_pdf(multi_page_images_pdf_file) is False
|
||||
def test_untagged_pdf_returns_false(self) -> None:
|
||||
assert is_tagged_pdf(SAMPLES / "multi-page-images.pdf") is False
|
||||
|
||||
def test_nonexistent_path_returns_false(self) -> None:
|
||||
assert is_tagged_pdf(Path("/nonexistent/file.pdf")) is False
|
||||
|
||||
@@ -1,48 +0,0 @@
|
||||
"""Test sample files shared across tests.
|
||||
|
||||
A sample used by a single app stays in that app's ``tests/samples`` tree.
|
||||
Samples needed by more than one app, or stored under more than one name, live
|
||||
here exactly once.
|
||||
"""
|
||||
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
|
||||
SHARED_SAMPLES_DIR = (Path(__file__).parent / "sample_files").resolve()
|
||||
|
||||
SIMPLE_DIGITAL_PDF = SHARED_SAMPLES_DIR / "simple-digital.pdf"
|
||||
MULTI_PAGE_DIGITAL_PDF = SHARED_SAMPLES_DIR / "multi-page-digital.pdf"
|
||||
WITH_FORM_PDF = SHARED_SAMPLES_DIR / "with-form.pdf"
|
||||
MULTI_PAGE_IMAGES_PDF = SHARED_SAMPLES_DIR / "multi-page-images.pdf"
|
||||
THUMBNAIL_WEBP = SHARED_SAMPLES_DIR / "thumbnail.webp"
|
||||
|
||||
# The fake media tree under documents/tests/samples/documents keeps only the
|
||||
# files unique to it. The files that duplicate a shared sample are laid back
|
||||
# down under their historical opaque names (DB rows and exporter manifests
|
||||
# refer to them) by install_document_samples().
|
||||
_DOCUMENT_SAMPLES_DIR = (
|
||||
Path(__file__).parent.parent / "documents" / "tests" / "samples" / "documents"
|
||||
).resolve()
|
||||
|
||||
_SHARED_AS_MEDIA = (
|
||||
("originals/0000001.pdf", SIMPLE_DIGITAL_PDF),
|
||||
("originals/0000002.pdf", MULTI_PAGE_DIGITAL_PDF),
|
||||
("originals/0000003.pdf", WITH_FORM_PDF),
|
||||
("archive/0000001.pdf", MULTI_PAGE_IMAGES_PDF),
|
||||
("thumbnails/0000001.webp", THUMBNAIL_WEBP),
|
||||
("thumbnails/0000002.webp", THUMBNAIL_WEBP),
|
||||
("thumbnails/0000003.webp", THUMBNAIL_WEBP),
|
||||
("thumbnails/0000004.webp", THUMBNAIL_WEBP),
|
||||
)
|
||||
|
||||
|
||||
def install_document_samples(dest: Path) -> None:
|
||||
"""Populate ``dest`` with the full fake media tree.
|
||||
|
||||
Replaces ``shutil.copytree(<samples>/documents, dest)`` in tests.
|
||||
"""
|
||||
shutil.copytree(_DOCUMENT_SAMPLES_DIR, dest, dirs_exist_ok=True)
|
||||
for relpath, source in _SHARED_AS_MEDIA:
|
||||
target = dest / relpath
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copy2(source, target)
|
||||