Compare commits

..
Author SHA1 Message Date
shamoon 6f7ed1d4a7 Bump version to 3.3.0 2026-10-05 20:34:24 -07:00
shamoon 0428bf6955 Merge branch 'dev' 2026-10-05 20:33:36 -07:00
shamoon c63afb47b2 Documentation: correct duplicates info (#14243) 2026-09-23 08:27:18 -07:00
31 changed files with 548 additions and 1150 deletions

No files matched your search

+3 -3
View File
@@ -80,7 +80,7 @@ jobs:
needs: changes needs: changes
if: needs.changes.outputs.backend_changed == 'true' if: needs.changes.outputs.backend_changed == 'true'
name: "Python ${{ matrix.python-version }}" name: "Python ${{ matrix.python-version }}"
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
strategy: strategy:
@@ -114,7 +114,7 @@ jobs:
packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils
- name: Configure ImageMagick - name: Configure ImageMagick
run: | run: |
sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-7/policy.xml sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-6/policy.xml
- name: Install Python dependencies - name: Install Python dependencies
env: env:
PYTHON_VERSION: ${{ steps.setup-python.outputs.python-version }} PYTHON_VERSION: ${{ steps.setup-python.outputs.python-version }}
@@ -158,7 +158,7 @@ jobs:
needs: changes needs: changes
if: needs.changes.outputs.backend_changed == 'true' if: needs.changes.outputs.backend_changed == 'true'
name: Check project typing name: Check project typing
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
env: env:
+3 -3
View File
@@ -24,10 +24,10 @@ jobs:
fail-fast: false fail-fast: false
matrix: matrix:
include: include:
- runner: ubuntu-26.04 - runner: ubuntu-24.04
arch: amd64 arch: amd64
platform: linux/amd64 platform: linux/amd64
- runner: ubuntu-26.04-arm - runner: ubuntu-24.04-arm
arch: arm64 arch: arm64
platform: linux/arm64 platform: linux/arm64
runs-on: ${{ matrix.runner }} runs-on: ${{ matrix.runner }}
@@ -163,7 +163,7 @@ jobs:
archive: false archive: false
merge-and-push: merge-and-push:
name: Merge and Push Manifest name: Merge and Push Manifest
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
needs: build-arch needs: build-arch
if: needs.build-arch.outputs.should-push == 'true' if: needs.build-arch.outputs.should-push == 'true'
environment: image-publishing environment: image-publishing
+2 -2
View File
@@ -65,7 +65,7 @@ jobs:
needs: changes needs: changes
if: needs.changes.outputs.docs_changed == 'true' if: needs.changes.outputs.docs_changed == 'true'
name: Build Documentation name: Build Documentation
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
steps: steps:
- uses: actions/configure-pages@45bfe0192ca1faeb007ade9deae92b16b8254a0d # v6.0.0 - uses: actions/configure-pages@45bfe0192ca1faeb007ade9deae92b16b8254a0d # v6.0.0
- name: Checkout - name: Checkout
@@ -102,7 +102,7 @@ jobs:
name: Deploy Documentation name: Deploy Documentation
needs: [changes, build] needs: [changes, build]
if: github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.changes.outputs.docs_changed == 'true' if: github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.changes.outputs.docs_changed == 'true'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
pages: write pages: write
id-token: write id-token: write
+6 -6
View File
@@ -72,7 +72,7 @@ jobs:
needs: changes needs: changes
if: needs.changes.outputs.frontend_changed == 'true' if: needs.changes.outputs.frontend_changed == 'true'
name: Install Dependencies name: Install Dependencies
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
steps: steps:
@@ -104,7 +104,7 @@ jobs:
name: Lint name: Lint
needs: [changes, install-dependencies] needs: [changes, install-dependencies]
if: needs.changes.outputs.frontend_changed == 'true' if: needs.changes.outputs.frontend_changed == 'true'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
steps: steps:
@@ -137,7 +137,7 @@ jobs:
name: "Unit Tests (${{ matrix.shard-index }}/${{ matrix.shard-count }})" name: "Unit Tests (${{ matrix.shard-index }}/${{ matrix.shard-count }})"
needs: [changes, install-dependencies] needs: [changes, install-dependencies]
if: needs.changes.outputs.frontend_changed == 'true' if: needs.changes.outputs.frontend_changed == 'true'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
strategy: strategy:
@@ -188,10 +188,10 @@ jobs:
name: E2E Tests name: E2E Tests
needs: [changes, install-dependencies] needs: [changes, install-dependencies]
if: needs.changes.outputs.frontend_changed == 'true' if: needs.changes.outputs.frontend_changed == 'true'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
container: mcr.microsoft.com/playwright:v1.62.1-resolute container: mcr.microsoft.com/playwright:v1.62.1-noble
env: env:
PLAYWRIGHT_BROWSERS_PATH: /ms-playwright PLAYWRIGHT_BROWSERS_PATH: /ms-playwright
PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD: 1 PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD: 1
@@ -246,7 +246,7 @@ jobs:
name: Frontend Build name: Frontend Build
needs: [changes, unit-tests, e2e-tests] needs: [changes, unit-tests, e2e-tests]
if: needs.changes.outputs.frontend_changed == 'true' if: needs.changes.outputs.frontend_changed == 'true'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
steps: steps:
+5 -5
View File
@@ -14,7 +14,7 @@ permissions: {}
jobs: jobs:
wait-for-docker: wait-for-docker:
name: Wait for Docker Build name: Wait for Docker Build
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
checks: read checks: read
statuses: read statuses: read
@@ -30,7 +30,7 @@ jobs:
build-release: build-release:
name: Build Release name: Build Release
needs: wait-for-docker needs: wait-for-docker
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
steps: steps:
@@ -73,7 +73,7 @@ jobs:
timeout-minutes: 12 timeout-minutes: 12
uses: $/.github/actions/apt-install uses: $/.github/actions/apt-install
with: with:
packages: gettext libleptonica6 packages: gettext liblept5
# ---- Build Documentation ---- # ---- Build Documentation ----
- name: Build documentation - name: Build documentation
env: env:
@@ -145,7 +145,7 @@ jobs:
publish-release: publish-release:
name: Publish Release name: Publish Release
needs: build-release needs: build-release
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: write contents: write
pull-requests: write pull-requests: write
@@ -197,7 +197,7 @@ jobs:
name: Append Changelog name: Append Changelog
needs: publish-release needs: publish-release
if: needs.publish-release.outputs.prerelease == 'false' if: needs.publish-release.outputs.prerelease == 'false'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: write contents: write
pull-requests: write pull-requests: write
+2 -2
View File
@@ -15,7 +15,7 @@ permissions:
jobs: jobs:
zizmor: zizmor:
name: Run zizmor name: Run zizmor
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
actions: read actions: read
@@ -29,7 +29,7 @@ jobs:
uses: zizmorcore/zizmor-action@cc914d7f3750a2d13d75c7f184a1060aa0e9d482 # v0.6.4 uses: zizmorcore/zizmor-action@cc914d7f3750a2d13d75c7f184a1060aa0e9d482 # v0.6.4
semgrep: semgrep:
name: Semgrep CE name: Semgrep CE
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
container: container:
image: semgrep/semgrep:1.155.0@sha256:cc869c685dcc0fe497c86258da9f205397d8108e56d21a86082ea4886e52784d image: semgrep/semgrep:1.155.0@sha256:cc869c685dcc0fe497c86258da9f205397d8108e56d21a86082ea4886e52784d
if: github.actor != 'dependabot[bot]' if: github.actor != 'dependabot[bot]'
+2 -2
View File
@@ -17,7 +17,7 @@ jobs:
cleanup-images: cleanup-images:
name: Cleanup Image Tags for ${{ matrix.primary-name }} name: Cleanup Image Tags for ${{ matrix.primary-name }}
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
environment: registry-maintenance environment: registry-maintenance
strategy: strategy:
fail-fast: false fail-fast: false
@@ -42,7 +42,7 @@ jobs:
cleanup-untagged-images: cleanup-untagged-images:
name: Cleanup Untagged Images Tags for ${{ matrix.primary-name }} name: Cleanup Untagged Images Tags for ${{ matrix.primary-name }}
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
needs: needs:
- cleanup-images - cleanup-images
environment: registry-maintenance environment: registry-maintenance
+1 -1
View File
@@ -21,7 +21,7 @@ on:
jobs: jobs:
analyze: analyze:
name: Analyze name: Analyze
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
actions: read actions: read
contents: read contents: read
+1 -1
View File
@@ -13,7 +13,7 @@ jobs:
synchronize-with-crowdin: synchronize-with-crowdin:
name: Crowdin Sync name: Crowdin Sync
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
environment: translation-sync environment: translation-sync
steps: steps:
- name: Checkout - name: Checkout
+1 -1
View File
@@ -7,7 +7,7 @@ jobs:
# Note: peakoss/anti-slop does not support the `issues` event yet (all of its # Note: peakoss/anti-slop does not support the `issues` event yet (all of its
# issue inputs are still commented out upstream), so the checks that the PR Bot # issue inputs are still commented out upstream), so the checks that the PR Bot
# workflow gets from the action are implemented manually here. # workflow gets from the action are implemented manually here.
runs-on: ubuntu-slim runs-on: ubuntu-latest
permissions: permissions:
issues: write issues: write
steps: steps:
+2 -2
View File
@@ -4,7 +4,7 @@ on:
types: [opened] types: [opened]
jobs: jobs:
Anti-slop: Anti-slop:
runs-on: ubuntu-slim runs-on: ubuntu-latest
permissions: permissions:
contents: read contents: read
issues: read issues: read
@@ -24,7 +24,7 @@ jobs:
ASLOP-PR-VERIFY ASLOP-PR-VERIFY
pr-bot: pr-bot:
name: Automated PR Bot name: Automated PR Bot
runs-on: ubuntu-slim runs-on: ubuntu-latest
# Runs after Anti-slop so the welcome comment can see whether the PR was closed # Runs after Anti-slop so the welcome comment can see whether the PR was closed
# instead of racing it. Still runs if that job fails, so labeling is not lost. # instead of racing it. Still runs if that job fails, so labeling is not lost.
needs: Anti-slop needs: Anti-slop
+1 -1
View File
@@ -12,7 +12,7 @@ permissions:
jobs: jobs:
pr_opened_or_reopened: pr_opened_or_reopened:
name: pr_opened_or_reopened name: pr_opened_or_reopened
runs-on: ubuntu-slim runs-on: ubuntu-24.04
permissions: permissions:
# write permission is required for autolabeler # write permission is required for autolabeler
pull-requests: write pull-requests: write
+5 -5
View File
@@ -9,7 +9,7 @@ jobs:
stale: stale:
name: 'Stale' name: 'Stale'
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
issues: write issues: write
pull-requests: write pull-requests: write
@@ -34,7 +34,7 @@ jobs:
lock-threads: lock-threads:
name: 'Lock Old Threads' name: 'Lock Old Threads'
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
issues: write issues: write
pull-requests: write pull-requests: write
@@ -58,7 +58,7 @@ jobs:
close-answered-discussions: close-answered-discussions:
name: 'Close Answered Discussions' name: 'Close Answered Discussions'
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-slim runs-on: ubuntu-24.04
permissions: permissions:
discussions: write discussions: write
steps: steps:
@@ -117,7 +117,7 @@ jobs:
close-outdated-discussions: close-outdated-discussions:
name: 'Close Outdated Discussions' name: 'Close Outdated Discussions'
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-slim runs-on: ubuntu-24.04
permissions: permissions:
discussions: write discussions: write
steps: steps:
@@ -211,7 +211,7 @@ jobs:
close-unsupported-feature-requests: close-unsupported-feature-requests:
name: 'Close Unsupported Feature Requests' name: 'Close Unsupported Feature Requests'
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-slim runs-on: ubuntu-24.04
permissions: permissions:
discussions: write discussions: write
steps: steps:
+1 -1
View File
@@ -8,7 +8,7 @@ env:
jobs: jobs:
generate-translate-strings: generate-translate-strings:
name: Generate Translation Strings name: Generate Translation Strings
runs-on: ubuntu-26.04 runs-on: ubuntu-latest
environment: translation-sync environment: translation-sync
permissions: permissions:
contents: write contents: write
+1
View File
@@ -38,6 +38,7 @@ src/documents/bulk_edit.py:0: error: Incompatible types in assignment (expressio
src/documents/bulk_edit.py:0: error: Invalid index type "str" for "dict[FieldDataType, str]"; expected type "FieldDataType" [index] src/documents/bulk_edit.py:0: error: Invalid index type "str" for "dict[FieldDataType, str]"; expected type "FieldDataType" [index]
src/documents/bulk_edit.py:0: error: List comprehension has incompatible type List[tuple[int, Any]]; expected List[int] [misc] src/documents/bulk_edit.py:0: error: List comprehension has incompatible type List[tuple[int, Any]]; expected List[int] [misc]
src/documents/bulk_edit.py:0: error: List comprehension has incompatible type List[tuple[int, None]]; expected List[int] [misc] src/documents/bulk_edit.py:0: error: List comprehension has incompatible type List[tuple[int, None]]; expected List[int] [misc]
src/documents/bulk_edit.py:0: error: Missing named argument "p" for "remove" of "PageList" [call-arg]
src/documents/bulk_edit.py:0: error: Missing type arguments for generic type "dict" [type-arg] src/documents/bulk_edit.py:0: error: Missing type arguments for generic type "dict" [type-arg]
src/documents/bulk_edit.py:0: error: Missing type arguments for generic type "dict" [type-arg] src/documents/bulk_edit.py:0: error: Missing type arguments for generic type "dict" [type-arg]
src/documents/bulk_edit.py:0: error: Need type annotation for "to_create" (hint: "to_create: list[<type>] = ...") [var-annotated] src/documents/bulk_edit.py:0: error: Need type annotation for "to_create" (hint: "to_create: list[<type>] = ...") [var-annotated]
+7
View File
@@ -91,6 +91,13 @@
"concise_description": "Argument `list[int]` is not assignable to parameter `args` with type `tuple[Any, ...] | None` in function `celery.app.task.Task.apply_async`", "concise_description": "Argument `list[int]` is not assignable to parameter `args` with type `tuple[Any, ...] | None` in function `celery.app.task.Task.apply_async`",
"severity": "error" "severity": "error"
}, },
{
"column": 33,
"path": "src/documents/bulk_edit.py",
"name": "missing-argument",
"concise_description": "Missing argument `p` in function `pikepdf._core.PageList.remove`",
"severity": "error"
},
{ {
"column": 25, "column": 25,
"path": "src/documents/caching.py", "path": "src/documents/caching.py",
+3 -1
View File
@@ -76,7 +76,9 @@ is not supported by any of the available parsers.
**A:** Not by default. As of v3, a file whose contents match an existing document is still **A:** Not by default. As of v3, a file whose contents match an existing document is still
consumed, and the duplicate is flagged in the UI — open the document and check the consumed, and the duplicate is flagged in the UI — open the document and check the
**Duplicates** tab to review documents that share the same content. If you prefer the old **Duplicates** tab to review documents that share the same content, or filter the document
list by **Duplicates** to find all of them (see
[Duplicate documents](usage.md#duplicate-documents)). If you prefer the old
behavior of rejecting duplicates during consumption, set behavior of rejecting duplicates during consumption, set
[`PAPERLESS_CONSUMER_DELETE_DUPLICATES`](configuration.md#PAPERLESS_CONSUMER_DELETE_DUPLICATES) [`PAPERLESS_CONSUMER_DELETE_DUPLICATES`](configuration.md#PAPERLESS_CONSUMER_DELETE_DUPLICATES)
to `true`. to `true`.
+7 -8
View File
@@ -299,19 +299,18 @@ for details.
### Duplicate documents ### Duplicate documents
By default, Paperless-ngx **does not reject duplicates**. If you consume a file whose By default, Paperless-ngx **does not reject duplicates**. If you consume a file whose
contents exactly match an existing document (same checksum), the new copy is still contents match an existing document (same original or archive checksum), the new copy is
consumed and a warning is logged. The task entry for the upload also flags that a still consumed and a warning is logged.
duplicate was detected and links to the existing document(s).
To review duplicates, open a document and switch to the **Duplicates** tab on the When a document has duplicates, a **Duplicates** tab appears on its detail page, listing
document detail page. It lists other documents that share the same content, including any the other documents you can view that share the same content (including any in the trash).
that are in the trash (shown with a badge), and links to each so you can decide which to To find all documents with duplicates, choose **Duplicates** in the document list's text
keep. filter dropdown, or use `has_duplicates=true` in the REST API.
If you would rather reject duplicates at consumption time (the pre-v3 behavior), set If you would rather reject duplicates at consumption time (the pre-v3 behavior), set
[`PAPERLESS_CONSUMER_DELETE_DUPLICATES`](configuration.md#PAPERLESS_CONSUMER_DELETE_DUPLICATES) [`PAPERLESS_CONSUMER_DELETE_DUPLICATES`](configuration.md#PAPERLESS_CONSUMER_DELETE_DUPLICATES)
to `true`. The duplicate file is then deleted instead of consumed, and the task fails with to `true`. The duplicate file is then deleted instead of consumed, and the task fails with
a "document already exists" message. a "Document already exists" message linking to the existing document.
## Document Suggestions ## Document Suggestions
+1 -1
View File
@@ -1,6 +1,6 @@
[project] [project]
name = "paperless-ngx" name = "paperless-ngx"
version = "3.2.1" version = "3.3.0"
description = """\ description = """\
A community-supported supercharged document management system: scan, index and archive all your physical documents\ A community-supported supercharged document management system: scan, index and archive all your physical documents\
""" """
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "paperless-ngx-ui", "name": "paperless-ngx-ui",
"version": "3.2.1", "version": "3.3.0",
"scripts": { "scripts": {
"preinstall": "npx only-allow pnpm", "preinstall": "npx only-allow pnpm",
"ng": "ng", "ng": "ng",
+1 -1
View File
@@ -8,7 +8,7 @@ export const environment = {
apiVersion: '10', // match src/paperless/settings.py apiVersion: '10', // match src/paperless/settings.py
appTitle: DEFAULT_APP_TITLE, appTitle: DEFAULT_APP_TITLE,
tag: 'prod', tag: 'prod',
version: '3.2.1', version: '3.3.0',
webSocketHost: window.location.host, webSocketHost: window.location.host,
webSocketProtocol: window.location.protocol == 'https:' ? 'wss:' : 'ws:', webSocketProtocol: window.location.protocol == 'https:' ? 'wss:' : 'ws:',
webSocketBaseUrl: base_url.pathname + 'ws/', webSocketBaseUrl: base_url.pathname + 'ws/',
+219 -184
View File
@@ -3,7 +3,6 @@ from __future__ import annotations
import logging import logging
import tempfile import tempfile
import uuid import uuid
from functools import partial
from pathlib import Path from pathlib import Path
from typing import TYPE_CHECKING from typing import TYPE_CHECKING
from typing import Literal from typing import Literal
@@ -18,7 +17,6 @@ from django.db.models import Max
from django.db.models import Q from django.db.models import Q
from django.utils import timezone from django.utils import timezone
from documents import pdf_ops
from documents.data_models import ConsumableDocument from documents.data_models import ConsumableDocument
from documents.data_models import DocumentMetadataOverrides from documents.data_models import DocumentMetadataOverrides
from documents.data_models import DocumentSource from documents.data_models import DocumentSource
@@ -118,11 +116,6 @@ def _resolve_root_and_source_doc(
) )
def _scratch_path(name: str) -> Path:
"""A path inside a fresh directory under SCRATCH_DIR."""
return Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR)) / name
def set_correspondent( def set_correspondent(
doc_ids: list[int], doc_ids: list[int],
correspondent: Correspondent, correspondent: Correspondent,
@@ -481,6 +474,8 @@ def rotate(
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode) pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
docs_by_root_id.setdefault(pair.root_doc.id, pair) docs_by_root_id.setdefault(pair.root_doc.id, pair)
import pikepdf
for pair in docs_by_root_id.values(): for pair in docs_by_root_id.values():
if pair.source_doc.mime_type != "application/pdf": if pair.source_doc.mime_type != "application/pdf":
logger.warning( logger.warning(
@@ -493,7 +488,11 @@ def rotate(
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR)) Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_rotated.pdf" / f"{pair.root_doc.id}_rotated.pdf"
) )
pdf_ops.rotate_pdf(pair.source_doc.source_path, filepath, degrees) with pikepdf.open(pair.source_doc.source_path) as pdf:
for page in pdf.pages:
page.rotate(degrees, relative=True)
pdf.remove_unreferenced_resources()
pdf.save(filepath)
# Preserve metadata/permissions via overrides; mark as new version # Preserve metadata/permissions via overrides; mark as new version
overrides = DocumentMetadataOverrides().from_document(pair.root_doc) overrides = DocumentMetadataOverrides().from_document(pair.root_doc)
@@ -536,45 +535,48 @@ def merge(
qs = Document.objects.select_related("root_document").filter(id__in=doc_ids) qs = Document.objects.select_related("root_document").filter(id__in=doc_ids)
docs_by_id = {doc.id: doc for doc in qs} docs_by_id = {doc.id: doc for doc in qs}
affected_docs: list[int] = [] affected_docs: list[int] = []
handoff_asn: int | None = None import pikepdf
with pdf_ops.PdfMerger() as merger:
# use doc_ids to preserve order
for doc_id in doc_ids:
doc = docs_by_id.get(doc_id)
if doc is None:
continue
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
try:
# archive_path is None when there is no archive version
archive_path = (
pair.source_doc.archive_path
if archive_fallback
and pair.source_doc.mime_type != "application/pdf"
else None
)
merger.add(
archive_path
if archive_path is not None
else pair.source_doc.source_path,
)
affected_docs.append(doc.id)
if handoff_asn is None and doc.archive_serial_number is not None:
handoff_asn = doc.archive_serial_number
except Exception as e:
logger.exception(
f"Error merging document {doc.id}, it will not be included in the merge: {e}",
)
if len(affected_docs) == 0:
logger.warning("No documents were merged")
return "OK"
filepath = ( merged_pdf = pikepdf.new()
Path( version: str = merged_pdf.pdf_version
tempfile.mkdtemp(dir=settings.SCRATCH_DIR), handoff_asn: int | None = None
# use doc_ids to preserve order
for doc_id in doc_ids:
doc = docs_by_id.get(doc_id)
if doc is None:
continue
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
try:
doc_path = (
pair.source_doc.archive_path
if archive_fallback
and pair.source_doc.mime_type != "application/pdf"
and pair.source_doc.has_archive_version
else pair.source_doc.source_path
) )
/ f"{'_'.join([str(doc_id) for doc_id in affected_docs])[:100]}_merged.pdf" with pikepdf.open(str(doc_path)) as pdf:
version = max(version, pdf.pdf_version)
merged_pdf.pages.extend(pdf.pages)
affected_docs.append(doc.id)
if handoff_asn is None and doc.archive_serial_number is not None:
handoff_asn = doc.archive_serial_number
except Exception as e:
logger.exception(
f"Error merging document {doc.id}, it will not be included in the merge: {e}",
)
if len(affected_docs) == 0:
logger.warning("No documents were merged")
return "OK"
filepath = (
Path(
tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
) )
merger.save(filepath) / f"{'_'.join([str(doc_id) for doc_id in affected_docs])[:100]}_merged.pdf"
)
merged_pdf.remove_unreferenced_resources()
merged_pdf.save(filepath, min_version=version)
merged_pdf.close()
if metadata_document_id: if metadata_document_id:
metadata_document = qs.get(id=metadata_document_id) metadata_document = qs.get(id=metadata_document_id)
@@ -750,60 +752,64 @@ def split(
) )
doc = Document.objects.select_related("root_document").get(id=doc_ids[0]) doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode) pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
import pikepdf
consume_tasks = [] consume_tasks = []
try: try:
outputs = [ with pikepdf.open(pair.source_doc.source_path) as pdf:
( for idx, split_doc in enumerate(pages):
[pdf_ops.PageSpec(page) for page in split_doc], dst: pikepdf.Pdf = pikepdf.new()
partial(_scratch_path, f"{doc.id}_{split_doc[0]}-{split_doc[-1]}.pdf"), for page in split_doc:
) dst.pages.append(pdf.pages[page - 1])
for split_doc in pages filepath: Path = (
] Path(
filepaths = pdf_ops.build_pdfs(pair.source_doc.source_path, outputs) tempfile.mkdtemp(dir=settings.SCRATCH_DIR),
)
/ f"{doc.id}_{split_doc[0]}-{split_doc[-1]}.pdf"
)
dst.remove_unreferenced_resources()
dst.save(filepath)
dst.close()
for idx, (split_doc, filepath) in enumerate( overrides: DocumentMetadataOverrides = (
zip(pages, filepaths, strict=True), DocumentMetadataOverrides().from_document(doc)
): )
overrides: DocumentMetadataOverrides = ( overrides.title = f"{doc.title} (split {idx + 1})"
DocumentMetadataOverrides().from_document(doc) if user is not None:
) overrides.owner_id = user.id
overrides.title = f"{doc.title} (split {idx + 1})" if not delete_originals:
if user is not None: overrides.skip_asn_if_exists = True
overrides.owner_id = user.id logger.info(
if not delete_originals: f"Adding split document with pages {split_doc} to the task queue.",
overrides.skip_asn_if_exists = True )
logger.info( consume_tasks.append(
f"Adding split document with pages {split_doc} to the task queue.", consume_file.s(
) input_doc=ConsumableDocument(
consume_tasks.append( source=DocumentSource.ConsumeFolder,
consume_file.s( original_file=filepath,
input_doc=ConsumableDocument( ),
source=DocumentSource.ConsumeFolder, overrides=overrides,
original_file=filepath, ).set(headers={"trigger_source": trigger_source}),
), )
overrides=overrides,
).set(headers={"trigger_source": trigger_source}),
)
if delete_originals: if delete_originals:
backup = release_archive_serial_numbers([doc.id]) backup = release_archive_serial_numbers([doc.id])
logger.info( logger.info(
"Queueing removal of original document after consumption of the split documents", "Queueing removal of original document after consumption of the split documents",
) )
try: try:
chord( chord(
header=consume_tasks, header=consume_tasks,
body=delete.si([doc.id]), body=delete.si([doc.id]),
).on_error( ).on_error(
restore_archive_serial_numbers_task.s(backup), restore_archive_serial_numbers_task.s(backup),
).apply_async() ).apply_async()
except Exception: except Exception:
restore_archive_serial_numbers(backup) restore_archive_serial_numbers(backup)
raise raise
else: else:
group(consume_tasks).delay() group(consume_tasks).delay()
except Exception as e: except Exception as e:
logger.exception(f"Error splitting document {doc.id}: {e}") logger.exception(f"Error splitting document {doc.id}: {e}")
@@ -824,6 +830,8 @@ def delete_pages(
) )
doc = Document.objects.select_related("root_document").get(id=doc_ids[0]) doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode) pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
pages = sorted(pages) # sort pages to avoid index issues
import pikepdf
try: try:
# Produce edited PDF to a temp file and create a new version # Produce edited PDF to a temp file and create a new version
@@ -831,7 +839,13 @@ def delete_pages(
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR)) Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_pages_deleted.pdf" / f"{pair.root_doc.id}_pages_deleted.pdf"
) )
pdf_ops.remove_pages(pair.source_doc.source_path, filepath, pages) with pikepdf.open(pair.source_doc.source_path) as pdf:
offset = 1 # pages are 1-indexed
for page_num in pages:
pdf.pages.remove(pdf.pages[page_num - offset])
offset += 1 # remove() changes the index of the pages
pdf.remove_unreferenced_resources()
pdf.save(filepath)
overrides = DocumentMetadataOverrides().from_document(pair.root_doc) overrides = DocumentMetadataOverrides().from_document(pair.root_doc)
if user is not None: if user is not None:
@@ -880,28 +894,47 @@ def edit_pdf(
) )
doc = Document.objects.select_related("root_document").get(id=doc_ids[0]) doc = Document.objects.select_related("root_document").get(id=doc_ids[0])
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode) pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
import pikepdf
pdf_docs: list[pikepdf.Pdf] = []
try: try:
output_count = pdf_ops.validate_page_operations( if not operations:
operations, raise ValueError("Output document index is out of bounds")
single_output=update_document,
) max_idx = max(op.get("doc", 0) for op in operations)
page_specs: list[list[pdf_ops.PageSpec]] = [[] for _ in range(output_count)] if update_document and max_idx > 0:
for op in operations: logger.error(
page_specs[op.get("doc", 0)].append( "Update requested but multiple output documents specified",
pdf_ops.PageSpec(op["page"], op.get("rotate", 0)),
) )
raise ValueError("Multiple output documents specified")
if any(
op.get("doc", 0) < 0 or op.get("doc", 0) >= len(operations)
for op in operations
):
raise ValueError("Output document index is out of bounds")
with pikepdf.open(pair.source_doc.source_path) as src:
# prepare output documents
pdf_docs = [pikepdf.new() for _ in range(max_idx + 1)]
for op in operations:
dst = pdf_docs[op.get("doc", 0)]
page = src.pages[op["page"] - 1]
dst.pages.append(page)
if op.get("rotate"):
dst.pages[-1].rotate(op["rotate"], relative=True)
if update_document: if update_document:
# Create a new version from the edited PDF rather than replacing in-place # Create a new version from the edited PDF rather than replacing in-place
(filepath,) = pdf_ops.build_pdfs( pdf = pdf_docs[0]
pair.source_doc.source_path, pdf.remove_unreferenced_resources()
[ filepath: Path = (
( Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
page_specs[0], / f"{pair.root_doc.id}_edited.pdf"
partial(_scratch_path, f"{pair.root_doc.id}_edited.pdf"),
),
],
) )
pdf.save(filepath)
overrides = ( overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc) DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata if include_metadata
@@ -922,19 +955,6 @@ def edit_pdf(
headers={"trigger_source": trigger_source}, headers={"trigger_source": trigger_source},
) )
else: else:
version_filepaths = pdf_ops.build_pdfs(
pair.source_doc.source_path,
[
(
specs,
partial(
_scratch_path,
f"{pair.root_doc.id}_edit_{idx}.pdf",
),
)
for idx, specs in enumerate(page_specs, start=1)
],
)
consume_tasks = [] consume_tasks = []
overrides = ( overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc) DocumentMetadataOverrides().from_document(pair.root_doc)
@@ -946,9 +966,15 @@ def edit_pdf(
overrides.actor_id = user.id overrides.actor_id = user.id
if not delete_original: if not delete_original:
overrides.skip_asn_if_exists = True overrides.skip_asn_if_exists = True
if delete_original and output_count == 1: if delete_original and len(pdf_docs) == 1:
overrides.asn = pair.root_doc.archive_serial_number overrides.asn = pair.root_doc.archive_serial_number
for version_filepath in version_filepaths: for idx, pdf in enumerate(pdf_docs, start=1):
version_filepath: Path = (
Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
/ f"{pair.root_doc.id}_edit_{idx}.pdf"
)
pdf.remove_unreferenced_resources()
pdf.save(version_filepath)
consume_tasks.append( consume_tasks.append(
consume_file.s( consume_file.s(
input_doc=ConsumableDocument( input_doc=ConsumableDocument(
@@ -998,6 +1024,8 @@ def remove_password(
""" """
Remove password protection from PDF documents. Remove password protection from PDF documents.
""" """
import pikepdf
for doc_id in doc_ids: for doc_id in doc_ids:
doc = Document.objects.select_related("root_document").get(id=doc_id) doc = Document.objects.select_related("root_document").get(id=doc_id)
pair = _resolve_root_and_source_doc(doc, source_mode=source_mode) pair = _resolve_root_and_source_doc(doc, source_mode=source_mode)
@@ -1011,69 +1039,76 @@ def remove_password(
doc.id, doc.id,
pair.source_doc.source_path, pair.source_doc.source_path,
) )
if not pdf_ops.needs_decrypt(source_path): try:
logger.info( with pikepdf.open(source_path) as pdf:
"Skipping password removal for document %s because the " if not pdf.is_encrypted:
"source PDF is not encrypted", logger.info(
pair.root_doc.id, "Skipping password removal for document %s because the "
) "source PDF is not encrypted",
continue pair.root_doc.id,
)
continue
except pikepdf.PasswordError:
# Password-protected PDFs need the supplied password below.
pass
filepath = pdf_ops.decrypt_pdf( with pikepdf.open(source_path, password=password) as pdf:
source_path, filepath: Path = (
partial(_scratch_path, f"{pair.root_doc.id}_unprotected.pdf"), Path(tempfile.mkdtemp(dir=settings.SCRATCH_DIR))
password, / f"{pair.root_doc.id}_unprotected.pdf"
) )
pdf.remove_unreferenced_resources()
pdf.save(filepath)
if update_document: if update_document:
# Create a new version rather than modifying the root/original in place. # Create a new version rather than modifying the root/original in place.
overrides = ( overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc) DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata if include_metadata
else DocumentMetadataOverrides() else DocumentMetadataOverrides()
) )
if user is not None: if user is not None:
overrides.owner_id = user.id overrides.owner_id = user.id
overrides.actor_id = user.id overrides.actor_id = user.id
consume_file.apply_async( consume_file.apply_async(
kwargs={ kwargs={
"input_doc": ConsumableDocument( "input_doc": ConsumableDocument(
source=DocumentSource.ConsumeFolder, source=DocumentSource.ConsumeFolder,
original_file=filepath, original_file=filepath,
root_document_id=pair.root_doc.id, root_document_id=pair.root_doc.id,
), ),
"overrides": overrides, "overrides": overrides,
}, },
headers={"trigger_source": trigger_source}, headers={"trigger_source": trigger_source},
) )
else:
consume_tasks = []
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata
else DocumentMetadataOverrides()
)
if user is not None:
overrides.owner_id = user.id
overrides.actor_id = user.id
consume_tasks.append(
consume_file.s(
input_doc=ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
),
overrides=overrides,
).set(headers={"trigger_source": trigger_source}),
)
if delete_original:
chord(
header=consume_tasks,
body=delete.si([doc.id]),
).delay()
else: else:
group(consume_tasks).delay() consume_tasks = []
overrides = (
DocumentMetadataOverrides().from_document(pair.root_doc)
if include_metadata
else DocumentMetadataOverrides()
)
if user is not None:
overrides.owner_id = user.id
overrides.actor_id = user.id
consume_tasks.append(
consume_file.s(
input_doc=ConsumableDocument(
source=DocumentSource.ConsumeFolder,
original_file=filepath,
),
overrides=overrides,
).set(headers={"trigger_source": trigger_source}),
)
if delete_original:
chord(
header=consume_tasks,
body=delete.si([doc.id]),
).delay()
else:
group(consume_tasks).delay()
except Exception as e: except Exception as e:
logger.exception( logger.exception(
-194
View File
@@ -1,194 +0,0 @@
"""
Pure PDF page operations used by documents.bulk_edit.
This module deliberately knows nothing about Django, Celery or the documents
app: callers resolve documents, choose output paths and queue work. Every
function that writes a PDF removes unreferenced resources before saving.
pikepdf is always called as ``pikepdf.open(...)`` / ``pikepdf.new()`` (never
``from pikepdf import open``) so tests can patch those module attributes.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
from typing import NamedTuple
import pikepdf
if TYPE_CHECKING:
from collections.abc import Callable
from collections.abc import Iterable
from collections.abc import Mapping
from collections.abc import Sequence
from pathlib import Path
from types import TracebackType
class PageSpec(NamedTuple):
"""One page of an output PDF: a 1-indexed source page, optionally rotated."""
page: int
rotate: int = 0 # relative degrees, 0 leaves the page alone
def _require_positive(pages: Iterable[int]) -> None:
for page in pages:
if page < 1:
raise ValueError(f"Page numbers start at 1, got {page}")
def rotate_pdf(src: Path, dst: Path, degrees: int) -> None:
"""
Rotate every page relatively on the opened document, not a rebuild, so Info,
XMP and outlines are kept. ``src`` is not modified.
"""
with pikepdf.open(src) as pdf:
for page in pdf.pages:
page.rotate(degrees, relative=True)
pdf.remove_unreferenced_resources()
pdf.save(dst)
def remove_pages(src: Path, dst: Path, pages: Iterable[int]) -> None:
"""
Remove 1-indexed pages from the opened document, not a rebuild, so Info, XMP
and outlines are kept. ``src`` is not modified.
Duplicates are ignored. Pages are removed highest first so earlier removals
never shift the index of later ones.
"""
unique = sorted(set(pages))
_require_positive(unique)
with pikepdf.open(src) as pdf:
for page_num in reversed(unique):
del pdf.pages[page_num - 1]
pdf.remove_unreferenced_resources()
pdf.save(dst)
def build_pdfs(
src: Path,
outputs: Sequence[tuple[Sequence[PageSpec], Callable[[], Path]]],
) -> list[Path]:
"""
Build one new PDF per output from pages of ``src``, opening ``src`` once.
Each output is ``(page_specs, make_dst)``. Every page number is checked against
``src`` before any output is built, and ``make_dst`` is called after that
output's pages are copied and immediately before it is saved, so a bad page
number in any output never leaves a destination behind. Document-level data
(Info, XMP, outlines) is not carried over. Returns the written paths in output
order.
"""
for specs, _ in outputs:
_require_positive(spec.page for spec in specs)
written: list[Path] = []
with pikepdf.open(src) as source:
page_count = len(source.pages)
for specs, _ in outputs:
for spec in specs:
if spec.page > page_count:
raise IndexError(
f"Page {spec.page} is out of range, the PDF has "
f"{page_count} pages",
)
for specs, make_dst in outputs:
dst = pikepdf.new()
for spec in specs:
dst.pages.append(source.pages[spec.page - 1])
if spec.rotate:
dst.pages[-1].rotate(spec.rotate, relative=True)
dst.remove_unreferenced_resources()
path = make_dst()
dst.save(path)
dst.close()
written.append(path)
return written
def validate_page_operations(
operations: Sequence[Mapping[str, int]],
*,
single_output: bool,
) -> int:
"""
Validate ``edit_pdf`` style operations and return the output document count.
Each operation has ``page`` and optionally ``rotate`` and ``doc`` (the output
document index, default 0). The bounds rule is kept as it was: a ``doc`` index
must be below the number of operations.
"""
if not operations:
raise ValueError("Output document index is out of bounds")
max_idx = max(op.get("doc", 0) for op in operations)
if single_output and max_idx > 0:
raise ValueError("Multiple output documents specified")
if any(
op.get("doc", 0) < 0 or op.get("doc", 0) >= len(operations) for op in operations
):
raise ValueError("Output document index is out of bounds")
return max_idx + 1
def needs_decrypt(src: Path) -> bool:
"""
True if ``src`` is encrypted. A PDF that needs a password to open at all
counts as encrypted.
"""
try:
with pikepdf.open(src) as pdf:
return bool(pdf.is_encrypted)
except pikepdf.PasswordError:
return True
def decrypt_pdf(src: Path, make_dst: Callable[[], Path], password: str) -> Path:
"""
Write an unencrypted copy of ``src`` and return its path.
``make_dst`` is only called once the password has been accepted, so a wrong
password never leaves a destination behind.
"""
with pikepdf.open(src, password=password) as pdf:
pdf.remove_unreferenced_resources()
dst = make_dst()
pdf.save(dst)
return dst
class PdfMerger:
"""
Accumulates the pages of several PDFs into one new PDF.
``add`` raises if a source cannot be read; deciding whether to skip it is the
caller's policy. Use as a context manager so the merged PDF is closed.
"""
def __init__(self) -> None:
self._pdf = pikepdf.new()
self._version: str = self._pdf.pdf_version
def __enter__(self) -> PdfMerger:
return self
def __exit__(
self,
exc_type: type[BaseException] | None,
exc: BaseException | None,
tb: TracebackType | None,
) -> None:
self._pdf.close()
def add(self, path: Path) -> None:
with pikepdf.open(str(path)) as pdf:
self._version = max(self._version, pdf.pdf_version)
self._pdf.pages.extend(pdf.pages)
def save(self, dst: Path) -> None:
self._pdf.remove_unreferenced_resources()
self._pdf.save(dst, min_version=self._version)
-2
View File
@@ -2144,8 +2144,6 @@ class BulkEditSerializer(
raise serializers.ValidationError("pages must be a list") raise serializers.ValidationError("pages must be a list")
if not all(isinstance(i, int) for i in parameters["pages"]): if not all(isinstance(i, int) for i in parameters["pages"]):
raise serializers.ValidationError("pages must be a list of integers") raise serializers.ValidationError("pages must be a list of integers")
if any(i < 1 for i in parameters["pages"]):
raise serializers.ValidationError("pages must be positive integers")
def _validate_parameters_merge(self, parameters) -> None: def _validate_parameters_merge(self, parameters) -> None:
if "delete_originals" in parameters: if "delete_originals" in parameters:
-30
View File
@@ -1843,36 +1843,6 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
m.assert_called_once() m.assert_called_once()
self.assertEqual(m.call_args.kwargs["pages"], [[1], [2, 3, 4], [5]]) self.assertEqual(m.call_args.kwargs["pages"], [[1], [2, 3, 4], [5]])
@mock.patch("documents.serialisers.bulk_edit.delete_pages")
def test_bulk_edit_delete_pages_rejects_pages_below_one(self, m) -> None:
"""
GIVEN:
- A legacy delete_pages bulk edit
WHEN:
- API to bulk edit is called with a page number below 1
THEN:
- API returns HTTP 400
- delete_pages is not called
"""
self.setup_mock(m, "delete_pages")
for pages in ([0], [-1], [1, 0]):
with self.subTest(pages=pages):
response = self.client.post(
"/api/documents/bulk_edit/",
json.dumps(
{
"documents": [self.doc2.id],
"method": "delete_pages",
"parameters": {"pages": pages},
},
),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn(b"pages must be positive integers", response.content)
m.assert_not_called()
@mock.patch("documents.views.bulk_edit.rotate") @mock.patch("documents.views.bulk_edit.rotate")
def test_rotate_insufficient_permissions(self, m) -> None: def test_rotate_insufficient_permissions(self, m) -> None:
self.doc1.owner = User.objects.get(username="temp_admin") self.doc1.owner = User.objects.get(username="temp_admin")
+231 -152
View File
@@ -1,9 +1,9 @@
import shutil import shutil
from collections.abc import Callable
from datetime import date from datetime import date
from pathlib import Path from pathlib import Path
from unittest import mock from unittest import mock
import pikepdf
from django.contrib.auth.models import Group from django.contrib.auth.models import Group
from django.contrib.auth.models import Permission from django.contrib.auth.models import Permission
from django.contrib.auth.models import User from django.contrib.auth.models import User
@@ -21,7 +21,6 @@ from documents.models import Document
from documents.models import DocumentType from documents.models import DocumentType
from documents.models import StoragePath from documents.models import StoragePath
from documents.models import Tag from documents.models import Tag
from documents.pdf_ops import PageSpec
from documents.permissions import set_permissions_for_objects from documents.permissions import set_permissions_for_objects
from paperless_testing.dirs import DirectoriesMixin from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.permissions import grant_object from paperless_testing.permissions import grant_object
@@ -794,14 +793,16 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.img_doc.save() self.img_doc.save()
@staticmethod @staticmethod
def fake_decrypt( def mock_password_required_pdf(
src: Path, mock_open: mock.Mock,
make_dst: Callable[[], Path], fake_pdf: mock.Mock,
password: str, ) -> None:
) -> Path: password_context = mock.MagicMock()
dst = make_dst() password_context.__enter__.return_value = fake_pdf
dst.write_bytes(b"password removed") mock_open.side_effect = [
return dst pikepdf.PasswordError("password required"),
password_context,
]
@mock.patch("documents.tasks.consume_file.s") @mock.patch("documents.tasks.consume_file.s")
def test_merge(self, mock_consume_file) -> None: def test_merge(self, mock_consume_file) -> None:
@@ -846,12 +847,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
@mock.patch("documents.pdf_ops.PdfMerger") @mock.patch("pikepdf.open")
@mock.patch("documents.tasks.consume_file.s") @mock.patch("documents.tasks.consume_file.s")
def test_merge_uses_latest_version_source_for_root_selection( def test_merge_uses_latest_version_source_for_root_selection(
self, self,
mock_consume_file, mock_consume_file,
mock_merger, mock_open_pdf,
) -> None: ) -> None:
version_file = self.dirs.scratch_dir / "sample2_version_merge.pdf" version_file = self.dirs.scratch_dir / "sample2_version_merge.pdf"
shutil.copy(self.doc2.source_path, version_file) shutil.copy(self.doc2.source_path, version_file)
@@ -862,14 +863,16 @@ class TestPDFActions(DirectoriesMixin, TestCase):
filename=version_file, filename=version_file,
mime_type="application/pdf", mime_type="application/pdf",
) )
merger = mock_merger.return_value.__enter__.return_value fake_pdf = mock.MagicMock()
merger.save.side_effect = lambda dst: shutil.copy(version.source_path, dst) fake_pdf.pdf_version = "1.7"
fake_pdf.pages = [mock.Mock()]
mock_open_pdf.return_value.__enter__.return_value = fake_pdf
result = bulk_edit.merge([self.doc2.id]) result = bulk_edit.merge([self.doc2.id])
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
merger.add.assert_called_once_with(version.source_path) mock_open_pdf.assert_called_once_with(str(version.source_path))
mock_consume_file.assert_called_once() mock_consume_file.assert_not_called()
@mock.patch("documents.bulk_edit.delete.si") @mock.patch("documents.bulk_edit.delete.si")
@mock.patch("documents.tasks.consume_file.s") @mock.patch("documents.tasks.consume_file.s")
@@ -1031,18 +1034,18 @@ class TestPDFActions(DirectoriesMixin, TestCase):
) )
@mock.patch("documents.tasks.consume_file.delay") @mock.patch("documents.tasks.consume_file.delay")
@mock.patch("documents.pdf_ops.PdfMerger.add") @mock.patch("pikepdf.open")
def test_merge_with_errors(self, mock_add, mock_consume_file) -> None: def test_merge_with_errors(self, mock_open_pdf, mock_consume_file) -> None:
""" """
GIVEN: GIVEN:
- Existing documents - Existing documents
WHEN: WHEN:
- Merge action is called with 2 documents - Merge action is called with 2 documents
- Error occurs when adding both files - Error occurs when opening both files
THEN: THEN:
- Consume file should not be called - Consume file should not be called
""" """
mock_add.side_effect = Exception("Error opening PDF") mock_open_pdf.side_effect = Exception("Error opening PDF")
doc_ids = [self.doc2.id, self.doc3.id] doc_ids = [self.doc2.id, self.doc3.id]
with self.assertLogs("paperless.bulk_edit", level="ERROR") as cm: with self.assertLogs("paperless.bulk_edit", level="ERROR") as cm:
@@ -1079,12 +1082,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
@mock.patch("documents.bulk_edit.group") @mock.patch("documents.bulk_edit.group")
@mock.patch("documents.pdf_ops.build_pdfs") @mock.patch("pikepdf.open")
@mock.patch("documents.tasks.consume_file.s") @mock.patch("documents.tasks.consume_file.s")
def test_split_uses_latest_version_source_for_root_selection( def test_split_uses_latest_version_source_for_root_selection(
self, self,
mock_consume_file, mock_consume_file,
mock_build_pdfs, mock_open_pdf,
mock_group, mock_group,
) -> None: ) -> None:
version_file = self.dirs.scratch_dir / "sample2_version_split.pdf" version_file = self.dirs.scratch_dir / "sample2_version_split.pdf"
@@ -1096,15 +1099,17 @@ class TestPDFActions(DirectoriesMixin, TestCase):
filename=version_file, filename=version_file,
mime_type="application/pdf", mime_type="application/pdf",
) )
mock_build_pdfs.return_value = [version.source_path, version.source_path] fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock(), mock.Mock()]
mock_open_pdf.return_value.__enter__.return_value = fake_pdf
mock_group.return_value.delay.return_value = None mock_group.return_value.delay.return_value = None
result = bulk_edit.split([self.doc2.id], [[1], [2]]) result = bulk_edit.split([self.doc2.id], [[1], [2]])
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
self.assertEqual(mock_build_pdfs.call_args.args[0], version.source_path) mock_open_pdf.assert_called_once_with(version.source_path)
self.assertEqual(mock_consume_file.call_count, 2) mock_consume_file.assert_not_called()
mock_group.return_value.delay.assert_called_once() mock_group.return_value.delay.assert_not_called()
@mock.patch("documents.bulk_edit.delete.si") @mock.patch("documents.bulk_edit.delete.si")
@mock.patch("documents.tasks.consume_file.s") @mock.patch("documents.tasks.consume_file.s")
@@ -1192,18 +1197,18 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(self.doc2.archive_serial_number, 222) self.assertEqual(self.doc2.archive_serial_number, 222)
@mock.patch("documents.tasks.consume_file.apply_async") @mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.pdf_ops.build_pdfs") @mock.patch("pikepdf.Pdf.save")
def test_split_with_errors(self, mock_build_pdfs, mock_consume_file) -> None: def test_split_with_errors(self, mock_save_pdf, mock_consume_file) -> None:
""" """
GIVEN: GIVEN:
- Existing documents - Existing documents
WHEN: WHEN:
- Split action is called with 1 document and 2 page groups - Split action is called with 1 document and 2 page groups
- Error occurs when building the files - Error occurs when saving the files
THEN: THEN:
- Consume file should not be called - Consume file should not be called
""" """
mock_build_pdfs.side_effect = Exception("Error building PDFs") mock_save_pdf.side_effect = Exception("Error saving PDF")
doc_ids = [self.doc2.id] doc_ids = [self.doc2.id]
pages = [[1, 2], [3]] pages = [[1, 2], [3]]
@@ -1238,10 +1243,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
@mock.patch("documents.tasks.consume_file.apply_async") @mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.pdf_ops.rotate_pdf") @mock.patch("pikepdf.Pdf.save")
def test_rotate_with_error( def test_rotate_with_error(
self, self,
mock_rotate_pdf, mock_pdf_save,
mock_consume_delay, mock_consume_delay,
) -> None: ) -> None:
""" """
@@ -1249,11 +1254,11 @@ class TestPDFActions(DirectoriesMixin, TestCase):
- Existing documents - Existing documents
WHEN: WHEN:
- Rotate action is called with 2 documents - Rotate action is called with 2 documents
- Rotating the PDF raises an error - PikePDF raises an error
THEN: THEN:
- Rotate action should be called 0 times - Rotate action should be called 0 times
""" """
mock_rotate_pdf.side_effect = Exception("Error rotating PDF") mock_pdf_save.side_effect = Exception("Error saving PDF")
doc_ids = [self.doc2.id, self.doc3.id] doc_ids = [self.doc2.id, self.doc3.id]
with self.assertLogs("paperless.bulk_edit", level="ERROR") as cm: with self.assertLogs("paperless.bulk_edit", level="ERROR") as cm:
@@ -1288,10 +1293,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf") @mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
@mock.patch("documents.tasks.consume_file.apply_async") @mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.pdf_ops.rotate_pdf") @mock.patch("pikepdf.open")
def test_rotate_explicit_selection_uses_root_source_when_root_selected( def test_rotate_explicit_selection_uses_root_source_when_root_selected(
self, self,
mock_rotate_pdf, mock_open,
mock_consume_delay, mock_consume_delay,
mock_magic, mock_magic,
) -> None: ) -> None:
@@ -1300,6 +1305,9 @@ class TestPDFActions(DirectoriesMixin, TestCase):
title="B version 1", title="B version 1",
root_document=self.doc2, root_document=self.doc2,
) )
fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock()]
mock_open.return_value.__enter__.return_value = fake_pdf
result = bulk_edit.rotate( result = bulk_edit.rotate(
[self.doc2.id], [self.doc2.id],
@@ -1308,35 +1316,26 @@ class TestPDFActions(DirectoriesMixin, TestCase):
) )
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
self.assertEqual(mock_rotate_pdf.call_args.args[0], self.doc2.source_path) mock_open.assert_called_once_with(self.doc2.source_path)
mock_consume_delay.assert_called_once() mock_consume_delay.assert_called_once()
@mock.patch("documents.tasks.consume_file.apply_async") @mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.pdf_ops.remove_pages") @mock.patch("pikepdf.Pdf.save")
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf") @mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
def test_delete_pages( def test_delete_pages(self, mock_magic, mock_pdf_save, mock_consume_delay) -> None:
self,
mock_magic,
mock_remove_pages,
mock_consume_delay,
) -> None:
""" """
GIVEN: GIVEN:
- Existing documents - Existing documents
WHEN: WHEN:
- Delete pages action is called with 1 document and 2 pages - Delete pages action is called with 1 document and 2 pages
THEN: THEN:
- The pages are removed from the document's source PDF - Save should be called once
- A new version should be enqueued via consume_file - A new version should be enqueued via consume_file
""" """
doc_ids = [self.doc2.id] doc_ids = [self.doc2.id]
pages = [1, 3] pages = [1, 3]
result = bulk_edit.delete_pages(doc_ids, pages) result = bulk_edit.delete_pages(doc_ids, pages)
mock_remove_pages.assert_called_once_with( mock_pdf_save.assert_called_once()
self.doc2.source_path,
mock.ANY,
[1, 3],
)
mock_consume_delay.assert_called_once() mock_consume_delay.assert_called_once()
task_kwargs = mock_consume_delay.call_args.kwargs["kwargs"] task_kwargs = mock_consume_delay.call_args.kwargs["kwargs"]
self.assertEqual(task_kwargs["input_doc"].root_document_id, self.doc2.id) self.assertEqual(task_kwargs["input_doc"].root_document_id, self.doc2.id)
@@ -1348,10 +1347,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf") @mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
@mock.patch("documents.tasks.consume_file.apply_async") @mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.pdf_ops.remove_pages") @mock.patch("pikepdf.open")
def test_delete_pages_explicit_selection_uses_root_source_when_root_selected( def test_delete_pages_explicit_selection_uses_root_source_when_root_selected(
self, self,
mock_remove_pages, mock_open,
mock_consume_delay, mock_consume_delay,
mock_magic, mock_magic,
) -> None: ) -> None:
@@ -1360,6 +1359,9 @@ class TestPDFActions(DirectoriesMixin, TestCase):
title="B version 1", title="B version 1",
root_document=self.doc2, root_document=self.doc2,
) )
fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock(), mock.Mock()]
mock_open.return_value.__enter__.return_value = fake_pdf
result = bulk_edit.delete_pages( result = bulk_edit.delete_pages(
[self.doc2.id], [self.doc2.id],
@@ -1368,26 +1370,23 @@ class TestPDFActions(DirectoriesMixin, TestCase):
) )
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
self.assertEqual(mock_remove_pages.call_args.args[0], self.doc2.source_path) mock_open.assert_called_once_with(self.doc2.source_path)
mock_consume_delay.assert_called_once() mock_consume_delay.assert_called_once()
@mock.patch("documents.tasks.consume_file.apply_async") @mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.pdf_ops.remove_pages") @mock.patch("pikepdf.Pdf.save")
def test_delete_pages_with_error( def test_delete_pages_with_error(self, mock_pdf_save, mock_consume_delay) -> None:
self,
mock_remove_pages,
mock_consume_delay,
) -> None:
""" """
GIVEN: GIVEN:
- Existing documents - Existing documents
WHEN: WHEN:
- Delete pages action is called with 1 document and 2 pages - Delete pages action is called with 1 document and 2 pages
- Removing the pages raises an error - PikePDF raises an error
THEN: THEN:
- Save should be called once
- No new version should be enqueued - No new version should be enqueued
""" """
mock_remove_pages.side_effect = Exception("Error removing pages") mock_pdf_save.side_effect = Exception("Error saving PDF")
doc_ids = [self.doc2.id] doc_ids = [self.doc2.id]
pages = [1, 3] pages = [1, 3]
@@ -1417,41 +1416,6 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
mock_group.return_value.delay.assert_called_once() mock_group.return_value.delay.assert_called_once()
@mock.patch("documents.bulk_edit.group")
@mock.patch("documents.pdf_ops.build_pdfs")
@mock.patch("documents.tasks.consume_file.s")
def test_edit_pdf_maps_operations_to_outputs(
self,
mock_consume_file: mock.Mock,
mock_build_pdfs: mock.Mock,
mock_group: mock.Mock,
) -> None:
"""
GIVEN:
- Existing document
WHEN:
- edit_pdf is called with operations interleaved across two outputs,
some of them rotated
THEN:
- Each output is built from its own operations, in operation order,
with the requested rotation
"""
mock_build_pdfs.return_value = [self.doc2.source_path, self.doc2.source_path]
mock_group.return_value.delay.return_value = None
operations = [
{"page": 3, "doc": 1},
{"page": 1, "doc": 0, "rotate": 90},
{"page": 2, "doc": 1, "rotate": 180},
]
bulk_edit.edit_pdf([self.doc2.id], operations)
outputs = mock_build_pdfs.call_args.args[1]
self.assertEqual(
[specs for specs, _ in outputs],
[[PageSpec(1, 90)], [PageSpec(3), PageSpec(2, 180)]],
)
@mock.patch("documents.bulk_edit.group") @mock.patch("documents.bulk_edit.group")
@mock.patch("documents.tasks.consume_file.s") @mock.patch("documents.tasks.consume_file.s")
def test_edit_pdf_with_user_override(self, mock_consume_file, mock_group) -> None: def test_edit_pdf_with_user_override(self, mock_consume_file, mock_group) -> None:
@@ -1567,10 +1531,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf") @mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
@mock.patch("documents.tasks.consume_file.apply_async") @mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.pdf_ops.build_pdfs") @mock.patch("pikepdf.new")
@mock.patch("pikepdf.open")
def test_edit_pdf_explicit_selection_uses_root_source_when_root_selected( def test_edit_pdf_explicit_selection_uses_root_source_when_root_selected(
self, self,
mock_build_pdfs, mock_open,
mock_new,
mock_consume_delay, mock_consume_delay,
mock_magic, mock_magic,
) -> None: ) -> None:
@@ -1579,7 +1545,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
title="B version 1", title="B version 1",
root_document=self.doc2, root_document=self.doc2,
) )
mock_build_pdfs.return_value = [Path("edited.pdf")] fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock()]
mock_open.return_value.__enter__.return_value = fake_pdf
output_pdf = mock.MagicMock()
output_pdf.pages = []
mock_new.return_value = output_pdf
result = bulk_edit.edit_pdf( result = bulk_edit.edit_pdf(
[self.doc2.id], [self.doc2.id],
@@ -1589,7 +1560,7 @@ class TestPDFActions(DirectoriesMixin, TestCase):
) )
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
self.assertEqual(mock_build_pdfs.call_args.args[0], self.doc2.source_path) mock_open.assert_called_once_with(self.doc2.source_path)
mock_consume_delay.assert_called_once() mock_consume_delay.assert_called_once()
@mock.patch("documents.bulk_edit.group") @mock.patch("documents.bulk_edit.group")
@@ -1615,6 +1586,31 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
mock_group.return_value.delay.assert_called_once() mock_group.return_value.delay.assert_called_once()
@mock.patch("documents.bulk_edit.group")
@mock.patch("documents.tasks.consume_file.s")
def test_edit_pdf_open_failure(
self,
mock_consume_file: mock.Mock,
mock_group: mock.Mock,
) -> None:
"""
GIVEN:
- Existing document
WHEN:
- edit_pdf fails to open PDF
THEN:
- Task group is not called
"""
doc_ids = [self.doc2.id]
operations = [
{"page": 9999}, # invalid page, forces error during PDF load
]
with self.assertLogs("paperless.bulk_edit", level="ERROR"):
with self.assertRaises(Exception):
bulk_edit.edit_pdf(doc_ids, operations)
mock_group.assert_not_called()
mock_consume_file.assert_not_called()
@mock.patch("documents.bulk_edit.group") @mock.patch("documents.bulk_edit.group")
@mock.patch("documents.tasks.consume_file.s") @mock.patch("documents.tasks.consume_file.s")
def test_edit_pdf_multiple_outputs_with_update_flag_errors( def test_edit_pdf_multiple_outputs_with_update_flag_errors(
@@ -1641,15 +1637,23 @@ class TestPDFActions(DirectoriesMixin, TestCase):
mock_group.assert_not_called() mock_group.assert_not_called()
mock_consume_file.assert_not_called() mock_consume_file.assert_not_called()
@mock.patch("pikepdf.open")
def test_edit_pdf_rejects_invalid_operations(self, mock_open) -> None:
for operations in ([], [{"page": 1, "doc": 2**32}]):
with self.subTest(operations=operations):
with self.assertLogs("paperless.bulk_edit", level="ERROR"):
with self.assertRaisesRegex(ValueError, "index is out of bounds"):
bulk_edit.edit_pdf([self.doc2.id], operations)
mock_open.assert_not_called()
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file.delay") @mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file.delay")
@mock.patch("documents.tasks.consume_file.apply_async") @mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.bulk_edit.tempfile.mkdtemp") @mock.patch("documents.bulk_edit.tempfile.mkdtemp")
@mock.patch("documents.pdf_ops.decrypt_pdf") @mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
def test_remove_password_update_document( def test_remove_password_update_document(
self, self,
mock_needs_decrypt, mock_open,
mock_decrypt,
mock_mkdtemp, mock_mkdtemp,
mock_consume_delay, mock_consume_delay,
mock_update_document, mock_update_document,
@@ -1658,7 +1662,16 @@ class TestPDFActions(DirectoriesMixin, TestCase):
temp_dir = self.dirs.scratch_dir / "remove-password-update" temp_dir = self.dirs.scratch_dir / "remove-password-update"
temp_dir.mkdir(parents=True, exist_ok=True) temp_dir.mkdir(parents=True, exist_ok=True)
mock_mkdtemp.return_value = str(temp_dir) mock_mkdtemp.return_value = str(temp_dir)
mock_decrypt.side_effect = self.fake_decrypt
fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock(), mock.Mock(), mock.Mock()]
fake_pdf.is_encrypted = True
def save_side_effect(target_path):
Path(target_path).write_bytes(b"new pdf content")
fake_pdf.save.side_effect = save_side_effect
mock_open.return_value.__enter__.return_value = fake_pdf
result = bulk_edit.remove_password( result = bulk_edit.remove_password(
[doc.id], [doc.id],
@@ -1667,8 +1680,14 @@ class TestPDFActions(DirectoriesMixin, TestCase):
) )
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
mock_needs_decrypt.assert_called_once_with(doc.source_path) self.assertEqual(
mock_decrypt.assert_called_once_with(doc.source_path, mock.ANY, "secret") mock_open.call_args_list,
[
mock.call(doc.source_path),
mock.call(doc.source_path, password="secret"),
],
)
fake_pdf.remove_unreferenced_resources.assert_called_once()
mock_update_document.assert_not_called() mock_update_document.assert_not_called()
mock_consume_delay.assert_called_once() mock_consume_delay.assert_called_once()
task_kwargs = mock_consume_delay.call_args.kwargs["kwargs"] task_kwargs = mock_consume_delay.call_args.kwargs["kwargs"]
@@ -1681,15 +1700,40 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertEqual(task_kwargs["input_doc"].root_document_id, doc.id) self.assertEqual(task_kwargs["input_doc"].root_document_id, doc.id)
self.assertIsNotNone(task_kwargs["overrides"]) self.assertIsNotNone(task_kwargs["overrides"])
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
@mock.patch("pikepdf.open")
def test_remove_password_update_document_skips_unencrypted_pdf(
self,
mock_open,
mock_mkdtemp,
mock_consume_delay,
) -> None:
doc = self.doc1
fake_pdf = mock.MagicMock()
fake_pdf.is_encrypted = False
mock_open.return_value.__enter__.return_value = fake_pdf
result = bulk_edit.remove_password(
[doc.id],
password="secret",
update_document=True,
)
self.assertEqual(result, "OK")
mock_open.assert_called_once_with(doc.source_path)
fake_pdf.remove_unreferenced_resources.assert_not_called()
fake_pdf.save.assert_not_called()
mock_mkdtemp.assert_not_called()
mock_consume_delay.assert_not_called()
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file.delay") @mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file.delay")
@mock.patch("documents.tasks.consume_file.apply_async") @mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.bulk_edit.tempfile.mkdtemp") @mock.patch("documents.bulk_edit.tempfile.mkdtemp")
@mock.patch("documents.pdf_ops.decrypt_pdf") @mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
def test_remove_password_update_document_uses_source_paths( def test_remove_password_update_document_uses_source_paths(
self, self,
mock_needs_decrypt, mock_open,
mock_decrypt,
mock_mkdtemp, mock_mkdtemp,
mock_consume_delay, mock_consume_delay,
mock_update_document, mock_update_document,
@@ -1700,7 +1744,14 @@ class TestPDFActions(DirectoriesMixin, TestCase):
temp_dir = self.dirs.scratch_dir / "remove-password-source-file" temp_dir = self.dirs.scratch_dir / "remove-password-source-file"
temp_dir.mkdir(parents=True, exist_ok=True) temp_dir.mkdir(parents=True, exist_ok=True)
mock_mkdtemp.return_value = str(temp_dir) mock_mkdtemp.return_value = str(temp_dir)
mock_decrypt.side_effect = self.fake_decrypt
fake_pdf = mock.MagicMock()
self.mock_password_required_pdf(mock_open, fake_pdf)
def save_side_effect(target_path):
Path(target_path).write_bytes(b"new pdf content")
fake_pdf.save.side_effect = save_side_effect
result = bulk_edit.remove_password( result = bulk_edit.remove_password(
[doc.id], [doc.id],
@@ -1710,19 +1761,22 @@ class TestPDFActions(DirectoriesMixin, TestCase):
) )
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
mock_needs_decrypt.assert_called_once_with(source_file) self.assertEqual(
mock_decrypt.assert_called_once_with(source_file, mock.ANY, "secret") mock_open.call_args_list,
[
mock.call(source_file),
mock.call(source_file, password="secret"),
],
)
mock_update_document.assert_not_called() mock_update_document.assert_not_called()
mock_consume_delay.assert_called_once() mock_consume_delay.assert_called_once()
@mock.patch("documents.data_models.magic.from_file", return_value="application/pdf") @mock.patch("documents.data_models.magic.from_file", return_value="application/pdf")
@mock.patch("documents.tasks.consume_file.apply_async") @mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.pdf_ops.decrypt_pdf") @mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
def test_remove_password_explicit_selection_uses_root_source_when_root_selected( def test_remove_password_explicit_selection_uses_root_source_when_root_selected(
self, self,
mock_needs_decrypt, mock_open,
mock_decrypt,
mock_consume_delay, mock_consume_delay,
mock_magic, mock_magic,
) -> None: ) -> None:
@@ -1731,7 +1785,8 @@ class TestPDFActions(DirectoriesMixin, TestCase):
title="A version 1", title="A version 1",
root_document=self.doc1, root_document=self.doc1,
) )
mock_decrypt.return_value = Path("unprotected.pdf") fake_pdf = mock.MagicMock()
self.mock_password_required_pdf(mock_open, fake_pdf)
result = bulk_edit.remove_password( result = bulk_edit.remove_password(
[self.doc1.id], [self.doc1.id],
@@ -1741,11 +1796,12 @@ class TestPDFActions(DirectoriesMixin, TestCase):
) )
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
mock_needs_decrypt.assert_called_once_with(self.doc1.source_path) self.assertEqual(
mock_decrypt.assert_called_once_with( mock_open.call_args_list,
self.doc1.source_path, [
mock.ANY, mock.call(self.doc1.source_path),
"secret", mock.call(self.doc1.source_path, password="secret"),
],
) )
mock_consume_delay.assert_called_once() mock_consume_delay.assert_called_once()
@@ -1753,12 +1809,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
@mock.patch("documents.bulk_edit.group") @mock.patch("documents.bulk_edit.group")
@mock.patch("documents.tasks.consume_file.s") @mock.patch("documents.tasks.consume_file.s")
@mock.patch("documents.bulk_edit.tempfile.mkdtemp") @mock.patch("documents.bulk_edit.tempfile.mkdtemp")
@mock.patch("documents.pdf_ops.decrypt_pdf") @mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
def test_remove_password_creates_consumable_document( def test_remove_password_creates_consumable_document(
self, self,
mock_needs_decrypt: mock.Mock, mock_open: mock.Mock,
mock_decrypt: mock.Mock,
mock_mkdtemp: mock.Mock, mock_mkdtemp: mock.Mock,
mock_consume_file: mock.Mock, mock_consume_file: mock.Mock,
mock_group: mock.Mock, mock_group: mock.Mock,
@@ -1768,7 +1822,15 @@ class TestPDFActions(DirectoriesMixin, TestCase):
temp_dir = self.dirs.scratch_dir / "remove-password" temp_dir = self.dirs.scratch_dir / "remove-password"
temp_dir.mkdir(parents=True, exist_ok=True) temp_dir.mkdir(parents=True, exist_ok=True)
mock_mkdtemp.return_value = str(temp_dir) mock_mkdtemp.return_value = str(temp_dir)
mock_decrypt.side_effect = self.fake_decrypt
fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock(), mock.Mock()]
self.mock_password_required_pdf(mock_open, fake_pdf)
def save_side_effect(target_path: Path) -> None:
target_path.write_bytes(b"password removed")
fake_pdf.save.side_effect = save_side_effect
mock_group.return_value.delay.return_value = None mock_group.return_value.delay.return_value = None
user = User.objects.create(username="owner") user = User.objects.create(username="owner")
@@ -1783,7 +1845,13 @@ class TestPDFActions(DirectoriesMixin, TestCase):
) )
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
mock_decrypt.assert_called_once_with(doc.source_path, mock.ANY, "secret") self.assertEqual(
mock_open.call_args_list,
[
mock.call(doc.source_path),
mock.call(doc.source_path, password="secret"),
],
)
mock_consume_file.assert_called_once() mock_consume_file.assert_called_once()
call_kwargs = mock_consume_file.call_args.kwargs call_kwargs = mock_consume_file.call_args.kwargs
consumable_document = call_kwargs["input_doc"] consumable_document = call_kwargs["input_doc"]
@@ -1805,18 +1873,21 @@ class TestPDFActions(DirectoriesMixin, TestCase):
@mock.patch("documents.bulk_edit.chord") @mock.patch("documents.bulk_edit.chord")
@mock.patch("documents.bulk_edit.group") @mock.patch("documents.bulk_edit.group")
@mock.patch("documents.tasks.consume_file.s") @mock.patch("documents.tasks.consume_file.s")
@mock.patch("documents.pdf_ops.decrypt_pdf") @mock.patch("documents.bulk_edit.tempfile.mkdtemp")
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=False) @mock.patch("pikepdf.open")
def test_remove_password_skips_unencrypted_pdf_without_queueing( def test_remove_password_skips_unencrypted_pdf_without_queueing(
self, self,
mock_needs_decrypt: mock.Mock, mock_open: mock.Mock,
mock_decrypt: mock.Mock, mock_mkdtemp: mock.Mock,
mock_consume_file: mock.Mock, mock_consume_file: mock.Mock,
mock_group: mock.Mock, mock_group: mock.Mock,
mock_chord: mock.Mock, mock_chord: mock.Mock,
mock_delete: mock.Mock, mock_delete: mock.Mock,
) -> None: ) -> None:
doc = self.doc2 doc = self.doc2
fake_pdf = mock.MagicMock()
fake_pdf.is_encrypted = False
mock_open.return_value.__enter__.return_value = fake_pdf
result = bulk_edit.remove_password( result = bulk_edit.remove_password(
[doc.id], [doc.id],
@@ -1826,8 +1897,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
) )
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
mock_needs_decrypt.assert_called_once_with(doc.source_path) mock_open.assert_called_once_with(doc.source_path)
mock_decrypt.assert_not_called() fake_pdf.remove_unreferenced_resources.assert_not_called()
fake_pdf.save.assert_not_called()
mock_mkdtemp.assert_not_called()
mock_consume_file.assert_not_called() mock_consume_file.assert_not_called()
mock_group.assert_not_called() mock_group.assert_not_called()
mock_chord.assert_not_called() mock_chord.assert_not_called()
@@ -1838,12 +1911,10 @@ class TestPDFActions(DirectoriesMixin, TestCase):
@mock.patch("documents.bulk_edit.group") @mock.patch("documents.bulk_edit.group")
@mock.patch("documents.tasks.consume_file.s") @mock.patch("documents.tasks.consume_file.s")
@mock.patch("documents.bulk_edit.tempfile.mkdtemp") @mock.patch("documents.bulk_edit.tempfile.mkdtemp")
@mock.patch("documents.pdf_ops.decrypt_pdf") @mock.patch("pikepdf.open")
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
def test_remove_password_deletes_original( def test_remove_password_deletes_original(
self, self,
mock_needs_decrypt: mock.Mock, mock_open: mock.Mock,
mock_decrypt: mock.Mock,
mock_mkdtemp: mock.Mock, mock_mkdtemp: mock.Mock,
mock_consume_file: mock.Mock, mock_consume_file: mock.Mock,
mock_group: mock.Mock, mock_group: mock.Mock,
@@ -1854,7 +1925,15 @@ class TestPDFActions(DirectoriesMixin, TestCase):
temp_dir = self.dirs.scratch_dir / "remove-password-delete" temp_dir = self.dirs.scratch_dir / "remove-password-delete"
temp_dir.mkdir(parents=True, exist_ok=True) temp_dir.mkdir(parents=True, exist_ok=True)
mock_mkdtemp.return_value = str(temp_dir) mock_mkdtemp.return_value = str(temp_dir)
mock_decrypt.side_effect = self.fake_decrypt
fake_pdf = mock.MagicMock()
fake_pdf.pages = [mock.Mock(), mock.Mock()]
self.mock_password_required_pdf(mock_open, fake_pdf)
def save_side_effect(target_path: Path) -> None:
target_path.write_bytes(b"password removed")
fake_pdf.save.side_effect = save_side_effect
mock_chord.return_value.delay.return_value = None mock_chord.return_value.delay.return_value = None
result = bulk_edit.remove_password( result = bulk_edit.remove_password(
@@ -1866,23 +1945,23 @@ class TestPDFActions(DirectoriesMixin, TestCase):
) )
self.assertEqual(result, "OK") self.assertEqual(result, "OK")
mock_decrypt.assert_called_once_with(doc.source_path, mock.ANY, "secret") self.assertEqual(
mock_open.call_args_list,
[
mock.call(doc.source_path),
mock.call(doc.source_path, password="secret"),
],
)
mock_consume_file.assert_called_once() mock_consume_file.assert_called_once()
mock_group.assert_not_called() mock_group.assert_not_called()
mock_chord.assert_called_once() mock_chord.assert_called_once()
mock_chord.return_value.delay.assert_called_once() mock_chord.return_value.delay.assert_called_once()
mock_delete.si.assert_called_once_with([doc.id]) mock_delete.si.assert_called_once_with([doc.id])
@mock.patch( @mock.patch("pikepdf.open")
"documents.pdf_ops.decrypt_pdf", def test_remove_password_open_failure(self, mock_open: mock.Mock) -> None:
side_effect=RuntimeError("wrong password"), mock_open.side_effect = RuntimeError("wrong password")
)
@mock.patch("documents.pdf_ops.needs_decrypt", return_value=True)
def test_remove_password_failure_raises_value_error(
self,
mock_needs_decrypt: mock.Mock,
mock_decrypt: mock.Mock,
) -> None:
with self.assertLogs("paperless.bulk_edit", level="ERROR") as cm: with self.assertLogs("paperless.bulk_edit", level="ERROR") as cm:
with self.assertRaises(ValueError) as exc: with self.assertRaises(ValueError) as exc:
bulk_edit.remove_password([self.doc1.id], password="secret") bulk_edit.remove_password([self.doc1.id], password="secret")
-520
View File
@@ -1,520 +0,0 @@
"""
Tests for documents.pdf_ops.
These use real PDFs from the sample directories. No database, Celery or mocks.
Pages are compared by a hash of their content stream, so page identity and order
are easy to assert.
"""
import hashlib
from collections.abc import Callable
from pathlib import Path
import pikepdf
import pytest
from documents import pdf_ops
from documents.pdf_ops import PageSpec
SRC_ROOT = Path(__file__).parents[2]
SAMPLES = Path(__file__).parent / "samples"
THREE_PAGES = SAMPLES / "documents" / "originals" / "0000002.pdf"
TWELVE_PAGES = SAMPLES / "barcodes" / "split-by-asn-2.pdf"
ENCRYPTED = SAMPLES / "password-is-test.pdf"
SIGNED = SRC_ROOT / "paperless" / "tests" / "samples" / "tesseract" / "signed.pdf"
def _page_fingerprint(page: pikepdf.Page) -> str:
contents = page.obj.get("/Contents")
assert contents is not None, "sample page has no /Contents"
streams = list(contents) if isinstance(contents, pikepdf.Array) else [contents]
return hashlib.sha256(b"".join(s.read_bytes() for s in streams)).hexdigest()
def fingerprints(path: Path) -> list[str]:
with pikepdf.open(path) as pdf:
return [_page_fingerprint(page) for page in pdf.pages]
def rotations(path: Path) -> list[int]:
with pikepdf.open(path) as pdf:
return [int(page.obj.get("/Rotate", 0)) for page in pdf.pages]
def docinfo_keys(path: Path) -> set[str]:
with pikepdf.open(path) as pdf:
return set(pdf.docinfo.keys())
def constant(path: Path) -> Callable[[], Path]:
return lambda: path
@pytest.fixture
def source_fingerprints() -> list[str]:
fps = fingerprints(THREE_PAGES)
assert len(set(fps)) == 3, "sample must have three distinct pages"
return fps
class TestRotatePdf:
def test_rotation_is_relative_and_applies_to_every_page(
self,
tmp_path: Path,
) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- It is rotated by 90 degrees, then the result is rotated by 90 again
THEN:
- Every page is rotated relative to its current rotation
- The page content itself is unchanged
"""
once = tmp_path / "once.pdf"
twice = tmp_path / "twice.pdf"
pdf_ops.rotate_pdf(THREE_PAGES, once, 90)
pdf_ops.rotate_pdf(once, twice, 90)
assert rotations(once) == [90, 90, 90]
assert rotations(twice) == [180, 180, 180]
assert fingerprints(twice) == fingerprints(THREE_PAGES)
def test_keeps_document_info(self, tmp_path: Path) -> None:
"""
GIVEN:
- A PDF with document info
WHEN:
- It is rotated
THEN:
- The document info is still present in the output
"""
dst = tmp_path / "out.pdf"
pdf_ops.rotate_pdf(THREE_PAGES, dst, 90)
assert "/Creator" in docinfo_keys(dst)
class TestRemovePages:
@pytest.mark.parametrize(
("pages", "kept"),
[
pytest.param([2], [0, 2], id="single"),
pytest.param([3, 1], [1], id="unordered"),
pytest.param([2, 2], [0, 2], id="duplicates-remove-once"),
pytest.param([], [0, 1, 2], id="empty-keeps-everything"),
pytest.param([1, 2, 3], [], id="every-page"),
],
)
def test_removes_only_the_requested_pages(
self,
tmp_path: Path,
source_fingerprints: list[str],
pages: list[int],
kept: list[int],
) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- Pages are removed, in any order and possibly repeated
THEN:
- Exactly the other pages remain, in their original order
"""
dst = tmp_path / "out.pdf"
pdf_ops.remove_pages(THREE_PAGES, dst, pages)
assert fingerprints(dst) == [source_fingerprints[i] for i in kept]
def test_keeps_document_info(self, tmp_path: Path) -> None:
"""
GIVEN:
- A PDF with document info
WHEN:
- A page is removed
THEN:
- The document info is still present in the output
"""
dst = tmp_path / "out.pdf"
pdf_ops.remove_pages(THREE_PAGES, dst, [1])
assert "/Creator" in docinfo_keys(dst)
@pytest.mark.parametrize(
"bad_page",
[
pytest.param(0, id="zero"),
pytest.param(-1, id="negative"),
],
)
def test_rejects_pages_below_one(self, tmp_path: Path, bad_page: int) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- Pages are removed and one of them is below 1
THEN:
- ValueError is raised
- No output file is written
"""
dst = tmp_path / "out.pdf"
with pytest.raises(ValueError, match="start at 1"):
pdf_ops.remove_pages(THREE_PAGES, dst, [1, bad_page])
assert not dst.exists()
class TestBuildPdfs:
def test_selects_and_orders_pages(
self,
tmp_path: Path,
source_fingerprints: list[str],
) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- One output is built from pages 3 then 1
THEN:
- The output has those pages in that order
- Its path is returned
"""
dst = tmp_path / "out.pdf"
written = pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(3), PageSpec(1)], constant(dst))],
)
assert written == [dst]
assert fingerprints(dst) == [source_fingerprints[2], source_fingerprints[0]]
def test_rotates_only_the_requested_pages(self, tmp_path: Path) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- One output is built with a different rotation on each page
THEN:
- Each output page has exactly the rotation requested for it
"""
dst = tmp_path / "out.pdf"
pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(1), PageSpec(2, 90), PageSpec(3, 180)], constant(dst))],
)
assert rotations(dst) == [0, 90, 180]
def test_writes_one_file_per_output_in_order(self, tmp_path: Path) -> None:
"""
GIVEN:
- A twelve page PDF
WHEN:
- Two outputs are built from different page ranges
THEN:
- Two files are written and returned in output order
- Each holds exactly its own pages
"""
first = tmp_path / "first.pdf"
second = tmp_path / "second.pdf"
source = fingerprints(TWELVE_PAGES)
written = pdf_ops.build_pdfs(
TWELVE_PAGES,
[
([PageSpec(p) for p in (1, 2, 3)], constant(first)),
([PageSpec(p) for p in range(4, 13)], constant(second)),
],
)
assert written == [first, second]
assert fingerprints(first) == source[:3]
assert fingerprints(second) == source[3:]
def test_empty_page_list_writes_a_zero_page_file(self, tmp_path: Path) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- An output with no pages is built
THEN:
- A PDF with zero pages is written
"""
dst = tmp_path / "out.pdf"
pdf_ops.build_pdfs(THREE_PAGES, [([], constant(dst))])
assert fingerprints(dst) == []
def test_destination_is_not_requested_when_a_page_is_out_of_range(
self,
tmp_path: Path,
) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- Several outputs are built
- A later output refers to a page past the end
THEN:
- IndexError is raised
- No destination was requested for any output, including earlier valid ones
"""
requested: list[Path] = []
def make_dst() -> Path:
requested.append(tmp_path / "out.pdf")
return requested[-1]
with pytest.raises(IndexError):
pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(1)], make_dst), ([PageSpec(99)], make_dst)],
)
assert requested == []
@pytest.mark.parametrize(
"bad_page",
[
pytest.param(0, id="zero"),
pytest.param(-1, id="negative"),
],
)
def test_rejects_pages_below_one_before_opening_anything(
self,
tmp_path: Path,
bad_page: int,
) -> None:
"""
GIVEN:
- A three page PDF
WHEN:
- Several outputs are built and a later one has a page below 1
THEN:
- ValueError is raised
- No destination was requested for any output
"""
requested: list[Path] = []
def make_dst() -> Path:
requested.append(tmp_path / "out.pdf")
return requested[-1]
with pytest.raises(ValueError, match="start at 1"):
pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(1)], make_dst), ([PageSpec(bad_page)], make_dst)],
)
assert requested == []
class TestValidatePageOperations:
def test_returns_the_output_count(self) -> None:
"""
GIVEN:
- Operations that all target the default output
WHEN:
- They are validated
THEN:
- One output document is reported
"""
operations = [{"page": 1}, {"page": 2}, {"page": 3}]
assert pdf_ops.validate_page_operations(operations, single_output=True) == 1
def test_gap_in_output_indices_counts_up_to_the_highest(self) -> None:
"""
GIVEN:
- Operations that target outputs 0 and 2 but never 1
WHEN:
- They are validated
THEN:
- Three output documents are reported
"""
operations = [
{"page": 1, "doc": 0},
{"page": 2, "doc": 2},
{"page": 3, "doc": 0},
]
count = pdf_ops.validate_page_operations(operations, single_output=False)
assert count == 3
def test_empty_operations_are_rejected(self) -> None:
"""
GIVEN:
- No operations
WHEN:
- They are validated
THEN:
- ValueError is raised
"""
with pytest.raises(ValueError, match="index is out of bounds"):
pdf_ops.validate_page_operations([], single_output=False)
def test_multiple_outputs_rejected_when_single_output_required(self) -> None:
"""
GIVEN:
- Operations that target two outputs
WHEN:
- They are validated with a single output required
THEN:
- ValueError is raised
"""
operations = [{"page": 1, "doc": 0}, {"page": 2, "doc": 1}]
with pytest.raises(ValueError, match="Multiple output documents"):
pdf_ops.validate_page_operations(operations, single_output=True)
@pytest.mark.parametrize(
"doc",
[
pytest.param(-1, id="negative"),
pytest.param(2, id="equal-to-operation-count"),
pytest.param(2**32, id="huge"),
],
)
def test_output_index_out_of_bounds(self, doc: int) -> None:
"""
GIVEN:
- Two operations, one with an output index that is out of bounds
WHEN:
- They are validated
THEN:
- ValueError is raised
"""
operations = [{"page": 1, "doc": 0}, {"page": 2, "doc": doc}]
with pytest.raises(ValueError, match="index is out of bounds"):
pdf_ops.validate_page_operations(operations, single_output=False)
class TestDecrypt:
@pytest.mark.parametrize(
("path", "expected"),
[
pytest.param(ENCRYPTED, True, id="password-required"),
pytest.param(SIGNED, True, id="opens-without-password-but-encrypted"),
pytest.param(THREE_PAGES, False, id="not-encrypted"),
],
)
def test_needs_decrypt(self, path: Path, *, expected: bool) -> None:
"""
GIVEN:
- A PDF that is encrypted, or encrypted but openable, or plain
WHEN:
- needs_decrypt is asked about it
THEN:
- Only the unencrypted PDF reports False
"""
assert pdf_ops.needs_decrypt(path) is expected
def test_decrypt_writes_an_unencrypted_copy(self, tmp_path: Path) -> None:
"""
GIVEN:
- A password protected PDF
WHEN:
- It is decrypted with the correct password
THEN:
- The path from make_dst is returned
- The written copy no longer needs decrypting
"""
dst = tmp_path / "out.pdf"
result = pdf_ops.decrypt_pdf(ENCRYPTED, constant(dst), "test")
assert result == dst
assert pdf_ops.needs_decrypt(dst) is False
def test_wrong_password_raises_and_never_requests_a_destination(
self,
tmp_path: Path,
) -> None:
"""
GIVEN:
- A password protected PDF
WHEN:
- It is decrypted with the wrong password
THEN:
- PasswordError is raised
- No destination was requested
"""
requested: list[Path] = []
def make_dst() -> Path:
requested.append(tmp_path / "out.pdf")
return requested[-1]
with pytest.raises(pikepdf.PasswordError):
pdf_ops.decrypt_pdf(ENCRYPTED, make_dst, "wrong")
assert requested == []
class TestPdfMerger:
def test_pages_are_appended_in_the_order_added(
self,
tmp_path: Path,
source_fingerprints: list[str],
) -> None:
"""
GIVEN:
- A reordered PDF and the original three page PDF
WHEN:
- Both are added to a merger in that order and saved
THEN:
- The output holds all pages in the order they were added
"""
reordered = tmp_path / "reordered.pdf"
merged = tmp_path / "merged.pdf"
pdf_ops.build_pdfs(
THREE_PAGES,
[([PageSpec(3), PageSpec(1)], constant(reordered))],
)
with pdf_ops.PdfMerger() as merger:
merger.add(reordered)
merger.add(THREE_PAGES)
merger.save(merged)
assert fingerprints(merged) == [
source_fingerprints[2],
source_fingerprints[0],
*source_fingerprints,
]
def test_output_version_is_at_least_the_highest_source_version(
self,
tmp_path: Path,
) -> None:
"""
GIVEN:
- Two PDFs with different PDF versions
WHEN:
- Both are added to a merger and saved
THEN:
- The output version is at least the highest source version
"""
merged = tmp_path / "merged.pdf"
with pikepdf.open(TWELVE_PAGES) as pdf:
source_versions = [pdf.pdf_version]
with pikepdf.open(THREE_PAGES) as pdf:
source_versions.append(pdf.pdf_version)
with pdf_ops.PdfMerger() as merger:
merger.add(TWELVE_PAGES)
merger.add(THREE_PAGES)
merger.save(merged)
with pikepdf.open(merged) as pdf:
assert pdf.pdf_version >= max(source_versions)
+19 -2
View File
@@ -2,7 +2,7 @@ msgid ""
msgstr "" msgstr ""
"Project-Id-Version: paperless-ngx\n" "Project-Id-Version: paperless-ngx\n"
"Report-Msgid-Bugs-To: \n" "Report-Msgid-Bugs-To: \n"
"POT-Creation-Date: 2026-10-06 15:12+0000\n" "POT-Creation-Date: 2026-10-05 16:26+0000\n"
"PO-Revision-Date: 2022-02-17 04:17\n" "PO-Revision-Date: 2022-02-17 04:17\n"
"Last-Translator: \n" "Last-Translator: \n"
"Language-Team: English\n" "Language-Team: English\n"
@@ -1941,8 +1941,25 @@ msgstr ""
msgid "As a final step, please complete the following form:" msgid "As a final step, please complete the following form:"
msgstr "" msgstr ""
#: documents/validators.py:24
#, python-brace-format
msgid "Unable to parse URI {value}, missing scheme"
msgstr ""
#: documents/validators.py:29
#, python-brace-format
msgid "Unable to parse URI {value}, missing net location or path"
msgstr ""
#: documents/validators.py:36 #: documents/validators.py:36
msgid ", " msgid ""
"URI scheme '{parts.scheme}' is not allowed. Allowed schemes: {', '."
"join(allowed_schemes)}"
msgstr ""
#: documents/validators.py:45
#, python-brace-format
msgid "Unable to parse URI {value}"
msgstr "" msgstr ""
#: documents/views.py:336 documents/views.py:2729 #: documents/views.py:336 documents/views.py:2729
@@ -137,20 +137,9 @@ class TestNginxService:
reason="No Gotenberg/Tika servers to test with", reason="No Gotenberg/Tika servers to test with",
) )
class TestParserLive: class TestParserLive:
# Rasterizer versions shift a few pixels, so compare perceptual hashes by @staticmethod
# Hamming distance (out of 18 * 18 = 324 bits) rather than for equality def imagehash(file: Path, hash_size: int = 18) -> str:
MAX_HASH_DISTANCE = 8 return f"{average_hash(Image.open(file), hash_size)}"
@classmethod
def assert_thumbnails_similar(cls, generated: Path, expected: Path) -> None:
distance = average_hash(Image.open(generated), 18) - average_hash(
Image.open(expected),
18,
)
assert distance <= cls.MAX_HASH_DISTANCE, (
f"Thumbnail {generated} differs from {expected} by {distance} bits "
f"(max {cls.MAX_HASH_DISTANCE})"
)
def test_get_thumbnail( def test_get_thumbnail(
self, self,
@@ -179,7 +168,12 @@ class TestParserLive:
assert thumb.exists() assert thumb.exists()
assert thumb.is_file() assert thumb.is_file()
self.assert_thumbnails_similar(thumb, simple_txt_email_thumbnail_file) assert self.imagehash(thumb) == self.imagehash(
simple_txt_email_thumbnail_file,
), (
f"Created thumbnail {thumb} differs from expected file "
f"{simple_txt_email_thumbnail_file}"
)
def test_tika_parse_successful(self, mail_parser: MailDocumentParser) -> None: def test_tika_parse_successful(self, mail_parser: MailDocumentParser) -> None:
""" """
@@ -261,7 +255,7 @@ class TestParserLive:
THEN: THEN:
- Gotenberg shall be called to generate the PDF - Gotenberg shall be called to generate the PDF
- The archive PDF shall contain the expected content - The archive PDF shall contain the expected content
- The generated thumbnail shall be perceptually close to the expected image - The generated thumbnail shall match the expected image hash
""" """
util_call_with_backoff(mail_parser.parse, [html_email_file, "message/rfc822"]) util_call_with_backoff(mail_parser.parse, [html_email_file, "message/rfc822"])
@@ -278,4 +272,14 @@ class TestParserLive:
html_email_file, html_email_file,
"message/rfc822", "message/rfc822",
) )
self.assert_thumbnails_similar(generated_thumbnail, html_email_thumbnail_file) generated_thumbnail_hash = self.imagehash(generated_thumbnail)
# The created PDF is not reproducible, but the converted image
# should always look the same
expected_hash = self.imagehash(html_email_thumbnail_file)
assert generated_thumbnail_hash == expected_hash, (
f"PDF thumbnail differs from expected. "
f"Generated: {generated_thumbnail}, "
f"Hash: {generated_thumbnail_hash} vs {expected_hash}"
)
+1 -1
View File
@@ -1,6 +1,6 @@
from typing import Final from typing import Final
__version__: Final[tuple[int, int, int]] = (3, 2, 1) __version__: Final[tuple[int, int, int]] = (3, 3, 0)
# Version string like X.Y.Z # Version string like X.Y.Z
__full_version_str__: Final[str] = ".".join(map(str, __version__)) __full_version_str__: Final[str] = ".".join(map(str, __version__))
# Version string like X.Y # Version string like X.Y
Generated
+1 -1
View File
@@ -2971,7 +2971,7 @@ wheels = [
[[package]] [[package]]
name = "paperless-ngx" name = "paperless-ngx"
version = "3.2.1" version = "3.3.0"
source = { virtual = "." } source = { virtual = "." }
dependencies = [ dependencies = [
{ name = "azure-ai-documentintelligence" }, { name = "azure-ai-documentintelligence" },