mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-10-08 09:07:13 +00:00
Compare commits
6
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f8876c55e3 | ||
|
|
e5b215f9d3 | ||
|
|
94f10b8622 | ||
|
|
6f7ed1d4a7 | ||
|
|
0428bf6955 | ||
|
|
c63afb47b2 |
No files matched your search
@@ -80,7 +80,7 @@ jobs:
|
||||
needs: changes
|
||||
if: needs.changes.outputs.backend_changed == 'true'
|
||||
name: "Python ${{ matrix.python-version }}"
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
contents: read
|
||||
strategy:
|
||||
@@ -111,10 +111,10 @@ jobs:
|
||||
timeout-minutes: 12
|
||||
uses: $/.github/actions/apt-install
|
||||
with:
|
||||
packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils qpdf
|
||||
packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils
|
||||
- name: Configure ImageMagick
|
||||
run: |
|
||||
sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-7/policy.xml
|
||||
sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-6/policy.xml
|
||||
- name: Install Python dependencies
|
||||
env:
|
||||
PYTHON_VERSION: ${{ steps.setup-python.outputs.python-version }}
|
||||
@@ -158,7 +158,7 @@ jobs:
|
||||
needs: changes
|
||||
if: needs.changes.outputs.backend_changed == 'true'
|
||||
name: Check project typing
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
contents: read
|
||||
env:
|
||||
|
||||
@@ -24,10 +24,10 @@ jobs:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- runner: ubuntu-26.04
|
||||
- runner: ubuntu-24.04
|
||||
arch: amd64
|
||||
platform: linux/amd64
|
||||
- runner: ubuntu-26.04-arm
|
||||
- runner: ubuntu-24.04-arm
|
||||
arch: arm64
|
||||
platform: linux/arm64
|
||||
runs-on: ${{ matrix.runner }}
|
||||
@@ -163,7 +163,7 @@ jobs:
|
||||
archive: false
|
||||
merge-and-push:
|
||||
name: Merge and Push Manifest
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
needs: build-arch
|
||||
if: needs.build-arch.outputs.should-push == 'true'
|
||||
environment: image-publishing
|
||||
|
||||
@@ -65,7 +65,7 @@ jobs:
|
||||
needs: changes
|
||||
if: needs.changes.outputs.docs_changed == 'true'
|
||||
name: Build Documentation
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
steps:
|
||||
- uses: actions/configure-pages@45bfe0192ca1faeb007ade9deae92b16b8254a0d # v6.0.0
|
||||
- name: Checkout
|
||||
@@ -102,7 +102,7 @@ jobs:
|
||||
name: Deploy Documentation
|
||||
needs: [changes, build]
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.changes.outputs.docs_changed == 'true'
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
pages: write
|
||||
id-token: write
|
||||
|
||||
@@ -72,7 +72,7 @@ jobs:
|
||||
needs: changes
|
||||
if: needs.changes.outputs.frontend_changed == 'true'
|
||||
name: Install Dependencies
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
@@ -104,7 +104,7 @@ jobs:
|
||||
name: Lint
|
||||
needs: [changes, install-dependencies]
|
||||
if: needs.changes.outputs.frontend_changed == 'true'
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
@@ -137,7 +137,7 @@ jobs:
|
||||
name: "Unit Tests (${{ matrix.shard-index }}/${{ matrix.shard-count }})"
|
||||
needs: [changes, install-dependencies]
|
||||
if: needs.changes.outputs.frontend_changed == 'true'
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
contents: read
|
||||
strategy:
|
||||
@@ -188,10 +188,10 @@ jobs:
|
||||
name: E2E Tests
|
||||
needs: [changes, install-dependencies]
|
||||
if: needs.changes.outputs.frontend_changed == 'true'
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
contents: read
|
||||
container: mcr.microsoft.com/playwright:v1.62.1-resolute
|
||||
container: mcr.microsoft.com/playwright:v1.62.1-noble
|
||||
env:
|
||||
PLAYWRIGHT_BROWSERS_PATH: /ms-playwright
|
||||
PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD: 1
|
||||
@@ -246,7 +246,7 @@ jobs:
|
||||
name: Frontend Build
|
||||
needs: [changes, unit-tests, e2e-tests]
|
||||
if: needs.changes.outputs.frontend_changed == 'true'
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
|
||||
@@ -14,7 +14,7 @@ permissions: {}
|
||||
jobs:
|
||||
wait-for-docker:
|
||||
name: Wait for Docker Build
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
checks: read
|
||||
statuses: read
|
||||
@@ -30,7 +30,7 @@ jobs:
|
||||
build-release:
|
||||
name: Build Release
|
||||
needs: wait-for-docker
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
@@ -73,7 +73,7 @@ jobs:
|
||||
timeout-minutes: 12
|
||||
uses: $/.github/actions/apt-install
|
||||
with:
|
||||
packages: gettext libleptonica6
|
||||
packages: gettext liblept5
|
||||
# ---- Build Documentation ----
|
||||
- name: Build documentation
|
||||
env:
|
||||
@@ -145,7 +145,7 @@ jobs:
|
||||
publish-release:
|
||||
name: Publish Release
|
||||
needs: build-release
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
@@ -197,7 +197,7 @@ jobs:
|
||||
name: Append Changelog
|
||||
needs: publish-release
|
||||
if: needs.publish-release.outputs.prerelease == 'false'
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
|
||||
@@ -15,7 +15,7 @@ permissions:
|
||||
jobs:
|
||||
zizmor:
|
||||
name: Run zizmor
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
contents: read
|
||||
actions: read
|
||||
@@ -29,7 +29,7 @@ jobs:
|
||||
uses: zizmorcore/zizmor-action@cc914d7f3750a2d13d75c7f184a1060aa0e9d482 # v0.6.4
|
||||
semgrep:
|
||||
name: Semgrep CE
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
container:
|
||||
image: semgrep/semgrep:1.155.0@sha256:cc869c685dcc0fe497c86258da9f205397d8108e56d21a86082ea4886e52784d
|
||||
if: github.actor != 'dependabot[bot]'
|
||||
|
||||
@@ -17,7 +17,7 @@ jobs:
|
||||
cleanup-images:
|
||||
name: Cleanup Image Tags for ${{ matrix.primary-name }}
|
||||
if: github.repository_owner == 'paperless-ngx'
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
environment: registry-maintenance
|
||||
strategy:
|
||||
fail-fast: false
|
||||
@@ -42,7 +42,7 @@ jobs:
|
||||
cleanup-untagged-images:
|
||||
name: Cleanup Untagged Images Tags for ${{ matrix.primary-name }}
|
||||
if: github.repository_owner == 'paperless-ngx'
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
needs:
|
||||
- cleanup-images
|
||||
environment: registry-maintenance
|
||||
|
||||
@@ -21,7 +21,7 @@ on:
|
||||
jobs:
|
||||
analyze:
|
||||
name: Analyze
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
@@ -13,7 +13,7 @@ jobs:
|
||||
synchronize-with-crowdin:
|
||||
name: Crowdin Sync
|
||||
if: github.repository_owner == 'paperless-ngx'
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
environment: translation-sync
|
||||
steps:
|
||||
- name: Checkout
|
||||
|
||||
@@ -7,7 +7,7 @@ jobs:
|
||||
# Note: peakoss/anti-slop does not support the `issues` event yet (all of its
|
||||
# issue inputs are still commented out upstream), so the checks that the PR Bot
|
||||
# workflow gets from the action are implemented manually here.
|
||||
runs-on: ubuntu-slim
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
issues: write
|
||||
steps:
|
||||
|
||||
@@ -4,7 +4,7 @@ on:
|
||||
types: [opened]
|
||||
jobs:
|
||||
Anti-slop:
|
||||
runs-on: ubuntu-slim
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
issues: read
|
||||
@@ -24,7 +24,7 @@ jobs:
|
||||
ASLOP-PR-VERIFY
|
||||
pr-bot:
|
||||
name: Automated PR Bot
|
||||
runs-on: ubuntu-slim
|
||||
runs-on: ubuntu-latest
|
||||
# Runs after Anti-slop so the welcome comment can see whether the PR was closed
|
||||
# instead of racing it. Still runs if that job fails, so labeling is not lost.
|
||||
needs: Anti-slop
|
||||
|
||||
@@ -12,7 +12,7 @@ permissions:
|
||||
jobs:
|
||||
pr_opened_or_reopened:
|
||||
name: pr_opened_or_reopened
|
||||
runs-on: ubuntu-slim
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
# write permission is required for autolabeler
|
||||
pull-requests: write
|
||||
|
||||
@@ -9,7 +9,7 @@ jobs:
|
||||
stale:
|
||||
name: 'Stale'
|
||||
if: github.repository_owner == 'paperless-ngx'
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
issues: write
|
||||
pull-requests: write
|
||||
@@ -34,7 +34,7 @@ jobs:
|
||||
lock-threads:
|
||||
name: 'Lock Old Threads'
|
||||
if: github.repository_owner == 'paperless-ngx'
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
issues: write
|
||||
pull-requests: write
|
||||
@@ -58,7 +58,7 @@ jobs:
|
||||
close-answered-discussions:
|
||||
name: 'Close Answered Discussions'
|
||||
if: github.repository_owner == 'paperless-ngx'
|
||||
runs-on: ubuntu-slim
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
discussions: write
|
||||
steps:
|
||||
@@ -117,7 +117,7 @@ jobs:
|
||||
close-outdated-discussions:
|
||||
name: 'Close Outdated Discussions'
|
||||
if: github.repository_owner == 'paperless-ngx'
|
||||
runs-on: ubuntu-slim
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
discussions: write
|
||||
steps:
|
||||
@@ -211,7 +211,7 @@ jobs:
|
||||
close-unsupported-feature-requests:
|
||||
name: 'Close Unsupported Feature Requests'
|
||||
if: github.repository_owner == 'paperless-ngx'
|
||||
runs-on: ubuntu-slim
|
||||
runs-on: ubuntu-24.04
|
||||
permissions:
|
||||
discussions: write
|
||||
steps:
|
||||
|
||||
@@ -8,7 +8,7 @@ env:
|
||||
jobs:
|
||||
generate-translate-strings:
|
||||
name: Generate Translation Strings
|
||||
runs-on: ubuntu-26.04
|
||||
runs-on: ubuntu-latest
|
||||
environment: translation-sync
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
@@ -40,7 +40,6 @@ services:
|
||||
volumes:
|
||||
- dbdata:/var/lib/mysql
|
||||
environment:
|
||||
MARIADB_HOST: paperless
|
||||
MARIADB_DATABASE: paperless
|
||||
MARIADB_USER: paperless
|
||||
MARIADB_PASSWORD: paperless
|
||||
|
||||
@@ -36,7 +36,6 @@ services:
|
||||
volumes:
|
||||
- dbdata:/var/lib/mysql
|
||||
environment:
|
||||
MARIADB_HOST: paperless
|
||||
MARIADB_DATABASE: paperless
|
||||
MARIADB_USER: paperless
|
||||
MARIADB_PASSWORD: paperless
|
||||
|
||||
@@ -1,5 +1,130 @@
|
||||
# Changelog
|
||||
|
||||
## paperless-ngx 3.3.0
|
||||
|
||||
### Features / Enhancements
|
||||
|
||||
- Enhancement: more control over suggestion requests [@shamoon](https://github.com/shamoon) ([#14258](https://github.com/paperless-ngx/paperless-ngx/pull/14258))
|
||||
- Chorehancement: set manifest CORS for credentials [@shamoon](https://github.com/shamoon) ([#14307](https://github.com/paperless-ngx/paperless-ngx/pull/14307))
|
||||
- Enhancement: include Django admin with 2FA [@shamoon](https://github.com/shamoon) ([#14270](https://github.com/paperless-ngx/paperless-ngx/pull/14270))
|
||||
- Feature: propagate resolved secrets to interactive container shells [@stumpylog](https://github.com/stumpylog) ([#14254](https://github.com/paperless-ngx/paperless-ngx/pull/14254))
|
||||
- Enhancement: support separate embedding API key [@furkanural](https://github.com/furkanural) ([#14067](https://github.com/paperless-ngx/paperless-ngx/pull/14067))
|
||||
- Enhancement: support passthrough extra params for LLMs [@shamoon](https://github.com/shamoon) ([#14202](https://github.com/paperless-ngx/paperless-ngx/pull/14202))
|
||||
- Feature: store barcode contents, list and search them [@jurassicparkicecream](https://github.com/jurassicparkicecream) ([#14276](https://github.com/paperless-ngx/paperless-ngx/pull/14276))
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
- Fix: Set the ProcessedMail owner based on the rule owner in all cases [@stumpylog](https://github.com/stumpylog) ([#14356](https://github.com/paperless-ngx/paperless-ngx/pull/14356))
|
||||
- Fix: Ensure log rotation settings are converted to integers [@stumpylog](https://github.com/stumpylog) ([#14343](https://github.com/paperless-ngx/paperless-ngx/pull/14343))
|
||||
- Fix: Wrap apt calls into a retry so we can ideally jump a slow mirror [@stumpylog](https://github.com/stumpylog) ([#14344](https://github.com/paperless-ngx/paperless-ngx/pull/14344))
|
||||
- Fix: ship pdf.js CMaps so CJK documents render in the viewer [@MrOggy85](https://github.com/MrOggy85) ([#14318](https://github.com/paperless-ngx/paperless-ngx/pull/14318))
|
||||
- Fix: use version page\_count for versioned document [@shamoon](https://github.com/shamoon) ([#14280](https://github.com/paperless-ngx/paperless-ngx/pull/14280))
|
||||
- Fix: allow pointer events for pdf links in pngx viewer [@shamoon](https://github.com/shamoon) ([#14264](https://github.com/paperless-ngx/paperless-ngx/pull/14264))
|
||||
- Fix: During a move to the trash directory, only attempt to copy metadata [@stumpylog](https://github.com/stumpylog) ([#14250](https://github.com/paperless-ngx/paperless-ngx/pull/14250))
|
||||
- Fix: convert file mtime to the configured time zone directly [@stumpylog](https://github.com/stumpylog) ([#14249](https://github.com/paperless-ngx/paperless-ngx/pull/14249))
|
||||
- Fix: ensure documentDeleted subscription is discarded [@shamoon](https://github.com/shamoon) ([#14247](https://github.com/paperless-ngx/paperless-ngx/pull/14247))
|
||||
- Chore: Fix bugs in the test suite [@stumpylog](https://github.com/stumpylog) ([#14244](https://github.com/paperless-ngx/paperless-ngx/pull/14244))
|
||||
- Fix: ensure bulk operations are checked against version root [@shamoon](https://github.com/shamoon) ([#14246](https://github.com/paperless-ngx/paperless-ngx/pull/14246))
|
||||
- Fix: indexing after document-added workflows signal [@shamoon](https://github.com/shamoon) ([#14242](https://github.com/paperless-ngx/paperless-ngx/pull/14242))
|
||||
- Fix: Record full tag and custom field lists in bulk edit audit log [@stumpylog](https://github.com/stumpylog) ([#14236](https://github.com/paperless-ngx/paperless-ngx/pull/14236))
|
||||
- Chore: update pikepdf for ocrmypdf requirement [@shamoon](https://github.com/shamoon) ([#14235](https://github.com/paperless-ngx/paperless-ngx/pull/14235))
|
||||
- Fix: handle legacy bulk edit split page range with missing page\_count [@shamoon](https://github.com/shamoon) ([#14212](https://github.com/paperless-ngx/paperless-ngx/pull/14212))
|
||||
- Fix: ignore invalid EXIF orientation when generating image archives [@zhzy0077](https://github.com/zhzy0077) ([#14203](https://github.com/paperless-ngx/paperless-ngx/pull/14203))
|
||||
|
||||
### Documentation
|
||||
|
||||
- Documentation: correct duplicates info [@shamoon](https://github.com/shamoon) ([#14243](https://github.com/paperless-ngx/paperless-ngx/pull/14243))
|
||||
|
||||
### Maintenance
|
||||
|
||||
- Chore(deps): Bump the actions group across 1 directory with 4 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14332](https://github.com/paperless-ngx/paperless-ngx/pull/14332))
|
||||
- Fix: Wrap apt calls into a retry so we can ideally jump a slow mirror [@stumpylog](https://github.com/stumpylog) ([#14344](https://github.com/paperless-ngx/paperless-ngx/pull/14344))
|
||||
- Chore(deps): Bump the actions group across 1 directory with 10 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14301](https://github.com/paperless-ngx/paperless-ngx/pull/14301))
|
||||
|
||||
### Dependencies
|
||||
|
||||
<details>
|
||||
<summary>27 changes</summary>
|
||||
|
||||
- Chore(deps): Bump django-filter from 25.2 to 26.1 @[dependabot[bot]](https://github.com/apps/dependabot) ([#14337](https://github.com/paperless-ngx/paperless-ngx/pull/14337))
|
||||
- Chore(deps): Bump the utilities-patch group across 1 directory with 2 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14338](https://github.com/paperless-ngx/paperless-ngx/pull/14338))
|
||||
- Chore(deps): Bump the pre-commit-dependencies group across 1 directory with 3 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14351](https://github.com/paperless-ngx/paperless-ngx/pull/14351))
|
||||
- Chore(deps-dev): Bump types-channels from 4.3.0.20260408 to 4.3.0.20260518 @[dependabot[bot]](https://github.com/apps/dependabot) ([#14335](https://github.com/paperless-ngx/paperless-ngx/pull/14335))
|
||||
- docker(deps): Bump astral-sh/uv from 0.12.20-python3.14-trixie-slim to 0.12.23-python3.14-trixie-slim @[dependabot[bot]](https://github.com/apps/dependabot) ([#14327](https://github.com/paperless-ngx/paperless-ngx/pull/14327))
|
||||
- Chore(deps): Bump the actions group across 1 directory with 4 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14332](https://github.com/paperless-ngx/paperless-ngx/pull/14332))
|
||||
- Chore(deps): Bump the uv group across 1 directory with 2 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14325](https://github.com/paperless-ngx/paperless-ngx/pull/14325))
|
||||
- Chore(deps): Bump the frontend-angular-dependencies group across 1 directory with 13 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14329](https://github.com/paperless-ngx/paperless-ngx/pull/14329))
|
||||
- Chore(deps-dev): Bump prettier from 3.9.8 to 3.9.9 in /src-ui @[dependabot[bot]](https://github.com/apps/dependabot) ([#14331](https://github.com/paperless-ngx/paperless-ngx/pull/14331))
|
||||
- Chore(deps-dev): Bump the frontend-eslint-dependencies group across 1 directory with 3 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14330](https://github.com/paperless-ngx/paperless-ngx/pull/14330))
|
||||
- Chore(deps-dev): Bump zensical from 0.0.64 to 0.0.65 in the development group @[dependabot[bot]](https://github.com/apps/dependabot) ([#14326](https://github.com/paperless-ngx/paperless-ngx/pull/14326))
|
||||
- Chore(deps): Bump the uv group across 1 directory with 2 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14314](https://github.com/paperless-ngx/paperless-ngx/pull/14314))
|
||||
- Chore(deps): Bump the utilities-minor group across 1 directory with 7 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14305](https://github.com/paperless-ngx/paperless-ngx/pull/14305))
|
||||
- docker-compose(deps): bump greenmail/standalone from 2.1.13 to 2.1.14 in /docker/compose @[dependabot[bot]](https://github.com/apps/dependabot) ([#14281](https://github.com/paperless-ngx/paperless-ngx/pull/14281))
|
||||
- docker(deps): Bump astral-sh/uv from 0.12.16-python3.14-trixie-slim to 0.12.20-python3.14-trixie-slim @[dependabot[bot]](https://github.com/apps/dependabot) ([#14282](https://github.com/paperless-ngx/paperless-ngx/pull/14282))
|
||||
- Chore(deps): Bump the pre-commit-dependencies group across 1 directory with 3 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14283](https://github.com/paperless-ngx/paperless-ngx/pull/14283))
|
||||
- Chore(deps): Bump the utilities-patch group across 1 directory with 6 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14297](https://github.com/paperless-ngx/paperless-ngx/pull/14297))
|
||||
- Chore(deps): Bump the actions group across 1 directory with 10 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14301](https://github.com/paperless-ngx/paperless-ngx/pull/14301))
|
||||
- Chore(deps-dev): Bump the frontend-jest-dependencies group across 1 directory with 2 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14286](https://github.com/paperless-ngx/paperless-ngx/pull/14286))
|
||||
- Chore(deps-dev): Bump eslint from 10.10.0 to 10.11.0 in /src-ui in the frontend-eslint-dependencies group across 1 directory @[dependabot[bot]](https://github.com/apps/dependabot) ([#14287](https://github.com/paperless-ngx/paperless-ngx/pull/14287))
|
||||
- Chore(deps-dev): Bump @types/node from 26.5.0 to 26.6.2 in /src-ui @[dependabot[bot]](https://github.com/apps/dependabot) ([#14288](https://github.com/paperless-ngx/paperless-ngx/pull/14288))
|
||||
- Chore(deps-dev): Bump prettier from 3.9.6 to 3.9.8 in /src-ui @[dependabot[bot]](https://github.com/apps/dependabot) ([#14289](https://github.com/paperless-ngx/paperless-ngx/pull/14289))
|
||||
- Chore(deps): Bump the frontend-angular-dependencies group across 1 directory with 10 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14285](https://github.com/paperless-ngx/paperless-ngx/pull/14285))
|
||||
- Chore: replace bleach with turbohtml [@gaborbernat](https://github.com/gaborbernat) ([#14269](https://github.com/paperless-ngx/paperless-ngx/pull/14269))
|
||||
- Chore(deps): Bump autobahn from 25.12.2 to 26.7.1 in the uv group across 1 directory @[dependabot[bot]](https://github.com/apps/dependabot) ([#14231](https://github.com/paperless-ngx/paperless-ngx/pull/14231))
|
||||
- Chore: update pikepdf for ocrmypdf requirement [@shamoon](https://github.com/shamoon) ([#14235](https://github.com/paperless-ngx/paperless-ngx/pull/14235))
|
||||
- Chore(deps): Bump the pre-commit-dependencies group across 1 directory with 2 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14133](https://github.com/paperless-ngx/paperless-ngx/pull/14133))
|
||||
|
||||
</details>
|
||||
|
||||
### All App Changes
|
||||
|
||||
<details>
|
||||
<summary>41 changes</summary>
|
||||
|
||||
- Feature: store barcode contents, list and search them [@jurassicparkicecream](https://github.com/jurassicparkicecream) ([#14276](https://github.com/paperless-ngx/paperless-ngx/pull/14276))
|
||||
- Fix: Set the ProcessedMail owner based on the rule owner in all cases [@stumpylog](https://github.com/stumpylog) ([#14356](https://github.com/paperless-ngx/paperless-ngx/pull/14356))
|
||||
- Chore(deps): Bump django-filter from 25.2 to 26.1 @[dependabot[bot]](https://github.com/apps/dependabot) ([#14337](https://github.com/paperless-ngx/paperless-ngx/pull/14337))
|
||||
- Chore(deps): Bump the utilities-patch group across 1 directory with 2 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14338](https://github.com/paperless-ngx/paperless-ngx/pull/14338))
|
||||
- Chore(deps-dev): Bump types-channels from 4.3.0.20260408 to 4.3.0.20260518 @[dependabot[bot]](https://github.com/apps/dependabot) ([#14335](https://github.com/paperless-ngx/paperless-ngx/pull/14335))
|
||||
- Chore(deps): Bump the uv group across 1 directory with 2 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14325](https://github.com/paperless-ngx/paperless-ngx/pull/14325))
|
||||
- Fix: Ensure log rotation settings are converted to integers [@stumpylog](https://github.com/stumpylog) ([#14343](https://github.com/paperless-ngx/paperless-ngx/pull/14343))
|
||||
- Chore(deps): Bump the frontend-angular-dependencies group across 1 directory with 13 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14329](https://github.com/paperless-ngx/paperless-ngx/pull/14329))
|
||||
- Chore(deps-dev): Bump prettier from 3.9.8 to 3.9.9 in /src-ui @[dependabot[bot]](https://github.com/apps/dependabot) ([#14331](https://github.com/paperless-ngx/paperless-ngx/pull/14331))
|
||||
- Chore(deps-dev): Bump the frontend-eslint-dependencies group across 1 directory with 3 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14330](https://github.com/paperless-ngx/paperless-ngx/pull/14330))
|
||||
- Fix: ship pdf.js CMaps so CJK documents render in the viewer [@MrOggy85](https://github.com/MrOggy85) ([#14318](https://github.com/paperless-ngx/paperless-ngx/pull/14318))
|
||||
- Chore(deps-dev): Bump zensical from 0.0.64 to 0.0.65 in the development group @[dependabot[bot]](https://github.com/apps/dependabot) ([#14326](https://github.com/paperless-ngx/paperless-ngx/pull/14326))
|
||||
- Enhancement: more control over suggestion requests [@shamoon](https://github.com/shamoon) ([#14258](https://github.com/paperless-ngx/paperless-ngx/pull/14258))
|
||||
- Chore(deps): Bump the uv group across 1 directory with 2 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14314](https://github.com/paperless-ngx/paperless-ngx/pull/14314))
|
||||
- Chore: anchor admin url pattern [@shamoon](https://github.com/shamoon) ([#14316](https://github.com/paperless-ngx/paperless-ngx/pull/14316))
|
||||
- Chorehancement: set manifest CORS for credentials [@shamoon](https://github.com/shamoon) ([#14307](https://github.com/paperless-ngx/paperless-ngx/pull/14307))
|
||||
- Chore(deps): Bump the utilities-minor group across 1 directory with 7 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14305](https://github.com/paperless-ngx/paperless-ngx/pull/14305))
|
||||
- Chore(deps): Bump the utilities-patch group across 1 directory with 6 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14297](https://github.com/paperless-ngx/paperless-ngx/pull/14297))
|
||||
- Chore(deps-dev): Bump the frontend-jest-dependencies group across 1 directory with 2 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14286](https://github.com/paperless-ngx/paperless-ngx/pull/14286))
|
||||
- Chore(deps-dev): Bump eslint from 10.10.0 to 10.11.0 in /src-ui in the frontend-eslint-dependencies group across 1 directory @[dependabot[bot]](https://github.com/apps/dependabot) ([#14287](https://github.com/paperless-ngx/paperless-ngx/pull/14287))
|
||||
- Chore(deps-dev): Bump @types/node from 26.5.0 to 26.6.2 in /src-ui @[dependabot[bot]](https://github.com/apps/dependabot) ([#14288](https://github.com/paperless-ngx/paperless-ngx/pull/14288))
|
||||
- Chore(deps-dev): Bump prettier from 3.9.6 to 3.9.8 in /src-ui @[dependabot[bot]](https://github.com/apps/dependabot) ([#14289](https://github.com/paperless-ngx/paperless-ngx/pull/14289))
|
||||
- Chore(deps): Bump the frontend-angular-dependencies group across 1 directory with 10 updates @[dependabot[bot]](https://github.com/apps/dependabot) ([#14285](https://github.com/paperless-ngx/paperless-ngx/pull/14285))
|
||||
- Fix: use version page\_count for versioned document [@shamoon](https://github.com/shamoon) ([#14280](https://github.com/paperless-ngx/paperless-ngx/pull/14280))
|
||||
- Chore: replace bleach with turbohtml [@gaborbernat](https://github.com/gaborbernat) ([#14269](https://github.com/paperless-ngx/paperless-ngx/pull/14269))
|
||||
- Enhancement: include Django admin with 2FA [@shamoon](https://github.com/shamoon) ([#14270](https://github.com/paperless-ngx/paperless-ngx/pull/14270))
|
||||
- Fix: allow pointer events for pdf links in pngx viewer [@shamoon](https://github.com/shamoon) ([#14264](https://github.com/paperless-ngx/paperless-ngx/pull/14264))
|
||||
- Feature: propagate resolved secrets to interactive container shells [@stumpylog](https://github.com/stumpylog) ([#14254](https://github.com/paperless-ngx/paperless-ngx/pull/14254))
|
||||
- Fix: During a move to the trash directory, only attempt to copy metadata [@stumpylog](https://github.com/stumpylog) ([#14250](https://github.com/paperless-ngx/paperless-ngx/pull/14250))
|
||||
- Fix: convert file mtime to the configured time zone directly [@stumpylog](https://github.com/stumpylog) ([#14249](https://github.com/paperless-ngx/paperless-ngx/pull/14249))
|
||||
- Fix: ensure documentDeleted subscription is discarded [@shamoon](https://github.com/shamoon) ([#14247](https://github.com/paperless-ngx/paperless-ngx/pull/14247))
|
||||
- Chore: Fix bugs in the test suite [@stumpylog](https://github.com/stumpylog) ([#14244](https://github.com/paperless-ngx/paperless-ngx/pull/14244))
|
||||
- Fix: ensure bulk operations are checked against version root [@shamoon](https://github.com/shamoon) ([#14246](https://github.com/paperless-ngx/paperless-ngx/pull/14246))
|
||||
- Enhancement: support separate embedding API key [@furkanural](https://github.com/furkanural) ([#14067](https://github.com/paperless-ngx/paperless-ngx/pull/14067))
|
||||
- Enhancement: support passthrough extra params for LLMs [@shamoon](https://github.com/shamoon) ([#14202](https://github.com/paperless-ngx/paperless-ngx/pull/14202))
|
||||
- Chore(deps): Bump autobahn from 25.12.2 to 26.7.1 in the uv group across 1 directory @[dependabot[bot]](https://github.com/apps/dependabot) ([#14231](https://github.com/paperless-ngx/paperless-ngx/pull/14231))
|
||||
- Fix: indexing after document-added workflows signal [@shamoon](https://github.com/shamoon) ([#14242](https://github.com/paperless-ngx/paperless-ngx/pull/14242))
|
||||
- Fix: Record full tag and custom field lists in bulk edit audit log [@stumpylog](https://github.com/stumpylog) ([#14236](https://github.com/paperless-ngx/paperless-ngx/pull/14236))
|
||||
- Chore: update pikepdf for ocrmypdf requirement [@shamoon](https://github.com/shamoon) ([#14235](https://github.com/paperless-ngx/paperless-ngx/pull/14235))
|
||||
- Fix: handle legacy bulk edit split page range with missing page\_count [@shamoon](https://github.com/shamoon) ([#14212](https://github.com/paperless-ngx/paperless-ngx/pull/14212))
|
||||
- Fix: ignore invalid EXIF orientation when generating image archives [@zhzy0077](https://github.com/zhzy0077) ([#14203](https://github.com/paperless-ngx/paperless-ngx/pull/14203))
|
||||
|
||||
</details>
|
||||
|
||||
## paperless-ngx 3.2.1
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
+18
-6
@@ -1315,17 +1315,29 @@ valid crontab(5) expression describing when to run.
|
||||
|
||||
#### [`PAPERLESS_CONVERT_MEMORY_LIMIT=<num>`](#PAPERLESS_CONVERT_MEMORY_LIMIT) {#PAPERLESS_CONVERT_MEMORY_LIMIT}
|
||||
|
||||
!!! warning
|
||||
: On smaller systems, or even in the case of Very Large Documents, the
|
||||
consumer may explode, complaining about how it's "unable to extend
|
||||
pixel cache". In such cases, try setting this to a reasonably low
|
||||
value, like 32. The default is to use whatever is necessary to do
|
||||
everything without writing to disk, and units are in megabytes.
|
||||
|
||||
Deprecated and has no effect, since PDF thumbnails no longer use
|
||||
ImageMagick. It will be removed in a future release.
|
||||
For more information on how to use this value, you should search the
|
||||
web for "MAGICK_MEMORY_LIMIT".
|
||||
|
||||
Defaults to 0, which disables the limit.
|
||||
|
||||
#### [`PAPERLESS_CONVERT_TMPDIR=<path>`](#PAPERLESS_CONVERT_TMPDIR) {#PAPERLESS_CONVERT_TMPDIR}
|
||||
|
||||
!!! warning
|
||||
: Similar to the memory limit, if you've got a small system and your
|
||||
OS mounts /tmp as tmpfs, you should set this to a path that's on a
|
||||
physical disk, like /home/your_user/tmp or something. ImageMagick
|
||||
will use this as scratch space when crunching through very large
|
||||
documents.
|
||||
|
||||
Deprecated and has no effect, since PDF thumbnails no longer use
|
||||
ImageMagick. It will be removed in a future release.
|
||||
For more information on how to use this value, you should search the
|
||||
web for "MAGICK_TMPDIR".
|
||||
|
||||
Default is none, which disables the temporary directory.
|
||||
|
||||
#### [`PAPERLESS_APPS=<string>`](#PAPERLESS_APPS) {#PAPERLESS_APPS}
|
||||
|
||||
|
||||
+3
-1
@@ -76,7 +76,9 @@ is not supported by any of the available parsers.
|
||||
|
||||
**A:** Not by default. As of v3, a file whose contents match an existing document is still
|
||||
consumed, and the duplicate is flagged in the UI — open the document and check the
|
||||
**Duplicates** tab to review documents that share the same content. If you prefer the old
|
||||
**Duplicates** tab to review documents that share the same content, or filter the document
|
||||
list by **Duplicates** to find all of them (see
|
||||
[Duplicate documents](usage.md#duplicate-documents)). If you prefer the old
|
||||
behavior of rejecting duplicates during consumption, set
|
||||
[`PAPERLESS_CONSUMER_DELETE_DUPLICATES`](configuration.md#PAPERLESS_CONSUMER_DELETE_DUPLICATES)
|
||||
to `true`.
|
||||
|
||||
+7
-4
@@ -177,12 +177,12 @@ to a positive number to enable polling and disable native filesystem notificatio
|
||||
- `pkg-config` for mysqlclient (python dependency)
|
||||
- `fonts-liberation` for generating thumbnails for plain text
|
||||
files
|
||||
- `imagemagick` >= 6 for image alpha handling
|
||||
- `imagemagick` >= 6 for PDF conversion
|
||||
- `gnupg` for decrypting GPG-encrypted email
|
||||
- `libpq-dev` for PostgreSQL
|
||||
- `libmagic-dev` for mime type detection
|
||||
- `mariadb-client` for MariaDB compile time
|
||||
- `poppler-utils` for thumbnail generation and barcode detection
|
||||
- `poppler-utils` for barcode detection
|
||||
|
||||
Use this list for your preferred package management:
|
||||
|
||||
@@ -416,8 +416,11 @@ to a positive number to enable polling and disable native filesystem notificatio
|
||||
You may need to change the path in the files. Example:
|
||||
`ExecStart=/opt/paperless/.local/bin/celery --app paperless worker --loglevel INFO`
|
||||
|
||||
12. Harden ImageMagick by disabling formats that Paperless-ngx does not use.
|
||||
PDF processing is not needed and should stay disabled.
|
||||
12. Configure ImageMagick to allow processing of PDF documents and disable
|
||||
formats that Paperless-ngx does not use. Most distributions disable PDF
|
||||
processing by default, since PDF documents can contain malware. If you
|
||||
don't enable it, Paperless-ngx will fall back to Ghostscript for certain
|
||||
steps such as thumbnail generation.
|
||||
|
||||
Configure the active ImageMagick policy file (commonly
|
||||
`/etc/ImageMagick-6/policy.xml` or `/etc/ImageMagick-7/policy.xml`) and
|
||||
|
||||
@@ -272,6 +272,65 @@ This error can occur in installations which have upgraded from a version of Pape
|
||||
$ python3 manage.py convert_mariadb_uuid
|
||||
```
|
||||
|
||||
## MariaDB/MySQL error "Illegal mix of collations"
|
||||
|
||||
Consumption or other operations fail with an error like:
|
||||
|
||||
```
|
||||
(1267, "Illegal mix of collations (utf8mb4_general_ci,IMPLICIT) and
|
||||
(utf8mb4_unicode_ci,IMPLICIT) for operation '='")
|
||||
```
|
||||
|
||||
This happens when the tables in your database do not all use the same
|
||||
collation. It is most often seen on databases that existed before a
|
||||
MariaDB/MySQL upgrade: older tables keep their original collation, while
|
||||
tables created afterwards use the new server default.
|
||||
|
||||
To work around it, set the collation your existing tables use (the one in the error
|
||||
that is not `utf8mb4_unicode_ci`) with
|
||||
[`PAPERLESS_DB_OPTIONS`](configuration.md#PAPERLESS_DB_OPTIONS):
|
||||
|
||||
```bash
|
||||
PAPERLESS_DB_OPTIONS="collation=utf8mb4_general_ci"
|
||||
```
|
||||
|
||||
To fix it permanently, back up your database, then convert the database and
|
||||
each table to a single collation and remove the override:
|
||||
|
||||
```sql
|
||||
ALTER DATABASE paperless CHARACTER SET utf8mb4 COLLATE utf8mb4_unicode_ci;
|
||||
ALTER TABLE <table_name> CONVERT TO CHARACTER SET utf8mb4 COLLATE utf8mb4_unicode_ci;
|
||||
```
|
||||
|
||||
## PostgreSQL warns about a "collation version mismatch"
|
||||
|
||||
The PostgreSQL log shows a warning like:
|
||||
|
||||
```
|
||||
WARNING: database "paperless" has a collation version mismatch
|
||||
DETAIL: The database was created using collation version 2.36, but the operating system provides version 2.41.
|
||||
HINT: Rebuild all objects in this database that use the default collation and run ALTER DATABASE paperless REFRESH COLLATION VERSION, or build PostgreSQL with the right library version.
|
||||
```
|
||||
|
||||
This comes from PostgreSQL, not Paperless-ngx. The `glibc` version that PostgreSQL
|
||||
runs against changed, and the existing database was created with an older one. With
|
||||
Docker, `glibc` comes from the PostgreSQL image's Debian base, not the host, so this
|
||||
commonly happens when a new image is pulled after its base Debian release changed.
|
||||
If PostgreSQL is installed directly on a host, it uses the host's `glibc` instead.
|
||||
|
||||
The warning is not an error and Paperless-ngx keeps working, but the database's text
|
||||
indexes may be built with outdated sorting rules. To resolve it, back up your
|
||||
database, then connect to it (for example with `psql -U paperless -d paperless`
|
||||
inside the database container) and run:
|
||||
|
||||
```sql
|
||||
REINDEX DATABASE paperless;
|
||||
ALTER DATABASE paperless REFRESH COLLATION VERSION;
|
||||
```
|
||||
|
||||
To avoid this in the future, pin your PostgreSQL image to a specific Debian release,
|
||||
for example `postgres:18-trixie`, rather than `postgres:18`.
|
||||
|
||||
## Platform-Specific Deployment Troubleshooting
|
||||
|
||||
A user-maintained wiki page is available to help troubleshoot issues that may arise when trying to deploy Paperless-ngx on specific platforms, for example SELinux. Please see [the wiki](https://github.com/paperless-ngx/paperless-ngx/wiki/Platform%E2%80%90Specific-Troubleshooting).
|
||||
+7
-8
@@ -299,19 +299,18 @@ for details.
|
||||
### Duplicate documents
|
||||
|
||||
By default, Paperless-ngx **does not reject duplicates**. If you consume a file whose
|
||||
contents exactly match an existing document (same checksum), the new copy is still
|
||||
consumed and a warning is logged. The task entry for the upload also flags that a
|
||||
duplicate was detected and links to the existing document(s).
|
||||
contents match an existing document (same original or archive checksum), the new copy is
|
||||
still consumed and a warning is logged.
|
||||
|
||||
To review duplicates, open a document and switch to the **Duplicates** tab on the
|
||||
document detail page. It lists other documents that share the same content, including any
|
||||
that are in the trash (shown with a badge), and links to each so you can decide which to
|
||||
keep.
|
||||
When a document has duplicates, a **Duplicates** tab appears on its detail page, listing
|
||||
the other documents you can view that share the same content (including any in the trash).
|
||||
To find all documents with duplicates, choose **Duplicates** in the document list's text
|
||||
filter dropdown, or use `has_duplicates=true` in the REST API.
|
||||
|
||||
If you would rather reject duplicates at consumption time (the pre-v3 behavior), set
|
||||
[`PAPERLESS_CONSUMER_DELETE_DUPLICATES`](configuration.md#PAPERLESS_CONSUMER_DELETE_DUPLICATES)
|
||||
to `true`. The duplicate file is then deleted instead of consumed, and the task fails with
|
||||
a "document already exists" message.
|
||||
a "Document already exists" message linking to the existing document.
|
||||
|
||||
## Document Suggestions
|
||||
|
||||
|
||||
@@ -50,6 +50,8 @@ PAPERLESS_SECRET_KEY=change-me
|
||||
#PAPERLESS_OCR_ROTATE_PAGES=true
|
||||
#PAPERLESS_OCR_ROTATE_PAGES_THRESHOLD=12.0
|
||||
#PAPERLESS_OCR_USER_ARGS={}
|
||||
#PAPERLESS_CONVERT_MEMORY_LIMIT=0
|
||||
#PAPERLESS_CONVERT_TMPDIR=/var/tmp/paperless
|
||||
|
||||
# Software tweaks
|
||||
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[project]
|
||||
name = "paperless-ngx"
|
||||
version = "3.2.1"
|
||||
version = "3.3.0"
|
||||
description = """\
|
||||
A community-supported supercharged document management system: scan, index and archive all your physical documents\
|
||||
"""
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "paperless-ngx-ui",
|
||||
"version": "3.2.1",
|
||||
"version": "3.3.0",
|
||||
"scripts": {
|
||||
"preinstall": "npx only-allow pnpm",
|
||||
"ng": "ng",
|
||||
|
||||
@@ -8,7 +8,7 @@ export const environment = {
|
||||
apiVersion: '10', // match src/paperless/settings.py
|
||||
appTitle: DEFAULT_APP_TITLE,
|
||||
tag: 'prod',
|
||||
version: '3.2.1',
|
||||
version: '3.3.0',
|
||||
webSocketHost: window.location.host,
|
||||
webSocketProtocol: window.location.protocol == 'https:' ? 'wss:' : 'ws:',
|
||||
webSocketBaseUrl: base_url.pathname + 'ws/',
|
||||
|
||||
+96
-159
@@ -1,8 +1,8 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import math
|
||||
import mimetypes
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import tempfile
|
||||
@@ -68,6 +68,58 @@ def get_supported_file_extensions() -> set[str]:
|
||||
return extensions
|
||||
|
||||
|
||||
def run_convert(
|
||||
input_file,
|
||||
output_file,
|
||||
*,
|
||||
density=None,
|
||||
scale=None,
|
||||
alpha=None,
|
||||
strip=False,
|
||||
trim=False,
|
||||
type=None,
|
||||
depth=None,
|
||||
auto_orient=False,
|
||||
use_cropbox=False,
|
||||
extra=None,
|
||||
logging_group=None,
|
||||
) -> None:
|
||||
environment = os.environ.copy()
|
||||
if settings.CONVERT_MEMORY_LIMIT:
|
||||
# MAGICK_MEMORY_LIMIT sets the maximum amount of RAM the pixel cache can use.
|
||||
# MAGICK_MAP_LIMIT sets the maximum amount of memory-mapped I/O allowed.
|
||||
#
|
||||
# For large-format documents ImageMagick will hit the RAM limit and
|
||||
# immediately try to "map" the remaining data. If MAGICK_MAP_LIMIT isn't
|
||||
# also set, the process may trigger an OOM kill because the default
|
||||
# system/policy map limit is often too restrictive for these massive bitmaps.
|
||||
environment["MAGICK_MEMORY_LIMIT"] = settings.CONVERT_MEMORY_LIMIT
|
||||
environment["MAGICK_MAP_LIMIT"] = settings.CONVERT_MEMORY_LIMIT
|
||||
if settings.CONVERT_TMPDIR:
|
||||
environment["MAGICK_TMPDIR"] = settings.CONVERT_TMPDIR
|
||||
|
||||
args = [settings.CONVERT_BINARY]
|
||||
args += ["-density", str(density)] if density else []
|
||||
args += ["-scale", str(scale)] if scale else []
|
||||
args += ["-alpha", str(alpha)] if alpha else []
|
||||
args += ["-strip"] if strip else []
|
||||
args += ["-trim"] if trim else []
|
||||
args += ["-type", str(type)] if type else []
|
||||
args += ["-depth", str(depth)] if depth else []
|
||||
args += ["-auto-orient"] if auto_orient else []
|
||||
args += ["-define", "pdf:use-cropbox=true"] if use_cropbox else []
|
||||
args += [str(input_file), str(output_file)]
|
||||
|
||||
logger.debug("Execute: " + " ".join(args), extra={"group": logging_group})
|
||||
|
||||
try:
|
||||
run_subprocess(args, environment, logger)
|
||||
except subprocess.CalledProcessError as e:
|
||||
raise ParseError(f"Convert failed at {args}") from e
|
||||
except Exception as e: # pragma: no cover
|
||||
raise ParseError("Unknown error running convert") from e
|
||||
|
||||
|
||||
def get_default_thumbnail() -> Path:
|
||||
"""
|
||||
Returns the path to a generic thumbnail
|
||||
@@ -75,168 +127,46 @@ def get_default_thumbnail() -> Path:
|
||||
return (Path(__file__).parent / "resources" / "document.webp").resolve()
|
||||
|
||||
|
||||
_THUMBNAIL_MAX_WIDTH = 500
|
||||
_THUMBNAIL_MAX_HEIGHT = 5000
|
||||
# Used only when the page geometry cannot be read
|
||||
_THUMBNAIL_FALLBACK_DPI = 150
|
||||
# Applied before supersampling, so tiny pages are not enlarged
|
||||
_THUMBNAIL_MAX_DPI = 300
|
||||
# Rendering at a multiple and downsampling keeps text crisper
|
||||
_THUMBNAIL_SUPERSAMPLE = 2
|
||||
|
||||
|
||||
def rasterize_pdf_page_to_png(
|
||||
in_path: Path,
|
||||
out_path: Path,
|
||||
*,
|
||||
dpi: int,
|
||||
logging_group=None,
|
||||
) -> None:
|
||||
"""
|
||||
Rasterizes the first page of a PDF to a PNG with pdftoppm.
|
||||
"""
|
||||
# -singlefile drops the page number and -png appends ".png", so pass the
|
||||
# path without its suffix
|
||||
args = [
|
||||
"pdftoppm",
|
||||
"-f",
|
||||
"1",
|
||||
"-l",
|
||||
"1",
|
||||
"-r",
|
||||
str(dpi),
|
||||
"-png",
|
||||
"-singlefile",
|
||||
"-cropbox",
|
||||
str(in_path),
|
||||
str(out_path.with_suffix("")),
|
||||
]
|
||||
|
||||
logger.debug("Execute: " + " ".join(args), extra={"group": logging_group})
|
||||
|
||||
try:
|
||||
run_subprocess(args, logger=logger)
|
||||
except subprocess.CalledProcessError as e:
|
||||
raise ParseError(f"pdftoppm failed at {args}") from e
|
||||
except Exception as e: # pragma: no cover
|
||||
raise ParseError("Unknown error running pdftoppm") from e
|
||||
|
||||
|
||||
def encode_thumbnail_webp(
|
||||
png_path: Path,
|
||||
out_path: Path,
|
||||
*,
|
||||
supersample: int = 1,
|
||||
) -> None:
|
||||
"""
|
||||
Flattens alpha onto white, undoes supersampling, shrinks to fit and saves as WebP.
|
||||
"""
|
||||
from PIL import Image
|
||||
|
||||
try:
|
||||
with Image.open(png_path) as im:
|
||||
if im.mode in ("RGBA", "LA"):
|
||||
flattened = Image.new("RGB", im.size, (255, 255, 255))
|
||||
flattened.paste(im, mask=im.split()[-1])
|
||||
else:
|
||||
flattened = im.convert("RGB")
|
||||
|
||||
if supersample > 1:
|
||||
flattened = flattened.resize(
|
||||
(
|
||||
max(1, round(flattened.width / supersample)),
|
||||
max(1, round(flattened.height / supersample)),
|
||||
),
|
||||
Image.Resampling.LANCZOS,
|
||||
)
|
||||
|
||||
flattened.thumbnail((_THUMBNAIL_MAX_WIDTH, _THUMBNAIL_MAX_HEIGHT))
|
||||
flattened.save(out_path, format="WEBP")
|
||||
except (OSError, Image.DecompressionBombError) as e:
|
||||
raise ParseError(f"Unable to encode thumbnail from {png_path}") from e
|
||||
|
||||
|
||||
def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> tuple[int, int]:
|
||||
"""
|
||||
Returns (dpi, supersample). Unknown geometry is not supersampled, since
|
||||
the render size cannot be bounded.
|
||||
"""
|
||||
from paperless.parsers.utils import get_pdf_first_page_size_points
|
||||
|
||||
size = get_pdf_first_page_size_points(in_path)
|
||||
if size is None:
|
||||
logger.debug(
|
||||
"Could not read PDF page size, using fallback DPI",
|
||||
extra={"group": logging_group},
|
||||
)
|
||||
return _THUMBNAIL_FALLBACK_DPI, 1
|
||||
|
||||
width_pts, height_pts = size
|
||||
dpi_for_width = _THUMBNAIL_MAX_WIDTH * 72 / width_pts
|
||||
dpi_for_height = _THUMBNAIL_MAX_HEIGHT * 72 / height_pts
|
||||
# Round up so the downsampled render is never a few pixels short of the
|
||||
# target; the shrink-only clamp in encode_thumbnail_webp trims the excess.
|
||||
dpi = max(
|
||||
1,
|
||||
math.ceil(min(_THUMBNAIL_MAX_DPI, dpi_for_width, dpi_for_height)),
|
||||
)
|
||||
# At the 1 DPI floor the page is already oversized, so do not supersample
|
||||
return dpi, 1 if dpi == 1 else _THUMBNAIL_SUPERSAMPLE
|
||||
|
||||
|
||||
def _render_pdf_thumbnail(
|
||||
in_path: Path,
|
||||
png_path: Path,
|
||||
out_path: Path,
|
||||
logging_group=None,
|
||||
) -> None:
|
||||
dpi, supersample = _compute_thumbnail_dpi(in_path, logging_group=logging_group)
|
||||
rasterize_pdf_page_to_png(
|
||||
in_path,
|
||||
png_path,
|
||||
dpi=dpi * supersample,
|
||||
logging_group=logging_group,
|
||||
)
|
||||
encode_thumbnail_webp(png_path, out_path, supersample=supersample)
|
||||
|
||||
|
||||
def _repair_pdf_with_qpdf(in_path: Path, out_path: Path) -> None:
|
||||
# qpdf exits 3 after a repair; --warning-exit-0 keeps that from failing
|
||||
try:
|
||||
shutil.copy(in_path, out_path)
|
||||
run_subprocess(
|
||||
["qpdf", "--warning-exit-0", "--replace-input", str(out_path)],
|
||||
logger=logger,
|
||||
)
|
||||
except (subprocess.CalledProcessError, OSError) as e:
|
||||
raise ParseError(f"qpdf repair failed for {in_path}") from e
|
||||
|
||||
|
||||
def make_thumbnail_from_pdf_qpdf_fallback(
|
||||
in_path: Path,
|
||||
temp_dir: Path,
|
||||
logging_group=None,
|
||||
) -> Path:
|
||||
png_path = temp_dir / "page1_repaired.png"
|
||||
out_path = temp_dir / "convert_qpdf.webp"
|
||||
repaired_path = temp_dir / "repaired.pdf"
|
||||
def make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group=None) -> Path:
|
||||
out_path: Path = Path(temp_dir) / "convert_gs.webp"
|
||||
|
||||
# if convert fails, fall back to extracting
|
||||
# the first PDF page as a PNG using Ghostscript
|
||||
logger.warning(
|
||||
"Thumbnail generation with pdftoppm failed, attempting qpdf repair and retry.",
|
||||
"Thumbnail generation with ImageMagick failed, falling back "
|
||||
"to ghostscript. Check your /etc/ImageMagick-x/policy.xml!",
|
||||
extra={"group": logging_group},
|
||||
)
|
||||
# Ghostscript doesn't handle WebP outputs
|
||||
gs_out_path: Path = Path(temp_dir) / "gs_out.png"
|
||||
cmd = [settings.GS_BINARY, "-q", "-sDEVICE=pngalpha", "-o", gs_out_path, in_path]
|
||||
|
||||
try:
|
||||
_repair_pdf_with_qpdf(in_path, repaired_path)
|
||||
_render_pdf_thumbnail(repaired_path, png_path, out_path, logging_group)
|
||||
try:
|
||||
run_subprocess(cmd, logger=logger)
|
||||
except subprocess.CalledProcessError as e:
|
||||
raise ParseError(f"Thumbnail (gs) failed at {cmd}") from e
|
||||
# then run convert on the output from gs to make WebP
|
||||
run_convert(
|
||||
density=300,
|
||||
scale="500x5000>",
|
||||
alpha="remove",
|
||||
strip=True,
|
||||
trim=False,
|
||||
auto_orient=True,
|
||||
input_file=gs_out_path,
|
||||
output_file=out_path,
|
||||
logging_group=logging_group,
|
||||
)
|
||||
|
||||
return out_path
|
||||
|
||||
except ParseError as e:
|
||||
logger.error(f"Unable to make thumbnail after qpdf repair: {e}")
|
||||
logger.error(f"Unable to make thumbnail with Ghostscript: {e}")
|
||||
# The caller might expect a generated thumbnail that can be moved,
|
||||
# so we need to copy it before it gets moved.
|
||||
# https://github.com/paperless-ngx/paperless-ngx/issues/3631
|
||||
default_thumbnail_path = temp_dir / "document.webp"
|
||||
default_thumbnail_path: Path = Path(temp_dir) / "document.webp"
|
||||
copy_file_with_basic_stats(get_default_thumbnail(), default_thumbnail_path)
|
||||
return default_thumbnail_path
|
||||
|
||||
@@ -245,18 +175,25 @@ def make_thumbnail_from_pdf(in_path: Path, temp_dir: Path, logging_group=None) -
|
||||
"""
|
||||
The thumbnail of a PDF is just a 500px wide image of the first page.
|
||||
"""
|
||||
png_path: Path = temp_dir / "page1.png"
|
||||
out_path: Path = temp_dir / "convert.webp"
|
||||
|
||||
# Run convert to get a decent thumbnail
|
||||
try:
|
||||
_render_pdf_thumbnail(in_path, png_path, out_path, logging_group)
|
||||
except ParseError as e:
|
||||
logger.error(f"Unable to make thumbnail with pdftoppm: {e}")
|
||||
out_path = make_thumbnail_from_pdf_qpdf_fallback(
|
||||
in_path,
|
||||
temp_dir,
|
||||
logging_group,
|
||||
run_convert(
|
||||
density=300,
|
||||
scale="500x5000>",
|
||||
alpha="remove",
|
||||
strip=True,
|
||||
trim=False,
|
||||
auto_orient=True,
|
||||
use_cropbox=True,
|
||||
input_file=f"{in_path}[0]",
|
||||
output_file=str(out_path),
|
||||
logging_group=logging_group,
|
||||
)
|
||||
except ParseError as e:
|
||||
logger.error(f"Unable to make thumbnail with convert: {e}")
|
||||
out_path = make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group)
|
||||
|
||||
return out_path
|
||||
|
||||
|
||||
@@ -1,22 +1,11 @@
|
||||
import subprocess
|
||||
from collections.abc import Generator
|
||||
from pathlib import Path
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
from PIL import Image
|
||||
from pytest_django.fixtures import Settings
|
||||
from pytest_mock import MockerFixture
|
||||
|
||||
from documents.parsers import ParseError
|
||||
from documents.parsers import _compute_thumbnail_dpi
|
||||
from documents.parsers import encode_thumbnail_webp
|
||||
from documents.parsers import get_default_file_extension
|
||||
from documents.parsers import get_default_thumbnail
|
||||
from documents.parsers import get_supported_file_extensions
|
||||
from documents.parsers import is_file_ext_supported
|
||||
from documents.parsers import make_thumbnail_from_pdf
|
||||
from documents.parsers import rasterize_pdf_page_to_png
|
||||
from paperless.parsers.registry import get_parser_registry
|
||||
from paperless.parsers.registry import reset_parser_registry
|
||||
from paperless.parsers.tesseract import RasterisedDocumentParser
|
||||
@@ -136,374 +125,3 @@ class TestParserAvailability:
|
||||
assert is_file_ext_supported(".pdf")
|
||||
assert not is_file_ext_supported(".hsdfh")
|
||||
assert not is_file_ext_supported("")
|
||||
|
||||
|
||||
class TestComputeThumbnailDpi:
|
||||
@pytest.mark.parametrize(
|
||||
("size", "expected"),
|
||||
[
|
||||
pytest.param((612.0, 792.0), (59, 2), id="letter-width-bound"),
|
||||
pytest.param((792.0, 612.0), (46, 2), id="landscape-rounded-up"),
|
||||
pytest.param((612.0, 100000.0), (4, 2), id="tall-strip-height-bound"),
|
||||
pytest.param((200.0, 300.0), (180, 2), id="small-page-width-bound"),
|
||||
pytest.param((72.0, 72.0), (300, 2), id="tiny-page-capped-at-300"),
|
||||
pytest.param(
|
||||
(1000000.0, 1000000.0),
|
||||
(1, 1),
|
||||
id="huge-page-minimum-one-unsupersampled",
|
||||
),
|
||||
pytest.param(None, (150, 1), id="unreadable-geometry-fallback"),
|
||||
],
|
||||
)
|
||||
def test_dpi_from_page_size(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tmp_path: Path,
|
||||
size: tuple[float, float] | None,
|
||||
expected: tuple[int, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A first page of the given size, or unreadable geometry
|
||||
WHEN:
|
||||
- The thumbnail DPI is computed
|
||||
THEN:
|
||||
- The expected DPI and supersample factor are returned
|
||||
"""
|
||||
mocker.patch(
|
||||
"paperless.parsers.utils.get_pdf_first_page_size_points",
|
||||
return_value=size,
|
||||
)
|
||||
assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected
|
||||
|
||||
|
||||
class TestMakeThumbnailFromPdf:
|
||||
@pytest.fixture
|
||||
def work_dir(self, tmp_path: Path) -> Path:
|
||||
path = tmp_path / "work"
|
||||
path.mkdir()
|
||||
return path
|
||||
|
||||
@staticmethod
|
||||
def _write_blank_pdf(path: Path, page_size: tuple[int, int] = (612, 792)) -> Path:
|
||||
pdf = pikepdf.new()
|
||||
pdf.add_blank_page(page_size=page_size)
|
||||
pdf.save(path, object_stream_mode=pikepdf.ObjectStreamMode.disable)
|
||||
return path
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("size", "expected_dpi", "expected_supersample"),
|
||||
[
|
||||
pytest.param((612.0, 792.0), 118, 2, id="known-geometry-2x"),
|
||||
pytest.param(None, 150, 1, id="unreadable-geometry-plain-fallback"),
|
||||
],
|
||||
)
|
||||
def test_render_dpi_requested(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tmp_path: Path,
|
||||
work_dir: Path,
|
||||
size: tuple[float, float] | None,
|
||||
expected_dpi: int,
|
||||
expected_supersample: int,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Readable or unreadable page geometry
|
||||
WHEN:
|
||||
- A thumbnail is made
|
||||
THEN:
|
||||
- Rasterize and encode get the matching DPI and supersample factor
|
||||
"""
|
||||
mocker.patch(
|
||||
"paperless.parsers.utils.get_pdf_first_page_size_points",
|
||||
return_value=size,
|
||||
)
|
||||
rasterize = mocker.patch("documents.parsers.rasterize_pdf_page_to_png")
|
||||
encode = mocker.patch("documents.parsers.encode_thumbnail_webp")
|
||||
|
||||
make_thumbnail_from_pdf(tmp_path / "in.pdf", work_dir)
|
||||
|
||||
assert rasterize.call_args.kwargs["dpi"] == expected_dpi
|
||||
assert encode.call_args.kwargs["supersample"] == expected_supersample
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("page_size", "expected_width"),
|
||||
[
|
||||
pytest.param((612, 792), 500, id="letter"),
|
||||
pytest.param((792, 612), 500, id="landscape-letter"),
|
||||
pytest.param((595, 842), 500, id="a4"),
|
||||
pytest.param((200, 300), 500, id="small-page"),
|
||||
pytest.param((72, 72), 300, id="tiny-page-capped"),
|
||||
],
|
||||
)
|
||||
def test_thumbnail_width(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
work_dir: Path,
|
||||
page_size: tuple[int, int],
|
||||
expected_width: int,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF whose first page has the given size in points
|
||||
WHEN:
|
||||
- A thumbnail is made from it
|
||||
THEN:
|
||||
- The thumbnail has the expected width
|
||||
"""
|
||||
pdf_path = self._write_blank_pdf(tmp_path / "in.pdf", page_size)
|
||||
|
||||
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
|
||||
|
||||
assert thumb == work_dir / "convert.webp"
|
||||
with Image.open(thumb) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.width == expected_width
|
||||
|
||||
@classmethod
|
||||
def _write_pdf_without_xref(cls, path: Path) -> Path:
|
||||
"""
|
||||
Cuts off the xref and trailer, which pdftoppm cannot recover from but qpdf can.
|
||||
"""
|
||||
cls._write_blank_pdf(path)
|
||||
data = path.read_bytes()
|
||||
path.write_bytes(data[: data.rindex(b"\nxref")])
|
||||
return path
|
||||
|
||||
def test_qpdf_repair_produces_real_thumbnail(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
work_dir: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF with its xref table and trailer cut off
|
||||
WHEN:
|
||||
- A thumbnail is made from it
|
||||
THEN:
|
||||
- The thumbnail is rendered from a qpdf repaired copy
|
||||
- The original file is unchanged
|
||||
"""
|
||||
pdf_path = self._write_pdf_without_xref(tmp_path / "broken.pdf")
|
||||
original_bytes = pdf_path.read_bytes()
|
||||
|
||||
with pytest.raises(ParseError):
|
||||
rasterize_pdf_page_to_png(pdf_path, work_dir / "probe.png", dpi=50)
|
||||
|
||||
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
|
||||
|
||||
assert thumb == work_dir / "convert_qpdf.webp"
|
||||
with Image.open(thumb) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.width == 500
|
||||
assert pdf_path.read_bytes() == original_bytes
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"qpdf_error",
|
||||
[
|
||||
pytest.param(subprocess.CalledProcessError(2, "qpdf"), id="qpdf-fails"),
|
||||
pytest.param(None, id="repaired-still-unrenderable"),
|
||||
],
|
||||
)
|
||||
def test_double_failure_uses_default_thumbnail(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tmp_path: Path,
|
||||
work_dir: Path,
|
||||
qpdf_error: subprocess.CalledProcessError | None,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF that cannot be rendered, even after qpdf repair
|
||||
WHEN:
|
||||
- A thumbnail is made from it
|
||||
THEN:
|
||||
- A copy of the default thumbnail is returned
|
||||
"""
|
||||
mocker.patch(
|
||||
"documents.parsers.rasterize_pdf_page_to_png",
|
||||
side_effect=ParseError("Does not compute."),
|
||||
)
|
||||
if qpdf_error is not None:
|
||||
mocker.patch("documents.parsers.run_subprocess", side_effect=qpdf_error)
|
||||
pdf_path = self._write_blank_pdf(tmp_path / "in.pdf")
|
||||
|
||||
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
|
||||
|
||||
assert thumb == work_dir / "document.webp"
|
||||
assert thumb.read_bytes() == get_default_thumbnail().read_bytes()
|
||||
|
||||
|
||||
class TestRasterizePdfPageToPng:
|
||||
@staticmethod
|
||||
def _write_pdf(
|
||||
path: Path,
|
||||
*,
|
||||
crop_box: tuple[float, float, float, float] | None = None,
|
||||
rotate: int | None = None,
|
||||
) -> Path:
|
||||
pdf = pikepdf.new()
|
||||
pdf.add_blank_page(page_size=(144, 72))
|
||||
pdf.add_blank_page(page_size=(300, 300))
|
||||
page = pdf.pages[0]
|
||||
if crop_box is not None:
|
||||
page.obj.CropBox = pikepdf.Array(crop_box)
|
||||
if rotate is not None:
|
||||
page.obj.Rotate = rotate
|
||||
pdf.save(path)
|
||||
return path
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("crop_box", "rotate", "expected_size"),
|
||||
[
|
||||
pytest.param(None, None, (144, 72), id="plain"),
|
||||
pytest.param(None, 90, (72, 144), id="rotated-90"),
|
||||
pytest.param((0, 0, 72, 36), None, (72, 36), id="crop-box"),
|
||||
],
|
||||
)
|
||||
def test_renders_first_page_to_exact_path(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
crop_box: tuple[float, float, float, float] | None,
|
||||
rotate: int | None,
|
||||
expected_size: tuple[int, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A two page PDF, first page optionally cropped or rotated
|
||||
WHEN:
|
||||
- The first page is rasterized at 72 DPI
|
||||
THEN:
|
||||
- Only out_path is written, sized to the first page's crop and rotation
|
||||
"""
|
||||
pdf_path = self._write_pdf(
|
||||
tmp_path / "in.pdf",
|
||||
crop_box=crop_box,
|
||||
rotate=rotate,
|
||||
)
|
||||
out_dir = tmp_path / "out"
|
||||
out_dir.mkdir()
|
||||
out_path = out_dir / "page1.png"
|
||||
|
||||
rasterize_pdf_page_to_png(pdf_path, out_path, dpi=72)
|
||||
|
||||
assert list(out_dir.iterdir()) == [out_path]
|
||||
with Image.open(out_path) as im:
|
||||
assert im.format == "PNG"
|
||||
assert im.size == expected_size
|
||||
|
||||
def test_failure_raises_parse_error(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A file that is not a PDF
|
||||
WHEN:
|
||||
- Rasterization is attempted
|
||||
THEN:
|
||||
- A ParseError is raised
|
||||
"""
|
||||
bad = tmp_path / "bad.pdf"
|
||||
bad.write_bytes(b"not a pdf")
|
||||
|
||||
with pytest.raises(ParseError):
|
||||
rasterize_pdf_page_to_png(bad, tmp_path / "page1.png", dpi=72)
|
||||
|
||||
|
||||
class TestEncodeThumbnailWebp:
|
||||
@pytest.mark.parametrize(
|
||||
("mode", "color"),
|
||||
[
|
||||
pytest.param("RGBA", (0, 0, 0, 0), id="rgba"),
|
||||
pytest.param("LA", (0, 0), id="la"),
|
||||
],
|
||||
)
|
||||
def test_alpha_flattened_onto_white(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
mode: str,
|
||||
color: tuple[int, ...],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A fully transparent PNG with an alpha channel
|
||||
WHEN:
|
||||
- It is encoded as a thumbnail
|
||||
THEN:
|
||||
- The WebP output is RGB with the transparency flattened to white
|
||||
"""
|
||||
png_path = tmp_path / "in.png"
|
||||
Image.new(mode, (20, 10), color).save(png_path)
|
||||
out_path = tmp_path / "out.webp"
|
||||
|
||||
encode_thumbnail_webp(png_path, out_path)
|
||||
|
||||
with Image.open(out_path) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.mode == "RGB"
|
||||
assert im.size == (20, 10)
|
||||
red, green, blue = im.getpixel((10, 5))
|
||||
assert min(red, green, blue) >= 250
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("in_size", "supersample", "expected_size"),
|
||||
[
|
||||
pytest.param((1000, 2000), 1, (500, 1000), id="too-wide-shrunk"),
|
||||
pytest.param((100, 10000), 1, (50, 5000), id="too-tall-shrunk"),
|
||||
pytest.param((100, 200), 1, (100, 200), id="small-not-enlarged"),
|
||||
pytest.param((1000, 1400), 2, (500, 700), id="2x-halved"),
|
||||
pytest.param((1001, 1401), 2, (500, 700), id="2x-odd-rounded"),
|
||||
pytest.param((600, 800), 2, (300, 400), id="2x-small-not-enlarged"),
|
||||
pytest.param((900, 1200), 2, (450, 600), id="2x-below-clamp"),
|
||||
],
|
||||
)
|
||||
def test_size_clamped_and_downsampled(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
in_size: tuple[int, int],
|
||||
supersample: int,
|
||||
expected_size: tuple[int, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A rendered image and its supersample factor
|
||||
WHEN:
|
||||
- It is encoded as a thumbnail with that factor
|
||||
THEN:
|
||||
- It is downsampled, fit within 500x5000 and never enlarged
|
||||
"""
|
||||
png_path = tmp_path / "in.png"
|
||||
Image.new("RGB", in_size, (255, 255, 255)).save(png_path)
|
||||
out_path = tmp_path / "out.webp"
|
||||
|
||||
encode_thumbnail_webp(png_path, out_path, supersample=supersample)
|
||||
|
||||
with Image.open(out_path) as im:
|
||||
assert im.size == expected_size
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"error",
|
||||
[
|
||||
pytest.param(OSError("broken image"), id="os-error"),
|
||||
pytest.param(Image.DecompressionBombError("too large"), id="bomb"),
|
||||
],
|
||||
)
|
||||
def test_decode_failure_raises_parse_error(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
mocker: MockerFixture,
|
||||
error: Exception,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Opening the rendered image fails
|
||||
WHEN:
|
||||
- It is encoded as a thumbnail
|
||||
THEN:
|
||||
- A ParseError is raised
|
||||
"""
|
||||
png_path = tmp_path / "in.png"
|
||||
Image.new("RGB", (10, 10)).save(png_path)
|
||||
mocker.patch("PIL.Image.open", side_effect=error)
|
||||
|
||||
with pytest.raises(ParseError):
|
||||
encode_thumbnail_webp(png_path, tmp_path / "out.webp")
|
||||
@@ -2,7 +2,7 @@ msgid ""
|
||||
msgstr ""
|
||||
"Project-Id-Version: paperless-ngx\n"
|
||||
"Report-Msgid-Bugs-To: \n"
|
||||
"POT-Creation-Date: 2026-10-06 15:12+0000\n"
|
||||
"POT-Creation-Date: 2026-10-05 16:26+0000\n"
|
||||
"PO-Revision-Date: 2022-02-17 04:17\n"
|
||||
"Last-Translator: \n"
|
||||
"Language-Team: English\n"
|
||||
@@ -1941,8 +1941,25 @@ msgstr ""
|
||||
msgid "As a final step, please complete the following form:"
|
||||
msgstr ""
|
||||
|
||||
#: documents/validators.py:24
|
||||
#, python-brace-format
|
||||
msgid "Unable to parse URI {value}, missing scheme"
|
||||
msgstr ""
|
||||
|
||||
#: documents/validators.py:29
|
||||
#, python-brace-format
|
||||
msgid "Unable to parse URI {value}, missing net location or path"
|
||||
msgstr ""
|
||||
|
||||
#: documents/validators.py:36
|
||||
msgid ", "
|
||||
msgid ""
|
||||
"URI scheme '{parts.scheme}' is not allowed. Allowed schemes: {', '."
|
||||
"join(allowed_schemes)}"
|
||||
msgstr ""
|
||||
|
||||
#: documents/validators.py:45
|
||||
#, python-brace-format
|
||||
msgid "Unable to parse URI {value}"
|
||||
msgstr ""
|
||||
|
||||
#: documents/views.py:336 documents/views.py:2729
|
||||
|
||||
@@ -265,52 +265,6 @@ def get_page_count_for_pdf(
|
||||
return None
|
||||
|
||||
|
||||
def get_pdf_first_page_size_points(
|
||||
path: Path,
|
||||
log: logging.Logger | None = None,
|
||||
) -> tuple[float, float] | None:
|
||||
"""Return the first page's (width, height) in PDF points, post-rotation.
|
||||
|
||||
Uses the CropBox (MediaBox if absent), which must match pdftoppm's
|
||||
``-cropbox`` or the computed DPI targets the wrong box.
|
||||
|
||||
Swaps width and height for 90/270 rotation. ``page.rotation`` resolves
|
||||
inherited and negative ``/Rotate`` values, a raw lookup does not.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
path:
|
||||
Absolute path to the PDF file.
|
||||
log:
|
||||
Logger for warnings. Falls back to the module-level logger when omitted.
|
||||
|
||||
Returns
|
||||
-------
|
||||
tuple[float, float] | None
|
||||
``(width_points, height_points)``, or ``None`` if the file cannot be
|
||||
opened, has no pages, or the page box is degenerate.
|
||||
"""
|
||||
import pikepdf
|
||||
|
||||
_log = log or logger
|
||||
try:
|
||||
with pikepdf.Pdf.open(path) as pdf:
|
||||
if len(pdf.pages) == 0:
|
||||
return None
|
||||
page = pdf.pages[0]
|
||||
llx, lly, urx, ury = (float(v) for v in page.cropbox)
|
||||
width = abs(urx - llx)
|
||||
height = abs(ury - lly)
|
||||
if width <= 0 or height <= 0:
|
||||
return None
|
||||
if page.rotation in (90, 270):
|
||||
width, height = height, width
|
||||
return width, height
|
||||
except Exception as e:
|
||||
_log.warning("Could not determine PDF page size for %s: %s", path, e)
|
||||
return None
|
||||
|
||||
|
||||
def extract_pdf_metadata(
|
||||
document_path: Path,
|
||||
log: logging.Logger | None = None,
|
||||
|
||||
@@ -986,6 +986,8 @@ GNUPG_HOME = os.getenv("HOME", "/tmp")
|
||||
|
||||
# Convert is part of the ImageMagick package
|
||||
CONVERT_BINARY = os.getenv("PAPERLESS_CONVERT_BINARY", "convert")
|
||||
CONVERT_TMPDIR = os.getenv("PAPERLESS_CONVERT_TMPDIR")
|
||||
CONVERT_MEMORY_LIMIT = os.getenv("PAPERLESS_CONVERT_MEMORY_LIMIT")
|
||||
|
||||
GS_BINARY = os.getenv("PAPERLESS_GS_BINARY", "gs")
|
||||
|
||||
|
||||
@@ -137,20 +137,9 @@ class TestNginxService:
|
||||
reason="No Gotenberg/Tika servers to test with",
|
||||
)
|
||||
class TestParserLive:
|
||||
# Rasterizer versions shift a few pixels, so compare perceptual hashes by
|
||||
# Hamming distance (out of 18 * 18 = 324 bits) rather than for equality
|
||||
MAX_HASH_DISTANCE = 8
|
||||
|
||||
@classmethod
|
||||
def assert_thumbnails_similar(cls, generated: Path, expected: Path) -> None:
|
||||
distance = average_hash(Image.open(generated), 18) - average_hash(
|
||||
Image.open(expected),
|
||||
18,
|
||||
)
|
||||
assert distance <= cls.MAX_HASH_DISTANCE, (
|
||||
f"Thumbnail {generated} differs from {expected} by {distance} bits "
|
||||
f"(max {cls.MAX_HASH_DISTANCE})"
|
||||
)
|
||||
@staticmethod
|
||||
def imagehash(file: Path, hash_size: int = 18) -> str:
|
||||
return f"{average_hash(Image.open(file), hash_size)}"
|
||||
|
||||
def test_get_thumbnail(
|
||||
self,
|
||||
@@ -179,7 +168,12 @@ class TestParserLive:
|
||||
assert thumb.exists()
|
||||
assert thumb.is_file()
|
||||
|
||||
self.assert_thumbnails_similar(thumb, simple_txt_email_thumbnail_file)
|
||||
assert self.imagehash(thumb) == self.imagehash(
|
||||
simple_txt_email_thumbnail_file,
|
||||
), (
|
||||
f"Created thumbnail {thumb} differs from expected file "
|
||||
f"{simple_txt_email_thumbnail_file}"
|
||||
)
|
||||
|
||||
def test_tika_parse_successful(self, mail_parser: MailDocumentParser) -> None:
|
||||
"""
|
||||
@@ -261,7 +255,7 @@ class TestParserLive:
|
||||
THEN:
|
||||
- Gotenberg shall be called to generate the PDF
|
||||
- The archive PDF shall contain the expected content
|
||||
- The generated thumbnail shall be perceptually close to the expected image
|
||||
- The generated thumbnail shall match the expected image hash
|
||||
"""
|
||||
util_call_with_backoff(mail_parser.parse, [html_email_file, "message/rfc822"])
|
||||
|
||||
@@ -278,4 +272,14 @@ class TestParserLive:
|
||||
html_email_file,
|
||||
"message/rfc822",
|
||||
)
|
||||
self.assert_thumbnails_similar(generated_thumbnail, html_email_thumbnail_file)
|
||||
generated_thumbnail_hash = self.imagehash(generated_thumbnail)
|
||||
|
||||
# The created PDF is not reproducible, but the converted image
|
||||
# should always look the same
|
||||
expected_hash = self.imagehash(html_email_thumbnail_file)
|
||||
|
||||
assert generated_thumbnail_hash == expected_hash, (
|
||||
f"PDF thumbnail differs from expected. "
|
||||
f"Generated: {generated_thumbnail}, "
|
||||
f"Hash: {generated_thumbnail_hash} vs {expected_hash}"
|
||||
)
|
||||
@@ -15,10 +15,9 @@ from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
from ocrmypdf import SubprocessOutputError
|
||||
from PIL import Image
|
||||
|
||||
from documents.parsers import ParseError
|
||||
from documents.parsers import rasterize_pdf_page_to_png
|
||||
from documents.parsers import run_convert
|
||||
from paperless.models import ModeChoices
|
||||
from paperless.parsers import ParserProtocol
|
||||
from paperless.parsers.tesseract import RasterisedDocumentParser
|
||||
@@ -281,71 +280,24 @@ class TestGetThumbnail:
|
||||
)
|
||||
assert thumb.is_file()
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("filename", "expected_height"),
|
||||
[
|
||||
pytest.param("simple-digital.pdf", 647, id="portrait-letter"),
|
||||
pytest.param("rotated.pdf", 386, id="landscape"),
|
||||
],
|
||||
)
|
||||
def test_thumbnail_is_correct_format_and_size(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
tesseract_samples_dir: Path,
|
||||
filename: str,
|
||||
expected_height: int,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A portrait or landscape PDF
|
||||
WHEN:
|
||||
- A thumbnail is generated
|
||||
THEN:
|
||||
- A 500px wide WebP keeping the page's aspect ratio
|
||||
"""
|
||||
thumb = tesseract_parser.get_thumbnail(
|
||||
tesseract_samples_dir / filename,
|
||||
"application/pdf",
|
||||
)
|
||||
with Image.open(thumb) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.width == 500
|
||||
assert im.height == pytest.approx(expected_height, abs=2)
|
||||
|
||||
def test_thumbnail_fallback_on_pdftoppm_error(
|
||||
def test_thumbnail_fallback_on_convert_error(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Rasterizing the original PDF fails
|
||||
WHEN:
|
||||
- A thumbnail is generated
|
||||
THEN:
|
||||
- The PDF is repaired with qpdf and a real thumbnail is rendered
|
||||
"""
|
||||
original = tesseract_samples_dir / "simple-digital.pdf"
|
||||
|
||||
def _fail_on_original(in_path: Path, out_path: Path, **kwargs) -> None:
|
||||
if in_path == original:
|
||||
def _raise_on_pdf(input_file, output_file, **kwargs) -> None:
|
||||
if ".pdf" in str(input_file):
|
||||
raise ParseError("Does not compute.")
|
||||
rasterize_pdf_page_to_png(in_path, out_path, **kwargs)
|
||||
run_convert(input_file=input_file, output_file=output_file, **kwargs)
|
||||
|
||||
rasterize = mocker.patch(
|
||||
"documents.parsers.rasterize_pdf_page_to_png",
|
||||
side_effect=_fail_on_original,
|
||||
mocker.patch("documents.parsers.run_convert", side_effect=_raise_on_pdf)
|
||||
|
||||
thumb = tesseract_parser.get_thumbnail(
|
||||
tesseract_samples_dir / "simple-digital.pdf",
|
||||
"application/pdf",
|
||||
)
|
||||
|
||||
thumb = tesseract_parser.get_thumbnail(original, "application/pdf")
|
||||
|
||||
assert rasterize.call_count == 2
|
||||
assert thumb.is_file()
|
||||
assert thumb.name == "convert_qpdf.webp"
|
||||
with Image.open(thumb) as im:
|
||||
assert im.format == "WEBP"
|
||||
assert im.width == 500
|
||||
|
||||
def test_thumbnail_encrypted_pdf(
|
||||
self,
|
||||
|
||||
@@ -6,10 +6,8 @@ import codecs
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
|
||||
from paperless.parsers.utils import get_pdf_first_page_size_points
|
||||
from paperless.parsers.utils import is_tagged_pdf
|
||||
from paperless.parsers.utils import pdf_born_digital_text
|
||||
from paperless.parsers.utils import post_process_text
|
||||
@@ -72,149 +70,6 @@ class TestIsTaggedPdf:
|
||||
assert is_tagged_pdf(bad) is False
|
||||
|
||||
|
||||
class TestGetPdfFirstPageSizePoints:
|
||||
@staticmethod
|
||||
def _write_pdf(
|
||||
path: Path,
|
||||
*,
|
||||
media_box: tuple[float, float, float, float] = (0, 0, 600, 800),
|
||||
crop_box: tuple[float, float, float, float] | None = None,
|
||||
page_rotate: int | None = None,
|
||||
inherited_rotate: int | None = None,
|
||||
) -> Path:
|
||||
pdf = pikepdf.new()
|
||||
pdf.add_blank_page(page_size=(media_box[2], media_box[3]))
|
||||
page = pdf.pages[0]
|
||||
page.obj.MediaBox = pikepdf.Array(media_box)
|
||||
if crop_box is not None:
|
||||
page.obj.CropBox = pikepdf.Array(crop_box)
|
||||
if page_rotate is not None:
|
||||
page.obj.Rotate = page_rotate
|
||||
if inherited_rotate is not None:
|
||||
pdf.Root.Pages.Rotate = inherited_rotate
|
||||
pdf.save(path)
|
||||
return path
|
||||
|
||||
def test_letter_sample(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A US Letter sample PDF with no CropBox and no rotation
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- The MediaBox size in points is returned
|
||||
"""
|
||||
assert get_pdf_first_page_size_points(SAMPLES / "simple-digital.pdf") == (
|
||||
612.0,
|
||||
792.0,
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("rotate", "expected"),
|
||||
[
|
||||
pytest.param(0, (600.0, 800.0), id="rotate-0"),
|
||||
pytest.param(90, (800.0, 600.0), id="rotate-90"),
|
||||
pytest.param(180, (600.0, 800.0), id="rotate-180"),
|
||||
pytest.param(270, (800.0, 600.0), id="rotate-270"),
|
||||
pytest.param(-90, (800.0, 600.0), id="rotate-negative-90"),
|
||||
],
|
||||
)
|
||||
def test_page_rotation_swaps_dimensions(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
rotate: int,
|
||||
expected: tuple[float, float],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A portrait PDF page with /Rotate set directly on the page
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- Width and height are swapped for quarter-turn rotations only
|
||||
"""
|
||||
pdf_path = self._write_pdf(tmp_path / "rotated.pdf", page_rotate=rotate)
|
||||
assert get_pdf_first_page_size_points(pdf_path) == expected
|
||||
|
||||
def test_inherited_rotation_swaps_dimensions(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A page inheriting /Rotate 90 from the /Pages node
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- Width and height are swapped
|
||||
"""
|
||||
pdf_path = self._write_pdf(tmp_path / "inherited.pdf", inherited_rotate=90)
|
||||
assert get_pdf_first_page_size_points(pdf_path) == (800.0, 600.0)
|
||||
|
||||
def test_crop_box_preferred_over_media_box(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF page with a CropBox smaller than its MediaBox
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- The CropBox size is returned
|
||||
"""
|
||||
pdf_path = self._write_pdf(
|
||||
tmp_path / "cropped.pdf",
|
||||
crop_box=(50, 100, 350, 500),
|
||||
)
|
||||
assert get_pdf_first_page_size_points(pdf_path) == (300.0, 400.0)
|
||||
|
||||
def test_degenerate_box_returns_none(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF page whose box has zero width
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- None is returned
|
||||
"""
|
||||
pdf_path = self._write_pdf(
|
||||
tmp_path / "degenerate.pdf",
|
||||
media_box=(0, 0, 600, 800),
|
||||
crop_box=(100, 0, 100, 800),
|
||||
)
|
||||
assert get_pdf_first_page_size_points(pdf_path) is None
|
||||
|
||||
def test_nonexistent_path_returns_none(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A path that does not exist
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- None is returned and nothing is raised
|
||||
"""
|
||||
assert get_pdf_first_page_size_points(Path("/nonexistent/file.pdf")) is None
|
||||
|
||||
def test_corrupt_pdf_returns_none(self, tmp_path: Path) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A file that is not a PDF
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- None is returned and nothing is raised
|
||||
"""
|
||||
bad = tmp_path / "bad.pdf"
|
||||
bad.write_bytes(b"not a pdf")
|
||||
assert get_pdf_first_page_size_points(bad) is None
|
||||
|
||||
def test_encrypted_pdf_returns_none(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A password protected PDF
|
||||
WHEN:
|
||||
- The first page size is requested
|
||||
THEN:
|
||||
- None is returned and nothing is raised
|
||||
"""
|
||||
assert get_pdf_first_page_size_points(SAMPLES / "encrypted.pdf") is None
|
||||
|
||||
|
||||
class TestPostProcessText:
|
||||
@pytest.mark.parametrize(
|
||||
("source", "expected"),
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
from typing import Final
|
||||
|
||||
__version__: Final[tuple[int, int, int]] = (3, 2, 1)
|
||||
__version__: Final[tuple[int, int, int]] = (3, 3, 0)
|
||||
# Version string like X.Y.Z
|
||||
__full_version_str__: Final[str] = ".".join(map(str, __version__))
|
||||
# Version string like X.Y
|
||||
|
||||
Reference in new issue
Block a user