Compare commits

..
Author SHA1 Message Date
shamoon 6f7ed1d4a7 Bump version to 3.3.0 2026-10-05 20:34:24 -07:00
shamoon 0428bf6955 Merge branch 'dev' 2026-10-05 20:33:36 -07:00
shamoon c63afb47b2 Documentation: correct duplicates info (#14243) 2026-09-23 08:27:18 -07:00
30 changed files with 158 additions and 404 deletions

No files matched your search

+3 -3
View File
@@ -80,7 +80,7 @@ jobs:
needs: changes needs: changes
if: needs.changes.outputs.backend_changed == 'true' if: needs.changes.outputs.backend_changed == 'true'
name: "Python ${{ matrix.python-version }}" name: "Python ${{ matrix.python-version }}"
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
strategy: strategy:
@@ -114,7 +114,7 @@ jobs:
packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils
- name: Configure ImageMagick - name: Configure ImageMagick
run: | run: |
sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-7/policy.xml sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-6/policy.xml
- name: Install Python dependencies - name: Install Python dependencies
env: env:
PYTHON_VERSION: ${{ steps.setup-python.outputs.python-version }} PYTHON_VERSION: ${{ steps.setup-python.outputs.python-version }}
@@ -158,7 +158,7 @@ jobs:
needs: changes needs: changes
if: needs.changes.outputs.backend_changed == 'true' if: needs.changes.outputs.backend_changed == 'true'
name: Check project typing name: Check project typing
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
env: env:
+3 -3
View File
@@ -24,10 +24,10 @@ jobs:
fail-fast: false fail-fast: false
matrix: matrix:
include: include:
- runner: ubuntu-26.04 - runner: ubuntu-24.04
arch: amd64 arch: amd64
platform: linux/amd64 platform: linux/amd64
- runner: ubuntu-26.04-arm - runner: ubuntu-24.04-arm
arch: arm64 arch: arm64
platform: linux/arm64 platform: linux/arm64
runs-on: ${{ matrix.runner }} runs-on: ${{ matrix.runner }}
@@ -163,7 +163,7 @@ jobs:
archive: false archive: false
merge-and-push: merge-and-push:
name: Merge and Push Manifest name: Merge and Push Manifest
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
needs: build-arch needs: build-arch
if: needs.build-arch.outputs.should-push == 'true' if: needs.build-arch.outputs.should-push == 'true'
environment: image-publishing environment: image-publishing
+2 -2
View File
@@ -65,7 +65,7 @@ jobs:
needs: changes needs: changes
if: needs.changes.outputs.docs_changed == 'true' if: needs.changes.outputs.docs_changed == 'true'
name: Build Documentation name: Build Documentation
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
steps: steps:
- uses: actions/configure-pages@45bfe0192ca1faeb007ade9deae92b16b8254a0d # v6.0.0 - uses: actions/configure-pages@45bfe0192ca1faeb007ade9deae92b16b8254a0d # v6.0.0
- name: Checkout - name: Checkout
@@ -102,7 +102,7 @@ jobs:
name: Deploy Documentation name: Deploy Documentation
needs: [changes, build] needs: [changes, build]
if: github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.changes.outputs.docs_changed == 'true' if: github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.changes.outputs.docs_changed == 'true'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
pages: write pages: write
id-token: write id-token: write
+6 -6
View File
@@ -72,7 +72,7 @@ jobs:
needs: changes needs: changes
if: needs.changes.outputs.frontend_changed == 'true' if: needs.changes.outputs.frontend_changed == 'true'
name: Install Dependencies name: Install Dependencies
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
steps: steps:
@@ -104,7 +104,7 @@ jobs:
name: Lint name: Lint
needs: [changes, install-dependencies] needs: [changes, install-dependencies]
if: needs.changes.outputs.frontend_changed == 'true' if: needs.changes.outputs.frontend_changed == 'true'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
steps: steps:
@@ -137,7 +137,7 @@ jobs:
name: "Unit Tests (${{ matrix.shard-index }}/${{ matrix.shard-count }})" name: "Unit Tests (${{ matrix.shard-index }}/${{ matrix.shard-count }})"
needs: [changes, install-dependencies] needs: [changes, install-dependencies]
if: needs.changes.outputs.frontend_changed == 'true' if: needs.changes.outputs.frontend_changed == 'true'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
strategy: strategy:
@@ -188,10 +188,10 @@ jobs:
name: E2E Tests name: E2E Tests
needs: [changes, install-dependencies] needs: [changes, install-dependencies]
if: needs.changes.outputs.frontend_changed == 'true' if: needs.changes.outputs.frontend_changed == 'true'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
container: mcr.microsoft.com/playwright:v1.62.1-resolute container: mcr.microsoft.com/playwright:v1.62.1-noble
env: env:
PLAYWRIGHT_BROWSERS_PATH: /ms-playwright PLAYWRIGHT_BROWSERS_PATH: /ms-playwright
PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD: 1 PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD: 1
@@ -246,7 +246,7 @@ jobs:
name: Frontend Build name: Frontend Build
needs: [changes, unit-tests, e2e-tests] needs: [changes, unit-tests, e2e-tests]
if: needs.changes.outputs.frontend_changed == 'true' if: needs.changes.outputs.frontend_changed == 'true'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
steps: steps:
+5 -5
View File
@@ -14,7 +14,7 @@ permissions: {}
jobs: jobs:
wait-for-docker: wait-for-docker:
name: Wait for Docker Build name: Wait for Docker Build
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
checks: read checks: read
statuses: read statuses: read
@@ -30,7 +30,7 @@ jobs:
build-release: build-release:
name: Build Release name: Build Release
needs: wait-for-docker needs: wait-for-docker
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
steps: steps:
@@ -73,7 +73,7 @@ jobs:
timeout-minutes: 12 timeout-minutes: 12
uses: $/.github/actions/apt-install uses: $/.github/actions/apt-install
with: with:
packages: gettext libleptonica6 packages: gettext liblept5
# ---- Build Documentation ---- # ---- Build Documentation ----
- name: Build documentation - name: Build documentation
env: env:
@@ -145,7 +145,7 @@ jobs:
publish-release: publish-release:
name: Publish Release name: Publish Release
needs: build-release needs: build-release
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: write contents: write
pull-requests: write pull-requests: write
@@ -197,7 +197,7 @@ jobs:
name: Append Changelog name: Append Changelog
needs: publish-release needs: publish-release
if: needs.publish-release.outputs.prerelease == 'false' if: needs.publish-release.outputs.prerelease == 'false'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: write contents: write
pull-requests: write pull-requests: write
+2 -2
View File
@@ -15,7 +15,7 @@ permissions:
jobs: jobs:
zizmor: zizmor:
name: Run zizmor name: Run zizmor
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
contents: read contents: read
actions: read actions: read
@@ -29,7 +29,7 @@ jobs:
uses: zizmorcore/zizmor-action@cc914d7f3750a2d13d75c7f184a1060aa0e9d482 # v0.6.4 uses: zizmorcore/zizmor-action@cc914d7f3750a2d13d75c7f184a1060aa0e9d482 # v0.6.4
semgrep: semgrep:
name: Semgrep CE name: Semgrep CE
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
container: container:
image: semgrep/semgrep:1.155.0@sha256:cc869c685dcc0fe497c86258da9f205397d8108e56d21a86082ea4886e52784d image: semgrep/semgrep:1.155.0@sha256:cc869c685dcc0fe497c86258da9f205397d8108e56d21a86082ea4886e52784d
if: github.actor != 'dependabot[bot]' if: github.actor != 'dependabot[bot]'
+2 -2
View File
@@ -17,7 +17,7 @@ jobs:
cleanup-images: cleanup-images:
name: Cleanup Image Tags for ${{ matrix.primary-name }} name: Cleanup Image Tags for ${{ matrix.primary-name }}
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
environment: registry-maintenance environment: registry-maintenance
strategy: strategy:
fail-fast: false fail-fast: false
@@ -42,7 +42,7 @@ jobs:
cleanup-untagged-images: cleanup-untagged-images:
name: Cleanup Untagged Images Tags for ${{ matrix.primary-name }} name: Cleanup Untagged Images Tags for ${{ matrix.primary-name }}
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
needs: needs:
- cleanup-images - cleanup-images
environment: registry-maintenance environment: registry-maintenance
+1 -1
View File
@@ -21,7 +21,7 @@ on:
jobs: jobs:
analyze: analyze:
name: Analyze name: Analyze
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
actions: read actions: read
contents: read contents: read
+1 -1
View File
@@ -13,7 +13,7 @@ jobs:
synchronize-with-crowdin: synchronize-with-crowdin:
name: Crowdin Sync name: Crowdin Sync
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
environment: translation-sync environment: translation-sync
steps: steps:
- name: Checkout - name: Checkout
+1 -1
View File
@@ -7,7 +7,7 @@ jobs:
# Note: peakoss/anti-slop does not support the `issues` event yet (all of its # Note: peakoss/anti-slop does not support the `issues` event yet (all of its
# issue inputs are still commented out upstream), so the checks that the PR Bot # issue inputs are still commented out upstream), so the checks that the PR Bot
# workflow gets from the action are implemented manually here. # workflow gets from the action are implemented manually here.
runs-on: ubuntu-slim runs-on: ubuntu-latest
permissions: permissions:
issues: write issues: write
steps: steps:
+2 -2
View File
@@ -4,7 +4,7 @@ on:
types: [opened] types: [opened]
jobs: jobs:
Anti-slop: Anti-slop:
runs-on: ubuntu-slim runs-on: ubuntu-latest
permissions: permissions:
contents: read contents: read
issues: read issues: read
@@ -24,7 +24,7 @@ jobs:
ASLOP-PR-VERIFY ASLOP-PR-VERIFY
pr-bot: pr-bot:
name: Automated PR Bot name: Automated PR Bot
runs-on: ubuntu-slim runs-on: ubuntu-latest
# Runs after Anti-slop so the welcome comment can see whether the PR was closed # Runs after Anti-slop so the welcome comment can see whether the PR was closed
# instead of racing it. Still runs if that job fails, so labeling is not lost. # instead of racing it. Still runs if that job fails, so labeling is not lost.
needs: Anti-slop needs: Anti-slop
+1 -1
View File
@@ -12,7 +12,7 @@ permissions:
jobs: jobs:
pr_opened_or_reopened: pr_opened_or_reopened:
name: pr_opened_or_reopened name: pr_opened_or_reopened
runs-on: ubuntu-slim runs-on: ubuntu-24.04
permissions: permissions:
# write permission is required for autolabeler # write permission is required for autolabeler
pull-requests: write pull-requests: write
+5 -5
View File
@@ -9,7 +9,7 @@ jobs:
stale: stale:
name: 'Stale' name: 'Stale'
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
issues: write issues: write
pull-requests: write pull-requests: write
@@ -34,7 +34,7 @@ jobs:
lock-threads: lock-threads:
name: 'Lock Old Threads' name: 'Lock Old Threads'
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04 runs-on: ubuntu-24.04
permissions: permissions:
issues: write issues: write
pull-requests: write pull-requests: write
@@ -58,7 +58,7 @@ jobs:
close-answered-discussions: close-answered-discussions:
name: 'Close Answered Discussions' name: 'Close Answered Discussions'
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-slim runs-on: ubuntu-24.04
permissions: permissions:
discussions: write discussions: write
steps: steps:
@@ -117,7 +117,7 @@ jobs:
close-outdated-discussions: close-outdated-discussions:
name: 'Close Outdated Discussions' name: 'Close Outdated Discussions'
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-slim runs-on: ubuntu-24.04
permissions: permissions:
discussions: write discussions: write
steps: steps:
@@ -211,7 +211,7 @@ jobs:
close-unsupported-feature-requests: close-unsupported-feature-requests:
name: 'Close Unsupported Feature Requests' name: 'Close Unsupported Feature Requests'
if: github.repository_owner == 'paperless-ngx' if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-slim runs-on: ubuntu-24.04
permissions: permissions:
discussions: write discussions: write
steps: steps:
+1 -1
View File
@@ -8,7 +8,7 @@ env:
jobs: jobs:
generate-translate-strings: generate-translate-strings:
name: Generate Translation Strings name: Generate Translation Strings
runs-on: ubuntu-26.04 runs-on: ubuntu-latest
environment: translation-sync environment: translation-sync
permissions: permissions:
contents: write contents: write
+3 -1
View File
@@ -76,7 +76,9 @@ is not supported by any of the available parsers.
**A:** Not by default. As of v3, a file whose contents match an existing document is still **A:** Not by default. As of v3, a file whose contents match an existing document is still
consumed, and the duplicate is flagged in the UI — open the document and check the consumed, and the duplicate is flagged in the UI — open the document and check the
**Duplicates** tab to review documents that share the same content. If you prefer the old **Duplicates** tab to review documents that share the same content, or filter the document
list by **Duplicates** to find all of them (see
[Duplicate documents](usage.md#duplicate-documents)). If you prefer the old
behavior of rejecting duplicates during consumption, set behavior of rejecting duplicates during consumption, set
[`PAPERLESS_CONSUMER_DELETE_DUPLICATES`](configuration.md#PAPERLESS_CONSUMER_DELETE_DUPLICATES) [`PAPERLESS_CONSUMER_DELETE_DUPLICATES`](configuration.md#PAPERLESS_CONSUMER_DELETE_DUPLICATES)
to `true`. to `true`.
+7 -8
View File
@@ -299,19 +299,18 @@ for details.
### Duplicate documents ### Duplicate documents
By default, Paperless-ngx **does not reject duplicates**. If you consume a file whose By default, Paperless-ngx **does not reject duplicates**. If you consume a file whose
contents exactly match an existing document (same checksum), the new copy is still contents match an existing document (same original or archive checksum), the new copy is
consumed and a warning is logged. The task entry for the upload also flags that a still consumed and a warning is logged.
duplicate was detected and links to the existing document(s).
To review duplicates, open a document and switch to the **Duplicates** tab on the When a document has duplicates, a **Duplicates** tab appears on its detail page, listing
document detail page. It lists other documents that share the same content, including any the other documents you can view that share the same content (including any in the trash).
that are in the trash (shown with a badge), and links to each so you can decide which to To find all documents with duplicates, choose **Duplicates** in the document list's text
keep. filter dropdown, or use `has_duplicates=true` in the REST API.
If you would rather reject duplicates at consumption time (the pre-v3 behavior), set If you would rather reject duplicates at consumption time (the pre-v3 behavior), set
[`PAPERLESS_CONSUMER_DELETE_DUPLICATES`](configuration.md#PAPERLESS_CONSUMER_DELETE_DUPLICATES) [`PAPERLESS_CONSUMER_DELETE_DUPLICATES`](configuration.md#PAPERLESS_CONSUMER_DELETE_DUPLICATES)
to `true`. The duplicate file is then deleted instead of consumed, and the task fails with to `true`. The duplicate file is then deleted instead of consumed, and the task fails with
a "document already exists" message. a "Document already exists" message linking to the existing document.
## Document Suggestions ## Document Suggestions
+1 -1
View File
@@ -1,6 +1,6 @@
[project] [project]
name = "paperless-ngx" name = "paperless-ngx"
version = "3.2.1" version = "3.3.0"
description = """\ description = """\
A community-supported supercharged document management system: scan, index and archive all your physical documents\ A community-supported supercharged document management system: scan, index and archive all your physical documents\
""" """
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "paperless-ngx-ui", "name": "paperless-ngx-ui",
"version": "3.2.1", "version": "3.3.0",
"scripts": { "scripts": {
"preinstall": "npx only-allow pnpm", "preinstall": "npx only-allow pnpm",
"ng": "ng", "ng": "ng",
+1 -1
View File
@@ -8,7 +8,7 @@ export const environment = {
apiVersion: '10', // match src/paperless/settings.py apiVersion: '10', // match src/paperless/settings.py
appTitle: DEFAULT_APP_TITLE, appTitle: DEFAULT_APP_TITLE,
tag: 'prod', tag: 'prod',
version: '3.2.1', version: '3.3.0',
webSocketHost: window.location.host, webSocketHost: window.location.host,
webSocketProtocol: window.location.protocol == 'https:' ? 'wss:' : 'ws:', webSocketProtocol: window.location.protocol == 'https:' ? 'wss:' : 'ws:',
webSocketBaseUrl: base_url.pathname + 'ws/', webSocketBaseUrl: base_url.pathname + 'ws/',
+2 -27
View File
@@ -13,14 +13,8 @@ from celery import group
from celery import shared_task from celery import shared_task
from django.conf import settings from django.conf import settings
from django.db import transaction from django.db import transaction
from django.db.models import Case
from django.db.models import F
from django.db.models import Max from django.db.models import Max
from django.db.models import OuterRef
from django.db.models import Q from django.db.models import Q
from django.db.models import Subquery
from django.db.models import When
from django.db.models.functions import Coalesce
from django.utils import timezone from django.utils import timezone
from documents.data_models import ConsumableDocument from documents.data_models import ConsumableDocument
@@ -42,7 +36,6 @@ from documents.tasks import remove_document_from_index
from documents.tasks import update_document_content_maybe_archive_file from documents.tasks import update_document_content_maybe_archive_file
from documents.versioning import get_latest_version_for_root from documents.versioning import get_latest_version_for_root
from documents.versioning import get_root_document from documents.versioning import get_root_document
from documents.versioning import versions_newest_first
if TYPE_CHECKING: if TYPE_CHECKING:
from collections.abc import Mapping from collections.abc import Mapping
@@ -415,28 +408,10 @@ def reprocess(doc_ids: list[int], *, remote_ocr: bool = False) -> Literal["OK"]:
Consumption workflows do not run here, so ``remote_ocr`` is how the user Consumption workflows do not run here, so ``remote_ocr`` is how the user
asks for the remote engine when it is not configured to handle everything. asks for the remote engine when it is not configured to handle everything.
A root document with versions reprocesses its latest version, which is the
file whose content, archive and thumbnail are shown for it.
""" """
latest_version = versions_newest_first( for document_id in doc_ids:
Document.objects.filter(root_document=OuterRef("pk")),
).values("id")[:1]
source_ids = (
Document.objects.filter(id__in=doc_ids)
.annotate(
source_id=Case(
When(root_document__isnull=False, then=F("id")),
default=Coalesce(Subquery(latest_version), F("id")),
),
)
.order_by()
.values_list("source_id", flat=True)
.distinct()
)
for source_id in source_ids:
update_document_content_maybe_archive_file.apply_async( update_document_content_maybe_archive_file.apply_async(
kwargs={"document_id": source_id, "remote_ocr": remote_ocr}, kwargs={"document_id": document_id, "remote_ocr": remote_ocr},
headers={"trigger_source": PaperlessTask.TriggerSource.MANUAL}, headers={"trigger_source": PaperlessTask.TriggerSource.MANUAL},
) )
+32 -38
View File
@@ -31,7 +31,6 @@ from documents.search._query import parse_user_query
from documents.search._schema import _write_sentinels from documents.search._schema import _write_sentinels
from documents.search._schema import build_schema from documents.search._schema import build_schema
from documents.search._schema import open_or_rebuild_index from documents.search._schema import open_or_rebuild_index
from documents.search._schema import rebuild_in_progress
from documents.search._schema import wipe_index from documents.search._schema import wipe_index
from documents.search._tokenizer import ascii_fold from documents.search._tokenizer import ascii_fold
from documents.search._tokenizer import autocomplete_tokens from documents.search._tokenizer import autocomplete_tokens
@@ -1111,44 +1110,39 @@ class TantivyBackend:
flushing a segment, deferring merge work; they do not avoid it. flushing a segment, deferring merge work; they do not avoid it.
""" """
wipe_index(self._path) wipe_index(self._path)
# The marker covers the window where the empty index is already stamped new_index = tantivy.Index(build_schema(), path=str(self._path))
# as current but not yet populated, so an interrupted rebuild is retried. _write_sentinels(self._path)
with rebuild_in_progress(self._path): register_tokenizers(new_index, settings.SEARCH_LANGUAGE)
new_index = tantivy.Index(build_schema(), path=str(self._path))
_write_sentinels(self._path)
register_tokenizers(new_index, settings.SEARCH_LANGUAGE)
# Point instance at the new index so _build_tantivy_doc uses it # Point instance at the new index so _build_tantivy_doc uses it
old_index, old_schema = self._raw_index, self._raw_schema old_index, old_schema = self._raw_index, self._raw_schema
self._raw_index = new_index self._raw_index = new_index
self._raw_schema = new_index.schema self._raw_schema = new_index.schema
# Stream documents one-by-one (so the progress bar advances per # Stream documents one-by-one (so the progress bar advances per
# document) while fetching viewer permissions one SQL query per # document) while fetching viewer permissions one SQL query per chunk.
# chunk. The stream is Sized, so iter_wrapper can still discover # The stream is Sized, so iter_wrapper can still discover the total.
# the total. documents_stream = _DocumentViewerStream(documents, chunk_size=1000)
documents_stream = _DocumentViewerStream(documents, chunk_size=1000) try:
try: writer = new_index.writer(heap_size=writer_heap_bytes)
writer = new_index.writer(heap_size=writer_heap_bytes) for document, (viewer_ids, viewer_group_ids) in iter_wrapper(
for document, (viewer_ids, viewer_group_ids) in iter_wrapper( documents_stream,
documents_stream, ):
): doc = self._build_tantivy_doc(
doc = self._build_tantivy_doc( document,
document, viewer_ids=viewer_ids,
viewer_ids=viewer_ids, viewer_group_ids=viewer_group_ids,
viewer_group_ids=viewer_group_ids, )
) writer.add_document(doc)
writer.add_document(doc) writer.commit()
writer.commit() # Wait for background merge threads to finish so all segments are
# Wait for background merge threads to finish so all segments # fully merged and persisted before the index is considered rebuilt.
# are fully merged and persisted before the index is considered writer.wait_merging_threads()
# rebuilt. new_index.reload()
writer.wait_merging_threads() except BaseException: # pragma: no cover
new_index.reload() # Restore old index on failure so the backend remains usable
except BaseException: # pragma: no cover self._raw_index = old_index
# Restore old index on failure so the backend remains usable self._raw_schema = old_schema
self._raw_index = old_index raise
self._raw_schema = old_schema
raise
def chunked(iterable, size): def chunked(iterable, size):
+4 -45
View File
@@ -4,7 +4,6 @@ import hashlib
import json import json
import logging import logging
import shutil import shutil
from contextlib import contextmanager
from typing import TYPE_CHECKING from typing import TYPE_CHECKING
from typing import Final from typing import Final
from typing import NamedTuple from typing import NamedTuple
@@ -17,7 +16,6 @@ from whoosh_compat import FieldKind
from documents.search._fields import PUBLIC_FIELDS from documents.search._fields import PUBLIC_FIELDS
if TYPE_CHECKING: if TYPE_CHECKING:
from collections.abc import Iterator
from pathlib import Path from pathlib import Path
logger = logging.getLogger("paperless.search") logger = logging.getLogger("paperless.search")
@@ -30,11 +28,6 @@ logger = logging.getLogger("paperless.search")
# v3 - barcodes JSON field for stored barcode contents # v3 - barcodes JSON field for stored barcode contents
SCHEMA_VERSION: Final[int] = 3 SCHEMA_VERSION: Final[int] = 3
# Present in the index directory from the moment a full rebuild starts until it
# finishes. If a rebuild is interrupted it is left behind, so the half-built
# index is not mistaken for a complete one.
REBUILD_MARKER: Final[str] = ".rebuilding"
class FieldDescriptor(NamedTuple): class FieldDescriptor(NamedTuple):
"""One tantivy field, in declaration order. """One tantivy field, in declaration order.
@@ -262,9 +255,9 @@ def needs_rebuild(index_dir: Path) -> bool:
""" """
Check if the search index needs rebuilding. Check if the search index needs rebuilding.
True if a previous full rebuild never finished (the rebuild marker is still Reads .index_settings.json to compare the stored schema version, search
present), or if the index's stamped settings no longer match the current language and schema fingerprint against the current configuration. Returns
configuration. See _settings_mismatch(). True if the file is missing, unparsable, or any value mismatches.
Args: Args:
index_dir: Path to the search index directory index_dir: Path to the search index directory
@@ -272,40 +265,6 @@ def needs_rebuild(index_dir: Path) -> bool:
Returns: Returns:
True if the index needs rebuilding, False if it's up to date True if the index needs rebuilding, False if it's up to date
""" """
if (index_dir / REBUILD_MARKER).exists():
logger.warning("Previous search index rebuild did not finish - rebuilding.")
return True
return _settings_mismatch(index_dir)
@contextmanager
def rebuild_in_progress(index_dir: Path) -> Iterator[None]:
"""
Flag the index as incomplete for the duration of a full rebuild.
The marker is cleared only if the block exits cleanly. There is deliberately
no try/finally: an exception must leave the marker behind so the next
needs_rebuild() check retries the rebuild.
"""
marker = index_dir / REBUILD_MARKER
marker.touch()
yield
marker.unlink(missing_ok=True)
def _settings_mismatch(index_dir: Path) -> bool:
"""
Check the stamped settings against the current configuration.
Reads .index_settings.json to compare the stored schema version, search
language and schema fingerprint. Returns True if the file is missing,
unparsable, or any value mismatches.
This deliberately ignores the rebuild marker: open_or_rebuild_index() uses it
so that a process opening the index while another process is mid-rebuild
(or after one died) does not wipe the partial index out from under it.
Repopulating is the job of ``document_index reindex``.
"""
settings_file = index_dir / ".index_settings.json" settings_file = index_dir / ".index_settings.json"
if not settings_file.exists(): if not settings_file.exists():
return True return True
@@ -374,7 +333,7 @@ def open_or_rebuild_index(index_dir: Path | None = None) -> tantivy.Index:
index_dir = cast("Path", settings.INDEX_DIR) index_dir = cast("Path", settings.INDEX_DIR)
if not index_dir.exists(): if not index_dir.exists():
return tantivy.Index(build_schema()) return tantivy.Index(build_schema())
if _settings_mismatch(index_dir): if needs_rebuild(index_dir):
wipe_index(index_dir) wipe_index(index_dir)
idx = tantivy.Index(build_schema(), path=str(index_dir)) idx = tantivy.Index(build_schema(), path=str(index_dir))
_write_sentinels(index_dir) _write_sentinels(index_dir)
+3 -8
View File
@@ -490,23 +490,18 @@ def update_document_content_maybe_archive_file(
shutil.move(thumbnail, document.thumbnail_path) shutil.move(thumbnail, document.thumbnail_path)
document.refresh_from_db() document.refresh_from_db()
root_document = (
document.root_document if document.root_document_id else document
)
logger.info( logger.info(
f"Updating index for document {root_document.pk} ({document.archive_checksum})", f"Updating index for document {document_id} ({document.archive_checksum})",
) )
from documents.search import get_backend from documents.search import get_backend
get_backend().add_or_update(root_document) get_backend().add_or_update(document)
ai_config = AIConfig() ai_config = AIConfig()
if ai_config.llm_index_enabled: if ai_config.llm_index_enabled:
llm_index_add_or_update_document(root_document) llm_index_add_or_update_document(document)
clear_document_caches(document.pk) clear_document_caches(document.pk)
if root_document.pk != document.pk:
clear_document_caches(root_document.pk)
except Exception: except Exception:
logger.exception( logger.exception(
@@ -16,10 +16,7 @@ from documents.search._backend import TantivyBackend
from documents.search._backend import WriteBatch from documents.search._backend import WriteBatch
from documents.search._backend import get_backend from documents.search._backend import get_backend
from documents.search._backend import reset_backend from documents.search._backend import reset_backend
from documents.search._schema import REBUILD_MARKER
from documents.search._schema import needs_rebuild
from documents.signals.handlers import add_to_index from documents.signals.handlers import add_to_index
from paperless_testing.dirs import PaperlessDirs
from paperless_testing.factories import CorrespondentFactory from paperless_testing.factories import CorrespondentFactory
from paperless_testing.factories import DocumentFactory from paperless_testing.factories import DocumentFactory
from paperless_testing.factories import DocumentTypeFactory from paperless_testing.factories import DocumentTypeFactory
@@ -826,53 +823,6 @@ class TestRebuild:
backend.rebuild(Document.objects.all(), iter_wrapper=wrapper) backend.rebuild(Document.objects.all(), iter_wrapper=wrapper)
assert 30 in seen assert 30 in seen
def test_successful_rebuild_leaves_index_up_to_date(
self,
backend: TantivyBackend,
paperless_dirs: PaperlessDirs,
) -> None:
"""
GIVEN:
- A backend and one document
WHEN:
- rebuild() completes
THEN:
- needs_rebuild() is False and no rebuild marker remains
"""
DocumentFactory.create()
backend.rebuild(Document.objects.all())
assert needs_rebuild(paperless_dirs.index_dir) is False
assert not (paperless_dirs.index_dir / REBUILD_MARKER).exists()
def test_interrupted_rebuild_is_retried(
self,
backend: TantivyBackend,
paperless_dirs: PaperlessDirs,
) -> None:
"""
GIVEN:
- A rebuild that dies while indexing documents (e.g. the database
connection is lost)
WHEN:
- needs_rebuild() is checked afterwards
THEN:
- It is True, even though the empty index was already stamped with
current settings, so the next start rebuilds instead of reporting
the index as up to date
"""
DocumentFactory.create()
def die(pairs):
raise RuntimeError("terminating connection due to administrator command")
yield # pragma: no cover
with pytest.raises(RuntimeError):
backend.rebuild(Document.objects.all(), iter_wrapper=die)
assert needs_rebuild(paperless_dirs.index_dir) is True
def test_includes_group_granted_viewers(self, backend: TantivyBackend) -> None: def test_includes_group_granted_viewers(self, backend: TantivyBackend) -> None:
"""Rebuild must index viewer ids for group-only grants, not just direct ones. """Rebuild must index viewer ids for group-only grants, not just direct ones.
+27 -91
View File
@@ -4,7 +4,6 @@ from pathlib import Path
from unittest import mock from unittest import mock
import pikepdf import pikepdf
import pytest
from django.contrib.auth.models import Group from django.contrib.auth.models import Group
from django.contrib.auth.models import Permission from django.contrib.auth.models import Permission
from django.contrib.auth.models import User from django.contrib.auth.models import User
@@ -13,7 +12,6 @@ from django.test import TestCase
from django.test.utils import CaptureQueriesContext from django.test.utils import CaptureQueriesContext
from guardian.shortcuts import get_groups_with_perms from guardian.shortcuts import get_groups_with_perms
from guardian.shortcuts import get_users_with_perms from guardian.shortcuts import get_users_with_perms
from pytest_mock import MockerFixture
from documents import bulk_edit from documents import bulk_edit
from documents.models import Correspondent from documents.models import Correspondent
@@ -25,7 +23,6 @@ from documents.models import StoragePath
from documents.models import Tag from documents.models import Tag
from documents.permissions import set_permissions_for_objects from documents.permissions import set_permissions_for_objects
from paperless_testing.dirs import DirectoriesMixin from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.factories import DocumentFactory
from paperless_testing.permissions import grant_object from paperless_testing.permissions import grant_object
@@ -1973,22 +1970,18 @@ class TestPDFActions(DirectoriesMixin, TestCase):
self.assertIn("Error removing password from document", cm.output[0]) self.assertIn("Error removing password from document", cm.output[0])
@pytest.mark.django_db class TestBulkEditReprocess(DirectoriesMixin, TestCase):
class TestBulkEditReprocess: def setUp(self) -> None:
@pytest.fixture super().setUp()
def mock_task(self, mocker: MockerFixture) -> mock.MagicMock:
return mocker.patch( self.doc = Document.objects.create(
"documents.bulk_edit.update_document_content_maybe_archive_file", title="test",
checksum="A",
mime_type="application/pdf",
) )
@staticmethod @mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file")
def _queued_ids(mock_task: mock.MagicMock) -> list[int]: def test_reprocess_defaults_to_local(self, mock_task: mock.Mock) -> None:
return [
call.kwargs["kwargs"]["document_id"]
for call in mock_task.apply_async.call_args_list
]
def test_reprocess_defaults_to_local(self, mock_task: mock.MagicMock) -> None:
""" """
GIVEN: GIVEN:
- A reprocess request that says nothing about remote OCR - A reprocess request that says nothing about remote OCR
@@ -1997,17 +1990,18 @@ class TestBulkEditReprocess:
THEN: THEN:
- The task is queued without asking for the remote engine - The task is queued without asking for the remote engine
""" """
doc = DocumentFactory() result = bulk_edit.reprocess([self.doc.id])
assert bulk_edit.reprocess([doc.id]) == "OK"
self.assertEqual(result, "OK")
mock_task.apply_async.assert_called_once() mock_task.apply_async.assert_called_once()
assert mock_task.apply_async.call_args.kwargs["kwargs"] == { _, kwargs = mock_task.apply_async.call_args
"document_id": doc.id, self.assertEqual(
"remote_ocr": False, kwargs["kwargs"],
} {"document_id": self.doc.id, "remote_ocr": False},
)
def test_reprocess_passes_remote_ocr(self, mock_task: mock.MagicMock) -> None: @mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file")
def test_reprocess_passes_remote_ocr(self, mock_task: mock.Mock) -> None:
""" """
GIVEN: GIVEN:
- A reprocess request that explicitly asks for remote OCR - A reprocess request that explicitly asks for remote OCR
@@ -2016,72 +2010,14 @@ class TestBulkEditReprocess:
THEN: THEN:
- The request is forwarded to the task for every document - The request is forwarded to the task for every document
""" """
docs = DocumentFactory.create_batch(2) other = Document.objects.create(
title="test2",
bulk_edit.reprocess([doc.id for doc in docs], remote_ocr=True) checksum="B",
mime_type="application/pdf",
assert mock_task.apply_async.call_count == 2
for call in mock_task.apply_async.call_args_list:
assert call.kwargs["kwargs"]["remote_ocr"]
def test_reprocess_root_uses_latest_version(
self,
mock_task: mock.MagicMock,
) -> None:
"""
GIVEN:
- A root document with two versions
WHEN:
- reprocess is called with the root document
THEN:
- The latest version is reprocessed, not the root's original file
"""
root = DocumentFactory()
DocumentFactory(root_document=root, version_index=1)
latest = DocumentFactory(root_document=root, version_index=2)
bulk_edit.reprocess([root.id])
assert self._queued_ids(mock_task) == [latest.id]
def test_reprocess_explicit_version(self, mock_task: mock.MagicMock) -> None:
"""
GIVEN:
- A root document with two versions
WHEN:
- reprocess is called with the older version
THEN:
- That version is reprocessed
"""
root = DocumentFactory()
older = DocumentFactory(root_document=root, version_index=1)
DocumentFactory(root_document=root, version_index=2)
bulk_edit.reprocess([older.id])
assert self._queued_ids(mock_task) == [older.id]
def test_reprocess_root_and_latest_version_dispatches_once(
self,
mock_task: mock.MagicMock,
) -> None:
"""
GIVEN:
- A root document with two versions, the latest created on a
different date than the root
WHEN:
- reprocess is called with both the root and its latest version
THEN:
- The latest version is reprocessed only once
"""
root = DocumentFactory(created=date(2024, 1, 1))
DocumentFactory(root_document=root, version_index=1)
latest = DocumentFactory(
root_document=root,
version_index=2,
created=date(2025, 1, 1),
) )
bulk_edit.reprocess([root.id, latest.id]) bulk_edit.reprocess([self.doc.id, other.id], remote_ocr=True)
assert self._queued_ids(mock_task) == [latest.id] self.assertEqual(mock_task.apply_async.call_count, 2)
for call in mock_task.apply_async.call_args_list:
self.assertTrue(call.kwargs["kwargs"]["remote_ocr"])
-77
View File
@@ -20,7 +20,6 @@ from documents.sanity_checker import SanityCheckMessages
from documents.tests.helpers import dummy_preprocess from documents.tests.helpers import dummy_preprocess
from paperless_testing.assertions import FileSystemAssertsMixin from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.factories import DocumentFactory
@pytest.mark.django_db @pytest.mark.django_db
@@ -288,82 +287,6 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
tasks.update_document_content_maybe_archive_file(doc.pk) tasks.update_document_content_maybe_archive_file(doc.pk)
self.assertNotEqual(Document.objects.get(pk=doc.pk).content, "test") self.assertNotEqual(Document.objects.get(pk=doc.pk).content, "test")
def _create_root_with_version(self) -> tuple[Document, Document]:
sample1 = self.dirs.scratch_dir / "sample.pdf"
shutil.copy(
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
sample1,
)
root = DocumentFactory(content="root content", mime_type="application/pdf")
version = DocumentFactory(
content="my document",
filename=sample1,
mime_type="application/pdf",
root_document=root,
version_index=1,
)
return root, version
@mock.patch("documents.tasks.clear_document_caches")
@mock.patch("documents.search.get_backend")
def test_update_content_version_indexes_root(
self,
mock_get_backend: mock.Mock,
mock_clear_caches: mock.Mock,
) -> None:
"""
GIVEN:
- A root document with a version
WHEN:
- Update content task is called for the version
THEN:
- The version's content is updated
- The root document is indexed rather than the version
- Caches are cleared for both
"""
root, version = self._create_root_with_version()
tasks.update_document_content_maybe_archive_file(version.pk)
self.assertNotEqual(
Document.objects.get(pk=version.pk).content,
"my document",
)
self.assertEqual(Document.objects.get(pk=root.pk).content, "root content")
indexed = mock_get_backend.return_value.add_or_update.call_args.args[0]
self.assertEqual(indexed.pk, root.pk)
mock_clear_caches.assert_has_calls(
[mock.call(version.pk), mock.call(root.pk)],
)
@override_settings(AI_ENABLED=True, LLM_EMBEDDING_BACKEND="huggingface")
@mock.patch("documents.tasks.llm_index_add_or_update_document")
@mock.patch("documents.search.get_backend")
def test_update_content_version_updates_llm_index_for_root(
self,
mock_get_backend: mock.Mock,
mock_llm_index: mock.Mock,
) -> None:
"""
GIVEN:
- A root document with a version
- The LLM index is enabled
WHEN:
- Update content task is called for the version
THEN:
- The LLM index is updated for the root document, not the version
"""
root, version = self._create_root_with_version()
tasks.update_document_content_maybe_archive_file(version.pk)
mock_llm_index.assert_called_once()
self.assertEqual(mock_llm_index.call_args.args[0].pk, root.pk)
class TestUpdateContentRemoteOCR(DirectoriesMixin, TestCase): class TestUpdateContentRemoteOCR(DirectoriesMixin, TestCase):
""" """
+19 -2
View File
@@ -2,7 +2,7 @@ msgid ""
msgstr "" msgstr ""
"Project-Id-Version: paperless-ngx\n" "Project-Id-Version: paperless-ngx\n"
"Report-Msgid-Bugs-To: \n" "Report-Msgid-Bugs-To: \n"
"POT-Creation-Date: 2026-10-06 15:12+0000\n" "POT-Creation-Date: 2026-10-05 16:26+0000\n"
"PO-Revision-Date: 2022-02-17 04:17\n" "PO-Revision-Date: 2022-02-17 04:17\n"
"Last-Translator: \n" "Last-Translator: \n"
"Language-Team: English\n" "Language-Team: English\n"
@@ -1941,8 +1941,25 @@ msgstr ""
msgid "As a final step, please complete the following form:" msgid "As a final step, please complete the following form:"
msgstr "" msgstr ""
#: documents/validators.py:24
#, python-brace-format
msgid "Unable to parse URI {value}, missing scheme"
msgstr ""
#: documents/validators.py:29
#, python-brace-format
msgid "Unable to parse URI {value}, missing net location or path"
msgstr ""
#: documents/validators.py:36 #: documents/validators.py:36
msgid ", " msgid ""
"URI scheme '{parts.scheme}' is not allowed. Allowed schemes: {', '."
"join(allowed_schemes)}"
msgstr ""
#: documents/validators.py:45
#, python-brace-format
msgid "Unable to parse URI {value}"
msgstr "" msgstr ""
#: documents/views.py:336 documents/views.py:2729 #: documents/views.py:336 documents/views.py:2729
@@ -137,20 +137,9 @@ class TestNginxService:
reason="No Gotenberg/Tika servers to test with", reason="No Gotenberg/Tika servers to test with",
) )
class TestParserLive: class TestParserLive:
# Rasterizer versions shift a few pixels, so compare perceptual hashes by @staticmethod
# Hamming distance (out of 18 * 18 = 324 bits) rather than for equality def imagehash(file: Path, hash_size: int = 18) -> str:
MAX_HASH_DISTANCE = 8 return f"{average_hash(Image.open(file), hash_size)}"
@classmethod
def assert_thumbnails_similar(cls, generated: Path, expected: Path) -> None:
distance = average_hash(Image.open(generated), 18) - average_hash(
Image.open(expected),
18,
)
assert distance <= cls.MAX_HASH_DISTANCE, (
f"Thumbnail {generated} differs from {expected} by {distance} bits "
f"(max {cls.MAX_HASH_DISTANCE})"
)
def test_get_thumbnail( def test_get_thumbnail(
self, self,
@@ -179,7 +168,12 @@ class TestParserLive:
assert thumb.exists() assert thumb.exists()
assert thumb.is_file() assert thumb.is_file()
self.assert_thumbnails_similar(thumb, simple_txt_email_thumbnail_file) assert self.imagehash(thumb) == self.imagehash(
simple_txt_email_thumbnail_file,
), (
f"Created thumbnail {thumb} differs from expected file "
f"{simple_txt_email_thumbnail_file}"
)
def test_tika_parse_successful(self, mail_parser: MailDocumentParser) -> None: def test_tika_parse_successful(self, mail_parser: MailDocumentParser) -> None:
""" """
@@ -261,7 +255,7 @@ class TestParserLive:
THEN: THEN:
- Gotenberg shall be called to generate the PDF - Gotenberg shall be called to generate the PDF
- The archive PDF shall contain the expected content - The archive PDF shall contain the expected content
- The generated thumbnail shall be perceptually close to the expected image - The generated thumbnail shall match the expected image hash
""" """
util_call_with_backoff(mail_parser.parse, [html_email_file, "message/rfc822"]) util_call_with_backoff(mail_parser.parse, [html_email_file, "message/rfc822"])
@@ -278,4 +272,14 @@ class TestParserLive:
html_email_file, html_email_file,
"message/rfc822", "message/rfc822",
) )
self.assert_thumbnails_similar(generated_thumbnail, html_email_thumbnail_file) generated_thumbnail_hash = self.imagehash(generated_thumbnail)
# The created PDF is not reproducible, but the converted image
# should always look the same
expected_hash = self.imagehash(html_email_thumbnail_file)
assert generated_thumbnail_hash == expected_hash, (
f"PDF thumbnail differs from expected. "
f"Generated: {generated_thumbnail}, "
f"Hash: {generated_thumbnail_hash} vs {expected_hash}"
)
+1 -1
View File
@@ -1,6 +1,6 @@
from typing import Final from typing import Final
__version__: Final[tuple[int, int, int]] = (3, 2, 1) __version__: Final[tuple[int, int, int]] = (3, 3, 0)
# Version string like X.Y.Z # Version string like X.Y.Z
__full_version_str__: Final[str] = ".".join(map(str, __version__)) __full_version_str__: Final[str] = ".".join(map(str, __version__))
# Version string like X.Y # Version string like X.Y
Generated
+1 -1
View File
@@ -2971,7 +2971,7 @@ wheels = [
[[package]] [[package]]
name = "paperless-ngx" name = "paperless-ngx"
version = "3.2.1" version = "3.3.0"
source = { virtual = "." } source = { virtual = "." }
dependencies = [ dependencies = [
{ name = "azure-ai-documentintelligence" }, { name = "azure-ai-documentintelligence" },