Compare commits

..
Author SHA1 Message Date
shamoon 6f7ed1d4a7 Bump version to 3.3.0 2026-10-05 20:34:24 -07:00
shamoon 0428bf6955 Merge branch 'dev' 2026-10-05 20:33:36 -07:00
shamoon c63afb47b2 Documentation: correct duplicates info (#14243) 2026-09-23 08:27:18 -07:00
30 changed files with 131 additions and 338 deletions

No files matched your search

+3 -3
View File
@@ -80,7 +80,7 @@ jobs:
needs: changes
if: needs.changes.outputs.backend_changed == 'true'
name: "Python ${{ matrix.python-version }}"
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
contents: read
strategy:
@@ -114,7 +114,7 @@ jobs:
packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils
- name: Configure ImageMagick
run: |
sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-7/policy.xml
sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-6/policy.xml
- name: Install Python dependencies
env:
PYTHON_VERSION: ${{ steps.setup-python.outputs.python-version }}
@@ -158,7 +158,7 @@ jobs:
needs: changes
if: needs.changes.outputs.backend_changed == 'true'
name: Check project typing
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
contents: read
env:
+3 -3
View File
@@ -24,10 +24,10 @@ jobs:
fail-fast: false
matrix:
include:
- runner: ubuntu-26.04
- runner: ubuntu-24.04
arch: amd64
platform: linux/amd64
- runner: ubuntu-26.04-arm
- runner: ubuntu-24.04-arm
arch: arm64
platform: linux/arm64
runs-on: ${{ matrix.runner }}
@@ -163,7 +163,7 @@ jobs:
archive: false
merge-and-push:
name: Merge and Push Manifest
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
needs: build-arch
if: needs.build-arch.outputs.should-push == 'true'
environment: image-publishing
+2 -2
View File
@@ -65,7 +65,7 @@ jobs:
needs: changes
if: needs.changes.outputs.docs_changed == 'true'
name: Build Documentation
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
steps:
- uses: actions/configure-pages@45bfe0192ca1faeb007ade9deae92b16b8254a0d # v6.0.0
- name: Checkout
@@ -102,7 +102,7 @@ jobs:
name: Deploy Documentation
needs: [changes, build]
if: github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.changes.outputs.docs_changed == 'true'
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
pages: write
id-token: write
+6 -6
View File
@@ -72,7 +72,7 @@ jobs:
needs: changes
if: needs.changes.outputs.frontend_changed == 'true'
name: Install Dependencies
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
contents: read
steps:
@@ -104,7 +104,7 @@ jobs:
name: Lint
needs: [changes, install-dependencies]
if: needs.changes.outputs.frontend_changed == 'true'
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
contents: read
steps:
@@ -137,7 +137,7 @@ jobs:
name: "Unit Tests (${{ matrix.shard-index }}/${{ matrix.shard-count }})"
needs: [changes, install-dependencies]
if: needs.changes.outputs.frontend_changed == 'true'
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
contents: read
strategy:
@@ -188,10 +188,10 @@ jobs:
name: E2E Tests
needs: [changes, install-dependencies]
if: needs.changes.outputs.frontend_changed == 'true'
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
contents: read
container: mcr.microsoft.com/playwright:v1.62.1-resolute
container: mcr.microsoft.com/playwright:v1.62.1-noble
env:
PLAYWRIGHT_BROWSERS_PATH: /ms-playwright
PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD: 1
@@ -246,7 +246,7 @@ jobs:
name: Frontend Build
needs: [changes, unit-tests, e2e-tests]
if: needs.changes.outputs.frontend_changed == 'true'
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
contents: read
steps:
+5 -5
View File
@@ -14,7 +14,7 @@ permissions: {}
jobs:
wait-for-docker:
name: Wait for Docker Build
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
checks: read
statuses: read
@@ -30,7 +30,7 @@ jobs:
build-release:
name: Build Release
needs: wait-for-docker
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
contents: read
steps:
@@ -73,7 +73,7 @@ jobs:
timeout-minutes: 12
uses: $/.github/actions/apt-install
with:
packages: gettext libleptonica6
packages: gettext liblept5
# ---- Build Documentation ----
- name: Build documentation
env:
@@ -145,7 +145,7 @@ jobs:
publish-release:
name: Publish Release
needs: build-release
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
contents: write
pull-requests: write
@@ -197,7 +197,7 @@ jobs:
name: Append Changelog
needs: publish-release
if: needs.publish-release.outputs.prerelease == 'false'
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
contents: write
pull-requests: write
+2 -2
View File
@@ -15,7 +15,7 @@ permissions:
jobs:
zizmor:
name: Run zizmor
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
contents: read
actions: read
@@ -29,7 +29,7 @@ jobs:
uses: zizmorcore/zizmor-action@cc914d7f3750a2d13d75c7f184a1060aa0e9d482 # v0.6.4
semgrep:
name: Semgrep CE
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
container:
image: semgrep/semgrep:1.155.0@sha256:cc869c685dcc0fe497c86258da9f205397d8108e56d21a86082ea4886e52784d
if: github.actor != 'dependabot[bot]'
+2 -2
View File
@@ -17,7 +17,7 @@ jobs:
cleanup-images:
name: Cleanup Image Tags for ${{ matrix.primary-name }}
if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
environment: registry-maintenance
strategy:
fail-fast: false
@@ -42,7 +42,7 @@ jobs:
cleanup-untagged-images:
name: Cleanup Untagged Images Tags for ${{ matrix.primary-name }}
if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
needs:
- cleanup-images
environment: registry-maintenance
+1 -1
View File
@@ -21,7 +21,7 @@ on:
jobs:
analyze:
name: Analyze
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
actions: read
contents: read
+1 -1
View File
@@ -13,7 +13,7 @@ jobs:
synchronize-with-crowdin:
name: Crowdin Sync
if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
environment: translation-sync
steps:
- name: Checkout
+1 -1
View File
@@ -7,7 +7,7 @@ jobs:
# Note: peakoss/anti-slop does not support the `issues` event yet (all of its
# issue inputs are still commented out upstream), so the checks that the PR Bot
# workflow gets from the action are implemented manually here.
runs-on: ubuntu-slim
runs-on: ubuntu-latest
permissions:
issues: write
steps:
+2 -2
View File
@@ -4,7 +4,7 @@ on:
types: [opened]
jobs:
Anti-slop:
runs-on: ubuntu-slim
runs-on: ubuntu-latest
permissions:
contents: read
issues: read
@@ -24,7 +24,7 @@ jobs:
ASLOP-PR-VERIFY
pr-bot:
name: Automated PR Bot
runs-on: ubuntu-slim
runs-on: ubuntu-latest
# Runs after Anti-slop so the welcome comment can see whether the PR was closed
# instead of racing it. Still runs if that job fails, so labeling is not lost.
needs: Anti-slop
+1 -1
View File
@@ -12,7 +12,7 @@ permissions:
jobs:
pr_opened_or_reopened:
name: pr_opened_or_reopened
runs-on: ubuntu-slim
runs-on: ubuntu-24.04
permissions:
# write permission is required for autolabeler
pull-requests: write
+5 -5
View File
@@ -9,7 +9,7 @@ jobs:
stale:
name: 'Stale'
if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
issues: write
pull-requests: write
@@ -34,7 +34,7 @@ jobs:
lock-threads:
name: 'Lock Old Threads'
if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-26.04
runs-on: ubuntu-24.04
permissions:
issues: write
pull-requests: write
@@ -58,7 +58,7 @@ jobs:
close-answered-discussions:
name: 'Close Answered Discussions'
if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-slim
runs-on: ubuntu-24.04
permissions:
discussions: write
steps:
@@ -117,7 +117,7 @@ jobs:
close-outdated-discussions:
name: 'Close Outdated Discussions'
if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-slim
runs-on: ubuntu-24.04
permissions:
discussions: write
steps:
@@ -211,7 +211,7 @@ jobs:
close-unsupported-feature-requests:
name: 'Close Unsupported Feature Requests'
if: github.repository_owner == 'paperless-ngx'
runs-on: ubuntu-slim
runs-on: ubuntu-24.04
permissions:
discussions: write
steps:
+1 -1
View File
@@ -8,7 +8,7 @@ env:
jobs:
generate-translate-strings:
name: Generate Translation Strings
runs-on: ubuntu-26.04
runs-on: ubuntu-latest
environment: translation-sync
permissions:
contents: write
+3 -1
View File
@@ -76,7 +76,9 @@ is not supported by any of the available parsers.
**A:** Not by default. As of v3, a file whose contents match an existing document is still
consumed, and the duplicate is flagged in the UI — open the document and check the
**Duplicates** tab to review documents that share the same content. If you prefer the old
**Duplicates** tab to review documents that share the same content, or filter the document
list by **Duplicates** to find all of them (see
[Duplicate documents](usage.md#duplicate-documents)). If you prefer the old
behavior of rejecting duplicates during consumption, set
[`PAPERLESS_CONSUMER_DELETE_DUPLICATES`](configuration.md#PAPERLESS_CONSUMER_DELETE_DUPLICATES)
to `true`.
+7 -8
View File
@@ -299,19 +299,18 @@ for details.
### Duplicate documents
By default, Paperless-ngx **does not reject duplicates**. If you consume a file whose
contents exactly match an existing document (same checksum), the new copy is still
consumed and a warning is logged. The task entry for the upload also flags that a
duplicate was detected and links to the existing document(s).
contents match an existing document (same original or archive checksum), the new copy is
still consumed and a warning is logged.
To review duplicates, open a document and switch to the **Duplicates** tab on the
document detail page. It lists other documents that share the same content, including any
that are in the trash (shown with a badge), and links to each so you can decide which to
keep.
When a document has duplicates, a **Duplicates** tab appears on its detail page, listing
the other documents you can view that share the same content (including any in the trash).
To find all documents with duplicates, choose **Duplicates** in the document list's text
filter dropdown, or use `has_duplicates=true` in the REST API.
If you would rather reject duplicates at consumption time (the pre-v3 behavior), set
[`PAPERLESS_CONSUMER_DELETE_DUPLICATES`](configuration.md#PAPERLESS_CONSUMER_DELETE_DUPLICATES)
to `true`. The duplicate file is then deleted instead of consumed, and the task fails with
a "document already exists" message.
a "Document already exists" message linking to the existing document.
## Document Suggestions
+1 -1
View File
@@ -1,6 +1,6 @@
[project]
name = "paperless-ngx"
version = "3.2.1"
version = "3.3.0"
description = """\
A community-supported supercharged document management system: scan, index and archive all your physical documents\
"""
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "paperless-ngx-ui",
"version": "3.2.1",
"version": "3.3.0",
"scripts": {
"preinstall": "npx only-allow pnpm",
"ng": "ng",
+1 -1
View File
@@ -8,7 +8,7 @@ export const environment = {
apiVersion: '10', // match src/paperless/settings.py
appTitle: DEFAULT_APP_TITLE,
tag: 'prod',
version: '3.2.1',
version: '3.3.0',
webSocketHost: window.location.host,
webSocketProtocol: window.location.protocol == 'https:' ? 'wss:' : 'ws:',
webSocketBaseUrl: base_url.pathname + 'ws/',
+2 -8
View File
@@ -408,16 +408,10 @@ def reprocess(doc_ids: list[int], *, remote_ocr: bool = False) -> Literal["OK"]:
Consumption workflows do not run here, so ``remote_ocr`` is how the user
asks for the remote engine when it is not configured to handle everything.
A root document with versions reprocesses its latest version, which is the
file whose content, archive and thumbnail are shown for it.
"""
for doc in Document.objects.select_related("root_document").filter(
id__in=doc_ids,
):
pair = _resolve_root_and_source_doc(doc)
for document_id in doc_ids:
update_document_content_maybe_archive_file.apply_async(
kwargs={"document_id": pair.source_doc.id, "remote_ocr": remote_ocr},
kwargs={"document_id": document_id, "remote_ocr": remote_ocr},
headers={"trigger_source": PaperlessTask.TriggerSource.MANUAL},
)
+32 -38
View File
@@ -31,7 +31,6 @@ from documents.search._query import parse_user_query
from documents.search._schema import _write_sentinels
from documents.search._schema import build_schema
from documents.search._schema import open_or_rebuild_index
from documents.search._schema import rebuild_in_progress
from documents.search._schema import wipe_index
from documents.search._tokenizer import ascii_fold
from documents.search._tokenizer import autocomplete_tokens
@@ -1111,44 +1110,39 @@ class TantivyBackend:
flushing a segment, deferring merge work; they do not avoid it.
"""
wipe_index(self._path)
# The marker covers the window where the empty index is already stamped
# as current but not yet populated, so an interrupted rebuild is retried.
with rebuild_in_progress(self._path):
new_index = tantivy.Index(build_schema(), path=str(self._path))
_write_sentinels(self._path)
register_tokenizers(new_index, settings.SEARCH_LANGUAGE)
new_index = tantivy.Index(build_schema(), path=str(self._path))
_write_sentinels(self._path)
register_tokenizers(new_index, settings.SEARCH_LANGUAGE)
# Point instance at the new index so _build_tantivy_doc uses it
old_index, old_schema = self._raw_index, self._raw_schema
self._raw_index = new_index
self._raw_schema = new_index.schema
# Stream documents one-by-one (so the progress bar advances per
# document) while fetching viewer permissions one SQL query per
# chunk. The stream is Sized, so iter_wrapper can still discover
# the total.
documents_stream = _DocumentViewerStream(documents, chunk_size=1000)
try:
writer = new_index.writer(heap_size=writer_heap_bytes)
for document, (viewer_ids, viewer_group_ids) in iter_wrapper(
documents_stream,
):
doc = self._build_tantivy_doc(
document,
viewer_ids=viewer_ids,
viewer_group_ids=viewer_group_ids,
)
writer.add_document(doc)
writer.commit()
# Wait for background merge threads to finish so all segments
# are fully merged and persisted before the index is considered
# rebuilt.
writer.wait_merging_threads()
new_index.reload()
except BaseException: # pragma: no cover
# Restore old index on failure so the backend remains usable
self._raw_index = old_index
self._raw_schema = old_schema
raise
# Point instance at the new index so _build_tantivy_doc uses it
old_index, old_schema = self._raw_index, self._raw_schema
self._raw_index = new_index
self._raw_schema = new_index.schema
# Stream documents one-by-one (so the progress bar advances per
# document) while fetching viewer permissions one SQL query per chunk.
# The stream is Sized, so iter_wrapper can still discover the total.
documents_stream = _DocumentViewerStream(documents, chunk_size=1000)
try:
writer = new_index.writer(heap_size=writer_heap_bytes)
for document, (viewer_ids, viewer_group_ids) in iter_wrapper(
documents_stream,
):
doc = self._build_tantivy_doc(
document,
viewer_ids=viewer_ids,
viewer_group_ids=viewer_group_ids,
)
writer.add_document(doc)
writer.commit()
# Wait for background merge threads to finish so all segments are
# fully merged and persisted before the index is considered rebuilt.
writer.wait_merging_threads()
new_index.reload()
except BaseException: # pragma: no cover
# Restore old index on failure so the backend remains usable
self._raw_index = old_index
self._raw_schema = old_schema
raise
def chunked(iterable, size):
+4 -45
View File
@@ -4,7 +4,6 @@ import hashlib
import json
import logging
import shutil
from contextlib import contextmanager
from typing import TYPE_CHECKING
from typing import Final
from typing import NamedTuple
@@ -17,7 +16,6 @@ from whoosh_compat import FieldKind
from documents.search._fields import PUBLIC_FIELDS
if TYPE_CHECKING:
from collections.abc import Iterator
from pathlib import Path
logger = logging.getLogger("paperless.search")
@@ -30,11 +28,6 @@ logger = logging.getLogger("paperless.search")
# v3 - barcodes JSON field for stored barcode contents
SCHEMA_VERSION: Final[int] = 3
# Present in the index directory from the moment a full rebuild starts until it
# finishes. If a rebuild is interrupted it is left behind, so the half-built
# index is not mistaken for a complete one.
REBUILD_MARKER: Final[str] = ".rebuilding"
class FieldDescriptor(NamedTuple):
"""One tantivy field, in declaration order.
@@ -262,9 +255,9 @@ def needs_rebuild(index_dir: Path) -> bool:
"""
Check if the search index needs rebuilding.
True if a previous full rebuild never finished (the rebuild marker is still
present), or if the index's stamped settings no longer match the current
configuration. See _settings_mismatch().
Reads .index_settings.json to compare the stored schema version, search
language and schema fingerprint against the current configuration. Returns
True if the file is missing, unparsable, or any value mismatches.
Args:
index_dir: Path to the search index directory
@@ -272,40 +265,6 @@ def needs_rebuild(index_dir: Path) -> bool:
Returns:
True if the index needs rebuilding, False if it's up to date
"""
if (index_dir / REBUILD_MARKER).exists():
logger.warning("Previous search index rebuild did not finish - rebuilding.")
return True
return _settings_mismatch(index_dir)
@contextmanager
def rebuild_in_progress(index_dir: Path) -> Iterator[None]:
"""
Flag the index as incomplete for the duration of a full rebuild.
The marker is cleared only if the block exits cleanly. There is deliberately
no try/finally: an exception must leave the marker behind so the next
needs_rebuild() check retries the rebuild.
"""
marker = index_dir / REBUILD_MARKER
marker.touch()
yield
marker.unlink(missing_ok=True)
def _settings_mismatch(index_dir: Path) -> bool:
"""
Check the stamped settings against the current configuration.
Reads .index_settings.json to compare the stored schema version, search
language and schema fingerprint. Returns True if the file is missing,
unparsable, or any value mismatches.
This deliberately ignores the rebuild marker: open_or_rebuild_index() uses it
so that a process opening the index while another process is mid-rebuild
(or after one died) does not wipe the partial index out from under it.
Repopulating is the job of ``document_index reindex``.
"""
settings_file = index_dir / ".index_settings.json"
if not settings_file.exists():
return True
@@ -374,7 +333,7 @@ def open_or_rebuild_index(index_dir: Path | None = None) -> tantivy.Index:
index_dir = cast("Path", settings.INDEX_DIR)
if not index_dir.exists():
return tantivy.Index(build_schema())
if _settings_mismatch(index_dir):
if needs_rebuild(index_dir):
wipe_index(index_dir)
idx = tantivy.Index(build_schema(), path=str(index_dir))
_write_sentinels(index_dir)
+3 -8
View File
@@ -490,23 +490,18 @@ def update_document_content_maybe_archive_file(
shutil.move(thumbnail, document.thumbnail_path)
document.refresh_from_db()
root_document = (
document.root_document if document.root_document_id else document
)
logger.info(
f"Updating index for document {root_document.pk} ({document.archive_checksum})",
f"Updating index for document {document_id} ({document.archive_checksum})",
)
from documents.search import get_backend
get_backend().add_or_update(root_document)
get_backend().add_or_update(document)
ai_config = AIConfig()
if ai_config.llm_index_enabled:
llm_index_add_or_update_document(root_document)
llm_index_add_or_update_document(document)
clear_document_caches(document.pk)
if root_document.pk != document.pk:
clear_document_caches(root_document.pk)
except Exception:
logger.exception(
@@ -16,10 +16,7 @@ from documents.search._backend import TantivyBackend
from documents.search._backend import WriteBatch
from documents.search._backend import get_backend
from documents.search._backend import reset_backend
from documents.search._schema import REBUILD_MARKER
from documents.search._schema import needs_rebuild
from documents.signals.handlers import add_to_index
from paperless_testing.dirs import PaperlessDirs
from paperless_testing.factories import CorrespondentFactory
from paperless_testing.factories import DocumentFactory
from paperless_testing.factories import DocumentTypeFactory
@@ -826,53 +823,6 @@ class TestRebuild:
backend.rebuild(Document.objects.all(), iter_wrapper=wrapper)
assert 30 in seen
def test_successful_rebuild_leaves_index_up_to_date(
self,
backend: TantivyBackend,
paperless_dirs: PaperlessDirs,
) -> None:
"""
GIVEN:
- A backend and one document
WHEN:
- rebuild() completes
THEN:
- needs_rebuild() is False and no rebuild marker remains
"""
DocumentFactory.create()
backend.rebuild(Document.objects.all())
assert needs_rebuild(paperless_dirs.index_dir) is False
assert not (paperless_dirs.index_dir / REBUILD_MARKER).exists()
def test_interrupted_rebuild_is_retried(
self,
backend: TantivyBackend,
paperless_dirs: PaperlessDirs,
) -> None:
"""
GIVEN:
- A rebuild that dies while indexing documents (e.g. the database
connection is lost)
WHEN:
- needs_rebuild() is checked afterwards
THEN:
- It is True, even though the empty index was already stamped with
current settings, so the next start rebuilds instead of reporting
the index as up to date
"""
DocumentFactory.create()
def die(pairs):
raise RuntimeError("terminating connection due to administrator command")
yield # pragma: no cover
with pytest.raises(RuntimeError):
backend.rebuild(Document.objects.all(), iter_wrapper=die)
assert needs_rebuild(paperless_dirs.index_dir) is True
def test_includes_group_granted_viewers(self, backend: TantivyBackend) -> None:
"""Rebuild must index viewer ids for group-only grants, not just direct ones.
-66
View File
@@ -2021,69 +2021,3 @@ class TestBulkEditReprocess(DirectoriesMixin, TestCase):
self.assertEqual(mock_task.apply_async.call_count, 2)
for call in mock_task.apply_async.call_args_list:
self.assertTrue(call.kwargs["kwargs"]["remote_ocr"])
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file")
def test_reprocess_root_uses_latest_version(self, mock_task: mock.Mock) -> None:
"""
GIVEN:
- A root document with two versions
WHEN:
- reprocess is called with the root document
THEN:
- The latest version is reprocessed, not the root's original file
"""
Document.objects.create(
title="test",
checksum="A-v1",
mime_type="application/pdf",
root_document=self.doc,
version_index=1,
)
latest = Document.objects.create(
title="test",
checksum="A-v2",
mime_type="application/pdf",
root_document=self.doc,
version_index=2,
)
bulk_edit.reprocess([self.doc.id])
mock_task.apply_async.assert_called_once()
self.assertEqual(
mock_task.apply_async.call_args.kwargs["kwargs"]["document_id"],
latest.id,
)
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file")
def test_reprocess_explicit_version(self, mock_task: mock.Mock) -> None:
"""
GIVEN:
- A root document with two versions
WHEN:
- reprocess is called with the older version
THEN:
- That version is reprocessed
"""
older = Document.objects.create(
title="test",
checksum="A-v1",
mime_type="application/pdf",
root_document=self.doc,
version_index=1,
)
Document.objects.create(
title="test",
checksum="A-v2",
mime_type="application/pdf",
root_document=self.doc,
version_index=2,
)
bulk_edit.reprocess([older.id])
mock_task.apply_async.assert_called_once()
self.assertEqual(
mock_task.apply_async.call_args.kwargs["kwargs"]["document_id"],
older.id,
)
-55
View File
@@ -287,61 +287,6 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
tasks.update_document_content_maybe_archive_file(doc.pk)
self.assertNotEqual(Document.objects.get(pk=doc.pk).content, "test")
@mock.patch("documents.tasks.clear_document_caches")
@mock.patch("documents.search.get_backend")
def test_update_content_version_indexes_root(
self,
mock_get_backend: mock.Mock,
mock_clear_caches: mock.Mock,
) -> None:
"""
GIVEN:
- A root document with a version
WHEN:
- Update content task is called for the version
THEN:
- The version's content is updated
- The root document is indexed rather than the version
- Caches are cleared for both
"""
sample1 = self.dirs.scratch_dir / "sample.pdf"
shutil.copy(
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
sample1,
)
root = Document.objects.create(
title="test",
content="root content",
checksum="root",
mime_type="application/pdf",
)
version = Document.objects.create(
title="test",
content="my document",
checksum="wow",
filename=sample1,
mime_type="application/pdf",
root_document=root,
version_index=1,
)
tasks.update_document_content_maybe_archive_file(version.pk)
self.assertNotEqual(
Document.objects.get(pk=version.pk).content,
"my document",
)
self.assertEqual(Document.objects.get(pk=root.pk).content, "root content")
indexed = mock_get_backend.return_value.add_or_update.call_args.args[0]
self.assertEqual(indexed.pk, root.pk)
mock_clear_caches.assert_has_calls(
[mock.call(version.pk), mock.call(root.pk)],
)
class TestUpdateContentRemoteOCR(DirectoriesMixin, TestCase):
"""
+19 -2
View File
@@ -2,7 +2,7 @@ msgid ""
msgstr ""
"Project-Id-Version: paperless-ngx\n"
"Report-Msgid-Bugs-To: \n"
"POT-Creation-Date: 2026-10-06 15:12+0000\n"
"POT-Creation-Date: 2026-10-05 16:26+0000\n"
"PO-Revision-Date: 2022-02-17 04:17\n"
"Last-Translator: \n"
"Language-Team: English\n"
@@ -1941,8 +1941,25 @@ msgstr ""
msgid "As a final step, please complete the following form:"
msgstr ""
#: documents/validators.py:24
#, python-brace-format
msgid "Unable to parse URI {value}, missing scheme"
msgstr ""
#: documents/validators.py:29
#, python-brace-format
msgid "Unable to parse URI {value}, missing net location or path"
msgstr ""
#: documents/validators.py:36
msgid ", "
msgid ""
"URI scheme '{parts.scheme}' is not allowed. Allowed schemes: {', '."
"join(allowed_schemes)}"
msgstr ""
#: documents/validators.py:45
#, python-brace-format
msgid "Unable to parse URI {value}"
msgstr ""
#: documents/views.py:336 documents/views.py:2729
@@ -137,20 +137,9 @@ class TestNginxService:
reason="No Gotenberg/Tika servers to test with",
)
class TestParserLive:
# Rasterizer versions shift a few pixels, so compare perceptual hashes by
# Hamming distance (out of 18 * 18 = 324 bits) rather than for equality
MAX_HASH_DISTANCE = 8
@classmethod
def assert_thumbnails_similar(cls, generated: Path, expected: Path) -> None:
distance = average_hash(Image.open(generated), 18) - average_hash(
Image.open(expected),
18,
)
assert distance <= cls.MAX_HASH_DISTANCE, (
f"Thumbnail {generated} differs from {expected} by {distance} bits "
f"(max {cls.MAX_HASH_DISTANCE})"
)
@staticmethod
def imagehash(file: Path, hash_size: int = 18) -> str:
return f"{average_hash(Image.open(file), hash_size)}"
def test_get_thumbnail(
self,
@@ -179,7 +168,12 @@ class TestParserLive:
assert thumb.exists()
assert thumb.is_file()
self.assert_thumbnails_similar(thumb, simple_txt_email_thumbnail_file)
assert self.imagehash(thumb) == self.imagehash(
simple_txt_email_thumbnail_file,
), (
f"Created thumbnail {thumb} differs from expected file "
f"{simple_txt_email_thumbnail_file}"
)
def test_tika_parse_successful(self, mail_parser: MailDocumentParser) -> None:
"""
@@ -261,7 +255,7 @@ class TestParserLive:
THEN:
- Gotenberg shall be called to generate the PDF
- The archive PDF shall contain the expected content
- The generated thumbnail shall be perceptually close to the expected image
- The generated thumbnail shall match the expected image hash
"""
util_call_with_backoff(mail_parser.parse, [html_email_file, "message/rfc822"])
@@ -278,4 +272,14 @@ class TestParserLive:
html_email_file,
"message/rfc822",
)
self.assert_thumbnails_similar(generated_thumbnail, html_email_thumbnail_file)
generated_thumbnail_hash = self.imagehash(generated_thumbnail)
# The created PDF is not reproducible, but the converted image
# should always look the same
expected_hash = self.imagehash(html_email_thumbnail_file)
assert generated_thumbnail_hash == expected_hash, (
f"PDF thumbnail differs from expected. "
f"Generated: {generated_thumbnail}, "
f"Hash: {generated_thumbnail_hash} vs {expected_hash}"
)
+1 -1
View File
@@ -1,6 +1,6 @@
from typing import Final
__version__: Final[tuple[int, int, int]] = (3, 2, 1)
__version__: Final[tuple[int, int, int]] = (3, 3, 0)
# Version string like X.Y.Z
__full_version_str__: Final[str] = ".".join(map(str, __version__))
# Version string like X.Y
Generated
+1 -1
View File
@@ -2971,7 +2971,7 @@ wheels = [
[[package]]
name = "paperless-ngx"
version = "3.2.1"
version = "3.3.0"
source = { virtual = "." }
dependencies = [
{ name = "azure-ai-documentintelligence" },