Merge branch 'dev' into feature/centralized-share-links

This commit is contained in:
shamoon
2026-09-15 07:52:12 -07:00
186 changed files with 13805 additions and 4182 deletions
+3 -3
View File
@@ -10,7 +10,7 @@ import pytest
from django.contrib.auth import get_user_model
from django.contrib.contenttypes.models import ContentType
from guardian.shortcuts import clear_ct_cache
from pytest_django.fixtures import SettingsWrapper
from pytest_django.fixtures import Settings
from rest_framework.test import APIClient
from documents.tests.factories import DocumentFactory
@@ -100,7 +100,7 @@ def sample_doc(
@pytest.fixture()
def _search_index(
tmp_path: Path,
settings: SettingsWrapper,
settings: Settings,
) -> Generator[None, None, None]:
"""Create a temp index directory and point INDEX_DIR at it.
@@ -118,7 +118,7 @@ def _search_index(
@pytest.fixture()
def settings_timezone(settings: SettingsWrapper) -> zoneinfo.ZoneInfo:
def settings_timezone(settings: Settings) -> zoneinfo.ZoneInfo:
return zoneinfo.ZoneInfo(settings.TIME_ZONE)
+1 -1
View File
@@ -70,7 +70,7 @@ def clear_lru_cache() -> Generator[None, None, None]:
@pytest.fixture
def mock_date_parser_settings(settings: pytest_django.fixtures.SettingsWrapper) -> Any:
def mock_date_parser_settings(settings: pytest_django.fixtures.Settings) -> Any:
"""
Override Django settings for the duration of date parser tests.
"""
+3 -3
View File
@@ -6,7 +6,7 @@ from pathlib import Path
import pytest
import pytest_mock
from pytest_django.fixtures import SettingsWrapper
from pytest_django.fixtures import Settings
from documents.export.sinks import DirectoryExportSink
from documents.export.sinks import ExportSink
@@ -242,7 +242,7 @@ class TestZipExportSink:
self,
tmp_path: Path,
source_file: Path,
settings: SettingsWrapper,
settings: Settings,
) -> None:
scratch_dir = tmp_path / "scratch"
settings.SCRATCH_DIR = scratch_dir
@@ -261,7 +261,7 @@ class TestZipExportSink:
def test_abort_after_manifest_written_cleans_up_pending_tmp(
self,
tmp_path: Path,
settings: SettingsWrapper,
settings: Settings,
) -> None:
scratch_dir = tmp_path / "scratch"
settings.SCRATCH_DIR = scratch_dir
+2 -14
View File
@@ -1,25 +1,21 @@
from __future__ import annotations
import tempfile
from typing import TYPE_CHECKING
import pytest
import tantivy
from documents.search._backend import TantivyBackend
from documents.search._backend import reset_backend
from documents.search._schema import build_schema
from documents.search._tokenizer import register_tokenizers
if TYPE_CHECKING:
from collections.abc import Generator
from pathlib import Path
from pytest_django.fixtures import SettingsWrapper
from pytest_django.fixtures import Settings
@pytest.fixture
def index_dir(tmp_path: Path, settings: SettingsWrapper) -> Path:
def index_dir(tmp_path: Path, settings: Settings) -> Path:
path = tmp_path / "index"
path.mkdir()
settings.INDEX_DIR = path
@@ -35,11 +31,3 @@ def backend() -> Generator[TantivyBackend, None, None]:
finally:
b.close()
reset_backend()
@pytest.fixture(scope="module")
def index() -> tantivy.Index:
"""A real Tantivy index for parse-acceptance tests (module scope for speed)."""
idx = tantivy.Index(build_schema(), path=tempfile.mkdtemp())
register_tokenizers(idx, "english")
return idx
@@ -0,0 +1,541 @@
"""Result-level acceptance corpus: real documents indexed via build_schema(),
real queries run through parse_user_query(), matched-document-ID sets
asserted, not intermediate ASTs or query strings. This is paperless-ngx's
analogue of whoosh-compat's own tests/emitter/test_acceptance_e2e.py.
Supersedes test_query.py's TestParseUserQuery result-level cases.
"""
from __future__ import annotations
from datetime import UTC
from datetime import datetime
from typing import TYPE_CHECKING
import pytest
import time_machine
from django.contrib.auth.models import User
from documents.models import CustomField
from documents.models import CustomFieldInstance
from documents.models import Document
from documents.models import DocumentType
from documents.models import Note
from documents.models import StoragePath
from documents.search._query import parse_user_query
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
FROZEN_NOW = datetime(2026, 6, 15, 12, 0, tzinfo=UTC)
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
"""Create a Document and index it in one step, for the common case
where nothing needs to happen between the two (no related Note/
CustomFieldInstance to attach first)."""
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
@pytest.fixture
def indexed_documents(backend: TantivyBackend) -> dict[str, int]:
"""Index a small fixture set, return {label: doc_id} for corpus queries."""
docs = {
"invoice_2020": _index(
backend,
title="Invoice 2020",
content="invoice total due",
checksum="acc-invoice-2020",
archive_serial_number=100,
),
"invoice_2021": _index(
backend,
title="Invoice 2021",
content="invoice total due",
checksum="acc-invoice-2021",
archive_serial_number=101,
),
"invoice_2023": _index(
backend,
title="Invoice 2023",
content="invoice total due",
checksum="acc-invoice-2023",
archive_serial_number=102,
),
"receipt_2022": _index(
backend,
title="Receipt 2022",
content="receipt total due",
checksum="acc-receipt-2022",
archive_serial_number=103,
),
}
return {label: doc.pk for label, doc in docs.items()}
class TestIssue13568BracketWildcard:
"""paperless-ngx#13568: title:202[0-3]* must keep its character class,
not fold to a prefix query that silently drops it."""
def test_bracket_class_wildcard_matches_only_in_range_years(
self,
backend: TantivyBackend,
indexed_documents: dict[str, int],
) -> None:
"""
GIVEN:
- Four indexed documents titled Invoice 2020/2021/2023 and
Receipt 2022
WHEN:
- "title:202[0-1]*" is searched ([0-1], not [0-3], is
deliberate: the fixture's trailing digits are 0/1/2/3, so a
[0-3] class would match all four and pass even if the
character class were silently dropped and folded to an
unconstrained "202*" prefix; [0-1] partitions the fixture
into a genuine in-range/out-of-range split)
THEN:
- Only the 2020 and 2021 documents match, proving the bracket
character class survived (issue #13568's original bug)
"""
matched = _matched_ids(backend, "title:202[0-1]*")
expected = {
indexed_documents["invoice_2020"],
indexed_documents["invoice_2021"],
}
assert matched == expected, (
"title:202[0-1]* must match 2020/2021 titles and exclude 2022/2023 "
"- if this matches everything, the wildcard's character class was "
"silently dropped (issue #13568's original bug)"
)
class TestFieldBoosts:
def test_title_boost_ranks_title_match_above_content_only_match(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- One document whose title contains the query word and another
whose content (not title) contains it
WHEN:
- The query word is searched unfielded
THEN:
- The title match ranks first, proving our title field boost
actually affects ranking
"""
title_match = _index(
backend,
title="urgent",
content="nothing else relevant",
checksum="acc-boost-title",
)
_index(
backend,
title="nothing",
content="urgent matter here",
checksum="acc-boost-content",
)
query = parse_user_query(backend._index, "urgent", UTC)
searcher = backend._index.searcher()
results = searcher.search(query, limit=10)
ranked_ids = [
searcher.doc(addr).to_dict()["id"][0] for _score, addr in results.hits
]
assert ranked_ids[0] == title_match.pk
class TestJsonSubpaths:
def test_notes_user_matches_document_with_that_note_author(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document with a Note authored by "alice" and a second,
unrelated document with no note
WHEN:
- "notes.user:alice" is searched
THEN:
- Only the document with alice's note matches
"""
alice = User.objects.create_user(username="alice")
doc_with_note = Document.objects.create(
title="Has note",
content="x",
checksum="acc-note-with",
)
Note.objects.create(document=doc_with_note, user=alice, note="reminder")
backend.add_or_update(doc_with_note)
_index(backend, title="No note", content="x", checksum="acc-note-without")
matched = _matched_ids(backend, "notes.user:alice")
assert matched == {doc_with_note.pk}
def test_custom_fields_name_and_value_combine(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document with a "Contract Number" custom field valued
"policy", and a second document with a differently-named
custom field also valued "policy"
WHEN:
- 'custom_fields.name:"Contract Number" custom_fields.value:policy'
is searched
THEN:
- Only the document whose field name AND value both match is
returned
"""
field = CustomField.objects.create(
name="Contract Number",
data_type=CustomField.FieldDataType.STRING,
)
other_field = CustomField.objects.create(
name="Other Field",
data_type=CustomField.FieldDataType.STRING,
)
matching = Document.objects.create(
title="Matching",
content="x",
checksum="acc-cf-matching",
)
CustomFieldInstance.objects.create(
document=matching,
field=field,
value_text="policy",
)
backend.add_or_update(matching)
non_matching = Document.objects.create(
title="Non-matching",
content="x",
checksum="acc-cf-nonmatching",
)
CustomFieldInstance.objects.create(
document=non_matching,
field=other_field,
value_text="policy",
)
backend.add_or_update(non_matching)
matched = _matched_ids(
backend,
'custom_fields.name:"Contract Number" custom_fields.value:policy',
)
assert matched == {matching.pk}
class TestUnregisteredIdFieldFoldsToLiteralText:
"""tag_id, owner_id, etc. are intentionally excluded from the
FieldRegistry - always internal index columns, never meant to be
query-addressable. Prove an unregistered field folds to a literal
text search that matches nothing, rather than erroring."""
def test_tag_id_query_matches_nothing(
self,
backend: TantivyBackend,
indexed_documents: dict[str, int],
) -> None:
"""
GIVEN:
- A real indexed corpus and "tag_id", a field intentionally
excluded from the FieldRegistry (an internal index column,
never meant to be query-addressable)
WHEN:
- "tag_id:5" is searched
THEN:
- It folds to a literal text search and matches nothing,
rather than erroring
"""
matched = _matched_ids(backend, "tag_id:5")
assert matched == set()
class TestFuzzyBlendSurvivesWhooshGrammar:
"""A query mixing whoosh-only grammar (a date keyword) with a typo'd
free-text word must still fuzzy-match the intended document when
ADVANCED_FUZZY_SEARCH_THRESHOLD is enabled. The fuzzy clause is built
from the parsed query's free-text tokens (whoosh_compat's
free_text_tokens), never from the raw query string, so whoosh grammar
that tantivy's own parser rejects cannot knock the fuzzy clause out."""
def test_typo_fuzzy_matches_alongside_date_keyword(
self,
backend: TantivyBackend,
settings,
) -> None:
"""
GIVEN:
- ADVANCED_FUZZY_SEARCH_THRESHOLD enabled, and a document
indexed with content "receipt total due"
WHEN:
- The query blends whoosh-only grammar tantivy's own parser
rejects ("added:today") with a one-transposition misspelling
of a word in the indexed content
THEN:
- The document still matches, because the fuzzy clause is
built from the parsed query's free-text tokens
(whoosh_compat's free_text_tokens), never from the raw
query string, so grammar tantivy's parser cannot handle
cannot knock the fuzzy clause out
"""
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.5
with time_machine.travel(FROZEN_NOW, tick=False):
doc = _index(
backend,
title="Receipt March",
content="receipt total due",
checksum="fuzzy-blend-1",
archive_serial_number=900,
)
# Sanity: the exact spelling matches through the exact clause.
assert doc.pk in _matched_ids(backend, "added:today receipt")
# The regression: the misspelling (one transposition) only
# matches via the fuzzy clause, and "added:today" is
# whoosh-only grammar tantivy's parser rejects, so raw-string
# fuzzy parsing skips the clause entirely and this returns
# nothing. The typo is deliberate; keep codespell away from it.
typo_query = "added:today reciept" # codespell:ignore reciept
assert doc.pk in _matched_ids(backend, typo_query)
def test_negated_words_do_not_fuzzy_match(
self,
backend: TantivyBackend,
settings,
) -> None:
"""
GIVEN:
- ADVANCED_FUZZY_SEARCH_THRESHOLD enabled, and a document
containing the NOT'd word ("receipt") but not the positive
word ("total"), so nothing matches the exact clause -- the
shape a naive fuzzy string built from ALL words (including
the NOT'd one) would make this document the sole hit,
normalize its score to 1.0, and survive any threshold (a
shape with an exact-matching sibling document would NOT
discriminate: normalization would rank the resurfaced
document far below the exact match and the threshold would
cut it even for a naive implementation)
WHEN:
- "added:today total NOT receipt" is searched
THEN:
- The document does not match; a term the user excluded must
not resurface through the fuzzy clause
"""
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.5
with time_machine.travel(FROZEN_NOW, tick=False):
_index(
backend,
title="Receipt Archive",
content="receipt archived stack",
checksum="fuzzy-blend-2",
archive_serial_number=901,
)
assert _matched_ids(backend, "added:today total NOT receipt") == set()
class TestUnquotedDateKeywordPhrases:
"""The unquoted spelling (added:previous month) is honored natively by
whoosh-compat's own grammar for this closed phrase vocabulary, no
app-level rewrite is involved. Pins that the historically supported
spelling keeps working now that paperless no longer pre-quotes it."""
@pytest.fixture
def period_documents(self, backend: TantivyBackend) -> dict[str, int]:
with time_machine.travel(FROZEN_NOW, tick=False):
in_may = _index(
backend,
title="May Doc",
content="statement",
checksum="kw-may",
archive_serial_number=910,
added=datetime(2026, 5, 20, 12, 0, tzinfo=UTC),
)
in_june = _index(
backend,
title="June Doc",
content="statement",
checksum="kw-june",
archive_serial_number=911,
added=datetime(2026, 6, 10, 12, 0, tzinfo=UTC),
)
return {"in_may": in_may.pk, "in_june": in_june.pk}
@pytest.mark.parametrize(
"query",
[
pytest.param("added:previous month", id="unquoted"),
pytest.param('added:"previous month"', id="quoted"),
pytest.param("added:Previous Month", id="unquoted-mixed-case"),
],
)
def test_unquoted_matches_the_same_documents_as_quoted(
self,
backend: TantivyBackend,
period_documents: dict[str, int],
query: str,
) -> None:
"""
GIVEN:
- Two documents added in different months, time frozen so
only one falls in "previous month"
WHEN:
- The same date-keyword phrase is spelled unquoted, quoted,
and unquoted with mixed case
THEN:
- All three spellings match the same document; paperless no
longer pre-quotes this phrase before parsing, relying on
whoosh-compat's own grammar to accept it unquoted natively
"""
with time_machine.travel(FROZEN_NOW, tick=False):
assert _matched_ids(backend, query) == {period_documents["in_may"]}
@pytest.mark.parametrize(
"query",
[
pytest.param("added:this month", id="this-month"),
pytest.param("added:this year", id="this-year"),
pytest.param("added:previous week", id="previous-week"),
pytest.param("added:previous quarter", id="previous-quarter"),
pytest.param("added:previous year", id="previous-year"),
pytest.param("created:previous month", id="created-field"),
pytest.param("modified:previous month", id="modified-field"),
],
)
def test_every_phrase_and_date_field_parses_without_error(
self,
backend: TantivyBackend,
period_documents: dict[str, int],
query: str,
) -> None:
"""
GIVEN:
- Our real schema and every date-keyword phrase in the
vocabulary, against every date field we expose (added,
created, modified)
WHEN:
- Each combination is searched
THEN:
- It parses and searches cleanly against our schema (no
SearchQueryError, so no HTTP 400); exact window semantics
are whoosh-compat's own and are pinned in its own suite
"""
with time_machine.travel(FROZEN_NOW, tick=False):
_matched_ids(backend, query)
def test_text_field_keyword_words_are_ordinary_text(
self,
backend: TantivyBackend,
period_documents: dict[str, int],
) -> None:
"""
GIVEN:
- period_documents (indexed by added-date) and a third
document whose title literally contains the words
"previous month"
WHEN:
- "title:previous month" is searched
THEN:
- Only the document whose title contains those words matches;
"previous month" after a TEXT field (or unfielded) is
ordinary text, not a date phrase, so the date-window
documents do not match
"""
with time_machine.travel(FROZEN_NOW, tick=False):
wordy = _index(
backend,
title="Notes from the previous month",
content="meeting notes",
checksum="kw-text",
archive_serial_number=912,
)
assert _matched_ids(backend, "title:previous month") == {wordy.pk}
class TestFieldAliases:
"""type:/path: are registry aliases for document_type:/storage_path:.
The only other alias coverage is parse-shape; these prove resolution
end-to-end against a real index."""
def test_type_alias_and_canonical_name_match_the_same_document(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document with document_type "invoice", and a decoy
document with no type whose content merely mentions
"invoice" (document_type is itself a default search field,
so if alias resolution ever broke and "type:invoice"
demoted to unfielded text, the token would STILL match the
typed document through the field value; the decoy carrying
the query word in content is what makes a demoted search
distinguishable, since it would then match both documents
and fail the exact-set assertion -- the title avoids
stemming to "type": English stems Typed -> type)
WHEN:
- "type:invoice" and "document_type:invoice" are each
searched
THEN:
- Both resolve to the same document, proving the "type" alias
and its canonical field name agree end-to-end against a
real index
"""
invoice_type = DocumentType.objects.create(name="invoice")
typed = _index(
backend,
title="First",
content="quarterly statement",
checksum="alias-type-1",
document_type=invoice_type,
)
_index(
backend,
title="Second",
content="invoice mentioned in body",
checksum="alias-type-2",
)
assert _matched_ids(backend, "type:invoice") == {typed.pk}
assert _matched_ids(backend, "document_type:invoice") == {typed.pk}
def test_path_alias_and_canonical_name_match_the_same_document(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document stored under storage_path "archive", and a decoy
document with no storage_path whose content merely mentions
"archive" (storage_path is NOT a default search field
today, so a demoted "path:archive" already matches nothing;
the content decoy keeps this test discriminating even if
storage_path ever joins the defaults)
WHEN:
- "path:archive" and "storage_path:archive" are each searched
THEN:
- Both resolve to the same document, proving the "path" alias
and its canonical field name agree end-to-end against a
real index
"""
archive = StoragePath.objects.create(name="archive", path="archive/{title}")
stored = _index(
backend,
title="Stored",
content="quarterly statement",
checksum="alias-path-1",
storage_path=archive,
)
_index(
backend,
title="Loose",
content="archive mentioned in body",
checksum="alias-path-2",
)
assert _matched_ids(backend, "path:archive") == {stored.pk}
assert _matched_ids(backend, "storage_path:archive") == {stored.pk}
+187
View File
@@ -4,6 +4,8 @@ from pathlib import Path
import pytest
from django.contrib.auth.models import Group
from django.contrib.auth.models import User
from django.db import connection
from django.test.utils import CaptureQueriesContext
from guardian.shortcuts import assign_perm
from pytest_mock import MockerFixture
@@ -102,6 +104,191 @@ class TestWriteBatch:
assert len(backend.search_ids("indexable", user=None)) == 1
class TestAddOrUpdateIds:
"""Test WriteBatch.add_or_update_ids(), the bulk id-based upsert path.
Unlike add_or_update() called once per document, this resolves viewer
permissions and effective (versioned) content in bulk against the ids as
a whole, so it must produce identical indexed output to the per-document
path while issuing a constant number of queries regardless of batch size.
"""
def test_missing_id_is_skipped_not_errored(
self,
backend: TantivyBackend,
) -> None:
doc = Document.objects.create(
title="doc",
content="present",
checksum="EXIST1",
pk=1,
)
missing_pk = 999
with backend.batch_update() as batch:
batch.add_or_update_ids([doc.pk, missing_pk])
assert backend.search_ids("present", user=None) == [doc.pk]
def test_query_count_does_not_scale_with_batch_size(
self,
backend: TantivyBackend,
) -> None:
"""Each query count must stay far below N, not merely match between
two runs -- an exact-equality assertion between two measurements is
at the mercy of incidental process-level caches (e.g. Django's
ContentType.objects.get_for_model) warming on whichever run happens
first, which makes counts differ by a query for reasons unrelated to
batch size. A generous fixed bound sidesteps that: the old
per-document path issued roughly 8 queries per document, so 50
documents under a bound this low proves the fix regardless of cache
state.
"""
max_queries_for_any_batch_size = 15
small_docs = [
Document.objects.create(
title="doc",
content=f"unique{i}",
checksum=f"SMALL{i}",
pk=i,
)
for i in range(1, 3)
]
with CaptureQueriesContext(connection) as ctx_small:
with backend.batch_update() as batch:
batch.add_or_update_ids([d.pk for d in small_docs])
assert len(ctx_small.captured_queries) <= max_queries_for_any_batch_size
large_docs = [
Document.objects.create(
title="doc",
content=f"unique{i}",
checksum=f"LARGE{i}",
pk=i,
)
for i in range(100, 150)
]
with CaptureQueriesContext(connection) as ctx_large:
with backend.batch_update() as batch:
batch.add_or_update_ids([d.pk for d in large_docs])
assert len(ctx_large.captured_queries) <= max_queries_for_any_batch_size
for doc in large_docs:
assert backend.search_ids(f"unique{doc.pk}", user=None) == [doc.pk]
def test_resolves_direct_user_grant_in_bulk(
self,
backend: TantivyBackend,
) -> None:
owner = UserFactory()
user = UserFactory()
doc = Document.objects.create(
title="doc",
checksum="PERM1",
pk=1,
owner=owner,
)
assign_perm("view_document", user, doc)
with backend.batch_update() as batch:
batch.add_or_update_ids([doc.pk])
assert backend.search_ids("doc", user=user) == [doc.pk]
other = UserFactory()
assert backend.search_ids("doc", user=other) == []
def test_resolves_group_grant_in_bulk(self, backend: TantivyBackend) -> None:
owner = UserFactory()
group = Group.objects.create(name="reviewers")
user = UserFactory()
user.groups.add(group)
doc = Document.objects.create(
title="doc",
checksum="GPERM1",
pk=1,
owner=owner,
)
assign_perm("view_document", group, doc)
with backend.batch_update() as batch:
batch.add_or_update_ids([doc.pk])
assert backend.search_ids("doc", user=user) == [doc.pk]
other = UserFactory()
assert backend.search_ids("doc", user=other) == []
def test_indexes_notes_and_custom_fields(self, backend: TantivyBackend) -> None:
note_author = UserFactory(username="noter")
field = CustomField.objects.create(
name="Invoice Number",
data_type=CustomField.FieldDataType.STRING,
)
doc = Document.objects.create(title="doc", checksum="RICH1", pk=1)
Note.objects.create(document=doc, note="Reviewed", user=note_author)
CustomFieldInstance.objects.create(
document=doc,
field=field,
value_text="INV-42",
)
with backend.batch_update() as batch:
batch.add_or_update_ids([doc.pk])
assert backend.search_ids("notes.user:noter", user=None) == [doc.pk]
assert backend.search_ids("custom_fields.value:INV-42", user=None) == [
doc.pk,
]
def test_uses_effective_content_for_versioned_documents(
self,
backend: TantivyBackend,
) -> None:
root = Document.objects.create(
title="Statement",
content="stale text",
checksum="ROOT1",
pk=1,
)
Document.objects.create(
title="Statement",
content="latest version text",
checksum="VER1",
pk=2,
root_document=root,
version_index=1,
)
with backend.batch_update() as batch:
batch.add_or_update_ids([root.pk])
assert backend.search_ids("latest", user=None) == [root.pk]
assert backend.search_ids("stale", user=None) == []
def test_reindexes_documents_already_in_the_index(
self,
backend: TantivyBackend,
) -> None:
"""add_or_update_ids must upsert, matching add_or_update's behaviour."""
doc = Document.objects.create(
title="doc",
content="original",
checksum="UP1",
pk=1,
)
backend.add_or_update(doc)
assert backend.search_ids("original", user=None) == [doc.pk]
doc.content = "updated"
doc.save()
with backend.batch_update() as batch:
batch.add_or_update_ids([doc.pk])
assert backend.search_ids("original", user=None) == []
assert backend.search_ids("updated", user=None) == [doc.pk]
class TestSearch:
"""Test search query parsing and matching via search_ids."""
@@ -0,0 +1,82 @@
"""``checksum`` wildcard patterns stay literal end to end, once user queries
route through whoosh-compat.
The registry-level fact (the pattern normalizer folds a KEYWORD pattern
rather than stemming it) is pinned on its own in
``test_keyword_pattern_literal.py``. This proves it actually reaches a real
query: ``checksum:ceded*`` must match only the document whose checksum
starts with "ceded", not the one whose checksum stems to the same run.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.models import Document
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
CEDEF00D = "cedef00ddeadbeef0123456789abcdef01234567"
CEDEDEAD = "cededeadbeef567801234567" + "89abcdef01234567"
class TestChecksumPrefixQueries:
@pytest.fixture
def indexed(self, backend: TantivyBackend) -> None:
for i, checksum in enumerate((CEDEF00D, CEDEDEAD)):
doc = Document.objects.create(
title=f"Checksum doc {i}",
content="invoices for the quarter",
checksum=checksum,
archive_serial_number=940 + i,
)
backend.add_or_update(doc)
def _ids(self, backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def test_prefix_matches_only_the_document_that_starts_with_it(
self,
backend: TantivyBackend,
indexed: None,
) -> None:
"""
GIVEN:
- Two documents indexed with checksums that share a stem when
run through the English stemmer ("cedef00d..." and
"cededead...") but only one literally starts with "ceded"
WHEN:
- "checksum:ceded*" is searched
THEN:
- Only the document whose checksum literally starts with
"ceded" matches; the pattern normalizer folds a KEYWORD
pattern rather than stemming it, so this reaches a real
query end to end
"""
matched = self._ids(backend, "checksum:ceded*")
expected = Document.objects.get(checksum=CEDEDEAD).pk
assert matched == {expected}
def test_text_prefix_still_reaches_the_stemmed_index(
self,
backend: TantivyBackend,
indexed: None,
) -> None:
"""
GIVEN:
- Two documents indexed with content "invoices for the
quarter"
WHEN:
- "invoice*" is searched against the TEXT content field
THEN:
- Both documents match, confirming the checksum field's
literal-pattern behavior is specific to KEYWORD fields and
does not affect TEXT field wildcard matching against
stemmed terms
"""
assert len(self._ids(backend, "invoice*")) == 2
@@ -0,0 +1,212 @@
"""The CJK bigram clause blended into QUERY-mode searches.
The clause exists so CJK runs are matchable at all (the default analyzers
keep a whitespace-free CJK run as one indivisible token), but it must not
widen the query beyond what the user asked for: a CJK term the query
excludes, or restricts to one field, must not come back through it.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.models import Document
if TYPE_CHECKING:
from pytest_django.fixtures import SettingsWrapper
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
class TestCjkParseFailureDegradesGracefully:
def test_a_cjk_run_tantivy_cannot_parse_drops_the_clause_only(self) -> None:
"""
GIVEN:
- A CJK run and an index-like object whose parse_query is
forced to raise
WHEN:
- _parse_cjk_text is called
THEN:
- It returns None instead of propagating, so a CJK run tantivy
cannot parse only drops the bigram clause rather than
failing the whole query. Broad on purpose (bare except
Exception), unlike the fuzzy blend's narrower ValueError
guard: a CJK run is not filtered to a guaranteed-safe token
set the way the fuzzy blend's word string is, so the exact
failure mode tantivy could raise here is not pinned down
"""
from documents.search._query import _parse_cjk_text
class _RaisingIndex:
def parse_query(self, *args: object, **kwargs: object) -> object:
raise RuntimeError("synthetic parse failure")
assert _parse_cjk_text(_RaisingIndex(), "東京", ["bigram_content"]) is None
def test_no_cjk_text_at_all_returns_none_without_parsing(self) -> None:
"""
GIVEN:
- A raw query string with no CJK characters at all
WHEN:
- _build_cjk_query (the simple TEXT/TITLE-mode builder) is
called directly
THEN:
- It returns None without ever attempting to parse anything.
The only real caller already guards this with _has_cjk(),
so this is defensive: it keeps the function safe to call on
its own, not a path a real search currently reaches
"""
from documents.search._query import _build_cjk_query
assert _build_cjk_query(None, "invoice total due", ["bigram_content"]) is None
class TestCjkClauseFollowsTheParsedQuery:
def test_negated_cjk_term_is_excluded(self, backend: TantivyBackend) -> None:
"""
GIVEN:
- Two documents both matching "invoice", one whose content
also contains 漢字
WHEN:
- "invoice NOT 漢字" is searched
THEN:
- Only the document without 漢字 matches; 'invoice NOT 漢字'
must not return the document containing 漢字
"""
with_cjk = _index(
backend,
title="Invoice A",
content="invoice total 漢字",
checksum="cjk-neg-1",
)
without_cjk = _index(
backend,
title="Invoice B",
content="invoice total only",
checksum="cjk-neg-2",
)
assert _matched_ids(backend, "invoice") == {with_cjk.pk, without_cjk.pk}
assert _matched_ids(backend, "invoice NOT 漢字") == {without_cjk.pk}
@pytest.mark.parametrize(
("threshold", "expected"),
[
pytest.param(None, {"titled"}, id="fuzzy_off"),
pytest.param(0.0, {"titled", "content_only"}, id="fuzzy_on"),
],
)
def test_fielded_cjk_term_searches_only_that_field(
self,
backend: TantivyBackend,
settings: SettingsWrapper,
threshold: float | None,
expected: set[str],
) -> None:
"""
GIVEN:
- One document with 東京 in its title, another with 東京 only
in its content, and ADVANCED_FUZZY_SEARCH_THRESHOLD either
off or on
WHEN:
- "title:東京" is searched
THEN:
- With fuzzy off, only the titled document matches: the CJK
clause honours the field, so 'title:東京' must not match a
document whose 東京 is only in the content. With fuzzy on,
the content-only document is also readmitted, because the
fuzzy clause contributes every free-text term UNFIELDED by
design (see _try_parse_fuzzy_query) on its own
0.1-boosted terms -- a documented trade-off, pinned here so
it stays deliberate
"""
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = threshold
content_only = _index(
backend,
title="Tokyo report",
content="東京都の人口は約1400万人です",
checksum="cjk-field-1",
)
titled = _index(
backend,
title="東京都の報告書",
content="an english summary",
checksum="cjk-field-2",
)
pks = {"titled": titled.pk, "content_only": content_only.pk}
assert _matched_ids(backend, "東京") == set(pks.values())
assert _matched_ids(backend, "title:東京") == {pks[label] for label in expected}
def test_cjk_on_a_non_default_field_builds_no_clause(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document with 東京 in its content
WHEN:
- "notes:東京" is searched (a field outside the default
search fields)
THEN:
- Nothing matches; a CJK term restricted to a field outside
the default search fields has nothing to contribute to the
bigram clause, so it must not fall back to matching 東京 in
the content
"""
_index(
backend,
title="Tokyo report",
content="東京都の人口は約1400万人です",
checksum="cjk-notes-1",
)
assert _matched_ids(backend, "notes:東京") == set()
def test_bare_cjk_term_still_matches_every_default_field(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- One document with 重要 in its content, another with 重要 in
its title
WHEN:
- "重要" and "重要 OR report" are each searched unfielded
THEN:
- Both documents match either way; the clause's reason for
existing is that an unfielded CJK run matches wherever it
is indexed, and does so alongside a latin term
"""
in_content = _index(
backend,
title="report",
content="本文に重要な情報",
checksum="cjk-bare-1",
)
in_title = _index(
backend,
title="重要な報告書",
content="english only",
checksum="cjk-bare-2",
)
assert _matched_ids(backend, "重要") == {in_content.pk, in_title.pk}
assert _matched_ids(backend, "重要 OR report") == {
in_content.pk,
in_title.pk,
}
@@ -0,0 +1,86 @@
"""Whoosh's compact, separator-free date spelling, resolved end to end.
whoosh-compat owns both widths of this spelling and asserts both forms'
bounds directly in its own test suite: the 8-digit form as a whole calendar
day (lower bound, upper bound and exclusivity), and the 14-digit form as a
single instant. The 14-digit form is kept here as the single representative
because it is the one that exercises paperless's ``added`` DATETIME fast
field at full precision: the corpus separates a document at the named
instant from one on the same calendar day at another hour and one on the
next day at the same hour, so a query that degrades into a whole-day
window, or drops the time of day, matches the wrong set rather than passing
on a corpus that could not tell the difference.
"""
from __future__ import annotations
from datetime import UTC
from datetime import datetime
from typing import TYPE_CHECKING
import pytest
from documents.models import Document
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
@pytest.fixture
def docs(backend: TantivyBackend) -> dict[str, int]:
return {
"instant": _index(
backend,
title="On the instant",
content="x",
checksum="compact-date-instant",
added=datetime(2005, 3, 4, 15, 30, tzinfo=UTC),
).pk,
"same_day": _index(
backend,
title="Same day, other hour",
content="x",
checksum="compact-date-same-day",
added=datetime(2005, 3, 4, 9, 0, tzinfo=UTC),
).pk,
"next_day": _index(
backend,
title="Next day, same hour",
content="x",
checksum="compact-date-next-day",
added=datetime(2005, 3, 5, 15, 30, tzinfo=UTC),
).pk,
}
def test_fourteen_digits_is_a_single_instant(
backend: TantivyBackend,
docs: dict[str, int],
) -> None:
"""
GIVEN:
- Three documents indexed on the ``added`` DATETIME fast field:
one at 2005-03-04T15:30:00, one on the same calendar day at a
different hour, and one on the next day at the same hour
WHEN:
- Searching with the 14-digit compact date form
``added:20050304153000``
THEN:
- Only the document at that exact instant matches; the same-day
document is what tells this apart from the 8-digit day-window
form, and the next-day document from a form that ignored the
time of day altogether
"""
assert _matched_ids(backend, "added:20050304153000") == {docs["instant"]}
@@ -0,0 +1,149 @@
"""_ConjunctiveNegations, the AST visitor that collects the subtrees a
query excludes from every document it matches, and _any_of, the clause-list
collapsing helper it feeds into.
Result-level proof that a negation reached through NOT/AND survives the
fuzzy/CJK blend lives in test_query_negation.py. These are direct unit
tests of the visitor's dispatch for the rarer grammar shapes
(AndNot/Boosted/AndMaybe/Require) that file's real-corpus queries don't
happen to exercise, plus the empty-clause-list case of _any_of.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
import whoosh_compat.ast as wc_ast
from documents.models import Document
from documents.search._query import _any_of
from documents.search._query import _ConjunctiveNegations
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
def _term(text: str) -> wc_ast.Term:
return wc_ast.Term(field=None, text=text)
class TestConjunctiveNegationsVisitor:
def test_visit_andnot_hoists_the_negative_branch(self) -> None:
"""
GIVEN:
- An AndNot(positive=a, negative=b) node
WHEN:
- _ConjunctiveNegations visits it
THEN:
- The negative branch is collected as an exclusion, since
AndNot requires positive and excludes negative
"""
negative = _term("b")
node = wc_ast.AndNot(positive=_term("a"), negative=negative)
assert _ConjunctiveNegations().visit(node) == (negative,)
def test_visit_andnot_also_collects_negations_already_in_the_positive_branch(
self,
) -> None:
"""
GIVEN:
- An AndNot node whose positive branch already contains a NOT
WHEN:
- _ConjunctiveNegations visits it
THEN:
- Both the positive branch's own negation and the AndNot's
negative branch are collected
"""
excluded_in_positive = _term("excluded")
negative = _term("negative")
node = wc_ast.AndNot(
positive=wc_ast.Not(child=excluded_in_positive),
negative=negative,
)
assert _ConjunctiveNegations().visit(node) == (excluded_in_positive, negative)
def test_visit_boosted_passes_through_to_the_child(self) -> None:
"""
GIVEN:
- A Boosted node (e.g. "(invoice NOT secret)^2") wrapping a
NOT
WHEN:
- _ConjunctiveNegations visits it
THEN:
- The negation inside the boosted child is still collected: a
boost must not shield an exclusion from being hoisted
"""
excluded = _term("secret")
node = wc_ast.Boosted(child=wc_ast.Not(child=excluded), boost=2.0)
assert _ConjunctiveNegations().visit(node) == (excluded,)
def test_visit_andmaybe_only_descends_into_required(self) -> None:
"""
GIVEN:
- An AndMaybe(required=a, optional=b) node where both required
and optional contain their own NOT
WHEN:
- _ConjunctiveNegations visits it
THEN:
- Only the negation in the required branch is collected. The
optional branch is not a conjunctive constraint on the whole
query (documents that fail it still match), so hoisting a
negation from it would exclude documents the query does not
actually exclude
"""
excluded_in_required = _term("excluded_in_required")
excluded_in_optional = _term("excluded_in_optional")
node = wc_ast.AndMaybe(
required=wc_ast.Not(child=excluded_in_required),
optional=wc_ast.Not(child=excluded_in_optional),
)
assert _ConjunctiveNegations().visit(node) == (excluded_in_required,)
def test_visit_require_descends_into_both_branches(self) -> None:
"""
GIVEN:
- A Require(scored=a, filter_only=b) node where both scored
and filter_only contain their own NOT
WHEN:
- _ConjunctiveNegations visits it
THEN:
- Both negations are collected: Require constrains the whole
query with both branches, one merely scored and the other
filter-only, so both are conjunctive
"""
excluded_in_scored = _term("excluded_in_scored")
excluded_in_filter = _term("excluded_in_filter")
node = wc_ast.Require(
scored=wc_ast.Not(child=excluded_in_scored),
filter_only=wc_ast.Not(child=excluded_in_filter),
)
assert _ConjunctiveNegations().visit(node) == (
excluded_in_scored,
excluded_in_filter,
)
class TestAnyOfEmptyClauseList:
def test_no_clauses_returns_a_query_that_matches_nothing(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- No clauses at all
WHEN:
- _any_of is called with an empty list
THEN:
- It returns tantivy's empty_query() rather than raising or
wrapping zero clauses in a boolean_query, and running it
against a real index matches no documents
"""
doc = Document.objects.create(title="x", content="x", checksum="any-of-empty")
backend.add_or_update(doc)
query = _any_of([])
results = backend._index.searcher().search(query, limit=10)
assert len(results.hits) == 0
@@ -0,0 +1,91 @@
"""Pins the correctness gained by deleting the pre-parse
_quote_date_keyword_phrases rewrite.
That rewrite matched date-keyword phrases (e.g. "previous month" after a
date field) anywhere in the raw query string, including inside an
unrelated quoted string, and inserted quotes mid-phrase there too. Its
own docstring gave ``title:"see added:previous month notes"`` as the
example of what it corrupted. whoosh-compat's grammar accepts the same
phrase vocabulary unquoted natively (see TestUnquotedDateKeywordPhrases
in test_acceptance.py), so the rewrite was redundant everywhere it was
safe and actively wrong everywhere it was not. This is the one case that
tells the two apart: a literal title phrase that happens to contain
"added:previous month" as running text.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.models import Document
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
class TestQuotedStringContainingDateKeywordText:
"""A quoted title phrase containing the literal text
"added:previous month" as running words must match on that literal
text alone, never spill into an unfielded search for "previous" and
"month" across the default search fields the way the deleted rewrite
would have decomposed it into."""
def test_matches_only_the_literal_phrase(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document whose title literally contains "see
added:previous month notes", and a decoy document whose
title/content carry the individual fragments the deleted
_quote_date_keyword_phrases rewrite would have decomposed
the phrase into (the decoy would incorrectly match under
the deleted rewrite: its title contains the "see added:"
and " notes" fragments the corrupted parse required as
title phrases, and its content supplies "previous" and
"month" as the decomposed word-match clauses the rewrite
turned the middle of the phrase into)
WHEN:
- 'title:"see added:previous month notes"' is searched
THEN:
- Only the document with the literal phrase matches; it must
never spill into an unfielded search for "previous" and
"month" across the default search fields
"""
literal = _index(
backend,
title="see added:previous month notes",
content="quarterly filing",
checksum="dkp-literal",
archive_serial_number=920,
)
# Under the deleted rewrite, this decoy would incorrectly match:
# its title contains the "see added:" and " notes" fragments the
# corrupted parse required as title phrases, and its content
# supplies "previous" and "month" as the decomposed word-match
# clauses the rewrite turned the middle of the phrase into.
decoy = _index(
backend,
title="see added: quarterly report notes",
content="we reviewed the previous statement about month end",
checksum="dkp-decoy",
archive_serial_number=921,
)
query = 'title:"see added:previous month notes"'
assert _matched_ids(backend, query) == {literal.pk}
assert decoy.pk not in _matched_ids(backend, query)
@@ -0,0 +1,100 @@
"""Date keyword phrases (``today``, etc.) resolved in a non-UTC timezone,
end to end.
paperless's own ``tz=get_current_timezone()`` plumbing
(``TantivyBackend._parse_query``) is exercised elsewhere only for
relative *ranges* (``added:[-1 week to now]``, in
documents/tests/test_api_search.py). This covers a date *keyword*
(``today``), whose day boundary depends on the active timezone the same
way but goes through whoosh-compat's DateParserPlugin resolution instead
of an explicit range.
Discriminating shape: frozen at 2026-06-15T02:00 UTC, which is
2026-06-14T22:00 in America/New_York -- still "today" (06-14) there, but
already "today" (06-15) in UTC. Two documents pin both directions of the
mistake a hardcoded-UTC bug would make:
- ``in_ny_today`` (added 2026-06-14T20:00 UTC = 2026-06-14T16:00 NY) is
inside New York's "today" window and outside a naive UTC-calendar-day
window. A ``tz``-ignoring bug would miss it.
- ``in_utc_calendar_day_only`` (added 2026-06-15T10:00 UTC =
2026-06-15T06:00 NY) is inside a naive UTC-calendar-day window but
outside New York's actual "today" window. A ``tz``-ignoring bug would
wrongly match it.
"""
from __future__ import annotations
from datetime import UTC
from datetime import datetime
from typing import TYPE_CHECKING
import pytest
import time_machine
from documents.models import Document
if TYPE_CHECKING:
from pytest_django.fixtures import SettingsWrapper
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
FROZEN_NOW = datetime(2026, 6, 15, 2, 0, tzinfo=UTC)
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
class TestDateKeywordUsesTheActiveTimezone:
def test_today_matches_the_new_york_calendar_day_not_the_utc_one(
self,
backend: TantivyBackend,
settings: SettingsWrapper,
) -> None:
"""
GIVEN:
- TIME_ZONE set to America/New_York, time frozen at
2026-06-15T02:00 UTC (2026-06-14T22:00 NY -- still "today"
there, but already "today" in UTC), and two documents: one
added inside New York's "today" window but outside a naive
UTC-calendar-day window, the other the reverse (inside a
naive UTC-calendar-day window but outside New York's actual
"today")
WHEN:
- "added:today" is searched
THEN:
- Only the document inside New York's actual "today" window
matches, proving our tz=get_current_timezone() plumbing
resolves the date keyword in the active timezone rather
than a hardcoded UTC calendar day
"""
settings.TIME_ZONE = "America/New_York"
with time_machine.travel(FROZEN_NOW, tick=False):
in_ny_today = _index(
backend,
title="NY today",
content="x",
checksum="tz-keyword-ny-today",
added=datetime(2026, 6, 14, 20, 0, tzinfo=UTC),
)
# Not captured: the exact-set assertion below already proves
# this document (inside a naive UTC-calendar-day window, but
# outside New York's actual "today") does not match.
_index(
backend,
title="UTC calendar day only",
content="x",
checksum="tz-keyword-utc-calendar-day-only",
added=datetime(2026, 6, 15, 10, 0, tzinfo=UTC),
)
assert _matched_ids(backend, "added:today") == {in_ny_today.pk}
@@ -0,0 +1,32 @@
"""``_DEFAULT_SEARCH_FIELDS`` must stay a subset of the registered public
field names.
Nothing enforced this before: a rename in PUBLIC_FIELDS not mirrored in
``_DEFAULT_SEARCH_FIELDS`` (documents/search/_query.py) would 400 every
unfielded search at request time, since ``index.parse_query`` and the
fuzzy/CJK clause builders are handed a field name the schema no longer
has.
"""
from __future__ import annotations
from documents.search._fields import PUBLIC_FIELDS
from documents.search._query import _DEFAULT_SEARCH_FIELDS
class TestDefaultSearchFieldsAreRegistered:
def test_every_default_search_field_is_a_public_field(self) -> None:
"""
GIVEN:
- PUBLIC_FIELDS and _DEFAULT_SEARCH_FIELDS, our own field
tables
WHEN:
- Every name in _DEFAULT_SEARCH_FIELDS is checked against the
registered public field names
THEN:
- Every one is present; a rename in PUBLIC_FIELDS not
mirrored here would 400 every unfielded search at request
time
"""
public_field_names = {f.name for f in PUBLIC_FIELDS}
assert set(_DEFAULT_SEARCH_FIELDS) <= public_field_names
@@ -0,0 +1,475 @@
"""Pins the search syntax that ``docs/usage.md`` promises users.
Every query here is syntax the "Document searches" section of
``docs/usage.md`` documents, either spelled as the docs spell it or as a
concrete instance of a form the docs describe. Each case indexes real
documents and asserts on matched document IDs rather than on the parsed
query, because a query that parses cleanly is not necessarily a query that
means what the documentation says it means: ``added:now`` parses without a
single diagnostic and then matches nothing, because it resolves to an
instant rather than to a span.
The negative cases matter as much as the positive ones. They pin the
behaviours the docs explicitly warn about, so that if any of them ever
starts working the warning can be removed deliberately rather than being
left standing as a lie.
"""
from __future__ import annotations
from datetime import UTC
from datetime import datetime
from typing import TYPE_CHECKING
import pytest
import time_machine
from documents.models import Document
from documents.models import Note
from documents.models import Tag
from documents.search._errors import InvalidDateQuery
if TYPE_CHECKING:
from collections.abc import Generator
from django.contrib.auth.models import User
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
# A Monday, so that "next monday"/"last monday" land a clean week either side.
FROZEN_NOW = datetime(2026, 6, 15, 12, 0, tzinfo=UTC)
# The checksum used in the docs' `checksum:` example.
DOC_CHECKSUM = "9f86d081884c7d659a2feaa0c55ad015a3bf4f1b2b0b822cd15d6c15b0f00a08"
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
class TestLogicalExpressions:
@pytest.fixture
def docs(self, backend: TantivyBackend) -> dict[str, int]:
return {
"secret": _index(
backend,
title="Invoice one",
content="invoice secret contents",
checksum="doc-syntax-secret",
).pk,
"plain": _index(
backend,
title="Invoice two",
content="invoice ordinary contents",
checksum="doc-syntax-plain",
).pk,
}
def test_not_excludes_a_term(
self,
backend: TantivyBackend,
docs: dict[str, int],
) -> None:
"""
GIVEN:
- Two indexed documents, one containing "secret" and one not
WHEN:
- "invoice NOT secret" is searched, as docs/usage.md documents
THEN:
- Only the document without "secret" matches
"""
assert _matched_ids(backend, "invoice NOT secret") == {docs["plain"]}
def test_leading_hyphen_requires_the_term_instead_of_excluding_it(
self,
backend: TantivyBackend,
docs: dict[str, int],
) -> None:
"""
GIVEN:
- Two indexed documents, one containing "secret" and one not
WHEN:
- "invoice -secret" is searched (a leading hyphen, not "NOT")
THEN:
- Only the document containing "secret" matches, because
separators are stripped at index time, so "-secret" is
indexed as the plain term "secret" and the query becomes an
AND rather than an exclusion, exactly as the docs warn
"""
assert _matched_ids(backend, "invoice -secret") == {docs["secret"]}
def test_or_inside_parentheses_matches_either_branch(
self,
backend: TantivyBackend,
docs: dict[str, int],
) -> None:
"""
GIVEN:
- Two indexed documents, one containing "secret" and one
containing "ordinary"
WHEN:
- "invoice AND (secret OR ordinary)" is searched
THEN:
- Both documents match
"""
matched = _matched_ids(backend, "invoice AND (secret OR ordinary)")
assert matched == {docs["secret"], docs["plain"]}
class TestPhraseSearch:
def test_quoted_phrase_requires_the_words_in_order(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document whose content contains "the quick brown fox jumps"
WHEN:
- A quoted phrase is searched, in order and out of order
THEN:
- The in-order phrase matches, and the same words reordered do
not
"""
doc = _index(
backend,
title="Phrase",
content="the quick brown fox jumps",
checksum="doc-syntax-phrase",
)
assert _matched_ids(backend, '"quick brown fox"') == {doc.pk}
assert _matched_ids(backend, '"brown quick fox"') == set()
class TestTagCommaList:
"""``tag:bills,unpaid`` is published syntax (docs/usage.md), so this checks
that the documented spelling still returns what the docs promise: only the
document carrying every listed tag.
It is deliberately not proof of paperless's field configuration, and must
not be read as such. Removing ``comma_values`` from the ``tag`` FieldSpec
leaves this test passing, because paperless's analyzer splits the literal
value "bills,unpaid" into the same two tokens the value-list reading
produces, so the two readings select the same documents. The registry fact
-- that ``tag`` opts in and no other field does -- is observable only at
the registry, and is owned by test_registry.py's
``test_tag_is_comma_values``/``test_correspondent_is_not_comma_values``.
"""
def test_comma_list_requires_every_listed_tag(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document carrying both "bills" and "unpaid" tags, and a
second document carrying only "bills" (plus "archived")
WHEN:
- "tag:bills,unpaid" is searched
THEN:
- Only the document carrying every listed tag matches, and a
single-tag "tag:bills" search still matches both documents
"""
bills = Tag.objects.create(name="bills")
unpaid = Tag.objects.create(name="unpaid")
archived = Tag.objects.create(name="archived")
both = Document.objects.create(
title="Both tags",
content="body",
checksum="doc-syntax-tag-both",
)
both.tags.add(bills, unpaid)
backend.add_or_update(both)
one = Document.objects.create(
title="One tag",
content="body",
checksum="doc-syntax-tag-one",
)
one.tags.add(bills, archived)
backend.add_or_update(one)
assert _matched_ids(backend, "tag:bills,unpaid") == {both.pk}
assert _matched_ids(backend, "tag:bills") == {both.pk, one.pk}
class TestArchiveMetadataFields:
@pytest.fixture
def doc(self, backend: TantivyBackend, admin_user: User) -> Document:
doc = Document.objects.create(
title="Metadata",
content="body",
checksum=DOC_CHECKSUM,
archive_serial_number=100,
page_count=12,
original_filename="invoice.pdf",
)
Note.objects.create(document=doc, user=admin_user, note="a note")
backend.add_or_update(doc)
return doc
@pytest.mark.parametrize(
"query",
[
"asn:100",
"asn:[50 to 150]",
"page_count:12",
"page_count:[10 to 20]",
"num_notes:1",
"num_notes:[1 to 5]",
"original_filename:invoice.pdf",
f"checksum:{DOC_CHECKSUM}",
"checksum:9f86d081*",
# A checksum term is stored verbatim, but a checksum *pattern* is
# lowercased before it is matched, so an uppercase prefix pattern
# still matches even though the uppercase term in the negative
# list below does not.
"checksum:9F86D081*",
],
)
def test_documented_metadata_query_matches(
self,
backend: TantivyBackend,
doc: Document,
query: str,
) -> None:
"""
GIVEN:
- A document with an ASN, page count, a note, an original
filename and a known checksum
WHEN:
- Every documented metadata-field spelling (exact value,
range, and, for checksum, a lowercase prefix pattern
regardless of the case the pattern itself is typed in) is
searched
THEN:
- Each one matches the document
"""
assert _matched_ids(backend, query) == {doc.pk}
@pytest.mark.parametrize(
"query",
[
# The docs say only a complete, lowercase checksum matches.
"checksum:9f86d081",
f"checksum:{DOC_CHECKSUM.upper()}",
],
)
def test_partial_or_uppercase_checksum_matches_nothing(
self,
backend: TantivyBackend,
doc: Document,
query: str,
) -> None:
"""
GIVEN:
- A document with a known, complete, lowercase checksum
WHEN:
- An exact-value search is run with a partial or uppercase
spelling of that checksum
THEN:
- Nothing matches, as the docs say only a complete, lowercase
checksum matches as an exact value
"""
assert _matched_ids(backend, query) == set()
class TestDocumentedDateForms:
@pytest.fixture(autouse=True)
def frozen_now(self) -> Generator[None, None, None]:
with time_machine.travel(FROZEN_NOW, tick=False):
yield
@pytest.fixture
def dated(self, backend: TantivyBackend) -> dict[str, int]:
stamps = {
"today": datetime(2026, 6, 15, 9, 0, tzinfo=UTC),
"yesterday": datetime(2026, 6, 14, 9, 0, tzinfo=UTC),
"tomorrow": datetime(2026, 6, 16, 9, 0, tzinfo=UTC),
"next_monday": datetime(2026, 6, 22, 10, 0, tzinfo=UTC),
"last_monday": datetime(2026, 6, 8, 10, 0, tzinfo=UTC),
"january": datetime(2026, 1, 10, 10, 0, tzinfo=UTC),
"old": datetime(2005, 3, 4, 15, 30, tzinfo=UTC),
}
return {
label: _index(
backend,
title=label,
content="dated body",
checksum=f"doc-syntax-date-{label}",
added=stamp,
).pk
for label, stamp in stamps.items()
}
@pytest.mark.parametrize(
("query", "label"),
[
("added:today", "today"),
("added:yesterday", "yesterday"),
("added:tomorrow", "tomorrow"),
('added:"next monday"', "next_monday"),
('added:"last monday"', "last_monday"),
("added:january", "january"),
("added:2005-03-04", "old"),
("added:2005-03", "old"),
("added:[2005-01-01 to 2005-12-31]", "old"),
("added:[2005 to 2009]", "old"),
# A full timestamp works, but only quoted when it stands alone,
# and only unquoted when it is a range bound. The bare standalone
# spelling is pinned as a non-match below.
('added:"2005-03-04T15:30:00Z"', "old"),
("added:[2005-03-04T09:00:00Z to 2005-03-04T17:00:00Z]", "old"),
# A quoted range bound works when the quotes are single ones; the
# double-quoted spelling is pinned as an error below.
("added:['2005-03-04' to 2005-03-05]", "old"),
],
)
def test_documented_date_form_matches_its_day_or_month(
self,
backend: TantivyBackend,
dated: dict[str, int],
query: str,
label: str,
) -> None:
"""
GIVEN:
- Documents dated today, yesterday, tomorrow, next/last
Monday, in January, and on an old fixed date, indexed
against a frozen "now" (a Monday)
WHEN:
- Every documented date-form spelling is searched: relative
keywords, quoted multi-word phrases, a bare year-month, an
explicit range, a quoted full timestamp standing alone, an
unquoted full timestamp as a range bound, and a
single-quoted range bound
THEN:
- Each form matches exactly the document dated on its day or
within its month
"""
assert _matched_ids(backend, query) == {dated[label]}
@pytest.mark.parametrize(
"query",
[
# Zero-width: these resolve to a single instant, not a span, so
# nothing in a realistic corpus lands on them. The docs warn
# about them rather than presenting them as usable.
"added:now",
"added:noon",
"added:midnight",
# Quoting is what rescues the other multi-word date expressions,
# so pin that it does not rescue these: the problem is the width
# of the resulting range, not the way the value is delimited.
# One quoted spelling is enough for that; which keyword sits
# inside the quotes is grammar whoosh-compat owns.
'added:"now"',
# A relative offset, which the warning in the docs names by this
# exact spelling. Standing alone it is an instant like the rest of
# this list; the same offset used as a range bound is a real
# window, pinned by the test below.
'added:"-1 week"',
],
)
def test_forms_the_docs_warn_about_match_nothing(
self,
backend: TantivyBackend,
dated: dict[str, int],
query: str,
) -> None:
"""
GIVEN:
- A realistic dated corpus (see the `dated` fixture)
WHEN:
- A zero-width date form ("now", "noon", "midnight", a quoted
"now") or a standalone relative offset ("-1 week") is
searched: each resolves to a single instant rather than a
span, and quoting does not rescue them the way it rescues
other multi-word date expressions, since the problem is the
width of the resulting range, not how the value is
delimited
THEN:
- Nothing matches, exactly as the docs warn, rather than
presenting these as usable spellings
"""
assert _matched_ids(backend, query) == set()
def test_bare_timestamp_is_rejected_rather_than_matching_nothing(
self,
backend: TantivyBackend,
dated: dict[str, int],
) -> None:
"""
GIVEN:
- A realistic dated corpus, including a document dated at a
known full timestamp
WHEN:
- The bare, unquoted spelling of that full timestamp is
searched (the quoted and range-bound spellings pinned above
do work and match this fixture's document)
THEN:
- `InvalidDateQuery` is raised rather than the query silently
matching nothing, since this is a user-fixable error the
docs tell the user to quote, and the reported value is the
whole contiguous fragment the user typed, not just the
prefix the date grammar's tokenizer first split on
"""
with pytest.raises(InvalidDateQuery) as exc_info:
_matched_ids(backend, "added:2005-03-04T15:30:00Z")
assert exc_info.value.field == "added"
assert exc_info.value.value == "2005-03-04T15:30:00Z"
def test_relative_offset_as_a_range_bound_is_a_real_window(
self,
backend: TantivyBackend,
dated: dict[str, int],
) -> None:
"""
GIVEN:
- A realistic dated corpus, including a document dated two
hours before a "last Monday to now" window opens, and
documents dated today and yesterday, inside that window
WHEN:
- "added:['-1 week' to now]" is searched: the same offset
that matches nothing standing alone (see the test above),
used here as a range bound instead
THEN:
- The window matches today and yesterday but excludes the
document two hours before it opens, showing the bound is
the offset itself and not a whole-day rounding of it, as
the docs say next to the warning about the standalone form
"""
assert _matched_ids(backend, "added:['-1 week' to now]") == {
dated["today"],
dated["yesterday"],
}
def test_double_quoted_range_bound_is_rejected(
self,
backend: TantivyBackend,
dated: dict[str, int],
) -> None:
"""
GIVEN:
- A realistic dated corpus
WHEN:
- A range bound is double-quoted rather than single-quoted
("added:[\"2005-03-04\" to 2005-03-05]")
THEN:
- `InvalidDateQuery` is raised, pinning which of the two
quote characters fails: quoting a range bound is allowed,
but only with single quotes, since the double-quoted
spelling reaches the date grammar with its quotes still
attached and is not a recognizable date
"""
with pytest.raises(InvalidDateQuery) as exc_info:
_matched_ids(backend, 'added:["2005-03-04" to 2005-03-05]')
assert exc_info.value.value == '"2005-03-04"'
@@ -0,0 +1,364 @@
"""Diagnostics route by Cause, and user-facing messages are host-owned.
whoosh-compat documents ``Diagnostic.message`` as developer output with no
stability guarantee, so it must never reach an HTTP response body.
"""
from __future__ import annotations
import logging
from datetime import UTC
import pytest
import tantivy
from whoosh_compat.errors import Diagnostic
from whoosh_compat.errors import DiagnosticKind
from whoosh_compat.errors import QueryError
from whoosh_compat.errors import cause_for
from whoosh_compat.fields import FieldKind
from whoosh_compat.fields import FieldRef
from documents.search._errors import SearchQueryError
from documents.search._query import _map_emit_error
from documents.search._query import _single_diagnostic_to_error
from documents.search._query import parse_user_query
from documents.search._schema import build_schema
from documents.search._tokenizer import register_tokenizers
pytestmark = pytest.mark.search
_LIBRARY_PROSE = "INTERNAL LIBRARY WORDING WITH raw tantivy detail"
@pytest.fixture(scope="module")
def query_index() -> tantivy.Index:
"""An in-memory, unstemmed index; these tests only parse, never index."""
idx = tantivy.Index(build_schema(), path=None)
register_tokenizers(idx, "")
return idx
def _diagnostic(
kind: DiagnosticKind,
*,
field: FieldRef | None = FieldRef("title"),
field_kind: FieldKind | None = FieldKind.TEXT,
) -> Diagnostic:
"""A Diagnostic shaped like the emitter's, with the library's own
kind -> cause mapping rather than a hand-picked cause."""
return Diagnostic(
kind=kind,
cause=cause_for(kind),
message=_LIBRARY_PROSE,
field=field,
field_kind=field_kind,
)
class TestEmitErrorRouting:
"""Every Cause gets a distinguishable treatment, not just "a 400"."""
@pytest.mark.parametrize(
"kind",
[
DiagnosticKind.BACKEND_REJECTED,
DiagnosticKind.AST_INVALID_SHAPE,
DiagnosticKind.AST_UNKNOWN_FIELD,
],
)
def test_internal_cause_is_not_converted(self, kind: DiagnosticKind) -> None:
"""
GIVEN:
- A QueryError wrapping a Diagnostic whose Cause is INTERNAL
(BACKEND_REJECTED/AST_INVALID_SHAPE/AST_UNKNOWN_FIELD)
WHEN:
- _map_emit_error processes it
THEN:
- The original QueryError propagates unchanged, so it surfaces
as a 500 monitoring can see, never a 400 blaming the user
"""
error = QueryError(_diagnostic(kind))
with pytest.raises(QueryError) as excinfo:
_map_emit_error(error)
assert excinfo.value is error
def test_misconfigured_cause_is_logged_and_reraised(
self,
caplog: pytest.LogCaptureFixture,
) -> None:
"""
GIVEN:
- A QueryError for SCHEMA_FIELD_MISSING naming field "asn"
WHEN:
- _map_emit_error processes it
THEN:
- Exactly one ERROR log record is emitted naming the field and
the diagnostic kind, and the original QueryError propagates
unchanged: a registry/schema disagreement is transient (the
exact same query succeeds once the index is rebuilt), so it
surfaces as a 500 an operator can see rather than a 400
telling the client their query is permanently invalid
"""
kind = DiagnosticKind.SCHEMA_FIELD_MISSING
error = QueryError(_diagnostic(kind, field=FieldRef("asn")))
with (
caplog.at_level(logging.ERROR, logger="paperless.search"),
pytest.raises(QueryError) as excinfo,
):
_map_emit_error(error)
assert excinfo.value is error
errors = [r for r in caplog.records if r.levelno == logging.ERROR]
assert len(errors) == 1
assert "asn" in errors[0].getMessage()
assert kind.name in errors[0].getMessage()
@pytest.mark.parametrize(
"kind",
[
DiagnosticKind.TEXT_RANGE,
DiagnosticKind.PATTERN_TOO_COMPLEX,
DiagnosticKind.EXISTS_REQUIRES_FAST,
],
)
def test_unsupported_cause_is_a_400_with_no_operator_log(
self,
kind: DiagnosticKind,
caplog: pytest.LogCaptureFixture,
) -> None:
"""
GIVEN:
- A QueryError for a query tantivy cannot run
(TEXT_RANGE/PATTERN_TOO_COMPLEX/EXISTS_REQUIRES_FAST)
WHEN:
- _map_emit_error processes it
THEN:
- It becomes a SearchQueryError with no log record at WARNING
or above; a query tantivy cannot run is the user's to fix,
not an operator alert. EXISTS_REQUIRES_FAST is nominally
MISCONFIGURED but belongs here: it is decided from the
registry's own FieldSpec, so it never reports a disagreement
anyone could resolve
"""
with caplog.at_level(logging.WARNING, logger="paperless.search"):
error = _map_emit_error(QueryError(_diagnostic(kind)))
assert isinstance(error, SearchQueryError)
assert caplog.records == []
@pytest.mark.parametrize(
"kind",
[
DiagnosticKind.TEXT_RANGE,
DiagnosticKind.PATTERN_TOO_COMPLEX,
DiagnosticKind.EXISTS_REQUIRES_FAST,
],
)
def test_user_facing_message_never_echoes_library_prose(
self,
kind: DiagnosticKind,
) -> None:
"""
GIVEN:
- A QueryError carrying whoosh-compat's own developer-facing
message text (SCHEMA_FIELD_MISSING excluded: it is now
re-raised rather than converted, so it never produces a
user-facing message at all, see
test_misconfigured_cause_is_logged_and_reraised)
WHEN:
- _map_emit_error processes it
THEN:
- The resulting error's string never contains that library
prose
"""
error = _map_emit_error(QueryError(_diagnostic(kind)))
assert _LIBRARY_PROSE not in str(error)
@pytest.mark.parametrize(
"kind",
[
DiagnosticKind.TEXT_RANGE,
DiagnosticKind.PATTERN_TOO_COMPLEX,
DiagnosticKind.EXISTS_REQUIRES_FAST,
],
)
def test_user_facing_message_names_the_field(
self,
kind: DiagnosticKind,
) -> None:
"""
GIVEN:
- A QueryError for a JSON subpath field (custom_fields.value)
WHEN:
- _map_emit_error processes it
THEN:
- The resulting error names the field using its canonical
dotted form, including the subpath (FieldRef.__str__ yields
this dotted name, so every user-reachable emit kind can name
it)
"""
diagnostic = _diagnostic(
kind,
field=FieldRef("custom_fields", "value"),
field_kind=FieldKind.JSON,
)
error = _map_emit_error(QueryError(diagnostic))
assert "custom_fields.value" in str(error)
class TestParseDiagnosticMessages:
"""Parse-time diagnostics are host-worded too, off field_kind."""
def test_too_deep_is_a_400_without_library_prose(self) -> None:
"""
GIVEN:
- A parse-time Diagnostic for TOO_DEEP with no field
WHEN:
- _single_diagnostic_to_error processes it
THEN:
- It becomes a SearchQueryError with no library prose in its
message
"""
error = _single_diagnostic_to_error(
_diagnostic(DiagnosticKind.TOO_DEEP, field=None, field_kind=None),
)
assert isinstance(error, SearchQueryError)
assert _LIBRARY_PROSE not in str(error)
@pytest.mark.parametrize(
("kind", "field_kind"),
[
(DiagnosticKind.PATTERN_ON_NUMERIC, FieldKind.U64),
(DiagnosticKind.PATTERN_ON_BOOLEAN_EXISTS, FieldKind.BOOLEAN_EXISTS),
(DiagnosticKind.PATTERN_ON_SUBPATH, FieldKind.JSON),
],
)
def test_pattern_on_kinds_name_the_field_and_its_kind(
self,
kind: DiagnosticKind,
field_kind: FieldKind,
) -> None:
"""
GIVEN:
- A parse-time Diagnostic for a pattern used against a kind
that cannot take one
(PATTERN_ON_NUMERIC/PATTERN_ON_BOOLEAN_EXISTS/PATTERN_ON_SUBPATH)
WHEN:
- _single_diagnostic_to_error processes it
THEN:
- The message names both the field and its kind, with no
library prose
"""
error = _single_diagnostic_to_error(
_diagnostic(kind, field=FieldRef("asn"), field_kind=field_kind),
)
message = str(error)
assert _LIBRARY_PROSE not in message
assert "asn" in message
assert field_kind.name.lower() in message
def test_single_char_bracket_range_names_the_field_and_the_value(self) -> None:
"""
GIVEN:
- A SINGLE_CHAR_BRACKET_RANGE diagnostic for "title" with
raw_value "200[1-9]"
WHEN:
- _single_diagnostic_to_error processes it
THEN:
- The resulting SearchQueryError names both the field and the
offending value, with no library prose
"""
diagnostic = Diagnostic(
kind=DiagnosticKind.SINGLE_CHAR_BRACKET_RANGE,
cause=cause_for(DiagnosticKind.SINGLE_CHAR_BRACKET_RANGE),
message=_LIBRARY_PROSE,
field=FieldRef("title"),
field_kind=FieldKind.TEXT,
raw_value="200[1-9]",
)
error = _single_diagnostic_to_error(diagnostic)
message = str(error)
assert isinstance(error, SearchQueryError)
assert _LIBRARY_PROSE not in message
assert "title" in message
assert "200[1-9]" in message
class TestRealQueriesRouteCorrectly:
"""The routing table against diagnostics emit() really produces."""
def test_text_range_is_a_400_naming_the_field(
self,
query_index: tantivy.Index,
) -> None:
"""
GIVEN:
- A real query index
WHEN:
- parse_user_query is called with a text-range query
("title:[a to b]")
THEN:
- It raises SearchQueryError naming "title"
"""
with pytest.raises(SearchQueryError) as excinfo:
parse_user_query(query_index, "title:[a to b]", UTC)
assert "title" in str(excinfo.value)
def test_wildcard_on_a_numeric_field_is_a_400_naming_the_field(
self,
query_index: tantivy.Index,
) -> None:
"""
GIVEN:
- A real query index
WHEN:
- parse_user_query is called with a wildcard on a numeric
field ("asn:12*")
THEN:
- It raises SearchQueryError naming "asn"
"""
with pytest.raises(SearchQueryError) as excinfo:
parse_user_query(query_index, "asn:12*", UTC)
assert "asn" in str(excinfo.value)
def test_single_char_bracket_range_is_a_400_naming_field_and_value(
self,
query_index: tantivy.Index,
) -> None:
"""
GIVEN:
- A real query index
WHEN:
- parse_user_query is called with "title:200[1-9]"
THEN:
- It raises SearchQueryError naming both "title" and
"200[1-9]"
"""
with pytest.raises(SearchQueryError) as excinfo:
parse_user_query(query_index, "title:200[1-9]", UTC)
message = str(excinfo.value)
assert "title" in message
assert "200[1-9]" in message
def test_internal_diagnostic_escapes_as_a_query_error(
self,
query_index: tantivy.Index,
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""
GIVEN:
- tantivy_emit monkeypatched to raise a QueryError with an
INTERNAL-cause diagnostic (BACKEND_REJECTED), the one case
with no query text of its own involved
WHEN:
- parse_user_query runs a normal query ("invoice")
THEN:
- The QueryError propagates unconverted; emit() reporting a
defect in itself must not become a user-facing 400
"""
import documents.search._query as query_mod
def raise_internal(*args: object, **kwargs: object) -> None:
raise QueryError(_diagnostic(DiagnosticKind.BACKEND_REJECTED))
monkeypatch.setattr(query_mod, "tantivy_emit", raise_internal)
with pytest.raises(QueryError):
parse_user_query(query_index, "invoice", UTC)
@@ -0,0 +1,124 @@
"""``field:*`` on a JSON field is user error, not an operator alert.
whoosh-compat classifies EXISTS_REQUIRES_FAST as MISCONFIGURED, and
_map_emit_error used to route every MISCONFIGURED diagnostic to an ERROR log.
But the kind is decided from the registry's own FieldSpec (kind plus fast)
without consulting the index schema, and field_descriptors() builds the JSON
fields non-fast deliberately, so nothing is misconfigured and no operator
action can clear the condition. Any authenticated user could otherwise emit
ERROR lines in a loop by repeating ``notes:*``.
SCHEMA_FIELD_MISSING, the other MISCONFIGURED kind, does compare the registry
against the live schema, so it stays an ERROR log. But it is not a 400
either: the exact same query would succeed once the index is rebuilt, so it
is a transient server-side condition, not a permanently bad request, and is
re-raised the same way an INTERNAL cause is.
"""
from __future__ import annotations
import logging
from datetime import UTC
import pytest
import tantivy
from whoosh_compat.errors import Diagnostic
from whoosh_compat.errors import DiagnosticKind
from whoosh_compat.errors import QueryError
from whoosh_compat.errors import cause_for
from whoosh_compat.fields import FieldKind
from whoosh_compat.fields import FieldRef
from documents.search._errors import SearchQueryError
from documents.search._query import _map_emit_error
from documents.search._query import parse_user_query
from documents.search._schema import build_schema
from documents.search._tokenizer import register_tokenizers
pytestmark = pytest.mark.search
# Every spelling of "does this JSON field have a value" a user can type.
EXISTS_QUERIES = [
"notes:*",
"notes.note:*",
"notes.user:*",
"custom_fields:*",
"custom_fields.name:*",
"custom_fields.value:*",
]
@pytest.fixture(scope="module")
def query_index() -> tantivy.Index:
idx = tantivy.Index(build_schema(), path=None)
register_tokenizers(idx, "")
return idx
class TestJsonExistsIsUserError:
@pytest.mark.parametrize("query", EXISTS_QUERIES)
def test_query_is_a_400_that_emits_no_error_log(
self,
query_index: tantivy.Index,
caplog: pytest.LogCaptureFixture,
query: str,
) -> None:
"""
GIVEN:
- A real query index, and every spelling of "does this JSON
field have a value" (notes:*, notes.note:*, custom_fields:*,
etc.)
WHEN:
- parse_user_query runs the query
THEN:
- It raises SearchQueryError naming the field, and no
ERROR-level log record is emitted; EXISTS_REQUIRES_FAST on a
JSON field is by design, not a misconfiguration an operator
could act on
"""
with caplog.at_level(logging.WARNING, logger="paperless.search"):
with pytest.raises(SearchQueryError) as excinfo:
parse_user_query(query_index, query, UTC)
assert query.split(":", maxsplit=1)[0] in str(excinfo.value)
assert [r for r in caplog.records if r.levelno >= logging.ERROR] == []
class TestGenuineMisconfigurationStillLogs:
def test_schema_field_missing_is_an_error_log_and_reraised(
self,
caplog: pytest.LogCaptureFixture,
) -> None:
"""
GIVEN:
- A QueryError for SCHEMA_FIELD_MISSING: the registry naming a
field the index schema does not have
WHEN:
- _map_emit_error processes it
THEN:
- It logs exactly one ERROR record naming the diagnostic kind
(a real mismatch an operator can fix, so it keeps the
alert), and the original QueryError propagates unchanged
rather than becoming a SearchQueryError: the exact same
query would succeed once the index is rebuilt, so this is a
transient server-side condition, not a permanently bad
request, and surfaces as a 500 rather than a 400
"""
kind = DiagnosticKind.SCHEMA_FIELD_MISSING
error = QueryError(
Diagnostic(
kind=kind,
cause=cause_for(kind),
message="field 'asn' is not defined in the index schema",
field=FieldRef("asn"),
field_kind=FieldKind.U64,
),
)
with (
caplog.at_level(logging.ERROR, logger="paperless.search"),
pytest.raises(QueryError) as excinfo,
):
_map_emit_error(error)
assert excinfo.value is error
records = [r for r in caplog.records if r.levelno == logging.ERROR]
assert len(records) == 1
assert kind.name in records[0].getMessage()
@@ -0,0 +1,261 @@
"""The words the fuzzy blend clause hands back to tantivy's parser.
The clause re-parses a word string through tantivy, which analyzes it
again, so the words must be the query's raw text rather than the analyzed
text (analysis is not idempotent), and must still be split into plain
words so that hyphenated, dotted and quoted terms keep contributing.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.models import Document
if TYPE_CHECKING:
from pytest_django.fixtures import SettingsWrapper
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
@pytest.fixture(autouse=True)
def fuzzy_enabled(settings: SettingsWrapper) -> None:
"""Enable the fuzzy blend clause. The threshold doubles as a minimum
score filter, so it is set to 0.0: every hit passes and the test sees
the clause's matching behaviour, not the filter's."""
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.0
class TestFuzzyClauseParseFailureDegradesGracefully:
def test_a_word_string_tantivy_rejects_drops_the_clause_only(self) -> None:
"""
GIVEN:
- A parsed query with free-text words, and an index-like
object whose parse_query is forced to raise ValueError
WHEN:
- _try_parse_fuzzy_query is called
THEN:
- It returns None instead of propagating, so a fuzzy word
string tantivy's own parser rejects only drops the fuzzy
clause: the exact/CJK clauses still stand rather than the
whole query failing. The ValueError guard is insurance (the
word string is plain tokens, so tantivy accepting it is
expected, not assumed)
"""
import whoosh_compat as wc
from documents.search._query import _DEFAULT_SEARCH_FIELDS
from documents.search._query import _try_parse_fuzzy_query
from documents.search._registry import get_field_registry
registry = get_field_registry(None)
result = wc.parse(
"invoice",
registry=registry,
default_fields=_DEFAULT_SEARCH_FIELDS,
)
class _RaisingIndex:
def parse_query(self, *args: object, **kwargs: object) -> object:
raise ValueError("synthetic parse failure")
assert _try_parse_fuzzy_query(_RaisingIndex(), result.ast, registry) is None
class TestFuzzyClauseWords:
def test_a_stemmed_word_is_not_stemmed_a_second_time(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- Documents whose content contains "universities", a
one-transposition typo of it ("universties"), and two
unrelated words that share its stem prefix ("univalent",
"unicycle")
WHEN:
- Searching for "universities" with the fuzzy blend enabled
THEN:
- Only the correctly-spelled document and its typo match; the
clause does not widen far enough to reach the unrelated
words. 'universities' stems to 'univers'; feeding that back
to tantivy would stem it again to 'univ', whose fuzzy prefix
reaches unrelated words - the clause must stay wide enough
for a typo and no wider
"""
wanted = _index(
backend,
title="A",
content="universities of europe",
checksum="fuzz-stem-1",
)
typo = _index(
backend,
title="B",
content="universties of europe",
checksum="fuzz-stem-2",
)
_index(
backend,
title="C",
content="univalent chemical bonds",
checksum="fuzz-stem-3",
)
_index(
backend,
title="D",
content="unicycle repair manual",
checksum="fuzz-stem-4",
)
assert _matched_ids(backend, "universities") == {wanted.pk, typo.pk}
def test_a_hyphenated_term_still_reaches_the_clause(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document whose content contains a near-miss of "COVID-19"
("covidx")
WHEN:
- Searching for "COVID-19" with the fuzzy blend enabled
THEN:
- The document matches; 'COVID-19' is one raw token, so unless
it is split into words, it carries characters the re-parse
would read as grammar, is dropped, and the whole query loses
its fuzzy clause
"""
misspelled = _index(
backend,
title="A",
content="covidx testing results",
checksum="fuzz-hyphen-1",
)
assert _matched_ids(backend, "COVID-19") == {misspelled.pk}
def test_a_phrase_still_reaches_the_clause(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document whose content near-misses a quoted phrase
WHEN:
- Searching for the quoted phrase '"tax reports"' with the
fuzzy blend enabled
THEN:
- The document matches; a phrase is one raw token carrying a
space, and is the whole query's only free text here, so it
must still reach the clause
"""
near_miss = _index(
backend,
title="A",
content="taxation reportage weekly",
checksum="fuzz-phrase-1",
)
assert _matched_ids(backend, '"tax reports"') == {near_miss.pk}
class TestBooleanKeywordsInRawText:
"""Tantivy's boolean keywords are word runs, so they survive the cut
into words and its own parser reads them as grammar. Raw query text
reaches that parser with its case intact, so a quoted phrase can carry
them in."""
@pytest.fixture
def corpus(self, backend: TantivyBackend) -> dict[str, int]:
both = _index(
backend,
title="A",
content="taxation reportage weekly",
checksum="fuzz-kw-1",
)
tax_only = _index(
backend,
title="B",
content="taxation only here",
checksum="fuzz-kw-2",
)
report_only = _index(
backend,
title="C",
content="reportage only here",
checksum="fuzz-kw-3",
)
return {
"both": both.pk,
"tax_only": tax_only.pk,
"report_only": report_only.pk,
}
@pytest.mark.parametrize(
"query",
[
pytest.param('"tax AND reports"', id="and"),
pytest.param('"tax OR reports"', id="or"),
pytest.param('"tax NOT reports"', id="not"),
pytest.param('"tax IN reports"', id="in"),
],
)
def test_a_keyword_inside_a_phrase_stays_an_ordinary_word(
self,
backend: TantivyBackend,
corpus: dict[str, int],
query: str,
) -> None:
"""
GIVEN:
- Three documents: one with both "taxation" and "reportage",
one with only "taxation", one with only "reportage"
WHEN:
- Searching for a quoted phrase carrying a tantivy boolean
keyword as one of its words (e.g. '"tax AND reports"')
THEN:
- The keyword stays an ordinary word inside the phrase, and
the fuzzy clause matches all three documents, the same
disjunction as the plain '"tax reports"' phrase: AND must
not turn it into a conjunction, NOT must not give it its own
exclusion, IN must not fail the parse
"""
assert _matched_ids(backend, '"tax reports"') == set(corpus.values())
assert _matched_ids(backend, query) == set(corpus.values())
def test_a_trailing_keyword_does_not_drop_the_clause(
self,
backend: TantivyBackend,
corpus: dict[str, int],
) -> None:
"""
GIVEN:
- Three documents: one with both "taxation" and "reportage",
one with only "taxation", one with only "reportage"
WHEN:
- Searching for '"tax AND"', a phrase ending in a tantivy
syntax error
THEN:
- The fuzzy clause still matches on "tax"; 'tax AND' alone is
a syntax error to tantivy's parser, which would otherwise
cost the whole query its fuzzy clause
"""
assert _matched_ids(backend, '"tax AND"') == {
corpus["both"],
corpus["tax_only"],
}
@@ -0,0 +1,249 @@
"""Regression coverage for the unguarded TEXT-mode highlight query.
parse_simple_text_highlight_query re-parses simple-search tokens through
Tantivy's query-string parser to build a SnippetGenerator-compatible query.
Simple-search tokens keep arbitrary punctuation (quotes, colons, brackets,
slashes), so any token carrying Tantivy query grammar raised an unguarded
ValueError. The search itself had already succeeded by the time this ran:
only the highlight step failed, and with the DocumentViewSet.list
exception handler narrowed elsewhere on this branch, that ValueError now
reaches the client as a bare 500 rather than a 400.
Covers three angles:
- the query builder itself: quoting each token as its own escaped phrase
should let it parse instead of raising, for every failure mode a plain-
text query can trigger (syntax error, unknown field, unsupported regex).
- highlight_hits: even when a token still can't be expressed as a
highlight query, the guard must fall back to a query that still
produces usable highlight HTML, not silently empty ones.
- the real API endpoint: pinning the previously-500 status to 200.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
import tantivy
from rest_framework import status
from documents.search._backend import SearchMode
from documents.search._query import parse_simple_text_highlight_query
from documents.search._schema import build_schema
from documents.search._tokenizer import register_tokenizers
from documents.tests.factories import DocumentFactory
if TYPE_CHECKING:
from rest_framework.test import APIClient
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
# Each spelling below trips a different Tantivy parser failure mode:
# 'a"b' -> Syntax Error (unterminated quote)
# foo:bar -> unknown field
# (a -> Syntax Error (unbalanced group)
# [a -> Syntax Error (unbalanced range)
# /a/ -> Unsupported query (regex queries disallowed)
_MALFORMED_QUERIES = [
pytest.param('a"b', id="unterminated_quote"),
pytest.param("foo:bar", id="unknown_field"),
pytest.param("(a", id="unbalanced_group"),
pytest.param("[a", id="unbalanced_range"),
pytest.param("/a/", id="unsupported_regex"),
]
@pytest.fixture(scope="module")
def query_index() -> tantivy.Index:
"""An in-memory, unstemmed index for parse-only tests."""
schema = build_schema()
idx = tantivy.Index(schema, path=None)
register_tokenizers(idx, "")
return idx
class TestParseSimpleTextHighlightQueryDoesNotRaise:
"""The query builder itself must tolerate Tantivy syntax in its tokens."""
@pytest.mark.parametrize("raw_query", _MALFORMED_QUERIES)
def test_malformed_token_does_not_raise(
self,
query_index: tantivy.Index,
raw_query: str,
) -> None:
"""
GIVEN:
- A simple-search query token carrying Tantivy query grammar
(unterminated quote, unknown field, unbalanced group/range,
or unsupported regex)
WHEN:
- parse_simple_text_highlight_query builds a highlight query
from it
THEN:
- It returns a tantivy.Query instead of raising, since each
token is quoted as its own escaped phrase rather than fed
to the parser raw
"""
assert isinstance(
parse_simple_text_highlight_query(query_index, raw_query),
tantivy.Query,
)
class TestHighlightHitsProducesUsableHighlights:
"""highlight_hits must keep producing real <b>-wrapped snippet HTML for
these queries, not merely avoid raising."""
@pytest.mark.parametrize(
"raw_query",
[*_MALFORMED_QUERIES, pytest.param("plain text", id="plain_text_sanity")],
)
def test_highlight_still_contains_matched_text(
self,
backend: TantivyBackend,
raw_query: str,
) -> None:
"""
GIVEN:
- A document whose content contains the raw query text
verbatim
WHEN:
- backend.highlight_hits builds highlights for a TEXT-mode
search using that same (possibly Tantivy-grammar-carrying)
query text
THEN:
- The hit still carries a content highlight with real
<b>-wrapped matched-term markup, not an empty fallback
"""
doc = DocumentFactory.create(
title="probe",
content=f"needle content containing {raw_query} literally here",
)
backend.add_or_update(doc)
hits = backend.highlight_hits(
raw_query,
[doc.pk],
search_mode=SearchMode.TEXT,
)
assert len(hits) == 1
highlights = hits[0]["highlights"]
assert "content" in highlights, (
f"Expected a content highlight for {raw_query!r}, got: {highlights!r}"
)
assert "<b>" in highlights["content"], (
f"Highlight for {raw_query!r} carries no matched-term markup: "
f"{highlights['content']!r}"
)
class TestHighlightGuardDiscriminatesOnValueError:
"""The guard added to highlight_hits must catch exactly ValueError, the
same shape as the sibling notes_text guard, and let anything else
through -- so a real library defect is never mistaken for a harmless
syntax error."""
def test_non_value_error_is_not_swallowed(
self,
backend: TantivyBackend,
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""
GIVEN:
- parse_simple_text_highlight_query patched to raise
RuntimeError instead of a syntax-related ValueError
WHEN:
- backend.highlight_hits is called
THEN:
- The RuntimeError propagates unguarded; the highlight guard
must catch exactly ValueError, the same shape as the
sibling notes_text guard, never mistaking a real library
defect for a harmless syntax error
"""
import documents.search._backend as backend_mod
def raise_runtime_error(*args: object, **kwargs: object) -> object:
raise RuntimeError("synthetic bug, unrelated to query syntax")
monkeypatch.setattr(
backend_mod,
"parse_simple_text_highlight_query",
raise_runtime_error,
)
doc = DocumentFactory.create(title="probe", content="anything here")
backend.add_or_update(doc)
with pytest.raises(RuntimeError):
backend.highlight_hits(
"anything",
[doc.pk],
search_mode=SearchMode.TEXT,
)
@pytest.mark.usefixtures("_search_index")
class TestApiNoLongerReturns500:
"""Pins the actual regression: a matching TEXT-mode search whose query
string carries Tantivy syntax must return results, not a server error."""
@pytest.mark.parametrize("raw_query", _MALFORMED_QUERIES)
def test_malformed_text_query_returns_200(
self,
admin_client: APIClient,
raw_query: str,
) -> None:
"""
GIVEN:
- A matching document whose content contains the raw query
text, indexed via the real search index fixture
WHEN:
- A TEXT-mode search is issued through the real API with a
query string carrying Tantivy syntax
THEN:
- The response is 200 with the expected result count, not a
500 (the regression this file exists to pin)
"""
from documents.search import get_backend
doc = DocumentFactory.create(
title="probe",
content=f"needle content containing {raw_query} literally here",
)
get_backend().add_or_update(doc)
response = admin_client.get(f"/api/documents/?text={raw_query}")
assert response.status_code == status.HTTP_200_OK
assert response.data["count"] == 1
def test_plain_text_query_still_returns_200(
self,
admin_client: APIClient,
) -> None:
"""
GIVEN:
- A matching document indexed via the real search index
fixture
WHEN:
- An ordinary TEXT-mode search (no Tantivy syntax) is issued
THEN:
- The response is 200 with the expected result count; sanity
check that the guard does not mask a total failure of the
ordinary highlight path
"""
from documents.search import get_backend
doc = DocumentFactory.create(
title="probe",
content="needle content containing plain text literally here",
)
get_backend().add_or_update(doc)
response = admin_client.get("/api/documents/?text=plain text")
assert response.status_code == status.HTTP_200_OK
assert response.data["count"] == 1
@@ -0,0 +1,206 @@
"""Bare notes:/custom_fields: prefix resolution.
"notes:foo"/"custom_fields:foo" were valid fielded searches before the
whoosh-compat migration. The registry only exposes them as JSON subpaths, so
each JSON FieldSpec declares a default subpath (SubpathSpec(default=True)):
notes: resolves to notes.note:, custom_fields: resolves to
custom_fields.value:. This replaced an earlier regex-based rewrite
(_rewrite_bare_json_field_prefixes) that ran on the raw query string before
parsing and was blind to quoting, so a phrase like
content:"payment notes: none" was silently corrupted into a notes-field
search and matched nothing. Resolving the default subpath inside the parser
instead means quoting is already understood by the time it happens.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from django.contrib.auth.models import User
from documents.models import CustomField
from documents.models import CustomFieldInstance
from documents.models import Document
from documents.models import Note
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
class TestBareJsonFieldPrefixes:
def test_bare_notes_prefix_searches_note_text(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document with a note whose text contains a word, and a
decoy document whose content (not notes) contains the same
word
WHEN:
- A bare "notes:" prefix query is run (notes declares "note"
as its default subpath)
THEN:
- Only the document whose note matches is returned; the
decoy's content match does not resurface through a demoted
text search
"""
alice = User.objects.create_user(username="alice")
with_note = Document.objects.create(
title="Has note",
content="x",
checksum="bare-notes-with",
)
Note.objects.create(document=with_note, user=alice, note="crocodile")
backend.add_or_update(with_note)
# This document's CONTENT contains the words a demoted text search
# would match; it must NOT match once the prefix addresses notes.
_index(
backend,
title="Notes about things",
content="notes crocodile mention",
checksum="bare-notes-decoy",
)
assert _matched_ids(backend, "notes:crocodile") == {with_note.pk}
def test_bare_custom_fields_prefix_searches_values(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document with a custom field instance whose value
contains a word, and a decoy document whose content (not a
custom field value) contains the same word
WHEN:
- A bare "custom_fields:" prefix query is run (custom_fields
declares "value" as its default subpath)
THEN:
- Only the document whose custom field value matches is
returned
"""
field = CustomField.objects.create(
name="Policy Number",
data_type=CustomField.FieldDataType.STRING,
)
with_value = Document.objects.create(
title="Has field",
content="x",
checksum="bare-cf-with",
)
CustomFieldInstance.objects.create(
document=with_value,
field=field,
value_text="crocodile",
)
backend.add_or_update(with_value)
_index(
backend,
title="Custom things",
content="custom fields crocodile",
checksum="bare-cf-decoy",
)
assert _matched_ids(backend, "custom_fields:crocodile") == {with_value.pk}
def test_subpath_spellings_are_untouched(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document with a note carrying both an author and note text
WHEN:
- The explicit subpath spellings "notes.user:" and
"notes.note:" are queried
THEN:
- Both resolve to their intended subpath and match the
document; the default-subpath resolution for the bare
prefix does not interfere with explicit subpath addressing
"""
bob = User.objects.create_user(username="bob")
doc = Document.objects.create(
title="Bob note",
content="x",
checksum="bare-subpath",
)
Note.objects.create(document=doc, user=bob, note="remark")
backend.add_or_update(doc)
assert _matched_ids(backend, "notes.user:bob") == {doc.pk}
assert _matched_ids(backend, "notes.note:remark") == {doc.pk}
class TestQuotedPhraseContainingNotesColonIsNotCorrupted:
"""The regex rewrite this migration removes was blind to quoting: it
matched "notes:" anywhere in the raw query string, including inside an
already-quoted phrase on an unrelated field, silently turning
content:"payment notes: none" into a notes-field search that matched
nothing. Resolving the default subpath during parsing (which is
quote-aware) fixes this."""
def test_quoted_phrase_with_notes_colon_matches_by_content(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document whose content literally contains the text
"payment notes: none" inside a quoted phrase
WHEN:
- A query quoting that exact phrase against the content
field is run
THEN:
- It matches by content, rather than the "notes:" substring
inside the quotes being corrupted into a notes-field search
that matches nothing (the bug the deleted regex rewrite
caused, since it was blind to quoting)
"""
target = _index(
backend,
title="Statement",
content="payment notes: none",
checksum="quoted-phrase-notes-colon",
)
assert _matched_ids(
backend,
'content:"payment notes: none"',
) == {target.pk}
def test_quoted_phrase_matches_the_same_document_unquoted(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document whose content contains the same words as the
previous test's phrase, but without the colon
WHEN:
- A query quoting that phrase against the content field is
run
THEN:
- It matches by content, proving the earlier fix is about
quote-awareness specifically, not about the words
themselves being unsearchable
"""
target = _index(
backend,
title="Statement",
content="payment notes none",
checksum="quoted-phrase-no-colon",
)
assert _matched_ids(
backend,
'content:"payment notes none"',
) == {target.pk}
@@ -0,0 +1,92 @@
"""Every declared JSON subpath must actually be written to the index.
PUBLIC_FIELDS declares each JSON field's subpaths (e.g. ``notes`` ->
{"user", "note"}), but nothing coupled that declaration to what
``_backend.py``'s document builder actually writes into the JSON blob at
index time. A subpath declared but never written would be
queryable-but-always-empty -- syntactically valid, silently matching
nothing -- with no test failure anywhere.
This indexes one real document carrying values for every JSON field
(a Note, a CustomFieldInstance) and inspects the document's own stored
JSON payload, rather than running field-specific queries: that way a
future JSON field's subpaths are covered automatically, without a new
per-subpath query having to be added by hand each time.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
import tantivy
from django.contrib.auth.models import User
from whoosh_compat import FieldKind
from documents.models import CustomField
from documents.models import CustomFieldInstance
from documents.models import Document
from documents.models import Note
from documents.search._fields import PUBLIC_FIELDS
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
class TestJsonSubpathsAreWrittenAtIndexTime:
def test_every_declared_json_subpath_appears_in_the_stored_document(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document with a Note and a CustomFieldInstance attached
WHEN:
- The document is indexed via TantivyBackend.add_or_update
THEN:
- Every subpath PUBLIC_FIELDS declares for notes/custom_fields
is present as a key in the document's stored JSON payload
"""
user = User.objects.create_user(username="completeness-user")
field = CustomField.objects.create(
name="Completeness Field",
data_type=CustomField.FieldDataType.STRING,
)
doc = Document.objects.create(
title="Completeness doc",
content="x",
checksum="json-subpath-completeness",
)
Note.objects.create(document=doc, user=user, note="a note")
CustomFieldInstance.objects.create(
document=doc,
field=field,
value_text="a value",
)
backend.add_or_update(doc)
index = backend._index
searcher = index.searcher()
hits = searcher.search(
tantivy.Query.term_query(index.schema, "id", doc.pk),
limit=1,
).hits
assert hits, "the document was not indexed"
stored = searcher.doc(hits[0][1]).to_dict()
json_fields = [f for f in PUBLIC_FIELDS if f.kind is FieldKind.JSON]
assert json_fields, "no JSON fields declared - fixture is stale"
for field_spec in json_fields:
stored_values = stored.get(field_spec.name)
assert stored_values, (
f"{field_spec.name} was not written to the index at all"
)
written_keys = stored_values[0].keys()
for subpath in field_spec.subpaths:
assert subpath in written_keys, (
f"{field_spec.name}.{subpath} is declared in PUBLIC_FIELDS "
"but _backend.py's document builder never writes it - it "
"would be queryable but always empty"
)
@@ -0,0 +1,62 @@
"""Wildcard patterns on KEYWORD fields must stay literal.
``checksum`` is the only KEYWORD field: it is indexed with the raw tokenizer,
so its terms are never lowercased, folded or stemmed. Running its wildcard
patterns through the stemming normalizer rewrote hex prefixes ("ceded" ->
"cede") and returned documents whose checksum did not start with what the user
typed, which for an identity field is a wrong answer.
This covers only the registry-level normalizer, which is all that exists to
prove at this point in the stack: user queries are not yet routed through
whoosh-compat (that lands with the query-layer PR), so the same fact proven
end to end against real indexed documents lives in
``test_checksum_prefix_queries.py``.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.search._registry import get_field_registry
if TYPE_CHECKING:
from whoosh_compat import FieldRegistry
from whoosh_compat import PatternNormalizer
pytestmark = [pytest.mark.search, pytest.mark.django_db]
def _normalizer(registry: FieldRegistry, name: str) -> PatternNormalizer:
ref = registry.make_ref(name)
assert ref is not None
resolved = registry.resolve(ref)
assert resolved is not None
assert resolved.spec.pattern_normalizer is not None
return resolved.spec.pattern_normalizer
class TestKeywordPatternNormalizer:
@pytest.mark.parametrize(
"run",
[
pytest.param("ceded", id="stems_to_cede"),
pytest.param("added", id="stems_to_ad"),
pytest.param("cafed", id="stems_to_cafe"),
],
)
def test_keyword_runs_are_folded_not_stemmed(self, run: str) -> None:
"""
GIVEN:
- The "checksum" field's registered pattern normalizer
(KEYWORD kind, "en" registry)
WHEN:
- A wildcard pattern run is normalized
THEN:
- The run is returned unchanged, never widened to a stem (which
would return checksums that do not start with what the user
typed)
"""
normalize = _normalizer(get_field_registry("en"), "checksum")
assert normalize(run) == run
@@ -0,0 +1,156 @@
"""The pattern normalizer's stem-alternates contract, and its consistency
with the index-side analyzer.
Query patterns are normalized but were not stemmed, while index terms are
stemmed, so the natural spelling of a prefix search matched nothing:
``invoice*`` found no document although ``invoic*`` did. v2's index was
UNSTEMMED (whoosh ``TEXT()`` defaults to ``StandardAnalyzer``), so this
regressed against both baselines.
These are pure unit tests against ``_make_pattern_normalizer`` and
``stem_pattern_text`` directly, no query routing involved. The end-to-end
proof that a real wildcard query actually reaches a stemmed index term
lives in ``test_pattern_stemming.py``.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.search._registry import _make_pattern_normalizer
from documents.search._tokenizer import ascii_fold
from documents.search._tokenizer import paperless_text_analyzer
from documents.search._tokenizer import stem_pattern_text
if TYPE_CHECKING:
from whoosh_compat import PatternNormalizer
class TestStemsMatchTheIndexAnalyzer:
"""stem_pattern_text rebuilds paperless_text_analyzer's stemming tail rather
than sharing it, so a filter added to the index analyzer alone would silently
stop patterns from reaching the terms it produces.
"""
@pytest.mark.parametrize(
"language",
["en", "de", "fr", "es", "sv", None, "klingon"],
)
@pytest.mark.parametrize(
"word",
["Copies", "copyright", "Companies", "Invoices", "laufen", "casas", "Straße"],
)
def test_stem_equals_the_index_term(self, word: str, language: str | None) -> None:
"""
GIVEN:
- A word, across several representative index languages
("en", "de", "fr", "es", "sv"), no language, and an
unsupported language ("klingon")
WHEN:
- `stem_pattern_text` (the pattern-side stemmer) processes the
folded word, and `paperless_text_analyzer` (the index-side
analyzer) independently processes the same word
THEN:
- The two produce the identical term. `stem_pattern_text`
rebuilds `paperless_text_analyzer`'s stemming tail rather
than sharing it, so a filter added to the index analyzer
alone would silently stop patterns from reaching the terms
it produces; this pins the two staying in sync
"""
indexed = paperless_text_analyzer(language).analyze(word)[0]
assert stem_pattern_text(ascii_fold(word.lower()), language) == indexed
def _forms(normalize: PatternNormalizer, text: str) -> tuple[str, ...]:
"""The distinct forms a term may match, in order, the way the emitter reads
the normalizer's answer (see whoosh_compat.PatternNormalizer)."""
result = normalize(text)
if isinstance(result, str):
return (result,)
return tuple(dict.fromkeys(result))
class TestPatternNormalizer:
@pytest.mark.parametrize(
("text", "expected"),
[
("Invoice", ("invoice", "invoic")),
("companies", ("companies", "compani")),
# y -> i is a substitution, so both forms are needed: the index
# holds "librari" for "library" and "library" for "librarian".
("library", ("library", "librari")),
# A run the stemmer leaves alone collapses back to one form, so it
# costs exactly the one regex branch it did before.
("invoic", ("invoic",)),
("Universit", ("universit",)),
("Café", ("cafe",)),
],
)
def test_offers_the_typed_run_and_its_stem(
self,
text: str,
expected: tuple[str, ...],
) -> None:
"""
GIVEN:
- The "en" pattern normalizer
WHEN:
- It processes a literal run (e.g. "Invoice", "library",
"Café")
THEN:
- It returns the folded run and, where it differs, the
stemmed form, as distinct alternatives; a run the stemmer
leaves alone (e.g. "invoic") collapses back to the single
folded form. "library" needs both forms since y -> i is a
substitution: the index holds "librari" for "library" and
"library" for "librarian"
"""
assert _forms(_make_pattern_normalizer("en"), text) == expected
def test_run_that_yields_no_token_falls_back_to_the_typed_run(self) -> None:
"""
GIVEN:
- The "en" pattern normalizer
WHEN:
- It processes a run past the analyzer's remove_long limit
THEN:
- The run analyzes to zero tokens, so there is no stem to
offer, and only the folded run remains
"""
over_long = "invoices" * 20
assert _forms(_make_pattern_normalizer("en"), over_long) == (over_long,)
@pytest.mark.parametrize("language", [None, "klingon"])
def test_unstemmed_language_folds_only(self, language: str | None) -> None:
"""
GIVEN:
- A pattern normalizer with no language configured, or one
this build has no stemmer for ("klingon")
WHEN:
- It processes "Invoices"
THEN:
- Only the folded form ("invoices") is offered, since with no
stemmer configured the index holds surface forms and the
pattern must keep them too
"""
assert _forms(_make_pattern_normalizer(language), "Invoices") == ("invoices",)
@pytest.mark.parametrize("char", ["a", "Z", "é"])
def test_a_single_character_collapses_to_one_folded_form(self, char: str) -> None:
"""
GIVEN:
- The "en" pattern normalizer
WHEN:
- It processes a single character
THEN:
- Exactly one, one-character form is returned. A bracket
class body is normalized one character at a time and the
answer is used only when it is a single one-character
form, so a stemmer that changed a lone character would
silently disable folding inside classes
"""
forms = _forms(_make_pattern_normalizer("en"), char)
assert len(forms) == 1
assert len(forms[0]) == 1
@@ -0,0 +1,221 @@
"""Wildcard patterns must match a stemmed index, end to end.
Query patterns are normalized but were not stemmed, while index terms are
stemmed, so the natural spelling of a prefix search matched nothing:
``invoice*`` found no document although ``invoic*`` did. v2's index was
UNSTEMMED (whoosh ``TEXT()`` defaults to ``StandardAnalyzer``), so this
regressed against both baselines.
These are end-to-end tests against a real indexed document and a real
query, proving the pattern normalizer's stem-alternates contract actually
reaches a stemmed index term. The pure unit tests against the normalizer
function itself live in ``test_pattern_normalizer.py``.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.models import Document
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
CONTENT = (
"invoice total due for electricity from both companies, "
"payments made to the university library, copies attached"
)
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
@pytest.fixture
def indexed_doc(backend: TantivyBackend) -> Document:
doc = Document.objects.create(
title="Invoice 2020 productname",
content=CONTENT,
checksum="pattern-stemming-1",
archive_serial_number=900,
)
backend.add_or_update(doc)
return doc
class TestPrefixStemming:
@pytest.mark.parametrize(
"query",
[
"invoice*",
"electricity*",
"companies*",
"payments*",
"library*",
"title:Invoice*",
],
)
def test_full_word_prefix_matches_its_stem(
self,
backend: TantivyBackend,
indexed_doc: Document,
query: str,
) -> None:
"""
GIVEN:
- A document indexed with content containing "invoice",
"electricity", "companies", "payments", "library" and title
"Invoice 2020 productname"
WHEN:
- A prefix wildcard on the full, unstemmed word is queried
(e.g. "invoice*", "title:Invoice*")
THEN:
- The document matches, since the pattern normalizer offers
the word's stem as an alternative alongside the typed run,
reaching the stemmed index term
"""
assert _matched_ids(backend, query) == {indexed_doc.id}
@pytest.mark.parametrize("query", ["invoic*", "electr*", "payment*"])
def test_already_stemmed_prefix_still_matches(
self,
backend: TantivyBackend,
indexed_doc: Document,
query: str,
) -> None:
"""
GIVEN:
- The same indexed document
WHEN:
- A prefix wildcard is typed already in its stemmed spelling
(e.g. "invoic*")
THEN:
- The document still matches, since the typed-run alternative
is itself a prefix of the stored stemmed term
"""
assert _matched_ids(backend, query) == {indexed_doc.id}
@pytest.mark.parametrize("query", ["univers*", "librar*"])
def test_partial_prefix_reaches_the_stemmed_term(
self,
backend: TantivyBackend,
indexed_doc: Document,
query: str,
) -> None:
"""
GIVEN:
- The same indexed document
WHEN:
- A prefix shorter than a whole word is queried ("univers*",
"librar*")
THEN:
- It still matches, and neither case needs the two-alternative
path to do it: measured under "en", the stemmer leaves
"librar" alone, so it has one form, and that form is a
prefix of the "librari" the index holds for "library";
"univers" stems to the *shorter* "univ", and the run as
typed and its stem are both prefixes of the "univers" the
index holds for "university". The case where the two forms
genuinely diverge, and only one of them matches, is
test_stem_substitution_reaches_both_the_inflection_and_the_compound
"""
assert _matched_ids(backend, query) == {indexed_doc.id}
def test_full_word_reaches_the_stem_but_a_fragment_of_it_does_not(
self,
backend: TantivyBackend,
indexed_doc: Document,
) -> None:
"""
GIVEN:
- The same indexed document, storing "university" as "univers"
WHEN:
- "universities*", "universit*" and "univers*" are each queried
THEN:
- "universities*" matches, since the stem of "universities" is
that same "univers"; "universit*" matches nothing, since
"universit" is a prefix of neither its own stem nor the
stored term. The alternatives widen recall without turning
a wildcard into a prefix search over the original text.
usage.md tells a reader whose `universit*` finds nothing to
shorten it to `univers*`, which matches
"""
assert _matched_ids(backend, "universities*") == {indexed_doc.id}
assert _matched_ids(backend, "universit*") == set()
assert _matched_ids(backend, "univers*") == {indexed_doc.id}
def test_pattern_past_the_stem_boundary_is_documented_not_fixed(
self,
backend: TantivyBackend,
indexed_doc: Document,
) -> None:
"""
GIVEN:
- The same indexed document, with "productname" indexed as
"productnam"
WHEN:
- "produ*name" (a pattern straddling the stem boundary) is
queried
THEN:
- It matches nothing; produ*name cannot match a stemmed
index, and usage.md must not advertise it. Pinned so the
limitation is deliberate, not accidental
"""
assert _matched_ids(backend, "produ*name") == set()
def test_stem_substitution_reaches_both_the_inflection_and_the_compound(
self,
backend: TantivyBackend,
indexed_doc: Document,
) -> None:
"""
GIVEN:
- The indexed document (containing "copies") plus a second
document titled "Copyright notice" with content "copyright
notice for the work"
WHEN:
- "copy*" and "copyright*" are each queried
THEN:
- "copy*" matches both documents, and "copyright*" matches
only the compound one. English stemming substitutes as well
as truncates: "copy" and "copies" both index as "copi",
while "copyright" keeps its literal "y". Neither form is a
prefix of the other, so no single normalized string reaches
both; the run is therefore emitted as a disjunction of the
folded and stemmed forms, and "copy*" reaches the base
word, its inflections and the compound alike
"""
compound = Document.objects.create(
title="Copyright notice",
content="copyright notice for the work",
checksum="pattern-stemming-2",
archive_serial_number=901,
)
backend.add_or_update(compound)
assert _matched_ids(backend, "copy*") == {indexed_doc.id, compound.id}
assert _matched_ids(backend, "copyright*") == {compound.id}
class TestBracketClassStillFolds:
def test_class_body_matches_case_insensitively(
self,
backend: TantivyBackend,
indexed_doc: Document,
) -> None:
"""
GIVEN:
- The indexed document, titled "Invoice 2020 productname"
WHEN:
- A bracket-class pattern mixing case is queried
("title:[IP]nvoice*")
THEN:
- It matches: the class body is folded per character, which
the alternatives contract preserves only because a lone
character stems to itself
"""
assert _matched_ids(backend, "title:[IP]nvoice*") == {indexed_doc.id}
@@ -0,0 +1,198 @@
"""Permission filtering must hold against the real indexed document shape.
Only three of the index's unsigned ``*_id`` columns are load-bearing:
``owner_id``, ``viewer_id`` and ``viewer_group_id``, all read by
build_permission_filter. The rest (correspondent/document_type/storage_path/tag
ids) were written on every document and read by nothing, and were dropped.
These tests index real Documents through the backend's own document builder and
assert result-level visibility per user, so a mistake about which columns are
load-bearing shows up as documents leaking across users rather than as a passing
unit test over a hand-built index.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from django.contrib.auth.models import Group
from django.contrib.auth.models import User
from guardian.shortcuts import assign_perm
from documents.models import Correspondent
from documents.models import Document
from documents.models import DocumentType
from documents.models import StoragePath
from documents.models import Tag
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
@pytest.fixture
def owner() -> User:
return User.objects.create_user(username="owner")
@pytest.fixture
def stranger() -> User:
return User.objects.create_user(username="stranger")
@pytest.fixture
def viewer() -> User:
return User.objects.create_user(username="viewer")
@pytest.fixture
def group_member() -> User:
user = User.objects.create_user(username="group_member")
user.groups.add(Group.objects.create(name="accounting"))
return user
class TestPermissionFilteringOnIndexedDocuments:
def test_unowned_document_is_visible_to_everyone(
self,
backend: TantivyBackend,
stranger: User,
) -> None:
"""
GIVEN:
- A document with no owner, indexed via the backend's real
document builder
WHEN:
- A stranger (no relation to the document) searches
THEN:
- The document is visible to them
"""
doc = Document.objects.create(
title="Public Invoice",
content="invoice total due",
checksum="perm-unowned",
)
backend.add_or_update(doc)
assert backend.search_ids("invoice", user=stranger) == [doc.pk]
def test_owned_document_is_visible_only_to_its_owner(
self,
backend: TantivyBackend,
owner: User,
stranger: User,
) -> None:
"""
GIVEN:
- A document owned by one user, indexed via the backend's
real document builder
WHEN:
- The owner and an unrelated stranger each search
THEN:
- The owner sees the document; the stranger does not
"""
doc = Document.objects.create(
title="Private Invoice",
content="invoice total due",
checksum="perm-owned",
owner=owner,
)
backend.add_or_update(doc)
assert backend.search_ids("invoice", user=owner) == [doc.pk]
assert backend.search_ids("invoice", user=stranger) == []
def test_explicitly_shared_document_is_visible_to_the_viewer(
self,
backend: TantivyBackend,
owner: User,
viewer: User,
stranger: User,
) -> None:
"""
GIVEN:
- A document owned by one user and explicitly shared with a
second user via guardian's view_document permission,
indexed via the backend's real document builder
WHEN:
- The shared viewer and an unrelated stranger each search
THEN:
- The viewer sees the document; the stranger does not
"""
doc = Document.objects.create(
title="Shared Invoice",
content="invoice total due",
checksum="perm-shared-user",
owner=owner,
)
assign_perm("view_document", viewer, doc)
backend.add_or_update(doc)
assert backend.search_ids("invoice", user=viewer) == [doc.pk]
assert backend.search_ids("invoice", user=stranger) == []
def test_group_shared_document_is_visible_to_group_members(
self,
backend: TantivyBackend,
owner: User,
group_member: User,
stranger: User,
) -> None:
"""
GIVEN:
- A document owned by one user and shared with a group via
guardian's view_document permission, indexed via the
backend's real document builder
WHEN:
- A member of that group and an unrelated stranger each
search
THEN:
- The group member sees the document; the stranger does not
"""
doc = Document.objects.create(
title="Group Invoice",
content="invoice total due",
checksum="perm-shared-group",
owner=owner,
)
assign_perm("view_document", group_member.groups.first(), doc)
backend.add_or_update(doc)
assert backend.search_ids("invoice", user=group_member) == [doc.pk]
assert backend.search_ids("invoice", user=stranger) == []
def test_metadata_does_not_widen_visibility(
self,
backend: TantivyBackend,
owner: User,
stranger: User,
) -> None:
"""
GIVEN:
- A document owned by one user and carrying
correspondent/document_type/storage_path/tag metadata,
indexed via the backend's real document builder
WHEN:
- The owner and an unrelated stranger each search
THEN:
- The owner sees the document; the stranger does not, since
the dropped, non-load-bearing metadata *_id columns must
not widen visibility beyond the owner_id/viewer_id/
viewer_group_id filter
"""
doc = Document.objects.create(
title="Tagged Invoice",
content="invoice total due",
checksum="perm-metadata",
owner=owner,
correspondent=Correspondent.objects.create(name="ACME"),
document_type=DocumentType.objects.create(name="Bill"),
storage_path=StoragePath.objects.create(name="Archive", path="archive/"),
)
doc.tags.add(Tag.objects.create(name="paid"))
backend.add_or_update(doc)
assert backend.search_ids("invoice", user=owner) == [doc.pk]
assert backend.search_ids("invoice", user=stranger) == []
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,198 @@
"""Negation must survive the blended query.
parse_user_query ORs an exact clause with optional fuzzy and CJK clauses.
Each of those is built from positive terms only, so unless the query's
exclusions are applied to the blend as a whole, a document the exact
clause excluded is re-admitted by whichever other clause is enabled.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.models import Document
if TYPE_CHECKING:
from pytest_django.fixtures import SettingsWrapper
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
@pytest.fixture
def fuzzy_enabled(settings: SettingsWrapper) -> None:
"""Enable the fuzzy blend clause. The threshold doubles as a minimum
score filter, so it is set to 0.0: every hit passes and the test sees
the clause's matching behaviour, not the filter's."""
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.0
class TestNegationConstrainsEveryClause:
@pytest.mark.usefixtures("fuzzy_enabled")
def test_fuzzy_clause_does_not_readmit_an_excluded_document(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- Two documents both matching a positive term, one of which
also contains a word the query excludes, with the fuzzy
blend clause enabled
WHEN:
- A query combining the positive term with a NOT exclusion is
run
THEN:
- Only the document without the excluded word is returned;
the fuzzy clause (built from positive terms only) does not
readmit the document the exact clause excluded
"""
secret = _index(
backend,
title="Invoice A",
content="invoice total secret",
checksum="neg-fuzzy-1",
)
public = _index(
backend,
title="Invoice B",
content="invoice total public",
checksum="neg-fuzzy-2",
)
assert _matched_ids(backend, "invoice") == {secret.pk, public.pk}
assert _matched_ids(backend, "invoice NOT secret") == {public.pk}
def test_cjk_clause_does_not_readmit_an_excluded_document(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- Two documents both containing a CJK run, one of which also
contains a word the query excludes
WHEN:
- A query combining the CJK term with a NOT exclusion is run
THEN:
- Only the document without the excluded word is returned;
the CJK clause legitimately carries the CJK run, so
rebuilding it from the AST cannot help here, only applying
the exclusion above the blend keeps the excluded document
out
"""
secret = _index(
backend,
title="Tokyo A",
content="東京都の秘密です secret",
checksum="neg-cjk-1",
)
public = _index(
backend,
title="Tokyo B",
content="東京都の報告書です public",
checksum="neg-cjk-2",
)
assert _matched_ids(backend, "東京") == {secret.pk, public.pk}
assert _matched_ids(backend, "東京 NOT secret") == {public.pk}
@pytest.mark.usefixtures("fuzzy_enabled")
def test_disjunctive_negation_still_admits_the_other_branch(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A document matching a positive term and also containing a
word a disjunctive NOT branch excludes, plus an unrelated
document
WHEN:
- A query of the shape "term OR NOT excluded_word" is run
THEN:
- Both documents are returned; "invoice OR NOT secret"
excludes nothing on its own, so a document matching the
left branch stays in even though it contains the excluded
word
"""
secret_invoice = _index(
backend,
title="Invoice A",
content="invoice total secret",
checksum="neg-or-1",
)
unrelated = _index(
backend,
title="Recipe",
content="flour and water",
checksum="neg-or-2",
)
assert _matched_ids(backend, "invoice OR NOT secret") == {
secret_invoice.pk,
unrelated.pk,
}
def test_a_negation_under_or_does_not_constrain_the_cjk_clause(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- Two CJK documents, one of which also contains a word an OR
branch's own NOT excludes, plus an unrelated latin document
WHEN:
- The exclusion is under a disjunctive OR branch, versus in
conjunctive position
THEN:
- Under OR, the excluded document still matches through the
CJK clause (an exclusion that is one branch's own condition
cannot be restated above the blend without dropping
documents the other branch matches, so it is left where it
is and the CJK clause stays unconstrained by it -- this
shows through here in a way it does not for latin text,
since the exact clause cannot match a CJK run at all, so
the CJK clause is the only thing matching the CJK
documents, and the excluded one comes with it)
- Under conjunctive "AND NOT", the same exclusion is hoisted
and does constrain the CJK clause, pinning the deliberate
limit of the hoist
"""
secret = _index(
backend,
title="Tokyo A",
content="東京都の秘密です secret",
checksum="neg-or-cjk-1",
)
public = _index(
backend,
title="Tokyo B",
content="東京都の報告書です public",
checksum="neg-or-cjk-2",
)
bill = _index(
backend,
title="Bill",
content="bill payment received",
checksum="neg-or-cjk-3",
)
assert _matched_ids(backend, "(東京 AND NOT secret) OR bill") == {
bill.pk,
public.pk,
secret.pk,
}
# The same exclusion in conjunctive position is hoisted, and does
# constrain the CJK clause.
assert _matched_ids(backend, "東京 AND NOT secret") == {public.pk}
+224
View File
@@ -0,0 +1,224 @@
from collections.abc import Sequence
import pytest
from whoosh_compat import FieldKind
from whoosh_compat import FieldRegistry
from whoosh_compat.fields import ResolvedField
from documents.search._fields import PUBLIC_FIELDS
from documents.search._registry import get_field_registry
@pytest.fixture
def registry() -> FieldRegistry:
return get_field_registry(None)
def _resolve(registry: FieldRegistry, name: str) -> ResolvedField:
ref = registry.make_ref(name)
assert ref is not None, f"{name} is not a valid field ref"
resolved = registry.resolve(ref)
assert resolved is not None, f"{name} did not resolve"
return resolved
def _distinct_forms(result: str | Sequence[str]) -> tuple[str, ...]:
"""The forms a term may match, in order, the way whoosh-compat's emitter
reads a pattern_normalizer's answer: a bare str is one form, a sequence is
several, deduplicated."""
if isinstance(result, str):
return (result,)
return tuple(dict.fromkeys(result))
class TestFieldRegistry:
def test_no_queryable_field_name_ends_in_id(self) -> None:
"""
GIVEN:
- PUBLIC_FIELDS, the canonical query-syntax field table
WHEN:
- Every declared field name is inspected
THEN:
- None of them end in "_id" (internal id columns, written for
permission filtering and joins, must never reach the query
surface; checked against PUBLIC_FIELDS rather than the
registry so a leak is caught where it is declared)
"""
leaked = [f.name for f in PUBLIC_FIELDS if f.name.endswith("_id")]
assert not leaked, f"internal id fields reached the query surface: {leaked}"
def test_type_alias_resolves_to_document_type(
self,
registry: FieldRegistry,
) -> None:
"""
GIVEN:
- The field registry
WHEN:
- The alias "type" is resolved
THEN:
- It resolves to the canonical "document_type" field
"""
assert _resolve(registry, "type").spec.name == "document_type"
def test_path_alias_resolves_to_storage_path(self, registry: FieldRegistry) -> None:
"""
GIVEN:
- The field registry
WHEN:
- The alias "path" is resolved
THEN:
- It resolves to the canonical "storage_path" field
"""
assert _resolve(registry, "path").spec.name == "storage_path"
def test_notes_json_subpaths_resolve(self, registry: FieldRegistry) -> None:
"""
GIVEN:
- The field registry
WHEN:
- "notes.user" is resolved
THEN:
- It resolves to the "notes" field with json_path "user"
"""
resolved = _resolve(registry, "notes.user")
assert resolved.spec.name == "notes"
assert resolved.json_path == "user"
assert resolved.is_subpath is True
def test_custom_fields_json_subpaths_resolve(self, registry: FieldRegistry) -> None:
"""
GIVEN:
- The field registry
WHEN:
- "custom_fields.name" and "custom_fields.value" are resolved
THEN:
- Both resolve without error
"""
for raw in ("custom_fields.name", "custom_fields.value"):
_resolve(registry, raw)
def test_tag_is_comma_values(self, registry: FieldRegistry) -> None:
"""
GIVEN:
- The field registry
WHEN:
- The "tag" field is resolved
THEN:
- It is marked comma_values=True
"""
assert _resolve(registry, "tag").spec.comma_values is True
def test_correspondent_is_not_comma_values(self, registry: FieldRegistry) -> None:
"""
GIVEN:
- The field registry
WHEN:
- The "correspondent" field is resolved
THEN:
- It is not marked comma_values ("tag" is the only field that
opts in; end to end the two readings of
"correspondent:foo,bar" agree anyway, since the analyzer
splits the literal value on the comma regardless, so this is
only observable at the registry level)
"""
assert _resolve(registry, "correspondent").spec.comma_values is False
def test_created_is_date_kind(self, registry: FieldRegistry) -> None:
"""
GIVEN:
- The field registry
WHEN:
- The "created" field is resolved
THEN:
- Its kind is DATE and date_only is True
"""
resolved = _resolve(registry, "created")
assert resolved.spec.kind is FieldKind.DATE
assert resolved.spec.date_only is True
def test_analyzer_lowercases_and_ascii_folds(self, registry: FieldRegistry) -> None:
"""
GIVEN:
- The field registry with no language configured (no stemmer
in the analyzer chain)
WHEN:
- The "title" field's analyzer processes "Café"
THEN:
- It is lowercased and ASCII-folded to the single token "cafe"
"""
resolved = _resolve(registry, "title")
assert resolved.spec.analyzer is not None
assert resolved.spec.analyzer("Café") == ["cafe"]
def test_checksum_analyzer_is_identity_single_token(
self,
registry: FieldRegistry,
) -> None:
"""
GIVEN:
- The field registry
WHEN:
- The "checksum" field's analyzer (raw tokenizer, no
splitting) processes "ABC-123"
THEN:
- It is returned unchanged as a single token
"""
resolved = _resolve(registry, "checksum")
assert resolved.spec.analyzer is not None
assert resolved.spec.analyzer("ABC-123") == ["ABC-123"]
def test_pattern_normalizer_follows_the_registry_language(
self,
registry: FieldRegistry,
) -> None:
"""
GIVEN:
- A registry with no language, and a registry built for "en"
WHEN:
- The "title" field's pattern normalizer processes "Running"
THEN:
- With no language, only the folded run is offered
("running"), since the index holds surface forms
- With "en", the stem is offered too ("run"), since indexed
terms are stemmed and the pattern has to reach them
"""
resolved = _resolve(registry, "title")
assert resolved.spec.pattern_normalizer is not None
assert _distinct_forms(resolved.spec.pattern_normalizer("Running")) == (
"running",
)
resolved_en = _resolve(get_field_registry("en"), "title")
assert resolved_en.spec.pattern_normalizer is not None
assert _distinct_forms(resolved_en.spec.pattern_normalizer("Running")) == (
"running",
"run",
)
def test_registry_is_cached_per_language(self) -> None:
"""
GIVEN:
- Two calls to get_field_registry("en")
WHEN:
- Both calls are made
THEN:
- They return the same registry instance
"""
a = get_field_registry("en")
b = get_field_registry("en")
assert a is b
def test_registry_rebuilds_on_language_change(self) -> None:
"""
GIVEN:
- A call to get_field_registry("en") and a call to
get_field_registry("de")
WHEN:
- Both calls are made
THEN:
- They return different registry instances
"""
a = get_field_registry("en")
b = get_field_registry("de")
assert a is not b
+77 -7
View File
@@ -5,13 +5,19 @@ from typing import TYPE_CHECKING
import pytest
from documents.search._fields import PUBLIC_FIELDS
from documents.search._schema import SCHEMA_VERSION
from documents.search._schema import build_schema
from documents.search._schema import field_descriptors
from documents.search._schema import needs_rebuild
from documents.search._schema import schema_fingerprint
if TYPE_CHECKING:
from pathlib import Path
from pytest_django.fixtures import SettingsWrapper
import tantivy
from pytest_django.fixtures import Settings
pytestmark = pytest.mark.search
@@ -25,18 +31,24 @@ class TestNeedsRebuild:
def test_returns_false_when_version_and_language_match(
self,
index_dir: Path,
settings: SettingsWrapper,
settings: Settings,
) -> None:
settings.SEARCH_LANGUAGE = "en"
(index_dir / ".index_settings.json").write_text(
json.dumps({"schema_version": SCHEMA_VERSION, "language": "en"}),
json.dumps(
{
"schema_version": SCHEMA_VERSION,
"language": "en",
"schema_fingerprint": schema_fingerprint(),
},
),
)
assert needs_rebuild(index_dir) is False
def test_returns_true_on_schema_version_mismatch(
self,
index_dir: Path,
settings: SettingsWrapper,
settings: Settings,
) -> None:
settings.SEARCH_LANGUAGE = None
(index_dir / ".index_settings.json").write_text(
@@ -47,7 +59,7 @@ class TestNeedsRebuild:
def test_returns_true_when_version_is_not_an_integer(
self,
index_dir: Path,
settings: SettingsWrapper,
settings: Settings,
) -> None:
settings.SEARCH_LANGUAGE = None
(index_dir / ".index_settings.json").write_text(
@@ -58,7 +70,7 @@ class TestNeedsRebuild:
def test_returns_true_when_language_key_missing(
self,
index_dir: Path,
settings: SettingsWrapper,
settings: Settings,
) -> None:
settings.SEARCH_LANGUAGE = "en"
(index_dir / ".index_settings.json").write_text(
@@ -69,10 +81,68 @@ class TestNeedsRebuild:
def test_returns_true_when_language_differs(
self,
index_dir: Path,
settings: SettingsWrapper,
settings: Settings,
) -> None:
settings.SEARCH_LANGUAGE = "de"
(index_dir / ".index_settings.json").write_text(
json.dumps({"schema_version": SCHEMA_VERSION, "language": "en"}),
)
assert needs_rebuild(index_dir) is True
def _schema_fields(schema: tantivy.Schema) -> dict[str, dict]:
"""{name: field-state} for every field declared on a tantivy Schema.
tantivy-py 0.26 exposes no public introspection API on Schema (no
__iter__, get_field, to_json, etc.) -- __reduce__() (used internally for
pickling) is the only way to recover the field list, so we lean on it
here for test assertions only.
"""
state = schema.__reduce__()[1][0]
return {field["name"]: field for field in state["inner"]}
class TestSchemaMatchesPublicFields:
def test_every_public_field_is_in_the_schema(self) -> None:
"""
GIVEN:
- PUBLIC_FIELDS and the tantivy schema built by build_schema()
WHEN:
- Every field declared in PUBLIC_FIELDS is checked against the
schema
THEN:
- Each one is present as a field in the built schema
"""
schema = build_schema()
schema_field_names = set(_schema_fields(schema))
for field in PUBLIC_FIELDS:
assert field.name in schema_field_names, (
f"{field.name} is in PUBLIC_FIELDS but missing from build_schema()"
)
class TestFastFlagAgreement:
def test_every_public_field_fast_flag_matches_the_built_schema(self) -> None:
"""
GIVEN:
- PUBLIC_FIELDS and field_descriptors() (the latter is exactly
the input build_schema()'s SchemaBuilder consumes for the
`fast` kwarg on every field kind, so it pins the agreement
without depending on a private tantivy-py pickled
representation)
WHEN:
- Every PUBLIC_FIELDS entry's fast flag is compared against
field_descriptors()' fast flag for the same field
THEN:
- They agree for every field, catching a fast=True
PUBLIC_FIELDS entry the builder silently ignores here
instead of at a user's field:* existence query, which
whoosh-compat's registry trusts PUBLIC_FIELDS' fast flag to
resolve
"""
descriptor_fast = {d.name: d.fast for d in field_descriptors()}
for public_field in PUBLIC_FIELDS:
assert descriptor_fast[public_field.name] == public_field.fast, (
f"{public_field.name}: PUBLIC_FIELDS says fast={public_field.fast} but"
f" field_descriptors() says fast={descriptor_fast[public_field.name]}"
)
@@ -0,0 +1,587 @@
"""The schema fingerprint stamped into .index_settings.json.
tantivy compares schemas by *ordered* field list, and `tantivy.Index(schema,
path=...)` (what every write path does) raises on any difference. SCHEMA_VERSION
is the manual guard against that, but build_schema() is edited for *parser*
reasons - adding an alias, flipping fast=True, adding a subpath - by people not
thinking about the on-disk index, and forgetting the bump is exactly how this
branch's bug happened.
The fingerprint is the automatic guard: it hashes the field descriptor list that
build_schema() itself iterates, so any change to a field's name, kind, options
or *position* forces a rebuild on its own.
"""
from __future__ import annotations
import hashlib
import json
from typing import TYPE_CHECKING
import pytest
import tantivy
from documents.search import _schema
from documents.search._schema import SCHEMA_VERSION
from documents.search._schema import FieldDescriptor
from documents.search._schema import _write_sentinels
from documents.search._schema import build_schema
from documents.search._schema import field_descriptors
from documents.search._schema import needs_rebuild
from documents.search._schema import schema_fingerprint
if TYPE_CHECKING:
from pathlib import Path
from pytest_django.fixtures import SettingsWrapper
pytestmark = pytest.mark.search
# The on-disk field layout of a v2 index, pinned as data. Any edit here is an
# index-format change: it must come with a rebuild, which the fingerprint now
# forces automatically. Reproduced from build_schema()'s output as it stood
# before the descriptor refactor, so it also pins that the refactor changed
# nothing.
PINNED_DESCRIPTORS: tuple[FieldDescriptor, ...] = (
FieldDescriptor("id", "u64", stored=True, indexed=True, fast=True, tokenizer=None),
FieldDescriptor(
"title",
"text",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
FieldDescriptor(
"content",
"text",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
FieldDescriptor(
"correspondent",
"text",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
FieldDescriptor(
"document_type",
"text",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
FieldDescriptor(
"storage_path",
"text",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
FieldDescriptor(
"original_filename",
"text",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
FieldDescriptor(
"tag",
"text",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
FieldDescriptor(
"checksum",
"text",
stored=True,
indexed=True,
fast=False,
tokenizer="raw",
),
FieldDescriptor("asn", "u64", stored=True, indexed=True, fast=True, tokenizer=None),
FieldDescriptor(
"page_count",
"u64",
stored=True,
indexed=True,
fast=True,
tokenizer=None,
),
FieldDescriptor(
"num_notes",
"u64",
stored=True,
indexed=True,
fast=True,
tokenizer=None,
),
FieldDescriptor(
"created",
"date",
stored=True,
indexed=True,
fast=True,
tokenizer=None,
),
FieldDescriptor(
"modified",
"date",
stored=True,
indexed=True,
fast=True,
tokenizer=None,
),
FieldDescriptor(
"added",
"date",
stored=True,
indexed=True,
fast=True,
tokenizer=None,
),
FieldDescriptor(
"notes",
"json",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
FieldDescriptor(
"notes_text",
"text",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
FieldDescriptor(
"custom_fields",
"json",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
FieldDescriptor(
"title_sort",
"text",
stored=False,
indexed=True,
fast=True,
tokenizer="simple_analyzer",
),
FieldDescriptor(
"correspondent_sort",
"text",
stored=False,
indexed=True,
fast=True,
tokenizer="simple_analyzer",
),
FieldDescriptor(
"type_sort",
"text",
stored=False,
indexed=True,
fast=True,
tokenizer="simple_analyzer",
),
FieldDescriptor(
"bigram_content",
"text",
stored=False,
indexed=True,
fast=False,
tokenizer="bigram_analyzer",
),
FieldDescriptor(
"bigram_title",
"text",
stored=False,
indexed=True,
fast=False,
tokenizer="bigram_analyzer",
),
FieldDescriptor(
"bigram_correspondent",
"text",
stored=False,
indexed=True,
fast=False,
tokenizer="bigram_analyzer",
),
FieldDescriptor(
"bigram_document_type",
"text",
stored=False,
indexed=True,
fast=False,
tokenizer="bigram_analyzer",
),
FieldDescriptor(
"bigram_tag",
"text",
stored=False,
indexed=True,
fast=False,
tokenizer="bigram_analyzer",
),
FieldDescriptor(
"simple_title",
"text",
stored=False,
indexed=True,
fast=False,
tokenizer="simple_search_analyzer",
),
FieldDescriptor(
"simple_content",
"text",
stored=False,
indexed=True,
fast=False,
tokenizer="simple_search_analyzer",
),
FieldDescriptor(
"autocomplete_word",
"text",
stored=False,
indexed=True,
fast=False,
tokenizer="raw",
),
FieldDescriptor(
"owner_id",
"u64",
stored=False,
indexed=True,
fast=True,
tokenizer=None,
),
FieldDescriptor(
"viewer_id",
"u64",
stored=False,
indexed=True,
fast=True,
tokenizer=None,
),
FieldDescriptor(
"viewer_group_id",
"u64",
stored=False,
indexed=True,
fast=True,
tokenizer=None,
),
)
def _schema_fields(schema: tantivy.Schema) -> list[dict]:
"""The tantivy-level field list, in declaration order.
tantivy-py 0.26 exposes no public introspection API on Schema, so
__reduce__() (its pickling hook) is the only way to recover the field list.
It is used here, in a test, precisely because it is the representation the
persisted fingerprint must NOT depend on.
"""
return schema.__reduce__()[1][0]["inner"]
def _sentinels(index_dir: Path, **overrides: object) -> None:
data = {
"schema_version": SCHEMA_VERSION,
"language": None,
"schema_fingerprint": schema_fingerprint(),
}
data.update(overrides)
(index_dir / ".index_settings.json").write_text(json.dumps(data))
class TestDescriptorsDescribeTheBuiltSchema:
def test_descriptors_match_the_pinned_field_layout(self) -> None:
"""
GIVEN:
- PINNED_DESCRIPTORS, a frozen snapshot of the v2 on-disk field
layout, reproduced from build_schema()'s output as it stood
before the descriptor refactor
WHEN:
- field_descriptors() is called
THEN:
- It matches the pinned layout exactly, in the same order,
pinning that the refactor changed nothing
"""
assert tuple(field_descriptors()) == PINNED_DESCRIPTORS
def test_built_schema_matches_the_descriptors(self) -> None:
"""
GIVEN:
- The schema built by build_schema()
WHEN:
- Its fields are read back via __reduce__() (schema.__reduce__(),
tantivy-py's pickling hook)
THEN:
- Every field's name, kind, stored/fast flags and tokenizer
match what field_descriptors() declared as input; the
descriptors are not a parallel description, they are the
input, so a descriptor edit cannot claim a shape the
SchemaBuilder did not actually build
"""
kinds = {"text": "text", "json": "json_object", "u64": "u64", "date": "date"}
built = [
(
field["name"],
field["type"],
field["options"]["stored"],
bool(field["options"].get("fast")),
(field["options"].get("indexing") or {}).get("tokenizer"),
)
for field in _schema_fields(build_schema())
]
expected = [
(
descriptor.name,
kinds[descriptor.kind],
descriptor.stored,
descriptor.fast,
descriptor.tokenizer,
)
for descriptor in field_descriptors()
]
assert built == expected
class TestFingerprintSensitivity:
def test_a_field_option_change_moves_the_fingerprint(
self,
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""
GIVEN:
- The current schema fingerprint
WHEN:
- A single field descriptor's "fast" option is changed, with
no other change
THEN:
- The fingerprint changes
"""
before = schema_fingerprint()
changed = field_descriptors()
changed[1] = changed[1]._replace(fast=True)
monkeypatch.setattr(_schema, "field_descriptors", lambda: changed)
assert schema_fingerprint() != before
def test_reordering_alone_moves_the_fingerprint(
self,
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""
GIVEN:
- The current schema fingerprint
WHEN:
- Two field descriptors are swapped, with no other change (the
original bug: same fields, different declaration order)
THEN:
- The fingerprint changes; a set- or dict-based fingerprint
would be blind to this, and tantivy would reject every write
against the existing index
"""
before = schema_fingerprint()
swapped = field_descriptors()
swapped[1], swapped[2] = swapped[2], swapped[1]
monkeypatch.setattr(_schema, "field_descriptors", lambda: swapped)
assert schema_fingerprint() != before
class TestFingerprintIsIndependentOfTantivy:
def test_a_tantivy_option_key_addition_would_not_move_it(self) -> None:
"""
GIVEN:
- The built schema's raw field list, and the same list with a
new tantivy-internal option key added (simulating a
tantivy-py upgrade)
WHEN:
- Both raw lists are hashed directly, and schema_fingerprint()
is compared against a hash of field_descriptors()
THEN:
- The raw hashes differ (hashing schema.__reduce__() would
force a global reindex on every tantivy-py upgrade), but
schema_fingerprint() is unaffected, since it hashes
field_descriptors(), never tantivy's own representation
"""
fields = _schema_fields(build_schema())
upgraded = [
{**field, "options": {**field["options"], "coerce": True}}
for field in fields
]
assert _hash(upgraded) != _hash(fields)
assert schema_fingerprint() == _fingerprint_of(field_descriptors())
def test_fingerprint_never_touches_the_schema_builder(
self,
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""
GIVEN:
- tantivy.SchemaBuilder replaced with a stand-in that raises if
constructed
WHEN:
- build_schema() is called (and raises), then
schema_fingerprint() is called again
THEN:
- schema_fingerprint() still matches its earlier value,
proving it never consults SchemaBuilder
"""
before = schema_fingerprint()
class _RemovedSchemaBuilder:
def __init__(self) -> None:
raise AssertionError("tantivy.SchemaBuilder was consulted")
monkeypatch.setattr(tantivy, "SchemaBuilder", _RemovedSchemaBuilder)
with pytest.raises(AssertionError):
build_schema()
assert schema_fingerprint() == before
def _hash(payload: object) -> str:
return hashlib.blake2b(json.dumps(payload).encode()).hexdigest()
def _fingerprint_of(descriptors: list[FieldDescriptor]) -> str:
return _hash([list(descriptor) for descriptor in descriptors])
class TestNeedsRebuildOnFingerprint:
def test_matching_fingerprint_does_not_rebuild(
self,
index_dir: Path,
settings: SettingsWrapper,
) -> None:
"""
GIVEN:
- An index directory whose sentinel file records the current
schema_fingerprint()
WHEN:
- needs_rebuild() is called
THEN:
- It returns False
"""
settings.SEARCH_LANGUAGE = None
_sentinels(index_dir)
assert needs_rebuild(index_dir) is False
def test_stale_fingerprint_rebuilds_despite_a_matching_version(
self,
index_dir: Path,
settings: SettingsWrapper,
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""
GIVEN:
- An index directory whose sentinel matches SCHEMA_VERSION,
but field_descriptors() is patched to add a field the
fingerprint never saw (schema edited, version not bumped)
WHEN:
- needs_rebuild() is called
THEN:
- It returns True; without the fingerprint check,
`reindex --if-needed` would report the index up to date and
every subsequent write would raise
"""
settings.SEARCH_LANGUAGE = None
_sentinels(index_dir)
extended = [
*field_descriptors(),
FieldDescriptor(
"new_field",
"u64",
stored=False,
indexed=True,
fast=True,
tokenizer=None,
),
]
monkeypatch.setattr(_schema, "field_descriptors", lambda: extended)
assert needs_rebuild(index_dir) is True
def test_reordered_schema_rebuilds(
self,
index_dir: Path,
settings: SettingsWrapper,
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""
GIVEN:
- An index directory whose sentinel matches the current
fingerprint, but field_descriptors() is patched to swap two
fields' order
WHEN:
- needs_rebuild() is called
THEN:
- It returns True
"""
settings.SEARCH_LANGUAGE = None
_sentinels(index_dir)
reordered = field_descriptors()
reordered[1], reordered[2] = reordered[2], reordered[1]
monkeypatch.setattr(_schema, "field_descriptors", lambda: reordered)
assert needs_rebuild(index_dir) is True
def test_missing_fingerprint_rebuilds(
self,
index_dir: Path,
settings: SettingsWrapper,
) -> None:
"""
GIVEN:
- An index directory whose sentinel has no "schema_fingerprint"
key at all
WHEN:
- needs_rebuild() is called
THEN:
- It returns True; an index whose schema shape nobody recorded
is rebuilt rather than trusted
"""
settings.SEARCH_LANGUAGE = None
(index_dir / ".index_settings.json").write_text(
json.dumps({"schema_version": SCHEMA_VERSION, "language": None}),
)
assert needs_rebuild(index_dir) is True
def test_written_sentinels_satisfy_the_check(
self,
index_dir: Path,
settings: SettingsWrapper,
) -> None:
"""
GIVEN:
- An index directory whose sentinels are written by
_write_sentinels() itself
WHEN:
- needs_rebuild() is called
THEN:
- It returns False
"""
settings.SEARCH_LANGUAGE = "en"
_write_sentinels(index_dir)
assert needs_rebuild(index_dir) is False
@@ -0,0 +1,178 @@
"""SCHEMA_VERSION must change whenever build_schema()'s field list or order does.
tantivy compares schemas by *ordered* field list. ``Index.open()`` loads the
schema from the index's own ``meta.json``, so reads against an index built by an
older release keep working after a field reorder. Writes do not:
``WriteBatch.__enter__`` calls ``tantivy.Index(build_schema(), path=...)``, an
open-or-create that raises ``ValueError`` on any schema difference. Nothing
catches that ValueError, so consumption, index_document and bulk edit all
hard-fail while ``/api/status/`` still reports the index healthy.
The only thing that saves such an install is ``needs_rebuild()`` noticing the
version stamped in ``.index_settings.json`` is stale.
"""
from __future__ import annotations
import json
from typing import TYPE_CHECKING
import pytest
import tantivy
from django.conf import settings as django_settings
from documents.search._schema import build_schema
from documents.search._schema import needs_rebuild
from documents.search._schema import open_or_rebuild_index
if TYPE_CHECKING:
from pathlib import Path
pytestmark = [pytest.mark.search]
RELEASED_V1_SCHEMA_VERSION = 1
def _build_released_v1_schema() -> tantivy.Schema:
"""Frozen copy of build_schema() as shipped in v3.0.x (schema version 1).
Deliberately duplicated rather than imported: it must keep describing the
on-disk layout of already-deployed indexes even as build_schema() evolves.
"""
sb = tantivy.SchemaBuilder()
sb.add_unsigned_field("id", stored=True, indexed=True, fast=True)
sb.add_text_field("checksum", stored=True, tokenizer_name="raw")
for field in (
"title",
"correspondent",
"document_type",
"storage_path",
"original_filename",
"content",
):
sb.add_text_field(field, stored=True, tokenizer_name="paperless_text")
for field in ("title_sort", "correspondent_sort", "type_sort"):
sb.add_text_field(
field,
stored=False,
tokenizer_name="simple_analyzer",
fast=True,
)
for field in (
"bigram_content",
"bigram_title",
"bigram_correspondent",
"bigram_document_type",
"bigram_tag",
):
sb.add_text_field(field, stored=False, tokenizer_name="bigram_analyzer")
for field in ("simple_title", "simple_content"):
sb.add_text_field(field, stored=False, tokenizer_name="simple_search_analyzer")
sb.add_text_field("autocomplete_word", stored=False, tokenizer_name="raw")
sb.add_text_field("tag", stored=True, tokenizer_name="paperless_text")
sb.add_json_field("notes", stored=True, tokenizer_name="paperless_text")
sb.add_text_field("notes_text", stored=True, tokenizer_name="paperless_text")
sb.add_json_field("custom_fields", stored=True, tokenizer_name="paperless_text")
for field in (
"correspondent_id",
"document_type_id",
"storage_path_id",
"tag_id",
"owner_id",
"viewer_id",
"viewer_group_id",
):
sb.add_unsigned_field(field, stored=False, indexed=True, fast=True)
for field in ("created", "modified", "added"):
sb.add_date_field(field, stored=True, indexed=True, fast=True)
for field in ("asn", "page_count", "num_notes"):
sb.add_unsigned_field(field, stored=True, indexed=True, fast=True)
return sb.build()
@pytest.fixture
def released_v1_index(tmp_path: Path) -> Path:
"""An index directory as a v3.0.x install would leave it on disk."""
index_dir = tmp_path / "index"
index_dir.mkdir()
tantivy.Index(_build_released_v1_schema(), path=str(index_dir))
(index_dir / ".index_settings.json").write_text(
json.dumps(
{
"schema_version": RELEASED_V1_SCHEMA_VERSION,
"language": django_settings.SEARCH_LANGUAGE,
},
),
)
return index_dir
class TestUpgradeFromReleasedV1Index:
def test_released_v1_index_is_flagged_for_rebuild(
self,
released_v1_index: Path,
) -> None:
"""
GIVEN:
- An index directory laid out exactly as a v3.0.x (schema
version 1) install would leave it
WHEN:
- needs_rebuild() is called
THEN:
- It returns True; if this fails,
`document_index reindex --if-needed` prints "Search index is
up to date" and skips, leaving the mismatched index in place
"""
assert needs_rebuild(released_v1_index) is True
def test_opening_a_v1_index_leaves_it_writable(
self,
released_v1_index: Path,
) -> None:
"""
GIVEN:
- A v1 index directory
WHEN:
- open_or_rebuild_index() is called against it
THEN:
- The directory can be reopened with the current schema
without raising; end to end, open_or_rebuild_index must
hand back an index the write path can reopen. Before the
version bump, needs_rebuild() returned False here, and the
stale directory survived untouched, so every subsequent
write against it raised tantivy's own schema-mismatch
ValueError
"""
open_or_rebuild_index(released_v1_index)
tantivy.Index(build_schema(), path=str(released_v1_index))
def test_rebuilt_index_is_not_rebuilt_again(
self,
released_v1_index: Path,
) -> None:
"""
GIVEN:
- A v1 index directory that has just been rebuilt by
open_or_rebuild_index()
WHEN:
- needs_rebuild() is called again
THEN:
- It returns False; the rebuild must stamp the version it
actually wrote, otherwise every startup wipes and reindexes
the whole corpus
"""
open_or_rebuild_index(released_v1_index)
assert needs_rebuild(released_v1_index) is False
@@ -0,0 +1,37 @@
from __future__ import annotations
import pytest
from documents.search._tokenizer import stem_pattern_text
pytestmark = pytest.mark.search
class TestStemPatternText:
def test_unsupported_language_returns_text_unchanged(self) -> None:
"""
GIVEN:
- A language code with no Snowball stemmer mapping
WHEN:
- A pattern run is stemmed for that language
THEN:
- The run is returned unchanged, since the stemming gate that
disables stemming for an unsupported language also disables
the pattern-side stemmer
"""
assert stem_pattern_text("running", "klingon") == "running"
def test_run_past_remove_long_limit_returns_text_unchanged(self) -> None:
"""
GIVEN:
- A supported language and a run longer than the remove_long
filter's limit (129 characters, matching Document.title's
max_length)
WHEN:
- The over-long run is stemmed
THEN:
- The remove_long filter drops the token entirely, leaving no
stem to substitute, so the run is returned unchanged
"""
long_run = "a" * 130
assert stem_pattern_text(long_run, "en") == long_run
+2 -2
View File
@@ -7,8 +7,8 @@ import pytest
import tantivy
from documents.search._tokenizer import _bigram_analyzer
from documents.search._tokenizer import _paperless_text
from documents.search._tokenizer import _simple_search_analyzer
from documents.search._tokenizer import paperless_text_analyzer
from documents.search._tokenizer import register_tokenizers
if TYPE_CHECKING:
@@ -25,7 +25,7 @@ class TestTokenizers:
sb.add_text_field("content", stored=True, tokenizer_name="paperless_text")
schema = sb.build()
idx = tantivy.Index(schema, path=None)
idx.register_tokenizer("paperless_text", _paperless_text(""))
idx.register_tokenizer("paperless_text", paperless_text_analyzer(""))
return idx
@pytest.fixture
@@ -1,810 +0,0 @@
from __future__ import annotations
from datetime import UTC
from datetime import datetime
from typing import TYPE_CHECKING
from zoneinfo import ZoneInfo
import pytest
import time_machine
from documents.search._dates import _precision_bounds
if TYPE_CHECKING:
import tantivy
from documents.search._query import _FIELD_BOOSTS
from documents.search._query import DEFAULT_SEARCH_FIELDS
from documents.search._translate import OPEN_HI
from documents.search._translate import OPEN_LO
from documents.search._translate import Comma
from documents.search._translate import FieldRange
from documents.search._translate import FieldValue
from documents.search._translate import FieldValueList
from documents.search._translate import InvalidDateQuery
from documents.search._translate import Passthrough
from documents.search._translate import resolve_commas
from documents.search._translate import scan
from documents.search._translate import translate_query
from documents.search._translate import translate_range
from documents.search._translate import translate_scalar
@pytest.mark.search
class TestPrecisionBounds:
@pytest.mark.parametrize(
("digits", "expected"),
[
("2020", ((2020, 1, 1), (2021, 1, 1))),
("202003", ((2020, 3, 1), (2020, 4, 1))),
("202012", ((2020, 12, 1), (2021, 1, 1))),
("20200115", ((2020, 1, 15), (2020, 1, 16))),
("20201231", ((2020, 12, 31), (2021, 1, 1))),
],
)
def test_valid(self, digits, expected):
lo, hi = _precision_bounds(digits)
assert (lo.year, lo.month, lo.day) == expected[0]
assert (hi.year, hi.month, hi.day) == expected[1]
@pytest.mark.parametrize("digits", ["202023", "20200230", "20201301", "20", "abcd"])
def test_invalid_returns_none(self, digits):
assert _precision_bounds(digits) is None
@pytest.mark.search
class TestScan:
def test_plain_words_are_passthrough(self):
assert scan("bank statement") == [Passthrough("bank statement")]
def test_field_value(self):
assert scan("created:2020") == [FieldValue("created", "2020")]
def test_field_value_in_boolean(self):
toks = scan("created:2020 OR foo")
assert toks == [
FieldValue("created", "2020"),
Passthrough(" OR foo"),
]
def test_field_value_in_parens(self):
toks = scan("(created:2020 OR foo)")
assert toks == [
Passthrough("("),
FieldValue("created", "2020"),
Passthrough(" OR foo)"),
]
def test_quoted_value(self):
assert scan('correspondent:"A B"') == [FieldValue("correspondent", '"A B"')]
def test_field_range(self):
assert scan("created:[2020 TO 2021]") == [
FieldRange("created", "[", "2020", "2021", "]"),
]
@pytest.mark.parametrize(
("query", "expected"),
[
pytest.param(
"created:[2020 to]",
FieldRange("created", "[", "2020", "", "]"),
id="open_upper",
),
pytest.param(
"created:[to 2020]",
FieldRange("created", "[", "", "2020", "]"),
id="open_lower",
),
],
)
def test_open_range(self, query, expected):
assert scan(query) == [expected]
def test_comma_inside_range_not_split(self):
# No depth-0 comma here; the whole thing is one range token.
toks = scan("created:[2020 TO 2021]")
assert len(toks) == 1
# --- Edge-case / regression tests (scan must never raise) ---
def test_url_is_passthrough(self):
# "http" is not a known field; the whole URL must pass through verbatim.
assert scan("http://example.com") == [Passthrough("http://example.com")]
def test_unterminated_quote_is_passthrough(self):
# title is a known field but the quoted value has no closing quote;
# _consume_value returns None so the whole string falls into passthrough.
assert scan('title:"abc') == [Passthrough('title:"abc')]
def test_unterminated_bracket_is_passthrough(self):
# created is a known field but the range bracket is never closed;
# _consume_range returns None so the whole string falls into passthrough.
assert scan("created:[2020") == [Passthrough("created:[2020")]
def test_empty_value_at_end_is_passthrough(self):
# created is a known field but there is no value after the colon
# (_consume_value returns None for start >= n), so passthrough.
assert scan("created:") == [Passthrough("created:")]
def test_value_containing_colon(self):
# The bare-word value reader stops at whitespace/paren, not at colon,
# so "2020:30" is consumed as a single value token.
assert scan("created:2020:30") == [FieldValue("created", "2020:30")]
def test_comma_followed_by_unconsumable_value_stops(self):
# A comma followed by whitespace is neither a value-list continuation nor a
# clause separator: the value stops and the comma stays as passthrough.
assert scan("tag:foo, bar") == [
FieldValue("tag", "foo"),
Passthrough(", bar"),
]
def test_bracket_without_to_is_open_upper_bound(self):
# A bracketed value with no TO falls back to (value, "") -> open upper bound.
assert scan("created:[2020]") == [
FieldRange("created", "[", "2020", "", "]"),
]
def test_known_field_name_midword_is_passthrough(self):
# A known field name embedded mid-word is not a field token (the
# word-boundary guard); the whole run stays passthrough.
assert scan("xtag:foo") == [Passthrough("xtag:foo")]
@pytest.mark.search
class TestCommaResolution:
def test_value_list_multi_value_field(self):
toks = resolve_commas(scan("tag:foo,bar"))
assert toks == [FieldValueList("tag", ("foo", "bar"))]
def test_value_list_three(self):
toks = resolve_commas(scan("tag_id:1,2,3"))
assert toks == [FieldValueList("tag_id", ("1", "2", "3"))]
def test_text_field_comma_is_literal(self):
# correspondent is not multi-value: comma stays inside the value.
toks = resolve_commas(scan("correspondent:foo,bar"))
assert toks == [FieldValue("correspondent", "foo,bar")]
def test_clause_separator_before_known_field(self):
toks = resolve_commas(scan("tag:foo,type:bar"))
assert toks == [FieldValue("tag", "foo"), Comma(), FieldValue("type", "bar")]
def test_clause_separator_after_range(self):
toks = resolve_commas(scan("created:[2020 TO 2021],added:[2022 TO 2023]"))
assert toks == [
FieldRange("created", "[", "2020", "2021", "]"),
Comma(),
FieldRange("added", "[", "2022", "2023", "]"),
]
def test_clause_separator_after_quote(self):
toks = resolve_commas(scan('correspondent:"A B",created:[2020 TO 2021]'))
assert toks == [
FieldValue("correspondent", '"A B"'),
Comma(),
FieldRange("created", "[", "2020", "2021", "]"),
]
def test_url_comma_is_literal_passthrough(self):
toks = resolve_commas(scan("http://example.com/a,b"))
assert toks == [Passthrough("http://example.com/a,b")]
def test_non_multi_value_comma_is_literal(self):
# title is not in MULTI_VALUE_FIELDS: comma stays inside the value.
toks = resolve_commas(scan("title:10,20"))
assert toks == [FieldValue("title", "10,20")]
def test_clause_separator_before_known_date_field(self):
# The comma between a bare value and a known date field acts as a
# clause separator; both sides survive as distinct tokens.
toks = resolve_commas(scan("correspondent:foo,created:[2020 TO 2021]"))
assert toks == [
FieldValue("correspondent", "foo"),
Comma(),
FieldRange("created", "[", "2020", "2021", "]"),
]
@pytest.mark.search
class TestTranslateScalar:
@pytest.mark.parametrize(
("field", "value", "expected"),
[
(
"created",
"2020",
"created:[2020-01-01T00:00:00Z TO 2021-01-01T00:00:00Z}",
),
(
"created",
"202003",
"created:[2020-03-01T00:00:00Z TO 2020-04-01T00:00:00Z}",
),
(
"created",
"20200115",
"created:[2020-01-15T00:00:00Z TO 2020-01-16T00:00:00Z}",
),
(
"created",
"2020-01-15",
"created:[2020-01-15T00:00:00Z TO 2020-01-16T00:00:00Z}",
),
(
"created",
"2020-03",
"created:[2020-03-01T00:00:00Z TO 2020-04-01T00:00:00Z}",
),
],
)
def test_partial_and_iso_dates(self, field: str, value: str, expected: str) -> None:
assert translate_scalar(field, value, UTC) == expected
def test_invalid_date_raises(self) -> None:
with pytest.raises(InvalidDateQuery) as exc_info:
translate_scalar("created", "202023", UTC)
assert exc_info.value.field == "created"
assert exc_info.value.value == "202023"
def test_keyword_delegates(self) -> None:
# keyword path produces a half-open range; just assert it is a created range
out = translate_scalar("created", "today", UTC)
assert out.startswith("created:[") and out.endswith("}")
def test_14digit_compact_datetime(self) -> None:
out = translate_scalar("created", "20240115120000", UTC)
assert "20240115120000" not in out
assert out.startswith("created:")
assert out == "created:[2024-01-15T12:00:00Z TO 2024-01-15T12:00:00Z]"
def test_14digit_invalid_month_raises(self) -> None:
with pytest.raises(InvalidDateQuery) as exc_info:
translate_scalar("created", "20231300120000", UTC)
assert exc_info.value.field == "created"
assert exc_info.value.value == "20231300120000"
def test_unrecognized_value_raises(self) -> None:
# A value that is not a keyword, digits, ISO date, or compact timestamp
# raises rather than producing invalid Tantivy syntax or silently matching
# nothing.
with pytest.raises(InvalidDateQuery) as exc_info:
translate_scalar("created", "garbage", UTC)
assert exc_info.value.field == "created"
assert exc_info.value.value == "garbage"
@pytest.mark.search
class TestTranslateRange:
@pytest.mark.parametrize(
("lo", "hi", "expected"),
[
("2005", "2009", "created:[2005-01-01T00:00:00Z TO 2010-01-01T00:00:00Z}"),
(
"202001",
"202006",
"created:[2020-01-01T00:00:00Z TO 2020-07-01T00:00:00Z}",
),
(
"20200101",
"20201231",
"created:[2020-01-01T00:00:00Z TO 2021-01-01T00:00:00Z}",
),
(
"2020-01-01",
"2020-12-31",
"created:[2020-01-01T00:00:00Z TO 2021-01-01T00:00:00Z}",
),
],
)
def test_absolute_ranges(self, lo, hi, expected):
assert translate_range("created", lo, hi, UTC) == expected
def test_reversed_swaps(self):
assert translate_range("created", "2009", "2005", UTC) == (
"created:[2005-01-01T00:00:00Z TO 2010-01-01T00:00:00Z}"
)
def test_open_upper(self):
out = translate_range("created", "2020", "", UTC)
assert out == f"created:[2020-01-01T00:00:00Z TO {OPEN_HI}]"
def test_open_lower(self):
out = translate_range("created", "", "2020", UTC)
assert out == f"created:[{OPEN_LO} TO 2021-01-01T00:00:00Z}}"
def test_invalid_bound_raises(self):
with pytest.raises(InvalidDateQuery) as exc_info:
translate_range("created", "202023", "2025", UTC)
assert exc_info.value.field == "created"
assert exc_info.value.value == "202023"
def test_invalid_high_bound_raises(self):
# Low bound parses, high bound does not -> raise on the high bound.
with pytest.raises(InvalidDateQuery) as exc_info:
translate_range("created", "2020", "garbage", UTC)
assert exc_info.value.field == "created"
assert exc_info.value.value == "garbage"
@pytest.mark.search
class TestTranslateQuery:
@pytest.mark.parametrize(
("raw", "expected"),
[
(
"created:2020",
"created:[2020-01-01T00:00:00Z TO 2021-01-01T00:00:00Z}",
),
("tag:foo,bar", "tag:foo AND tag:bar"),
# 'type' is a user-facing alias rewritten to 'document_type' (the real schema field)
("tag:foo,type:bar", "tag:foo AND document_type:bar"),
(
"created:[2020 TO 2021],added:[2022 TO 2023]",
(
"created:[2020-01-01T00:00:00Z TO 2022-01-01T00:00:00Z}"
" AND "
"added:[2022-01-01T00:00:00Z TO 2024-01-01T00:00:00Z}"
),
),
# correspondent is not multi-value: comma stays literal inside the value
("correspondent:foo,bar", "correspondent:foo,bar"),
],
)
def test_golden(self, raw: str, expected: str) -> None:
assert translate_query(raw, UTC) == expected
@pytest.mark.parametrize(
"raw",
[
"created:2020",
"created:202003",
"created:[20200101 TO 20201231]",
"created:[2020-01-01 TO 2020-12-31]",
"created:[2020 to]",
"created:[to 2020]",
"title:x,created:[2020 TO 2021]",
"created:2020 OR foo",
"(created:2020 OR invoice)",
"tag:foo,type:bar",
"bank statement",
],
)
def test_parse_acceptance(self, index: tantivy.Index, raw: str) -> None:
translated = translate_query(raw, UTC)
# Must not raise:
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
@pytest.mark.search
class TestFieldAliasing:
"""Whoosh->Tantivy field-name aliasing (type/path -> document_type/storage_path)."""
def test_type_alias(self) -> None:
assert translate_query("type:invoice", UTC) == "document_type:invoice"
def test_path_alias(self) -> None:
assert translate_query("path:/foo/bar", UTC) == "storage_path:/foo/bar"
def test_type_id_alias(self) -> None:
assert translate_query("type_id:5", UTC) == "document_type_id:5"
def test_path_id_alias(self) -> None:
assert translate_query("path_id:7", UTC) == "storage_path_id:7"
def test_clause_separator_plus_alias(self) -> None:
# Comma between known fields acts as AND separator; alias still applied.
assert (
translate_query("tag:foo,type:bar", UTC) == "tag:foo AND document_type:bar"
)
def test_type_range_alias(self) -> None:
# type is not a date field; range passes through verbatim with alias applied.
assert (
translate_query("type:[2020 TO 2021]", UTC)
== "document_type:[2020 TO 2021]"
)
def test_parse_acceptance_type(self, index: tantivy.Index) -> None:
# Translated output must be accepted by the real Tantivy parser.
translated = translate_query("type:invoice", UTC)
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
def test_parse_acceptance_path(self, index: tantivy.Index) -> None:
translated = translate_query("path:foo", UTC)
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
# Freeze time so relative-date tests are deterministic.
_FROZEN_NOW = datetime(2026, 3, 28, 12, 0, 0, tzinfo=UTC)
@pytest.mark.search
class TestRelativeRanges:
"""Relative date-range tokens resolved against a frozen clock."""
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_minus_7_days_to_now(self) -> None:
assert translate_query("added:[-7 days to now]", UTC) == (
"added:[2026-03-21T12:00:00Z TO 2026-03-28T12:00:00Z]"
)
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_minus_1_week_to_now(self) -> None:
assert translate_query("added:[-1 week to now]", UTC) == (
"added:[2026-03-21T12:00:00Z TO 2026-03-28T12:00:00Z]"
)
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_minus_1_month_to_now(self) -> None:
assert translate_query("created:[-1 month to now]", UTC) == (
"created:[2026-02-28T12:00:00Z TO 2026-03-28T12:00:00Z]"
)
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_minus_1_year_to_now(self) -> None:
assert translate_query("modified:[-1 year to now]", UTC) == (
"modified:[2025-03-28T12:00:00Z TO 2026-03-28T12:00:00Z]"
)
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_minus_3_hours_to_now(self) -> None:
assert translate_query("added:[-3 hours to now]", UTC) == (
"added:[2026-03-28T09:00:00Z TO 2026-03-28T12:00:00Z]"
)
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_uppercase_units(self) -> None:
assert translate_query("added:[-1 WEEK TO NOW]", UTC) == (
"added:[2026-03-21T12:00:00Z TO 2026-03-28T12:00:00Z]"
)
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_now_minus_7d_compact(self) -> None:
assert translate_query("added:[now-7d TO now]", UTC) == (
"added:[2026-03-21T12:00:00Z TO 2026-03-28T12:00:00Z]"
)
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_reversed_range_swapped(self) -> None:
# now+1h TO now-1h is reversed; translate_range swaps -> lo=now-1h, hi=now+1h
assert translate_query("added:[now+1h TO now-1h]", UTC) == (
"added:[2026-03-28T11:00:00Z TO 2026-03-28T13:00:00Z]"
)
@pytest.mark.parametrize(
"raw",
[
"added:[-7 days to now]",
"added:[-1 week to now]",
"created:[-1 month to now]",
"modified:[-1 year to now]",
"added:[-3 hours to now]",
"added:[now-7d TO now]",
"added:[now+1h TO now-1h]",
],
)
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_parse_acceptance(self, index: tantivy.Index, raw: str) -> None:
translated = translate_query(raw, UTC)
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
@pytest.mark.search
class TestWhooshUnitAbbreviations:
"""
Whoosh's PlusMinus date grammar accepted abbreviated unit spellings
(e.g. "yrs", "mos", "wks", "hrs", "mins", "secs"); saved views/searches
created under the old Whoosh backend can contain those tokens (see
https://github.com/paperless-ngx/paperless-ngx/issues/13482), so the
Tantivy translator must still accept them.
"""
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_minus_999_yrs(self) -> None:
assert translate_query("created:[-999yrs to now]", UTC) == (
"created:[1027-03-28T12:00:00Z TO 2026-03-28T12:00:00Z]"
)
@pytest.mark.parametrize(
("token", "expected_lo"),
[
("-1y", "2025-03-28T12:00:00Z"),
("-1yr", "2025-03-28T12:00:00Z"),
("-3mos", "2025-12-28T12:00:00Z"),
("-3mo", "2025-12-28T12:00:00Z"),
("-2wks", "2026-03-14T12:00:00Z"),
("-2wk", "2026-03-14T12:00:00Z"),
("-5dys", "2026-03-23T12:00:00Z"),
("-5dy", "2026-03-23T12:00:00Z"),
("-1hrs", "2026-03-28T11:00:00Z"),
("-1hr", "2026-03-28T11:00:00Z"),
("-10mins", "2026-03-28T11:50:00Z"),
("-10min", "2026-03-28T11:50:00Z"),
("-30secs", "2026-03-28T11:59:30Z"),
("-30sec", "2026-03-28T11:59:30Z"),
],
)
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_abbreviated_units(self, token: str, expected_lo: str) -> None:
assert translate_query(f"added:[{token} to now]", UTC) == (
f"added:[{expected_lo} TO 2026-03-28T12:00:00Z]"
)
@pytest.mark.parametrize(
"raw",
[
"created:[-999yrs to now]",
"added:[-1y to now]",
"created:[-3mos to now]",
"added:[-2wks to now]",
"added:[-5dys to now]",
"added:[-1hrs to now]",
"added:[-10mins to now]",
"added:[-30secs to now]",
],
)
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_parse_acceptance(self, index: tantivy.Index, raw: str) -> None:
translated = translate_query(raw, UTC)
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
@pytest.mark.search
class TestOperatorNormalization:
"""Post-render operator normalization in translate_query."""
def test_spaced_dash_removed(self) -> None:
assert (
translate_query("H52.1 - Kurzsichtigkeit", UTC) == "H52.1 Kurzsichtigkeit"
)
def test_spaced_dash_simple(self) -> None:
assert translate_query("bar - baz", UTC) == "bar baz"
def test_trailing_operator_stripped(self) -> None:
assert translate_query("foo -", UTC) == "foo"
def test_date_range_preserved(self) -> None:
out = translate_query("created:[2020 TO 2021]", UTC)
# Must not corrupt the ISO range
assert out == "created:[2020-01-01T00:00:00Z TO 2022-01-01T00:00:00Z}"
def test_date_scalar_with_or(self) -> None:
out = translate_query("created:2020 OR foo", UTC)
# The created scalar becomes a range; " OR foo" passes through verbatim.
assert out.startswith("created:[")
assert "OR foo" in out
def test_parse_acceptance_spaced_dash(self, index: tantivy.Index) -> None:
translated = translate_query("H52.1 - Kurzsichtigkeit", UTC)
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
def test_parse_acceptance_trailing_op(self, index: tantivy.Index) -> None:
translated = translate_query("foo -", UTC)
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
@pytest.mark.search
class TestMultiWordDateKeywords:
"""scan() must consume multi-word date keywords as a single value."""
def test_scan_previous_week_as_single_token(self) -> None:
# "created:previous week" must produce one FieldValue with value "previous week",
# not FieldValue("created","previous") + Passthrough(" week").
toks = scan("created:previous week")
assert toks == [FieldValue("created", "previous week")]
def test_scan_this_month_as_single_token(self) -> None:
toks = scan("added:this month")
assert toks == [FieldValue("added", "this month")]
def test_scan_previous_month_as_single_token(self) -> None:
toks = scan("created:previous month")
assert toks == [FieldValue("created", "previous month")]
def test_scan_this_year_as_single_token(self) -> None:
toks = scan("added:this year")
assert toks == [FieldValue("added", "this year")]
def test_scan_previous_year_as_single_token(self) -> None:
toks = scan("created:previous year")
assert toks == [FieldValue("created", "previous year")]
def test_scan_previous_quarter_as_single_token(self) -> None:
toks = scan("created:previous quarter")
assert toks == [FieldValue("created", "previous quarter")]
def test_quoted_multi_word_keyword_still_works(self) -> None:
# The quoted form must continue to work as before.
toks = scan('created:"previous week"')
assert toks == [FieldValue("created", '"previous week"')]
def test_non_date_field_not_affected(self) -> None:
# "previous" stops at the space for non-date fields; " week" passes through.
toks = scan("correspondent:previous week")
assert toks == [
FieldValue("correspondent", "previous"),
Passthrough(" week"),
]
@pytest.mark.search
class TestKeywordDateResolution:
"""Relative date keywords resolve to exact ISO ranges against a frozen clock.
Frozen at 2026-03-28 12:00 UTC (a Saturday in Q1) so the week, month,
quarter and year rollovers are all exercised by a single anchor.
"""
# created is a DateField: bounds are UTC midnight, no timezone offset.
@pytest.mark.parametrize(
("keyword", "expected"),
[
pytest.param(
"today",
"created:[2026-03-28T00:00:00Z TO 2026-03-29T00:00:00Z}",
id="today",
),
pytest.param(
"yesterday",
"created:[2026-03-27T00:00:00Z TO 2026-03-28T00:00:00Z}",
id="yesterday",
),
pytest.param(
"previous week",
"created:[2026-03-16T00:00:00Z TO 2026-03-23T00:00:00Z}",
id="previous-week",
),
pytest.param(
"this month",
"created:[2026-03-01T00:00:00Z TO 2026-04-01T00:00:00Z}",
id="this-month",
),
pytest.param(
"previous month",
"created:[2026-02-01T00:00:00Z TO 2026-03-01T00:00:00Z}",
id="previous-month",
),
pytest.param(
"this year",
"created:[2026-01-01T00:00:00Z TO 2027-01-01T00:00:00Z}",
id="this-year",
),
pytest.param(
"previous year",
"created:[2025-01-01T00:00:00Z TO 2026-01-01T00:00:00Z}",
id="previous-year",
),
pytest.param(
"previous quarter",
"created:[2025-10-01T00:00:00Z TO 2026-01-01T00:00:00Z}",
id="previous-quarter",
),
],
)
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_date_only_field_keyword_ranges(
self,
keyword: str,
expected: str,
) -> None:
assert translate_query(f"created:{keyword}", UTC) == expected
# added is a DateTimeField: local-tz midnight converted to UTC. Tokyo
# (+09:00, no DST) shifts each midnight boundary back to 15:00Z the day
# before, so this also exercises the local-midnight offset path.
@pytest.mark.parametrize(
("keyword", "expected"),
[
pytest.param(
"today",
"added:[2026-03-27T15:00:00Z TO 2026-03-28T15:00:00Z}",
id="today",
),
pytest.param(
"yesterday",
"added:[2026-03-26T15:00:00Z TO 2026-03-27T15:00:00Z}",
id="yesterday",
),
pytest.param(
"previous week",
"added:[2026-03-15T15:00:00Z TO 2026-03-22T15:00:00Z}",
id="previous-week",
),
pytest.param(
"this month",
"added:[2026-02-28T15:00:00Z TO 2026-03-31T15:00:00Z}",
id="this-month",
),
pytest.param(
"previous month",
"added:[2026-01-31T15:00:00Z TO 2026-02-28T15:00:00Z}",
id="previous-month",
),
pytest.param(
"this year",
"added:[2025-12-31T15:00:00Z TO 2026-12-31T15:00:00Z}",
id="this-year",
),
pytest.param(
"previous year",
"added:[2024-12-31T15:00:00Z TO 2025-12-31T15:00:00Z}",
id="previous-year",
),
pytest.param(
"previous quarter",
"added:[2025-09-30T15:00:00Z TO 2025-12-31T15:00:00Z}",
id="previous-quarter",
),
],
)
@time_machine.travel(_FROZEN_NOW, tick=False)
def test_datetime_field_keyword_ranges_local_tz(
self,
keyword: str,
expected: str,
) -> None:
assert translate_query(f"added:{keyword}", ZoneInfo("Asia/Tokyo")) == expected
@pytest.mark.search
class TestISODatetimeBounds:
"""Full ISO datetime tokens in range bounds must be parsed directly."""
def test_translate_range_iso_bounds_passthrough(self) -> None:
# Already-ISO datetime bounds must pass through as-is (exact instant).
result = translate_range(
"created",
"2020-01-01T00:00:00Z",
"2021-01-01T00:00:00Z",
UTC,
)
assert result == "created:[2020-01-01T00:00:00Z TO 2021-01-01T00:00:00Z]"
def test_translate_query_iso_range_preserved(self) -> None:
q = "created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
assert translate_query(q, UTC) == q
def test_translate_query_comma_separated_iso_ranges(self) -> None:
q = (
"created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z],"
"added:[2026-05-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
)
result = translate_query(q, UTC)
assert result == (
"created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
" AND "
"added:[2026-05-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
)
def test_translate_query_text_before_comma_separated_date_clause(self) -> None:
result = translate_query("schäfersee,created:previous year", UTC)
assert result == (
"schäfersee AND created:[2025-01-01T00:00:00Z TO 2026-01-01T00:00:00Z}"
)
def test_invalid_iso_datetime_raises(self) -> None:
# A token with "T" that is not valid ISO datetime -> raise.
with pytest.raises(InvalidDateQuery) as exc_info:
translate_range(
"created",
"2020-01-01T99:00:00Z",
"2021-01-01T00:00:00Z",
UTC,
)
assert exc_info.value.field == "created"
assert exc_info.value.value == "2020-01-01T99:00:00Z"
def test_parse_acceptance_iso_bounds(self, index: tantivy.Index) -> None:
q = "created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
translated = translate_query(q, UTC)
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
def test_parse_acceptance_comma_iso_ranges(self, index: tantivy.Index) -> None:
q = (
"created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z],"
"added:[2026-05-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
)
translated = translate_query(q, UTC)
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
+49 -2
View File
@@ -35,7 +35,8 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
THEN:
- Existing config
"""
response = self.client.get(self.ENDPOINT, format="json")
with patch.dict("os.environ", {}, clear=True):
response = self.client.get(self.ENDPOINT, format="json")
self.assertEqual(response.status_code, status.HTTP_200_OK)
@@ -45,6 +46,7 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
response.data[0],
{
"id": 1,
"externally_configured_variables": [],
"output_type": None,
"pages": None,
"language": None,
@@ -76,7 +78,7 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
"remote_ocr_api_key": None,
"remote_ocr_endpoint": None,
"remote_ocr_mode": None,
"ai_enabled": False,
"ai_enabled": None,
"llm_embedding_backend": None,
"llm_embedding_model": None,
"llm_embedding_endpoint": None,
@@ -91,6 +93,31 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
},
)
def test_api_get_config_reports_external_configuration_without_values(self) -> None:
with patch.dict(
"os.environ",
{
"PAPERLESS_OCR_LANGUAGE": "eng",
"PAPERLESS_REMOTE_OCR_API_KEY": "secret-value",
"PAPERLESS_FUTURE_SETTING": "future-value",
"UNRELATED_SETTING": "unrelated-value",
},
clear=True,
):
response = self.client.get(self.ENDPOINT, format="json")
self.assertCountEqual(
response.data[0]["externally_configured_variables"],
[
"PAPERLESS_FUTURE_SETTING",
"PAPERLESS_OCR_LANGUAGE",
"PAPERLESS_REMOTE_OCR_API_KEY",
],
)
self.assertNotContains(response, "secret-value")
self.assertNotContains(response, "future-value")
self.assertNotContains(response, "UNRELATED_SETTING")
def test_api_get_ui_settings_with_config(self) -> None:
"""
GIVEN:
@@ -949,6 +976,26 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
)
mock_update.assert_called_once()
@override_settings(AI_ENABLED=True, LLM_EMBEDDING_BACKEND=None)
def test_external_ai_setting_triggers_index_update(self) -> None:
config = ApplicationConfiguration.objects.first()
assert config is not None
config.ai_enabled = None
config.llm_embedding_backend = None
config.save()
with (
patch("documents.tasks.llmindex_index.apply_async") as mock_update,
patch("paperless.views.llm_index_exists", return_value=False),
):
self.client.patch(
f"{self.ENDPOINT}1/",
json.dumps({"llm_embedding_backend": "openai-like"}),
content_type="application/json",
)
mock_update.assert_called_once()
def test_update_llm_embedding_chunk_size_triggers_rebuild(self) -> None:
config = ApplicationConfiguration.objects.first()
assert config is not None
@@ -4,6 +4,7 @@ import json
import shutil
import zipfile
from django.contrib.auth.models import Permission
from django.contrib.auth.models import User
from django.test import override_settings
from django.utils import timezone
@@ -326,6 +327,9 @@ class TestBulkDownload(DirectoriesMixin, SampleDirMixin, APITestCase):
def test_download_insufficient_permissions(self) -> None:
user = User.objects.create_user(username="temp_user")
user.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
self.client.force_authenticate(user=user)
self.doc2.owner = self.user
@@ -339,3 +343,29 @@ class TestBulkDownload(DirectoriesMixin, SampleDirMixin, APITestCase):
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
self.assertEqual(response.content, b"Insufficient permissions")
def test_bad_search_query_returns_400(self) -> None:
"""
GIVEN:
- Bulk download request selects documents via a saved-search
query filter
WHEN:
- The query contains a malformed field value (an invalid date)
THEN:
- The response is a 400 naming the bad value, exactly like the
search list endpoint, never a 500
"""
response = self.client.post(
self.ENDPOINT,
json.dumps(
{
"all": True,
"filters": {"query": "added:notadate"},
"content": "originals",
},
),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn(b"notadate", response.content)
+135 -1
View File
@@ -717,6 +717,44 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
self.assertEqual(args[0], [self.doc2.id])
self.assertEqual(kwargs["storage_path"], self.sp1.id)
@mock.patch("documents.serialisers.bulk_edit.set_storage_path")
def test_api_bulk_edit_with_all_true_resolves_owned_duplicates(self, m) -> None:
self.setup_mock(m, "set_storage_path")
user = User.objects.create_user(username="duplicate-owner")
user.user_permissions.add(
Permission.objects.get(codename="change_document"),
)
first_duplicate = Document.objects.create(
checksum="owned-duplicate",
title="First duplicate",
owner=user,
)
second_duplicate = Document.objects.create(
checksum="owned-duplicate",
title="Second duplicate",
owner=user,
)
self.client.force_authenticate(user=user)
response = self.client.post(
"/api/documents/bulk_edit/",
json.dumps(
{
"all": True,
"filters": {"has_duplicates": True},
"method": "set_storage_path",
"parameters": {"storage_path": self.sp1.id},
},
),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_200_OK)
m.assert_called_once()
args, kwargs = m.call_args
self.assertCountEqual(args[0], [first_duplicate.id, second_duplicate.id])
self.assertEqual(kwargs["storage_path"], self.sp1.id)
@mock.patch("documents.search.get_backend")
@mock.patch("documents.serialisers.bulk_edit.set_storage_path")
def test_api_bulk_edit_with_all_true_resolves_documents_from_search_filters(
@@ -1046,6 +1084,8 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
user1 = User.objects.create(username="user1")
self.client.force_authenticate(user=user1)
assign_perm("view_document", user1, self.doc2)
response = self.client.post(
"/api/documents/selection_data/",
json.dumps({"documents": [self.doc2.id]}),
@@ -1053,7 +1093,18 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
)
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
self.assertEqual(response.content, b"Insufficient permissions")
user1.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
user1 = User.objects.get(pk=user1.pk)
self.client.force_authenticate(user=user1)
response = self.client.post(
"/api/documents/selection_data/",
json.dumps({"documents": [self.doc2.id]}),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_200_OK)
@mock.patch("documents.serialisers.bulk_edit.set_permissions")
def test_set_permissions(self, m) -> None:
@@ -1598,6 +1649,40 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
def test_legacy_bulk_edit_rejects_out_of_bounds_pdf_doc_index(self) -> None:
response = self.client.post(
"/api/documents/bulk_edit/",
json.dumps(
{
"documents": [self.doc2.id],
"method": "edit_pdf",
"parameters": {
"operations": [{"page": 1, "doc": 2**32}],
},
},
),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn(b"doc index is out of bounds", response.content)
def test_legacy_bulk_edit_rejects_empty_pdf_operations(self) -> None:
response = self.client.post(
"/api/documents/bulk_edit/",
json.dumps(
{
"documents": [self.doc2.id],
"method": "edit_pdf",
"parameters": {"operations": []},
},
),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn(b"operations must not be empty", response.content)
@mock.patch("documents.views.bulk_edit.edit_pdf")
def test_edit_pdf(self, m) -> None:
self.setup_mock(m, "edit_pdf")
@@ -1648,6 +1733,13 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn(b"Expected a list of items", response.content)
response = self.client.post(
"/api/documents/edit_pdf/",
{"documents": [self.doc2.id], "operations": []},
format="json",
)
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
response = self.client.post(
"/api/documents/edit_pdf/",
json.dumps(
@@ -1700,6 +1792,21 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn(b"doc must be an integer", response.content)
for doc_index in (-1, 2**32):
with self.subTest(doc_index=doc_index):
response = self.client.post(
"/api/documents/edit_pdf/",
json.dumps(
{
"documents": [self.doc2.id],
"operations": [{"page": 1, "doc": doc_index}],
},
),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn(b"doc index is out of bounds", response.content)
response = self.client.post(
"/api/documents/edit_pdf/",
json.dumps(
@@ -2021,3 +2128,30 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
self.assertEqual(response.status_code, status.HTTP_200_OK)
self.assertEqual(LogEntry.objects.filter(object_pk=self.doc1.id).count(), 2)
def test_api_bulk_edit_with_bad_search_query_returns_400(self) -> None:
"""
GIVEN:
- Bulk edit request selects documents via a saved-search query
filter
WHEN:
- The query contains a malformed field value (an invalid date)
THEN:
- The response is a 400 naming the bad value, exactly like the
search list endpoint, never a 500
"""
response = self.client.post(
"/api/documents/bulk_edit/",
json.dumps(
{
"all": True,
"filters": {"query": "added:notadate"},
"method": "set_storage_path",
"parameters": {"storage_path": self.sp1.id},
},
),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn(b"notadate", response.content)
+36
View File
@@ -38,6 +38,42 @@ class TestChatStreamingViewInputValidation(APITestCase):
)
assert resp.status_code == status.HTTP_400_BAD_REQUEST
def test_answer_is_not_compressed(self) -> None:
"""
GIVEN:
- A client that accepts compressed responses
WHEN:
- It asks the chat endpoint a question
THEN:
- The answer is streamed unencoded, chunk for chunk
The stream compressors buffer, so a compressed answer arrives in one
piece. The view cannot opt out by flagging the request: DRF's request
wrapper proxies reads but keeps writes to itself, so the flag never
reaches the Django request the middleware sees.
"""
chunks = [f"token{i} " for i in range(40)]
with (
mock.patch(
"documents.views.AIConfig",
return_value=self._mock_ai_enabled(),
),
mock.patch(
"documents.views.stream_chat_with_documents",
return_value=iter(chunks),
),
):
resp = self.client.post(
"/api/documents/chat/",
{"q": "What is in my archive?"},
format="json",
HTTP_ACCEPT_ENCODING="gzip, deflate, br, zstd",
)
assert resp.status_code == status.HTTP_200_OK
assert not resp.has_header("Content-Encoding")
assert list(resp.streaming_content) == [c.encode() for c in chunks]
def test_missing_question_is_rejected(self) -> None:
with mock.patch(
"documents.views.AIConfig",
@@ -2,14 +2,12 @@ from __future__ import annotations
import datetime
from typing import TYPE_CHECKING
from unittest import TestCase
from unittest import mock
from auditlog.models import LogEntry # type: ignore[import-untyped]
from django.contrib.auth.models import Permission
from django.contrib.auth.models import User
from django.contrib.contenttypes.models import ContentType
from django.core.exceptions import FieldError
from django.core.files.uploadedfile import SimpleUploadedFile
from django.test import TestCase as DjangoTestCase
from django.utils import timezone
@@ -22,6 +20,7 @@ from documents.filters import TitleContentFilter
from documents.models import Document
from documents.tests.utils import DirectoriesMixin
from documents.tests.utils import read_streaming_response
from documents.versioning import annotate_effective_content
from documents.views import DocumentSelectionMixin
if TYPE_CHECKING:
@@ -598,6 +597,7 @@ class TestDocumentVersioningApi(DirectoriesMixin, APITestCase):
self.assertEqual(input_doc.root_document_id, root.id)
self.assertEqual(input_doc.source, DocumentSource.ApiUpload)
self.assertEqual(overrides.version_label, "New Version")
self.assertEqual(overrides.owner_id, self.user.id)
self.assertEqual(overrides.actor_id, self.user.id)
def test_update_version_with_version_pk_normalizes_to_root(self) -> None:
@@ -891,32 +891,104 @@ class TestDocumentVersioningApi(DirectoriesMixin, APITestCase):
)
class TestVersionAwareFilters(TestCase):
def test_title_content_filter_falls_back_to_content(self) -> None:
queryset = mock.Mock()
fallback_queryset = mock.Mock()
queryset.filter.side_effect = [FieldError("missing field"), fallback_queryset]
class TestVersionAwareFilters(DjangoTestCase):
"""
The filters annotate effective_content themselves rather than relying on
the caller's queryset carrying it, so they stay version-aware on a plain
Document queryset (e.g. the bulk-edit "select all matching" path).
"""
result = TitleContentFilter().filter(queryset, " latest ")
def setUp(self) -> None:
super().setUp()
self.root = Document.objects.create(
title="root",
checksum="root",
mime_type="application/pdf",
content="superseded-content",
)
Document.objects.create(
title="version",
checksum="version",
mime_type="application/pdf",
root_document=self.root,
version_index=1,
content="latest-content",
)
self.unversioned = Document.objects.create(
title="unversioned",
checksum="unversioned",
mime_type="application/pdf",
content="latest-content",
)
self.assertIs(result, fallback_queryset)
self.assertEqual(queryset.filter.call_count, 2)
def test_effective_content_filter_falls_back_to_content_lookup(self) -> None:
queryset = mock.Mock()
fallback_queryset = mock.Mock()
queryset.filter.side_effect = [FieldError("missing field"), fallback_queryset]
result = EffectiveContentFilter(lookup_expr="icontains").filter(
queryset,
def test_title_content_filter_matches_latest_version_content(self) -> None:
result = TitleContentFilter().filter(
Document.objects.filter(root_document__isnull=True),
" latest ",
)
self.assertIs(result, fallback_queryset)
first_kwargs = queryset.filter.call_args_list[0].kwargs
second_kwargs = queryset.filter.call_args_list[1].kwargs
self.assertEqual(first_kwargs, {"effective_content__icontains": "latest"})
self.assertEqual(second_kwargs, {"content__icontains": "latest"})
self.assertCountEqual(
[doc.id for doc in result],
[self.root.id, self.unversioned.id],
)
def test_effective_content_filter_matches_latest_version_content(self) -> None:
result = EffectiveContentFilter(lookup_expr="icontains").filter(
Document.objects.filter(root_document__isnull=True),
" latest ",
)
self.assertCountEqual(
[doc.id for doc in result],
[self.root.id, self.unversioned.id],
)
def test_effective_content_filter_ignores_superseded_content(self) -> None:
result = EffectiveContentFilter(lookup_expr="icontains").filter(
Document.objects.filter(root_document__isnull=True),
"superseded",
)
self.assertEqual(list(result), [])
def test_filters_reuse_an_existing_annotation(self) -> None:
"""
Annotating twice under the same alias is an error, so an already
annotated queryset (the search path) has to be left alone.
"""
annotated = annotate_effective_content(
Document.objects.filter(root_document__isnull=True),
)
self.assertIs(annotate_effective_content(annotated), annotated)
result = EffectiveContentFilter(lookup_expr="icontains").filter(
annotated,
"latest",
)
self.assertCountEqual(
[doc.id for doc in result],
[self.root.id, self.unversioned.id],
)
def test_bulk_selection_does_not_match_superseded_content(self) -> None:
"""
Bulk edit's "select all matching" builds its own queryset, so before
the filters annotated for themselves it matched the root document's
superseded content -- selecting documents the list view, filtered by
the same term, does not show.
"""
user = User.objects.create_superuser(username="bulk_selection")
selected = DocumentSelectionMixin()._resolve_document_ids(
user=user,
validated_data={
"all": True,
"filters": {"content__icontains": "superseded"},
},
)
self.assertEqual(selected, [])
def test_effective_content_filter_returns_input_for_empty_values(self) -> None:
queryset = mock.Mock()
+186
View File
@@ -981,6 +981,128 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
self.assertEqual(len(results), 1)
self.assertEqual(results[0]["id"], doc.id)
def test_has_duplicates_filter(self) -> None:
original_match = Document.objects.create(
title="original match",
checksum="same-original",
)
second_original_match = Document.objects.create(
title="second original match",
checksum="same-original",
)
archive_match = Document.objects.create(
title="archive match",
checksum="archive-source",
archive_checksum="same-archive",
)
original_to_archive_match = Document.objects.create(
title="original to archive match",
checksum="same-archive",
)
first_archive_match = Document.objects.create(
title="first archive match",
checksum="first-archive-source",
archive_checksum="same-archive-only",
)
second_archive_match = Document.objects.create(
title="second archive match",
checksum="second-archive-source",
archive_checksum="same-archive-only",
)
first_empty_archive = Document.objects.create(
title="first empty archive",
checksum="first-empty-archive",
archive_checksum="",
)
second_empty_archive = Document.objects.create(
title="second empty archive",
checksum="second-empty-archive",
archive_checksum="",
)
unique = Document.objects.create(title="unique", checksum="unique")
version_root = Document.objects.create(
title="version root",
checksum="version-root",
)
Document.objects.create(
title="version",
checksum=unique.checksum,
root_document=version_root,
version_index=1,
)
trash_match = Document.objects.create(
title="trash match",
checksum="trash-match",
)
trashed_duplicate = Document.objects.create(
title="trashed duplicate",
checksum="trash-match",
)
trashed_duplicate.delete()
response = self.client.get("/api/documents/?has_duplicates=true")
self.assertEqual(response.status_code, status.HTTP_200_OK)
self.assertCountEqual(
[document["id"] for document in response.data["results"]],
[
original_match.id,
second_original_match.id,
archive_match.id,
original_to_archive_match.id,
first_archive_match.id,
second_archive_match.id,
trash_match.id,
],
)
response = self.client.get("/api/documents/?has_duplicates=false")
self.assertEqual(response.status_code, status.HTTP_200_OK)
self.assertCountEqual(
[document["id"] for document in response.data["results"]],
[
unique.id,
version_root.id,
first_empty_archive.id,
second_empty_archive.id,
],
)
response = self.client.get(f"/api/documents/{first_empty_archive.id}/")
self.assertEqual(response.data["duplicate_documents"], [])
def test_has_duplicates_filter_respects_document_permissions(self) -> None:
owner = User.objects.create_user(username="duplicate-owner")
requester = User.objects.create_user(username="duplicate-requester")
requester.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
visible_document = Document.objects.create(
title="visible document",
checksum="permission-match",
owner=requester,
)
hidden_duplicate = Document.objects.create(
title="hidden duplicate",
checksum="permission-match",
owner=owner,
)
self.client.force_authenticate(user=requester)
response = self.client.get("/api/documents/?has_duplicates=true")
self.assertNotIn(
visible_document.id,
[document["id"] for document in response.data["results"]],
)
assign_perm("view_document", requester, hidden_duplicate)
response = self.client.get("/api/documents/?has_duplicates=true")
self.assertIn(
visible_document.id,
[document["id"] for document in response.data["results"]],
)
def test_custom_fields_icontains_filter_no_duplicates(self) -> None:
"""
GIVEN:
@@ -3493,6 +3615,55 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
self.assertEqual(response.content, b"Insufficient permissions to delete notes")
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
def test_notes_require_global_document_permissions(self) -> None:
user = User.objects.create_user(username="note_editor")
user.user_permissions.add(
*Permission.objects.filter(
codename__in=["view_note", "add_note", "delete_note"],
),
)
doc = Document.objects.create(
title="test",
mime_type="application/pdf",
content="notes",
owner=user,
)
note = Note.objects.create(note="Existing", document=doc, user=user)
self.client.force_authenticate(user)
response = self.client.get(f"/api/documents/{doc.pk}/notes/")
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
user.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
user = User.objects.get(pk=user.pk)
self.client.force_authenticate(user)
response = self.client.get(f"/api/documents/{doc.pk}/notes/")
self.assertEqual(response.status_code, status.HTTP_200_OK)
response = self.client.post(
f"/api/documents/{doc.pk}/notes/",
data={"note": "New"},
)
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
user.user_permissions.add(
Permission.objects.get(codename="change_document"),
)
user = User.objects.get(pk=user.pk)
self.client.force_authenticate(user)
response = self.client.post(
f"/api/documents/{doc.pk}/notes/",
data={"note": "New"},
)
self.assertEqual(response.status_code, status.HTTP_200_OK)
response = self.client.delete(
f"/api/documents/{doc.pk}/notes/?id={note.pk}",
)
self.assertEqual(response.status_code, status.HTTP_200_OK)
def test_delete_note(self) -> None:
"""
GIVEN:
@@ -3891,6 +4062,21 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
assign_perm("view_document", user1, doc)
create_resp = self.client.post(
"/api/share_links/",
data={
"document": doc.pk,
"file_version": "original",
},
format="json",
)
self.assertEqual(create_resp.status_code, status.HTTP_403_FORBIDDEN)
user1.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
user1 = User.objects.get(pk=user1.pk)
self.client.force_authenticate(user1)
create_resp = self.client.post(
"/api/share_links/",
data={
+97
View File
@@ -2,10 +2,15 @@ import datetime
import json
from unittest import mock
from django.contrib.auth.models import Group
from django.contrib.auth.models import Permission
from django.contrib.auth.models import User
from django.db import connection
from django.test import override_settings
from django.test.utils import CaptureQueriesContext
from guardian.shortcuts import assign_perm
from guardian.shortcuts import get_groups_with_perms
from guardian.shortcuts import get_users_with_perms
from rest_framework import status
from rest_framework.test import APITestCase
@@ -452,6 +457,9 @@ class TestApiStoragePaths(DirectoriesMixin, APITestCase):
def test_test_storage_path_requires_document_view_permission(self) -> None:
owner = User.objects.create_user(username="owner")
unprivileged = User.objects.create_user(username="unprivileged")
unprivileged.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
document = Document.objects.create(
mime_type="application/pdf",
owner=owner,
@@ -483,6 +491,23 @@ class TestApiStoragePaths(DirectoriesMixin, APITestCase):
)
assign_perm("view_document", viewer, document)
self.client.force_authenticate(user=viewer)
response = self.client.post(
f"{self.ENDPOINT}test/",
json.dumps(
{
"document": document.id,
"path": "path/{{ title }}",
},
),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
viewer.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
viewer = User.objects.get(pk=viewer.pk)
self.client.force_authenticate(user=viewer)
response = self.client.post(
f"{self.ENDPOINT}test/",
@@ -525,6 +550,9 @@ class TestApiStoragePaths(DirectoriesMixin, APITestCase):
password="password",
email="owner@example.com",
)
owner.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
document = Document.objects.create(
mime_type="application/pdf",
owner=owner,
@@ -600,6 +628,9 @@ class TestApiStoragePaths(DirectoriesMixin, APITestCase):
checksum="123",
)
assign_perm("view_document", viewer, document)
viewer.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
self.client.force_authenticate(user=viewer)
response = self.client.post(
@@ -687,6 +718,9 @@ class TestApiStoragePaths(DirectoriesMixin, APITestCase):
)
document.tags.add(private_tag)
assign_perm("view_document", viewer, document)
viewer.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
self.client.force_authenticate(user=viewer)
response = self.client.post(
@@ -740,6 +774,9 @@ class TestApiStoragePaths(DirectoriesMixin, APITestCase):
value_int=42,
)
assign_perm("view_document", viewer, document)
viewer.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
self.client.force_authenticate(user=viewer)
response = self.client.post(
@@ -842,6 +879,66 @@ class TestBulkEditObjects(APITestCase):
self.assertEqual(response.status_code, status.HTTP_200_OK)
self.assertEqual(StoragePath.objects.count(), 0)
def test_bulk_objects_set_permissions_batched_across_object_count(
self,
) -> None:
"""
GIVEN:
- Many tags are being bulk-edited to set permissions at once
WHEN:
- bulk_edit_objects API endpoint is called with set_permissions
operation over a small batch vs. a much larger one
THEN:
- Permissions are applied correctly at both scales
- Query count does not grow with the number of tags, i.e. each
user/group is applied across all tags with one batched call
rather than one call per (tag, identity) pair
"""
group1 = Group.objects.create(name="perm-group")
permissions = {
"view": {"users": [self.user1.id, self.user2.id], "groups": [group1.id]},
"change": {"users": [self.user1.id], "groups": [group1.id]},
}
def run_with_n_tags(n: int) -> int:
tags = [Tag.objects.create(name=f"perm-tag-{n}-{i}") for i in range(n)]
with CaptureQueriesContext(connection) as ctx:
response = self.client.post(
"/api/bulk_edit_objects/",
json.dumps(
{
"objects": [t.id for t in tags],
"object_type": "tags",
"operation": "set_permissions",
"permissions": permissions,
"merge": False,
},
),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_200_OK)
for tag in tags:
self.assertEqual(get_users_with_perms(tag).count(), 2)
self.assertEqual(get_groups_with_perms(tag).count(), 1)
return len(ctx.captured_queries)
small_batch_queries = run_with_n_tags(5)
large_batch_queries = run_with_n_tags(50)
# A tolerance rather than equality, matching the N+1 check in
# test_views.py: bulk_create's batch_size caps rows per INSERT, so a
# large enough selection does legitimately add statements, and the
# per-process ContentType cache makes the first run carry an extra
# query. Neither can hide a regression to per-object assignment,
# which would be ~10x the small-batch count here.
self.assertLessEqual(
large_batch_queries,
small_batch_queries + 5,
"Permission assignment appears to scale with object count: "
f"{small_batch_queries} queries for 5 tags vs. "
f"{large_batch_queries} for 50",
)
def test_bulk_objects_delete_all_filtered(self) -> None:
"""
GIVEN:
+144
View File
@@ -786,6 +786,10 @@ class TestDocumentSearchApi(DirectoriesMixin, APITestCase):
tick=False,
):
response = self.client.get("/api/documents/?query=added:previous month")
assert response.status_code == 200, (
f"expected a successful search response, got {response.status_code}: "
f"{response.data!r}"
)
results = response.data["results"]
self.assertEqual(len(results), 1)
@@ -818,6 +822,26 @@ class TestDocumentSearchApi(DirectoriesMixin, APITestCase):
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn("invalid-date", str(response.data["query"]))
def test_search_multiple_bad_fields_returns_all_messages(self) -> None:
"""
GIVEN:
- One document added
WHEN:
- Query with multiple bad fields (e.g. invalid date and invalid number)
THEN:
- 400 Bad Request with error messages for every bad field,
so the user can fix them all in one round-trip
"""
response = self.client.get(
"/api/documents/",
{"query": "created:notadate AND asn:notanumber"},
)
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
messages = response.data["query"]
self.assertEqual(len(messages), 2)
self.assertTrue(any("created" in m for m in messages))
self.assertTrue(any("asn" in m for m in messages))
@override_settings(
TIME_ZONE="UTC",
)
@@ -861,6 +885,29 @@ class TestDocumentSearchApi(DirectoriesMixin, APITestCase):
results = response.data["results"]
self.assertEqual({r["id"] for r in results}, {1, 2})
@mock.patch("documents.search._backend.parse_user_query")
def test_search_parser_bug_surfaces_as_500_not_400(self, m) -> None:
"""
GIVEN:
- The query parser itself fails (a whoosh-compat bug, per
QueryParserError's own contract: not user-fixable input)
WHEN:
- Any search request runs
THEN:
- The error surfaces as a 500 monitoring can see, never a 400
blaming the user for a library defect
"""
from whoosh_compat.errors import QueryParserError
m.side_effect = QueryParserError("synthetic parser bug")
self.client.raise_request_exception = False
response = self.client.get("/api/documents/?query=anything")
self.assertEqual(
response.status_code,
status.HTTP_500_INTERNAL_SERVER_ERROR,
)
@mock.patch("documents.search._backend.TantivyBackend.autocomplete")
def test_search_autocomplete_limits(self, m) -> None:
"""
@@ -1947,6 +1994,29 @@ class TestDocumentSearchApi(DirectoriesMixin, APITestCase):
self.assertEqual(len(response.data["documents"]), 1)
self.assertEqual(response.data["documents"][0]["id"], title_match.id)
def test_global_search_returns_latest_version_content(self) -> None:
root = Document.objects.create(
title="bank statement",
content="superseded content",
checksum="GSV1",
pk=23,
)
Document.objects.create(
title="bank statement v2",
content="latest content",
checksum="GSV2",
pk=24,
root_document=root,
version_index=1,
)
self.client.force_authenticate(self.user)
response = self.client.get("/api/search/?query=bank&db_only=true")
self.assertEqual(response.status_code, status.HTTP_200_OK)
returned = {doc["id"]: doc["content"] for doc in response.data["documents"]}
self.assertEqual(returned.get(root.id), "latest content")
def test_global_search_filters_owned_mail_objects(self) -> None:
user1 = User.objects.create_user("mail-search-user")
user2 = User.objects.create_user("other-mail-search-user")
@@ -2035,3 +2105,77 @@ class TestDocumentSearchApi(DirectoriesMixin, APITestCase):
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
response = self.client.get("/api/search/?query=no")
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
def _assert_query_finds(self, doc: Document, query: str) -> None:
get_backend().add_or_update(doc)
response = self.client.get("/api/documents/", {"query": query})
self.assertEqual(response.status_code, status.HTTP_200_OK)
ids = [r["id"] for r in response.data["results"]]
self.assertIn(doc.id, ids)
def test_search_by_asn(self) -> None:
"""
GIVEN:
- A document with an archive serial number, indexed
WHEN:
- A query filters by "asn:<value>"
THEN:
- The document is found
"""
doc = Document.objects.create(
title="Has ASN",
content="content",
checksum="asn-checksum",
archive_serial_number=555,
)
self._assert_query_finds(doc, "asn:555")
def test_search_by_page_count(self) -> None:
"""
GIVEN:
- A document with a page count, indexed
WHEN:
- A query filters by "page_count:<value>"
THEN:
- The document is found
"""
doc = Document.objects.create(
title="Multi-page",
content="content",
checksum="page-count-checksum",
page_count=42,
)
self._assert_query_finds(doc, "page_count:42")
def test_search_by_original_filename(self) -> None:
"""
GIVEN:
- A document with an original filename, indexed
WHEN:
- A query filters by "original_filename:<value>"
THEN:
- The document is found
"""
doc = Document.objects.create(
title="Named file",
content="content",
checksum="filename-checksum",
original_filename="quarterly-report.pdf",
)
self._assert_query_finds(doc, "original_filename:quarterly-report.pdf")
def test_search_by_checksum(self) -> None:
"""
GIVEN:
- A document with a checksum, indexed
WHEN:
- A query filters by "checksum:<value>"
THEN:
- The document is found
"""
doc = Document.objects.create(
title="Checksum doc",
content="content",
checksum="deadbeef1234",
)
self._assert_query_finds(doc, "checksum:deadbeef1234")
@@ -0,0 +1,344 @@
"""The search list endpoint's exception handling: what becomes a 400 and
what a library defect surfaces as instead.
Companion to documents/tests/search/test_error_routing.py, which pins the
Cause -> SearchQueryError/QueryError routing inside documents/search/_query.py.
These tests pin the layer above it: DocumentViewSet.list's own except clauses,
which decide what an already-routed error becomes on the wire.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from rest_framework import status
from whoosh_compat.errors import Cause
from whoosh_compat.errors import Diagnostic
from whoosh_compat.errors import DiagnosticKind
from whoosh_compat.errors import QueryError
from documents.search import SearchQueryError
from documents.tests.factories import DocumentFactory
if TYPE_CHECKING:
from rest_framework.test import APIClient
from documents.models import Document
pytestmark = [pytest.mark.django_db, pytest.mark.usefixtures("_search_index")]
@pytest.fixture
def indexed_document() -> Document:
from documents.search import get_backend
doc = DocumentFactory.create(title="quarterly invoice", content="acme corp")
get_backend().add_or_update(doc)
return doc
class TestSearchQueryErrorStillBecomesA400:
def test_search_query_error_becomes_a_400_naming_the_field(
self,
admin_client: APIClient,
monkeypatch: pytest.MonkeyPatch,
indexed_document: Document,
) -> None:
"""
GIVEN:
- parse_user_query() raising a SearchQueryError naming a field
WHEN:
- The document list endpoint is queried
THEN:
- The response is a 400 whose body names the field
"""
import documents.search._backend as backend_mod
def raise_search_query_error(*args: object, **kwargs: object) -> object:
raise SearchQueryError("bad value for field 'added'")
monkeypatch.setattr(
backend_mod,
"parse_user_query",
raise_search_query_error,
)
response = admin_client.get("/api/documents/?query=anything")
assert response.status_code == status.HTTP_400_BAD_REQUEST
assert "added" in str(response.data["query"])
class TestLibraryDefectsPropagate:
"""The exact regression this task exists to fix: an unexpected or
INTERNAL-cause library error must not be relabeled a 400."""
def test_unexpected_exception_is_not_converted_to_a_400(
self,
admin_client: APIClient,
monkeypatch: pytest.MonkeyPatch,
indexed_document: Document,
) -> None:
"""
GIVEN:
- parse_user_query() raising an unrelated exception
(ZeroDivisionError), not a SearchQueryError
WHEN:
- The document list endpoint is queried
THEN:
- The exception propagates unconverted, rather than being
relabeled a 400
"""
import documents.search._backend as backend_mod
def raise_zero_division(*args: object, **kwargs: object) -> object:
raise ZeroDivisionError("synthetic bug, unrelated to search grammar")
monkeypatch.setattr(
backend_mod,
"parse_user_query",
raise_zero_division,
)
with pytest.raises(ZeroDivisionError):
admin_client.get("/api/documents/?query=anything")
def test_internal_cause_query_error_is_not_converted_to_a_400(
self,
admin_client: APIClient,
monkeypatch: pytest.MonkeyPatch,
indexed_document: Document,
) -> None:
"""
GIVEN:
- A real query string running through the real parse and
routing pipeline (pre-parse rewrites, wc.parse(), and
_map_emit_error's own Cause routing all run for real), except
the final emit call (tantivy_emit) is forced to report a
library-internal defect (Cause.INTERNAL) - the one
library-internal failure mode reachable from a real query
WHEN:
- The document list endpoint is queried
THEN:
- The QueryError propagates unconverted, rather than being
relabeled a 400
"""
import documents.search._query as query_mod
def raise_internal(*args: object, **kwargs: object) -> object:
raise QueryError(
Diagnostic(
kind=DiagnosticKind.BACKEND_REJECTED,
cause=Cause.INTERNAL,
message="synthetic whoosh-compat emitter defect",
),
)
monkeypatch.setattr(query_mod, "tantivy_emit", raise_internal)
with pytest.raises(QueryError):
admin_client.get("/api/documents/?query=invoice")
class TestSelectionPathsAgreeWithSearch:
"""DocumentSelectionMixin backs bulk edit, bulk download, and a
more_like_id selection filter. It catches only SearchQueryError -- the
same contract the search list endpoint enforces above -- so all three
must map SearchQueryError to a 400 and let anything else surface."""
def test_bulk_edit_maps_search_query_error_to_a_400(
self,
admin_client: APIClient,
monkeypatch: pytest.MonkeyPatch,
indexed_document: Document,
) -> None:
"""
GIVEN:
- parse_user_query() raising a SearchQueryError naming a field
WHEN:
- The bulk_edit endpoint is called with a query filter
THEN:
- The response is a 400 whose body names the field
"""
import documents.search._backend as backend_mod
def raise_search_query_error(*args: object, **kwargs: object) -> object:
raise SearchQueryError("bad value for field 'added'")
monkeypatch.setattr(
backend_mod,
"parse_user_query",
raise_search_query_error,
)
response = admin_client.post(
"/api/documents/bulk_edit/",
{
"documents": [],
"all": True,
"filters": {"query": "anything"},
"method": "set_document_type",
"parameters": {"document_type": None},
},
format="json",
)
assert response.status_code == status.HTTP_400_BAD_REQUEST
assert "added" in str(response.data["query"])
def test_bulk_edit_lets_an_unexpected_exception_surface(
self,
admin_client: APIClient,
monkeypatch: pytest.MonkeyPatch,
indexed_document: Document,
) -> None:
"""
GIVEN:
- parse_user_query() raising an unrelated exception
(ZeroDivisionError), not a SearchQueryError
WHEN:
- The bulk_edit endpoint is called with a query filter
THEN:
- The exception propagates unconverted, rather than being
relabeled a 400
"""
import documents.search._backend as backend_mod
def raise_zero_division(*args: object, **kwargs: object) -> object:
raise ZeroDivisionError("synthetic bug, unrelated to search grammar")
monkeypatch.setattr(
backend_mod,
"parse_user_query",
raise_zero_division,
)
with pytest.raises(ZeroDivisionError):
admin_client.post(
"/api/documents/bulk_edit/",
{
"documents": [],
"all": True,
"filters": {"query": "anything"},
"method": "set_document_type",
"parameters": {"document_type": None},
},
format="json",
)
def test_bulk_download_maps_search_query_error_to_a_400(
self,
admin_client: APIClient,
monkeypatch: pytest.MonkeyPatch,
indexed_document: Document,
) -> None:
"""
GIVEN:
- parse_user_query() raising a SearchQueryError naming a field
WHEN:
- The bulk_download endpoint is called with a query filter
THEN:
- The response is a 400 whose body names the field
"""
import documents.search._backend as backend_mod
def raise_search_query_error(*args: object, **kwargs: object) -> object:
raise SearchQueryError("bad value for field 'added'")
monkeypatch.setattr(
backend_mod,
"parse_user_query",
raise_search_query_error,
)
response = admin_client.post(
"/api/documents/bulk_download/",
{
"documents": [],
"all": True,
"filters": {"query": "anything"},
},
format="json",
)
assert response.status_code == status.HTTP_400_BAD_REQUEST
assert "added" in str(response.data["query"])
def test_more_like_id_selection_filter_maps_search_query_error_to_a_400(
self,
admin_client: APIClient,
monkeypatch: pytest.MonkeyPatch,
indexed_document: Document,
) -> None:
"""
GIVEN:
- TantivyBackend.more_like_this_ids() raising a
SearchQueryError
WHEN:
- The bulk_download endpoint is called with a more_like_id
filter
THEN:
- The response is a 400
"""
import documents.search._backend as backend_mod
def raise_search_query_error(*args: object, **kwargs: object) -> object:
raise SearchQueryError("similar-document lookup is unavailable")
monkeypatch.setattr(
backend_mod.TantivyBackend,
"more_like_this_ids",
raise_search_query_error,
)
response = admin_client.post(
"/api/documents/bulk_download/",
{
"documents": [],
"all": True,
"filters": {"more_like_id": indexed_document.pk},
},
format="json",
)
assert response.status_code == status.HTTP_400_BAD_REQUEST
def test_more_like_id_selection_filter_lets_an_unexpected_exception_surface(
self,
admin_client: APIClient,
monkeypatch: pytest.MonkeyPatch,
indexed_document: Document,
) -> None:
"""
GIVEN:
- TantivyBackend.more_like_this_ids() raising an unrelated
exception (ZeroDivisionError), not a SearchQueryError
WHEN:
- The bulk_download endpoint is called with a more_like_id
filter
THEN:
- The exception propagates unconverted, rather than being
relabeled a 400
"""
import documents.search._backend as backend_mod
def raise_zero_division(*args: object, **kwargs: object) -> object:
raise ZeroDivisionError("synthetic bug, unrelated to similarity lookup")
monkeypatch.setattr(
backend_mod.TantivyBackend,
"more_like_this_ids",
raise_zero_division,
)
with pytest.raises(ZeroDivisionError):
admin_client.post(
"/api/documents/bulk_download/",
{
"documents": [],
"all": True,
"filters": {"more_like_id": indexed_document.pk},
},
format="json",
)
@@ -0,0 +1,287 @@
"""The query-length cap in ``_get_tantivy_query_and_mode``.
whoosh-compat's fieldname tagger is O(n^2) in plain word characters, so an
unbounded ``query`` (SearchMode.QUERY) string is a CPU-exhaustion vector
against a single request handler. The GET search endpoint is incidentally
bounded by the web server's header limit, but the POST selection-filter
path (bulk edit, bulk download) is not -- that is the real vector, so it
must be pinned here too, not just the GET path.
The cap is enforced once, in the shared helper both entry points call, so
these tests exercise the real endpoints rather than the helper directly:
a construct that looks right in isolation has repeatedly behaved
differently end to end on this branch.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
from unittest import mock
import pytest
from rest_framework import status
import documents.search._backend
from documents.tests.factories import DocumentFactory
from documents.views import _MAX_QUERY_LENGTH
if TYPE_CHECKING:
from rest_framework.test import APIClient
from documents.models import Document
pytestmark = [pytest.mark.django_db, pytest.mark.usefixtures("_search_index")]
@pytest.fixture
def indexed_document() -> Document:
from documents.search import get_backend
doc = DocumentFactory.create(title="quarterly invoice", content="acme corp")
get_backend().add_or_update(doc)
return doc
class TestGetSearchEndpointEnforcesTheCap:
def test_query_one_over_the_cap_is_a_400(
self,
admin_client: APIClient,
indexed_document: Document,
) -> None:
"""
GIVEN:
- The GET search endpoint
WHEN:
- A query one character over `_MAX_QUERY_LENGTH` is submitted
THEN:
- The response is a 400 naming both the actual length and the
cap, and the query is rejected before it ever reaches the
parser -- the 400 alone doesn't prove that, since the parser
could run first and the view could discard the result
"""
query = "a" * (_MAX_QUERY_LENGTH + 1)
with mock.patch(
"documents.search._backend.parse_user_query",
wraps=documents.search._backend.parse_user_query,
) as parse_spy:
response = admin_client.get("/api/documents/", {"query": query})
assert response.status_code == status.HTTP_400_BAD_REQUEST
message = str(response.data["query"])
assert str(_MAX_QUERY_LENGTH) in message
assert str(_MAX_QUERY_LENGTH + 1) in message
parse_spy.assert_not_called()
def test_query_at_exactly_the_cap_is_accepted(
self,
admin_client: APIClient,
indexed_document: Document,
) -> None:
"""
GIVEN:
- The GET search endpoint
WHEN:
- A query exactly `_MAX_QUERY_LENGTH` characters long is
submitted
THEN:
- The response is a 200 (the cap is inclusive, not exclusive)
"""
query = "a" * _MAX_QUERY_LENGTH
response = admin_client.get("/api/documents/", {"query": query})
assert response.status_code == status.HTTP_200_OK
def test_an_ordinary_query_is_unaffected(
self,
admin_client: APIClient,
indexed_document: Document,
) -> None:
"""
GIVEN:
- The GET search endpoint and an indexed document
WHEN:
- An ordinary, well-under-the-cap query is submitted
THEN:
- The cap has no effect on a normal search: the matching
document is returned
"""
response = admin_client.get("/api/documents/", {"query": "invoice"})
assert response.status_code == status.HTTP_200_OK
assert response.data["count"] == 1
class TestPostSelectionPathsEnforceTheCap:
"""The bulk-edit and bulk-download selection filters share the same
helper the GET search path uses. This is the path that actually
matters: it is not bounded by a web server's header-length limit the
way the GET path incidentally is."""
def test_bulk_edit_query_one_over_the_cap_is_a_400(
self,
admin_client: APIClient,
indexed_document: Document,
) -> None:
"""
GIVEN:
- The bulk-edit selection-filter endpoint
WHEN:
- Its `filters.query` is one character over `_MAX_QUERY_LENGTH`
THEN:
- The response is a 400 naming both the actual length and the
cap, and the query is rejected before it ever reaches the
parser -- this is the path with no web-server header-length
limit to fall back on, so this is the invariant that matters
"""
query = "a" * (_MAX_QUERY_LENGTH + 1)
with mock.patch(
"documents.search._backend.parse_user_query",
wraps=documents.search._backend.parse_user_query,
) as parse_spy:
response = admin_client.post(
"/api/documents/bulk_edit/",
{
"documents": [],
"all": True,
"filters": {"query": query},
"method": "set_document_type",
"parameters": {"document_type": None},
},
format="json",
)
assert response.status_code == status.HTTP_400_BAD_REQUEST
message = str(response.data["query"])
assert str(_MAX_QUERY_LENGTH) in message
assert str(_MAX_QUERY_LENGTH + 1) in message
parse_spy.assert_not_called()
@mock.patch("documents.bulk_edit.bulk_update_documents.apply_async")
def test_bulk_edit_query_at_exactly_the_cap_is_accepted(
self,
bulk_update_task_mock: mock.MagicMock,
admin_client: APIClient,
indexed_document: Document,
) -> None:
"""
GIVEN:
- The bulk-edit selection-filter endpoint
WHEN:
- Its `filters.query` is exactly `_MAX_QUERY_LENGTH` characters
long
THEN:
- The cap check accepts it and the request reaches the real
bulk-edit method (its Celery dispatch is mocked out here,
same as every other bulk-edit test, since nothing here is
testing that method itself)
"""
query = "a" * _MAX_QUERY_LENGTH
response = admin_client.post(
"/api/documents/bulk_edit/",
{
"documents": [],
"all": True,
"filters": {"query": query},
"method": "set_document_type",
"parameters": {"document_type": None},
},
format="json",
)
assert response.status_code == status.HTTP_200_OK
def test_bulk_download_query_one_over_the_cap_is_a_400(
self,
admin_client: APIClient,
indexed_document: Document,
) -> None:
"""
GIVEN:
- The bulk-download selection-filter endpoint
WHEN:
- Its `filters.query` is one character over `_MAX_QUERY_LENGTH`
THEN:
- The response is a 400 naming both the actual length and the
cap, and the query is rejected before it ever reaches the
parser
"""
query = "a" * (_MAX_QUERY_LENGTH + 1)
with mock.patch(
"documents.search._backend.parse_user_query",
wraps=documents.search._backend.parse_user_query,
) as parse_spy:
response = admin_client.post(
"/api/documents/bulk_download/",
{
"documents": [],
"all": True,
"filters": {"query": query},
},
format="json",
)
assert response.status_code == status.HTTP_400_BAD_REQUEST
message = str(response.data["query"])
assert str(_MAX_QUERY_LENGTH) in message
assert str(_MAX_QUERY_LENGTH + 1) in message
parse_spy.assert_not_called()
class TestGlobalSearchEnforcesTheCapToo:
"""GlobalSearchView calls the backend directly, not through the shared helper.
It hardcodes SearchMode.TEXT, which is linear rather than quadratic, so it
was never the CPU-exhaustion vector. It is capped anyway so that "every
user query string reaching the backend passes a length check" is an
invariant rather than a claim with an exception: the view already bounds
the query from below, and a later change letting it select a mode would
otherwise reopen the hole silently.
"""
def test_query_one_over_the_cap_is_a_400(
self,
admin_client: APIClient,
indexed_document: Document,
) -> None:
"""
GIVEN:
- GlobalSearchView, which calls the backend directly in
SearchMode.TEXT rather than through the shared cap-checking
helper
WHEN:
- Its query is one character over `_MAX_QUERY_LENGTH`
THEN:
- The response is still a 400, keeping "every user query
string reaching the backend passes a length check" an
invariant with no exception, even though TEXT mode is
linear and was never itself the CPU-exhaustion vector
"""
response = admin_client.get(
"/api/search/",
{"query": "a" * (_MAX_QUERY_LENGTH + 1)},
)
assert response.status_code == status.HTTP_400_BAD_REQUEST
def test_query_at_exactly_the_cap_is_accepted(
self,
admin_client: APIClient,
indexed_document: Document,
) -> None:
"""
GIVEN:
- GlobalSearchView
WHEN:
- Its query is exactly `_MAX_QUERY_LENGTH` characters long
THEN:
- The response is a 200 (the cap is inclusive, not exclusive)
"""
response = admin_client.get(
"/api/search/",
{"query": "a" * _MAX_QUERY_LENGTH},
)
assert response.status_code == status.HTTP_200_OK
@@ -0,0 +1,88 @@
"""An unterminated ``[`` date range bracket at the API level.
``created:[2020`` (with or without a dangling ``to <value>``) now raises
BAD_DATE and the search endpoint returns HTTP 400, where it used to parse
past the missing ``]`` and silently pass the malformed range through.
A 400 is correct: malformed input should fail loudly rather than silently
matching an unintended query. Pinned at the API level -- the layer a user
or client actually sees -- rather than only against the parser directly.
The properly closed decoy proves the bracket is what matters, not
whoosh-compat's date grammar generally: ``created:[2020 to 2021]`` parses
and searches cleanly.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from rest_framework import status
from documents.tests.factories import DocumentFactory
if TYPE_CHECKING:
from rest_framework.test import APIClient
from documents.models import Document
pytestmark = [pytest.mark.django_db, pytest.mark.usefixtures("_search_index")]
@pytest.fixture
def indexed_document() -> Document:
from documents.search import get_backend
doc = DocumentFactory.create(title="quarterly invoice", content="acme corp")
get_backend().add_or_update(doc)
return doc
class TestUnterminatedBracketReturnsA400:
@pytest.mark.parametrize(
"query",
[
pytest.param("created:[2020", id="missing_upper_bound_and_bracket"),
pytest.param("created:[2020 to 2021", id="missing_closing_bracket"),
],
)
def test_unterminated_bracket_is_a_400(
self,
admin_client: APIClient,
indexed_document: Document,
query: str,
) -> None:
"""
GIVEN:
- The search endpoint
WHEN:
- A date-range query with a missing closing `]` (with or
without a dangling upper bound) is submitted
THEN:
- The response is a 400 naming the field, rather than parsing
past the missing bracket and silently passing the malformed
range through
"""
response = admin_client.get(f"/api/documents/?query={query}")
assert response.status_code == status.HTTP_400_BAD_REQUEST
assert "created" in str(response.data["query"])
def test_properly_closed_bracket_still_searches_cleanly(
self,
admin_client: APIClient,
indexed_document: Document,
) -> None:
"""
GIVEN:
- The search endpoint
WHEN:
- A properly closed date-range query is submitted
THEN:
- The response is a 200 (the decoy proving the missing
bracket, not whoosh-compat's date grammar generally, is
what the 400 above is about)
"""
response = admin_client.get(
"/api/documents/?query=created:[2020 to 2021]",
)
assert response.status_code == status.HTTP_200_OK
+72
View File
@@ -69,6 +69,16 @@ class TestTrashAPI(DirectoriesMixin, APITestCase):
self.assertEqual(resp.status_code, status.HTTP_200_OK)
self.assertEqual(Document.global_objects.count(), 0)
def test_trash_list_requires_global_document_view_permission(self) -> None:
user = User.objects.create_user(username="trash_owner")
document = Document.objects.create(title="Owned", owner=user)
document.delete()
self.client.force_authenticate(user)
response = self.client.get("/api/trash/")
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
def test_trash_api_empty_all(self) -> None:
"""
GIVEN:
@@ -207,3 +217,65 @@ class TestTrashAPI(DirectoriesMixin, APITestCase):
)
self.assertEqual(resp.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn("have not yet been deleted", resp.data["documents"][0])
def _make_versioned_document(self) -> tuple[Document, list[Document]]:
root = Document.objects.create(
title="root",
content="root-content",
checksum="root",
mime_type="application/pdf",
)
versions = [
Document.objects.create(
title=f"v{index}",
content=f"v{index}-content",
checksum=f"v{index}",
mime_type="application/pdf",
root_document=root,
version_index=index,
)
for index in range(1, 3)
]
return root, versions
def test_api_trash_restore_document_restores_its_versions(self) -> None:
"""
GIVEN:
- Existing document with two versions
WHEN:
- API request to delete the document
- API request to restore it from the trash
THEN:
- Only the document itself is listed in the trash
- A version cannot be restored without its root
- The document is restored together with all of its versions
"""
root, versions = self._make_versioned_document()
self.client.force_login(user=self.user)
self.client.delete(f"/api/documents/{root.pk}/")
self.assertEqual(Document.deleted_objects.count(), 3)
resp = self.client.get("/api/trash/")
self.assertEqual(resp.status_code, status.HTTP_200_OK)
self.assertEqual(resp.data["count"], 1)
self.assertEqual(resp.data["results"][0]["id"], root.pk)
# A version cannot be restored while its root remains in the trash.
resp = self.client.post(
"/api/trash/",
{"action": "restore", "documents": [versions[0].pk]},
)
self.assertEqual(resp.status_code, status.HTTP_400_BAD_REQUEST)
self.assertIn("Restore the root document", resp.data["documents"][0])
resp = self.client.post(
"/api/trash/",
{"action": "restore", "documents": [root.pk]},
)
self.assertEqual(resp.status_code, status.HTTP_200_OK)
self.assertEqual(Document.deleted_objects.count(), 0)
self.assertCountEqual(
Document.objects.filter(root_document=root).values_list("id", flat=True),
[version.pk for version in versions],
)
+42
View File
@@ -194,6 +194,48 @@ class TestApiWorkflows(DirectoriesMixin, APITestCase):
self.assertEqual(response.status_code, status.HTTP_201_CREATED)
self.assertEqual(Workflow.objects.count(), 2)
def test_api_create_workflow_ignores_nested_action_id(self) -> None:
"""
GIVEN:
- An existing workflow action
WHEN:
- API request to create a workflow includes that action's ID
THEN:
- A new action is created without changing the existing action
"""
original_title = self.action.assign_title
response = self.client.post(
self.ENDPOINT,
json.dumps(
{
"name": "Workflow 2",
"order": 1,
"triggers": [
{
"sources": [DocumentSource.ApiUpload],
"type": WorkflowTrigger.WorkflowTriggerType.CONSUMPTION,
"filter_filename": "*",
},
],
"actions": [
{
"id": self.action.id,
"assign_title": "New Action Title",
},
],
},
),
content_type="application/json",
)
self.assertEqual(response.status_code, status.HTTP_201_CREATED)
self.action.refresh_from_db()
self.assertEqual(self.action.assign_title, original_title)
new_action = Workflow.objects.get(name="Workflow 2").actions.get()
self.assertNotEqual(new_action.id, self.action.id)
self.assertEqual(new_action.assign_title, "New Action Title")
def test_api_create_workflow_nested(self) -> None:
"""
GIVEN:
+191
View File
@@ -5,8 +5,11 @@ from unittest import mock
import pikepdf
from django.contrib.auth.models import Group
from django.contrib.auth.models import Permission
from django.contrib.auth.models import User
from django.db import connection
from django.test import TestCase
from django.test.utils import CaptureQueriesContext
from guardian.shortcuts import assign_perm
from guardian.shortcuts import get_groups_with_perms
from guardian.shortcuts import get_users_with_perms
@@ -19,6 +22,7 @@ from documents.models import Document
from documents.models import DocumentType
from documents.models import StoragePath
from documents.models import Tag
from documents.permissions import set_permissions_for_objects
from documents.tests.utils import DirectoriesMixin
@@ -392,6 +396,11 @@ class TestBulkEdit(DirectoriesMixin, TestCase):
self.assertFalse(Document.objects.filter(id=self.doc1.id).exists())
self.assertFalse(Document.objects.filter(id=version.id).exists())
Document.deleted_objects.get(id=self.doc1.id).restore(strict=False)
self.assertTrue(Document.objects.filter(id=self.doc1.id).exists())
self.assertTrue(Document.objects.filter(id=version.id).exists())
def test_delete_version_document_keeps_root(self) -> None:
version = Document.objects.create(
checksum="A-v1",
@@ -510,6 +519,178 @@ class TestBulkEdit(DirectoriesMixin, TestCase):
)
self.assertEqual(groups_with_perms.count(), 2)
@mock.patch("documents.tasks.bulk_update_documents.apply_async")
def test_set_permissions_batched_across_document_count(
self,
m,
) -> None:
"""
GIVEN:
- Many documents are being bulk-edited to set permissions at once
WHEN:
- set_permissions runs over a small batch vs. a much larger one
THEN:
- Permissions are applied correctly at both scales
- Query count does not grow with the number of documents, i.e.
each user/group is applied across all documents with one
batched call rather than one call per (document, identity)
pair
"""
permissions = {
"view": {
"users": [self.user1.id, self.user2.id],
"groups": [self.group2.id],
},
"change": {
"users": [self.user1.id],
"groups": [self.group2.id],
},
}
def run_with_n_documents(n: int) -> int:
docs = [
Document.objects.create(checksum=f"perm-{n}-{i}", title=f"perm-{n}-{i}")
for i in range(n)
]
with CaptureQueriesContext(connection) as ctx:
bulk_edit.set_permissions(
[doc.id for doc in docs],
set_permissions=permissions,
owner=self.owner,
merge=False,
)
for doc in docs:
self.assertEqual(get_users_with_perms(doc).count(), 2)
self.assertEqual(get_groups_with_perms(doc).count(), 1)
return len(ctx.captured_queries)
small_batch_queries = run_with_n_documents(5)
large_batch_queries = run_with_n_documents(50)
# A tolerance rather than equality, matching the N+1 check in
# test_views.py: bulk_create's batch_size caps rows per INSERT, so a
# large enough selection does legitimately add statements, and the
# per-process ContentType cache makes the first run carry an extra
# query. Neither can hide a regression to per-document assignment,
# which would be ~10x the small-batch count here.
self.assertLessEqual(
large_batch_queries,
small_batch_queries + 5,
"Permission assignment appears to scale with document count: "
f"{small_batch_queries} queries for 5 documents vs. "
f"{large_batch_queries} for 50",
)
@mock.patch("documents.tasks.bulk_update_documents.apply_async")
def test_set_permissions_grants_direct_perm_even_if_already_granted_via_group(
self,
m,
) -> None:
"""
GIVEN:
- A user already has view access to a document via group
membership, with no direct grant of their own
WHEN:
- set_permissions explicitly grants that same user direct view
access via bulk_edit
THEN:
- A direct permission grant is created for the user, not skipped
because they already have equivalent access via the group
Regression test: guardian's queryset-aware assign_perm() (routed to
when the target is a list/queryset) skips creating a direct row for
anyone whose ObjectPermissionChecker.has_perm() already returns True
-- which includes group-derived access. The single-object assign_perm
this bulk path replaces has no such check; it always ensures a
direct row via get_or_create. Losing that guarantee would mean
revoking the group's grant later silently strips access that was
supposed to be explicit.
"""
self.doc1.owner = self.user1
self.doc1.save()
self.user1.groups.add(self.group1)
assign_perm("view_document", self.group1, self.doc1)
bulk_edit.set_permissions(
[self.doc1.id],
set_permissions={
"view": {"users": [self.user1.id], "groups": []},
},
merge=True,
)
direct_users = get_users_with_perms(
self.doc1,
only_with_perms_in=["view_document"],
with_group_users=False,
)
self.assertIn(self.user1, direct_users)
def test_set_permissions_for_objects_raises_for_unknown_action(self) -> None:
"""
GIVEN:
- An unrecognized permission action name with users to grant it
to
WHEN:
- set_permissions_for_objects is called
THEN:
- Permission.DoesNotExist is raised, not a silent no-op
Regression test: the endpoint that calls this
(BulkEditObjectPermissionsView) never actually validates action
names against the raw client-supplied permissions dict --
BulkEditObjectsSerializer._validate_permissions calls
validate_set_permissions() only for its side-effecting user/group id
checks and discards the filtered dict it returns -- so a bogus
action key reaches this function as-is. Resolving the Permission via
a bare `.filter()` (which returns empty instead of raising) would
silently drop the grant and report success.
"""
with self.assertRaises(Permission.DoesNotExist):
set_permissions_for_objects(
{"not_a_real_action": {"users": [self.user1.id], "groups": []}},
Document,
[self.doc1.pk],
)
def test_set_permissions_for_objects_unknown_action_applies_nothing(
self,
) -> None:
"""
GIVEN:
- A permissions dict with a valid action ordered ahead of an
unrecognized one
WHEN:
- set_permissions_for_objects is called
THEN:
- Permission.DoesNotExist is raised
- The valid action ahead of it is not applied either
Every action is resolved before any row is written, so a bad action
name cannot leave a half-applied change behind. That matters because
BulkEditObjectsView turns this exception into a 400: without the
up-front resolution the client would be told the request failed
while the leading action had already been committed.
"""
with self.assertRaises(Permission.DoesNotExist):
set_permissions_for_objects(
{
"view": {"users": [self.user1.id], "groups": []},
"not_a_real_action": {"users": [self.user1.id], "groups": []},
},
Document,
[self.doc1.pk],
)
self.assertNotIn(
self.user1,
get_users_with_perms(
self.doc1,
only_with_perms_in=["view_document"],
with_group_users=False,
),
)
@mock.patch("documents.models.Document.delete")
def test_delete_documents_old_uuid_field(self, m) -> None:
m.side_effect = Exception("Data too long for column 'transaction_id' at row 1")
@@ -1461,6 +1642,16 @@ class TestPDFActions(DirectoriesMixin, TestCase):
mock_group.assert_not_called()
mock_consume_file.assert_not_called()
@mock.patch("pikepdf.open")
def test_edit_pdf_rejects_invalid_operations(self, mock_open) -> None:
for operations in ([], [{"page": 1, "doc": 2**32}]):
with self.subTest(operations=operations):
with self.assertLogs("paperless.bulk_edit", level="ERROR"):
with self.assertRaisesRegex(ValueError, "index is out of bounds"):
bulk_edit.edit_pdf([self.doc2.id], operations)
mock_open.assert_not_called()
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file.delay")
@mock.patch("documents.tasks.consume_file.apply_async")
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
+2 -2
View File
@@ -1,7 +1,7 @@
import pytest
from django.core.checks import Error
from django.core.checks import Warning
from pytest_django.fixtures import SettingsWrapper
from pytest_django.fixtures import Settings
from pytest_mock import MockerFixture
from documents.checks import filename_format_check
@@ -47,7 +47,7 @@ class TestFilenameFormatCheck:
)
def test_warns_on_old_style_format(
self,
settings: SettingsWrapper,
settings: Settings,
filename_format: str,
expected_hint: str,
) -> None:
+145
View File
@@ -3,6 +3,7 @@ import warnings
from pathlib import Path
from unittest import mock
import numpy as np
import pytest
from django.conf import settings
from django.test import TestCase
@@ -11,6 +12,7 @@ from django.test import override_settings
from documents.classifier import ClassifierModelCorruptError
from documents.classifier import DocumentClassifier
from documents.classifier import IncompatibleClassifierVersionError
from documents.classifier import _predict_with_threshold
from documents.classifier import load_classifier
from documents.models import Correspondent
from documents.models import Document
@@ -625,6 +627,103 @@ class TestClassifier(DirectoriesMixin, TestCase):
self.assertEqual(self.classifier.predict_storage_path(doc1.content), sp.pk)
self.assertIsNone(self.classifier.predict_storage_path(doc2.content))
def test_predict_rejects_prediction_below_match_threshold(self) -> None:
"""
GIVEN:
- Classifiers trained against test data with confident predictions
WHEN:
- CLASSIFIER_MATCH_THRESHOLD exceeds the model's confidence
THEN:
- Every predict_* method discards the match in favor of no match
"""
c1 = Correspondent.objects.create(
name="c1",
matching_algorithm=Correspondent.MATCH_AUTO,
)
dt1 = DocumentType.objects.create(
name="dt1",
matching_algorithm=DocumentType.MATCH_AUTO,
)
sp1 = StoragePath.objects.create(
name="sp1",
matching_algorithm=StoragePath.MATCH_AUTO,
)
doc1 = Document.objects.create(
title="doc1",
content="this is a document from c1",
correspondent=c1,
document_type=dt1,
storage_path=sp1,
checksum="A",
)
Document.objects.create(
title="doc2",
content="this is a document from no one",
checksum="B",
)
self.classifier.train()
predictors = {
"correspondent": self.classifier.predict_correspondent,
"document_type": self.classifier.predict_document_type,
"storage_path": self.classifier.predict_storage_path,
}
# No real prediction can reach a confidence this high, so this
# isolates the threshold check from the model's actual output.
with override_settings(CLASSIFIER_MATCH_THRESHOLD=0.999999):
for name, predict in predictors.items():
with self.subTest(field=name):
self.assertIsNone(predict(doc1.content))
def test_train_uses_balanced_sample_weight(self) -> None:
"""
GIVEN:
- A training set with correspondents, document types and storage paths
WHEN:
- The classifier is trained
THEN:
- Each MLP classifier is fit with balanced sample weights, so that
over-represented classes don't dominate predictions
"""
c1 = Correspondent.objects.create(
name="c1",
matching_algorithm=Correspondent.MATCH_AUTO,
)
dt1 = DocumentType.objects.create(
name="dt1",
matching_algorithm=DocumentType.MATCH_AUTO,
)
sp1 = StoragePath.objects.create(
name="sp1",
matching_algorithm=StoragePath.MATCH_AUTO,
)
Document.objects.create(
title="doc1",
content="this is a document from c1",
correspondent=c1,
document_type=dt1,
storage_path=sp1,
checksum="A",
)
Document.objects.create(
title="doc2",
content="this is a document from no one",
checksum="B",
)
with mock.patch(
"sklearn.utils.class_weight.compute_sample_weight",
return_value=None,
) as mocked_compute_sample_weight:
self.classifier.train()
self.assertEqual(mocked_compute_sample_weight.call_count, 3)
for call in mocked_compute_sample_weight.call_args_list:
self.assertEqual(call.args[0], "balanced")
def test_one_tag_predict(self) -> None:
t1 = Tag.objects.create(name="t1", matching_algorithm=Tag.MATCH_AUTO, pk=12)
@@ -810,6 +909,52 @@ class TestClassifier(DirectoriesMixin, TestCase):
load_classifier(raise_exception=True)
class _StubProbaClassifier:
"""
A fake scikit-learn classifier exposing just enough of the API for
`_predict_with_threshold`: `classes_` and `predict_proba`.
"""
def __init__(self, classes: list[int], probabilities: list[float]) -> None:
self.classes_ = np.array(classes)
self._probabilities = np.array([probabilities])
def predict_proba(self, X) -> np.ndarray:
return self._probabilities
@pytest.mark.parametrize(
("classes", "probabilities", "threshold", "expected"),
[
# confident prediction above the threshold is returned
([-1, 3], [0.1, 0.9], 0.6, 3),
# prediction below the threshold is discarded
([-1, 3], [0.45, 0.55], 0.6, None),
# boundary: exactly at the threshold is accepted, not discarded
([-1, 3], [0.4, 0.6], 0.6, 3),
# the winning class is the "no match" pseudo-class, regardless of its
# own confidence
([-1, 3], [0.99, 0.01], 0.0, None),
# threshold of 0.0 disables the confidence check entirely
([-1, 3], [0.45, 0.55], 0.0, 3),
],
)
def test_predict_with_threshold(classes, probabilities, threshold, expected) -> None:
classifier = _StubProbaClassifier(classes, probabilities)
result = _predict_with_threshold(classifier, X=None, threshold=threshold)
assert result == expected
def test_classifier_match_threshold_default() -> None:
"""
GIVEN:
- No PAPERLESS_CLASSIFIER_MATCH_THRESHOLD environment variable is set
THEN:
- The classifier match threshold defaults to 0.6
"""
assert settings.CLASSIFIER_MATCH_THRESHOLD == 0.6
def test_preprocess_content() -> None:
"""
GIVEN:
@@ -0,0 +1,457 @@
from __future__ import annotations
from types import SimpleNamespace
from typing import TYPE_CHECKING
import pytest
from django.db import connection
from django.test.utils import CaptureQueriesContext
from rest_framework import status
from documents.models import Document
from documents.tests.factories import DocumentFactory
from documents.versioning import LATEST_VERSION_CONTENT_PREFETCH_ATTR
from documents.versioning import has_prefetched_effective_content
from documents.versioning import latest_version_content_prefetch
from documents.views import DocumentViewSet
if TYPE_CHECKING:
from rest_framework.test import APIClient
class TestNeedsEffectiveContentAnnotation:
"""
DocumentViewSet._needs_effective_content_annotation() decides whether
the effective_content correlated subquery is worth attaching to the
queryset at all -- see TestDocumentListEffectiveContentAnnotation below
for why. This only checks that decision's own logic (a plain query-param
membership test), not that Django/DRF's filtering machinery works.
"""
@pytest.mark.parametrize(
("params", "expected"),
[
({}, False),
({"ordering": "-added"}, False),
({"tags__id__in": "1,2"}, False),
({"search": ""}, False),
({"search": " "}, False),
({"content__icontains": ""}, False),
({"search": "foo"}, True),
({"title_content": "foo"}, True),
({"content__istartswith": "foo"}, True),
({"content__iendswith": "foo"}, True),
({"content__icontains": "foo"}, True),
({"content__iexact": "foo"}, True),
],
)
def test_detects_content_filter_params(
self,
params: dict[str, str],
expected: bool, # noqa: FBT001
) -> None:
"""
GIVEN:
- A view bound to a request carrying the given query params
WHEN:
- Checking whether the effective_content annotation is needed
THEN:
- It is needed only for requests that actually filter on it
"""
view = DocumentViewSet()
view.request = SimpleNamespace(query_params=params)
assert view._needs_effective_content_annotation() is expected
class TestNeedsEffectiveContentPrefetch:
"""
DocumentViewSet._needs_effective_content_prefetch() decides whether the
single-version content prefetch is worth attaching. It has to read the
`fields` param exactly the way get_serializer() does, or a request whose
response includes content ends up without the prefetch and pays
get_effective_content()'s per-instance fallback instead.
"""
@pytest.mark.parametrize(
("params", "expected"),
[
pytest.param({}, True, id="no-fields-param-keeps-every-field"),
pytest.param({"fields": ""}, True, id="blank-fields-keeps-every-field"),
pytest.param(
{"fields": "id,content"},
True,
id="content-among-requested-fields",
),
pytest.param({"fields": "content"}, True, id="content-only"),
pytest.param({"fields": "id"}, False, id="content-not-requested"),
pytest.param(
{"fields": "id,title"},
False,
id="several-fields-without-content",
),
],
)
def test_detects_whether_content_can_reach_the_response(
self,
params: dict[str, str],
expected: bool, # noqa: FBT001
) -> None:
"""
GIVEN:
- A view bound to a request carrying the given query params
WHEN:
- Checking whether the content prefetch is needed
THEN:
- It is needed exactly when get_serializer() would emit content,
which treats a blank `fields` the same as an absent one
"""
view = DocumentViewSet()
view.request = SimpleNamespace(query_params=params)
assert view._needs_effective_content_prefetch() is expected
@pytest.mark.django_db
class TestDocumentListEffectiveContentAnnotation:
"""
DocumentViewSet.get_queryset() only attaches the effective_content
correlated subquery when a request actually filters on it. Attaching it
unconditionally re-executes it once per candidate row before the page's
LIMIT is applied -- fine on SQLite/Postgres, but pathological on
MariaDB's default cardinality estimation for the root_document_id
self-join once candidate counts get large (see the root_document_id /
effective_content perf investigation).
"""
def test_list_without_content_filter_skips_annotation_but_returns_latest_content(
self,
admin_client: APIClient,
) -> None:
"""
GIVEN:
- A root document whose latest version has different content
WHEN:
- Listing documents with no search/content-filter param
THEN:
- The response still reflects the latest version's content
- The database never evaluates effective_content per row
"""
root = DocumentFactory(content="old-root-content")
DocumentFactory(
root_document=root,
version_index=1,
content="new-version-content",
)
with CaptureQueriesContext(connection) as ctx:
response = admin_client.get("/api/documents/?fields=id,content")
assert response.status_code == status.HTTP_200_OK
assert response.data["results"] == [
{"id": root.id, "content": "new-version-content"},
]
assert not any(
"effective_content" in query["sql"] for query in ctx.captured_queries
)
@pytest.mark.parametrize(
"fields_param",
[
pytest.param("", id="blank-fields"),
pytest.param("id,content", id="content-requested"),
],
)
def test_content_resolves_without_a_query_per_document(
self,
admin_client: APIClient,
fields_param: str,
) -> None:
"""
GIVEN:
- One versioned root document, then two more
WHEN:
- Listing documents with a `fields` param that keeps content
THEN:
- Every root's content resolves to its latest version's
- The query count does not grow with the number of documents,
i.e. a blank `fields` does not skip the prefetch and fall back
to loading each root's deferred version content
"""
first = DocumentFactory(content="first-root-content")
DocumentFactory(
root_document=first,
version_index=1,
content="first-version-content",
)
with CaptureQueriesContext(connection) as one_document:
response = admin_client.get(f"/api/documents/?fields={fields_param}")
assert response.status_code == status.HTTP_200_OK
assert [r["content"] for r in response.data["results"]] == [
"first-version-content",
]
for index in range(2):
root = DocumentFactory(content=f"root-content-{index}")
DocumentFactory(
root_document=root,
version_index=1,
content=f"version-content-{index}",
)
with CaptureQueriesContext(connection) as three_documents:
response = admin_client.get(f"/api/documents/?fields={fields_param}")
assert response.status_code == status.HTTP_200_OK
assert sorted(r["content"] for r in response.data["results"]) == [
"first-version-content",
"version-content-0",
"version-content-1",
]
assert len(_get_document_queries(three_documents)) == len(
_get_document_queries(one_document),
)
def test_list_without_content_field_skips_prefetch_and_omits_content(
self,
admin_client: APIClient,
) -> None:
"""
GIVEN:
- A versioned root document
WHEN:
- Listing documents without asking for content
THEN:
- Content is neither serialized nor resolved
- Nothing pays for the prefetch or the per-instance fallback
"""
root = DocumentFactory(content="root-content")
DocumentFactory(
root_document=root,
version_index=1,
content="version-content",
)
with CaptureQueriesContext(connection) as ctx:
response = admin_client.get("/api/documents/?fields=id")
assert response.status_code == status.HTTP_200_OK
assert response.data["results"] == [{"id": root.id}]
assert _get_effective_content_fallback_queries(ctx) == []
# Only the list query itself reads a content column: no extra query
# for the skipped prefetch, none for a per-instance fallback
content_queries = [
query
for query in ctx.captured_queries
if '"documents_document"."content"' in query["sql"]
]
assert len(content_queries) == 1
def test_latest_version_content_prefetch_carries_only_the_newest_version(
self,
) -> None:
"""
GIVEN:
- A root document with two versions
WHEN:
- Fetching the root through latest_version_content_prefetch()
THEN:
- The prefetch carries only the single newest version, not every
historical version's content (the whole point of not reusing
the metadata-only "versions" prefetch for this)
"""
root = DocumentFactory(content="root-content")
DocumentFactory(
root_document=root,
version_index=1,
content="older-version-content",
)
DocumentFactory(
root_document=root,
version_index=2,
content="newest-version-content",
)
fetched_root = (
Document.objects.filter(pk=root.pk)
.prefetch_related(
latest_version_content_prefetch(),
)
.get()
)
latest = getattr(fetched_root, LATEST_VERSION_CONTENT_PREFETCH_ATTR)
assert [v.content for v in latest] == ["newest-version-content"]
class TestHasPrefetchedEffectiveContent:
"""
DocumentSerializer.to_representation() only calls get_effective_content()
when has_prefetched_effective_content() says it's cheap -- otherwise a
caller that never set up an annotation or prefetch (TrashView,
GlobalSearchView, which build their own querysets and don't display
content at all) would pay for a per-instance query nobody asked for.
"""
def test_false_with_no_annotation_or_prefetch(self) -> None:
"""
GIVEN:
- A document the ORM never annotated or prefetched for
WHEN:
- Asking whether its effective content is already resolved
THEN:
- It is not, so the serializer must leave it alone
"""
document = DocumentFactory.build()
assert has_prefetched_effective_content(document) is False
def test_true_with_effective_content_annotation(self) -> None:
"""
GIVEN:
- A document carrying the queryset's effective_content annotation
WHEN:
- Asking whether its effective content is already resolved
THEN:
- It is, straight off the annotation
"""
document = DocumentFactory.build()
document.effective_content = "resolved"
assert has_prefetched_effective_content(document) is True
def test_true_with_lean_prefetch_attr_even_when_empty(self) -> None:
"""
GIVEN:
- A document the lean content prefetch ran for, finding no versions
WHEN:
- Asking whether its effective content is already resolved
THEN:
- It is: an empty prefetch is an answer, not a missing one
"""
document = DocumentFactory.build()
setattr(document, LATEST_VERSION_CONTENT_PREFETCH_ATTR, [])
assert has_prefetched_effective_content(document) is True
def test_true_with_metadata_versions_prefetch_cache(self) -> None:
"""
GIVEN:
- A document carrying only the metadata "versions" prefetch
WHEN:
- Asking whether its effective content is already resolved
THEN:
- It is, via get_effective_content()'s prefetch-cache branch
"""
document = DocumentFactory.build()
document._prefetched_objects_cache = {"versions": []}
assert has_prefetched_effective_content(document) is True
def _get_document_queries(
ctx: CaptureQueriesContext,
) -> list[dict[str, str]]:
"""
The queries a list request spends on the documents themselves, i.e.
everything but the one-time django_content_type lookup guardian's
permission filtering makes. That lookup is process-cached, and the
autouse fixture in conftest clears the cache before every test, so it
lands in whichever request happens to run first and never repeats --
counting it makes a request look like it costs one query more than the
identical request after it.
"""
return [q for q in ctx.captured_queries if '"django_content_type"' not in q["sql"]]
def _get_effective_content_fallback_queries(
ctx: CaptureQueriesContext,
) -> list[dict[str, str]]:
"""
Document.get_effective_content()'s per-instance fallback (no annotation,
no prefetch) is a `.values_list("content", flat=True).first()` query --
a SELECT of just the content column. Distinct from get_versions()'s own,
unrelated per-instance metadata query (id/checksum/added/etc, no
content) run to build the "versions" response field, which isn't part
of what this test file covers.
"""
return [
q
for q in ctx.captured_queries
if q["sql"].startswith('SELECT "documents_document"."content" FROM')
]
@pytest.mark.django_db
class TestTrashAndGlobalSearchEffectiveContentIsNeverPerInstance:
"""
TrashView and GlobalSearchView serialize Document instances with
DocumentSerializer too, but build their querysets independently of
DocumentViewSet.get_queryset(). TrashView doesn't display content at all,
so it keeps the document's own unresolved content; GlobalSearchView
annotates effective_content itself, so it shows the latest version's.
Neither should ever fall back to a per-instance query.
"""
def test_trash_list_shows_unresolved_content_with_no_extra_query(
self,
admin_client: APIClient,
) -> None:
"""
GIVEN:
- A trashed root document whose own content differs from what a
version would have had (also trashed, deletion cascades)
WHEN:
- Listing trash
THEN:
- The response shows the document's own content
- Nothing ever queries for versions to resolve it
"""
root = DocumentFactory(content="own-content")
DocumentFactory(
root_document=root,
version_index=1,
content="version-content",
)
root.delete()
with CaptureQueriesContext(connection) as ctx:
response = admin_client.get("/api/trash/")
assert response.status_code == status.HTTP_200_OK
[result] = [r for r in response.data["results"] if r["id"] == root.id]
assert result["content"] == "own-content"
assert _get_effective_content_fallback_queries(ctx) == []
def test_global_search_db_only_shows_latest_version_content_with_no_extra_query(
self,
admin_client: APIClient,
) -> None:
"""
GIVEN:
- A root document, findable by title, whose own content differs
from its latest version's
WHEN:
- Using the global search endpoint's db_only mode
THEN:
- The response shows the latest version's content, resolved by
GlobalSearchView's own effective_content annotation
- There is no per-instance fallback query
"""
root = DocumentFactory(title="findme", content="own-content")
DocumentFactory(
root_document=root,
version_index=1,
content="version-content",
)
with CaptureQueriesContext(connection) as ctx:
response = admin_client.get(
"/api/search/?query=findme&db_only=true",
)
assert response.status_code == status.HTTP_200_OK
[result] = [d for d in response.data["documents"] if d["id"] == root.id]
assert result["content"] == "version-content"
assert _get_effective_content_fallback_queries(ctx) == []
+5 -1
View File
@@ -110,7 +110,7 @@ class TestDocument(TestCase):
checksum="checksum",
mime_type="application/pdf",
)
Document.objects.create(
version = Document.objects.create(
root_document=root,
correspondent=root.correspondent,
title="Version",
@@ -124,6 +124,10 @@ class TestDocument(TestCase):
self.assertEqual(Document.objects.count(), 0)
self.assertEqual(Document.deleted_objects.count(), 2)
root.restore(strict=False)
self.assertTrue(Document.objects.filter(pk=version.pk).exists())
def test_file_name(self) -> None:
doc = Document(
mime_type="application/pdf",
+118 -6
View File
@@ -43,7 +43,7 @@ if TYPE_CHECKING:
from collections.abc import Generator
from unittest.mock import MagicMock
from pytest_django.fixtures import SettingsWrapper
from pytest_django.fixtures import Settings
from pytest_mock import MockerFixture
@@ -136,6 +136,23 @@ def wait_for_mock_call(
return False
def sleep_past_stability(
owner: FileStabilityTracker | ConsumerThread,
*,
windows: float = 1.5,
) -> None:
"""
Block until a tracked file's stability window has certainly elapsed.
Args:
owner: The tracker, or the consumer thread running one, whose
configured stability delay sets the wait.
windows: How many stability windows to wait, giving slop for a slow
or loaded test runner.
"""
sleep(owner.stability_delay * windows)
class TestTrackedFile:
"""Tests for the TrackedFile dataclass."""
@@ -261,6 +278,56 @@ class TestFileStabilityTracker:
assert len(stable) == 0
assert stability_tracker.pending_count == 1
def test_get_stable_files_skips_empty_file(
self,
stability_tracker: FileStabilityTracker,
tmp_path: Path,
) -> None:
"""
GIVEN:
- A zero byte file, tracked and past its stability delay
WHEN:
- Stable files are collected
THEN:
- The file is not yielded for consumption
- The file is dropped from tracking rather than held, so an
abandoned placeholder does not keep the watch loop awake
"""
empty = tmp_path / "scan.pdf"
empty.write_bytes(b"")
stability_tracker.track(empty, Change.added)
sleep_past_stability(stability_tracker)
stable = list(stability_tracker.get_stable_files())
assert stable == []
assert stability_tracker.pending_count == 0
def test_empty_file_is_yielded_once_content_arrives(
self,
stability_tracker: FileStabilityTracker,
tmp_path: Path,
) -> None:
"""
GIVEN:
- A zero byte file which was dropped from tracking while empty
WHEN:
- The writer fills the file and a new event re-tracks it
THEN:
- The file is yielded for consumption once it is stable
"""
target = tmp_path / "scan.pdf"
target.write_bytes(b"")
stability_tracker.track(target, Change.added)
sleep_past_stability(stability_tracker)
assert list(stability_tracker.get_stable_files()) == []
target.write_bytes(b"%PDF-1.4 content")
stability_tracker.track(target, Change.modified)
sleep_past_stability(stability_tracker)
assert list(stability_tracker.get_stable_files()) == [target]
def test_get_stable_files_deleted_during_check(self, temp_file: Path) -> None:
"""Test deleted file is not returned during stability check."""
tracker = FileStabilityTracker(stability_delay=0.1)
@@ -605,7 +672,7 @@ class TestCommandValidation:
def test_raises_for_missing_consumption_dir(
self,
settings: SettingsWrapper,
settings: Settings,
) -> None:
"""Test command raises error when directory is not provided."""
settings.CONSUMPTION_DIR = None
@@ -639,7 +706,7 @@ class TestCommandOneshot:
scratch_dir: Path,
sample_pdf: Path,
mock_consume_file_delay: MagicMock,
settings: SettingsWrapper,
settings: Settings,
) -> None:
"""Test oneshot mode processes existing files."""
target = consumption_dir / "document.pdf"
@@ -659,7 +726,7 @@ class TestCommandOneshot:
scratch_dir: Path,
sample_pdf: Path,
mock_consume_file_delay: MagicMock,
settings: SettingsWrapper,
settings: Settings,
) -> None:
"""Test oneshot mode processes files recursively."""
subdir = consumption_dir / "subdir"
@@ -681,7 +748,7 @@ class TestCommandOneshot:
consumption_dir: Path,
scratch_dir: Path,
mock_consume_file_delay: MagicMock,
settings: SettingsWrapper,
settings: Settings,
) -> None:
"""Test oneshot mode ignores unsupported file extensions."""
target = consumption_dir / "document.xyz"
@@ -879,6 +946,51 @@ class TestCommandWatch:
mock_consume_file_delay.apply_async.assert_called()
def test_scanner_placeholder_is_not_consumed_while_empty(
self,
consumption_dir: Path,
sample_pdf: Path,
mock_consume_file_delay: MagicMock,
start_consumer: Callable[..., ConsumerThread],
) -> None:
"""
GIVEN:
- A scanner which creates a zero byte placeholder and only writes
the page some time later (GH discussion #13969)
WHEN:
- The placeholder sits untouched well past the stability delay
- The scanner then writes the real content
THEN:
- The empty placeholder is never queued, as it could only fail
with "Unsupported mime type inode/x-empty"
- The file is queued exactly once, when the content lands
"""
thread = start_consumer(stability_delay=0.2)
target = consumption_dir / "scan.pdf"
target.write_bytes(b"") # the scanner's placeholder
# Well past the stability delay: the old behaviour queued it here.
sleep_past_stability(thread, windows=5)
if thread.exception:
raise thread.exception
assert mock_consume_file_delay.apply_async.call_count == 0
shutil.copy(sample_pdf, target) # the scanner finishes the page
assert wait_for_mock_call(
mock_consume_file_delay.apply_async,
timeout_s=5.0,
)
if thread.exception:
raise thread.exception
assert mock_consume_file_delay.apply_async.call_count == 1
queued_doc = mock_consume_file_delay.apply_async.call_args.kwargs["kwargs"][
"input_doc"
]
assert queued_doc.original_file.name == "scan.pdf"
def test_ignores_macos_files(
self,
consumption_dir: Path,
@@ -1256,7 +1368,7 @@ class TestProcessExistingFilesQueued:
consumption_dir: Path,
sample_pdf: Path,
mock_consume_file_delay: MagicMock,
settings: SettingsWrapper,
settings: Settings,
) -> None:
"""The set returned seeds the rescan's queued set, avoiding re-queue."""
target = consumption_dir / "document.pdf"
+2 -2
View File
@@ -1,7 +1,7 @@
from collections.abc import Generator
import pytest
from pytest_django.fixtures import SettingsWrapper
from pytest_django.fixtures import Settings
from documents.parsers import get_default_file_extension
from documents.parsers import get_supported_file_extensions
@@ -14,7 +14,7 @@ from paperless.parsers.tika import TikaDocumentParser
@pytest.fixture()
def _tika_registry(settings: SettingsWrapper) -> Generator[None, None, None]:
def _tika_registry(settings: Settings) -> Generator[None, None, None]:
"""
Rebuild the parser registry with Tika enabled for the duration of the
test, then reset on exit so other tests see the default (Tika-disabled)
@@ -309,6 +309,9 @@ class TestEmailDocumentPermissionBoundary:
):
owner = User.objects.create_user(username="owner")
requester = User.objects.create_user(username="requester")
requester.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
rest_api_client.force_authenticate(user=requester)
hidden = DocumentFactory(owner=owner)
@@ -364,6 +367,27 @@ class TestBulkEditChangePermissionBoundary:
@pytest.mark.django_db
class TestBulkDownloadPermissionChecksRootDocument:
def test_download_requires_global_view_permission(
self,
rest_api_client,
paperless_dirs,
_media_settings,
):
owner = User.objects.create_user(username="owner")
requester = User.objects.create_user(username="requester")
root = DocumentFactory(owner=owner)
root.source_path.write_bytes(b"%PDF-1.4 test")
assign_perm("view_document", requester, root)
rest_api_client.force_authenticate(user=requester)
response = rest_api_client.post(
"/api/documents/bulk_download/",
{"documents": [root.pk]},
format="json",
)
assert response.status_code == HTTPStatus.FORBIDDEN
def test_permission_checked_on_root_not_on_version(
self,
rest_api_client,
@@ -372,6 +396,9 @@ class TestBulkDownloadPermissionChecksRootDocument:
):
owner = User.objects.create_user(username="owner")
requester = User.objects.create_user(username="requester")
requester.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
rest_api_client.force_authenticate(user=requester)
root = DocumentFactory(owner=owner)
# a version of root that the requester has NOT been individually granted
@@ -396,6 +423,9 @@ class TestBulkDownloadPermissionChecksRootDocument:
# `stranger` case) can't tell the two apart, since they're denied
# either way.
version_only_grantee = User.objects.create_user(username="version_only_grantee")
version_only_grantee.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
assign_perm("view_document", version_only_grantee, version)
rest_api_client.force_authenticate(user=version_only_grantee)
response = rest_api_client.post(
@@ -417,6 +447,9 @@ class TestTrashRestorePermissionBoundary:
):
owner = User.objects.create_user(username="owner")
requester = User.objects.create_user(username="requester")
requester.user_permissions.add(
Permission.objects.get(codename="delete_document"),
)
rest_api_client.force_authenticate(user=requester)
doc = DocumentFactory(owner=owner)
assign_perm("view_document", requester, doc) # view only, NOT delete
@@ -435,6 +468,9 @@ class TestTrashRestorePermissionBoundary:
):
owner = User.objects.create_user(username="owner")
requester = User.objects.create_user(username="requester")
requester.user_permissions.add(
Permission.objects.get(codename="delete_document"),
)
rest_api_client.force_authenticate(user=requester)
doc = DocumentFactory(owner=owner)
assign_perm("delete_document", requester, doc)
@@ -447,6 +483,22 @@ class TestTrashRestorePermissionBoundary:
)
assert response.status_code == HTTPStatus.OK
def test_restore_requires_global_delete_permission(self, rest_api_client):
owner = User.objects.create_user(username="owner")
requester = User.objects.create_user(username="requester")
rest_api_client.force_authenticate(user=requester)
doc = DocumentFactory(owner=owner)
assign_perm("delete_document", requester, doc)
doc.delete()
response = rest_api_client.post(
"/api/trash/",
{"documents": [doc.pk], "action": "restore"},
format="json",
)
assert response.status_code == HTTPStatus.FORBIDDEN
@pytest.mark.django_db
class TestTrashViewExcludesExplicitlyGrantedDocuments:
@@ -463,6 +515,9 @@ class TestTrashViewExcludesExplicitlyGrantedDocuments:
def test_explicit_grant_does_not_leak_trashed_document(self, rest_api_client):
owner = User.objects.create_user(username="trash_owner")
grantee = User.objects.create_user(username="trash_grantee")
grantee.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
doc = DocumentFactory(owner=owner)
doc.delete() # soft delete
assign_perm("view_document", grantee, doc)
+7
View File
@@ -1,5 +1,6 @@
import pytest
import regex
from django.conf import settings
from pytest_mock import MockerFixture
from documents.regex import safe_regex_finditer
@@ -9,6 +10,12 @@ from documents.regex import safe_regex_sub
from documents.regex import validate_regex_pattern
def test_regex_timeout_uses_configured_setting() -> None:
from documents.regex import REGEX_TIMEOUT_SECONDS
assert REGEX_TIMEOUT_SECONDS == settings.MATCH_REGEX_TIMEOUT_SECONDS
class TestValidateRegexPattern:
def test_valid_pattern(self) -> None:
validate_regex_pattern(r"\d+")
@@ -6,8 +6,10 @@ from pathlib import Path
from unittest import mock
from django.conf import settings
from django.contrib.auth.models import Permission
from django.contrib.auth.models import User
from django.utils import timezone
from guardian.shortcuts import assign_perm
from rest_framework import serializers
from rest_framework import status
from rest_framework.test import APITestCase
@@ -48,6 +50,37 @@ class ShareLinkBundleAPITests(DirectoriesMixin, APITestCase):
delay_mock.assert_called_once()
self.assertEqual(delay_mock.call_args.kwargs["kwargs"]["bundle_id"], bundle.pk)
@mock.patch("documents.views.build_share_link_bundle.apply_async")
def test_create_bundle_requires_global_document_view_permission(
self,
delay_mock,
) -> None:
owner = User.objects.create_user(username="document_owner")
requester = User.objects.create_user(username="bundle_creator")
requester.user_permissions.add(
Permission.objects.get(codename="add_sharelinkbundle"),
)
document = DocumentFactory.create(owner=owner)
assign_perm("view_document", requester, document)
self.client.force_authenticate(requester)
payload = {
"document_ids": [document.pk],
"file_version": ShareLink.FileVersion.ARCHIVE,
"expiration_days": 7,
}
response = self.client.post(self.ENDPOINT, payload, format="json")
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
requester.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
requester = User.objects.get(pk=requester.pk)
self.client.force_authenticate(requester)
response = self.client.post(self.ENDPOINT, payload, format="json")
self.assertEqual(response.status_code, status.HTTP_201_CREATED)
delay_mock.assert_called_once()
def test_create_bundle_rejects_missing_documents(self) -> None:
payload = {
"document_ids": [9999],
+39
View File
@@ -2,6 +2,7 @@ from unittest import mock
from django.contrib.auth.models import Permission
from django.contrib.auth.models import User
from rest_framework import status
from rest_framework.test import APITestCase
from documents import bulk_edit
@@ -108,6 +109,44 @@ class TestTagHierarchy(DirectoriesMixin, APITestCase):
self.document.refresh_from_db()
assert self.document.tags.count() == 0
def test_remove_inbox_tags_removes_nested_children(self) -> None:
inbox = Tag.objects.create(name="Inbox", is_inbox_tag=True)
nested = Tag.objects.create(name="Nested", tn_parent=inbox)
self.document.add_nested_tags([nested])
resp = self.client.patch(
f"/api/documents/{self.document.pk}/",
{"title": "new title", "remove_inbox_tags": True},
format="json",
)
assert resp.status_code == status.HTTP_200_OK
self.document.refresh_from_db()
assert self.document.tags.count() == 0
# A subsequent save must not re-add the inbox tag as an ancestor
resp = self.client.patch(
f"/api/documents/{self.document.pk}/",
{"title": "another title", "tags": [], "remove_inbox_tags": True},
format="json",
)
assert resp.status_code == status.HTTP_200_OK
self.document.refresh_from_db()
assert self.document.tags.count() == 0
def test_remove_inbox_tags_keeps_inbox_when_nested_child_added(self) -> None:
inbox = Tag.objects.create(name="Inbox", is_inbox_tag=True)
nested = Tag.objects.create(name="Nested", tn_parent=inbox)
self.document.add_nested_tags([inbox])
self.client.patch(
f"/api/documents/{self.document.pk}/",
{"tags": [nested.pk], "remove_inbox_tags": True},
format="json",
)
self.document.refresh_from_db()
tags = set(self.document.tags.values_list("pk", flat=True))
assert tags == {inbox.pk, nested.pk}
def test_bulk_edit_respects_hierarchy(self) -> None:
bulk_edit.add_tag([self.document.pk], self.child.pk)
self.document.refresh_from_db()
+11
View File
@@ -106,6 +106,17 @@ class TestBeforeTaskPublishHandler:
assert task.task_type == PaperlessTask.TaskType.TRAIN_CLASSIFIER
assert task.trigger_source == PaperlessTask.TriggerSource.MANUAL
# A Celery retry republishes with the same task_id; this must not
# raise a duplicate-key IntegrityError, and must leave the original
# PENDING record alone.
send_publish(
"documents.tasks.train_classifier",
(),
{},
headers={"id": task_id},
)
assert PaperlessTask.objects.filter(task_id=task_id).count() == 1
def test_creates_task_for_sanity_check(self) -> None:
task_id = send_publish("documents.tasks.sanity_check", (), {})
task = PaperlessTask.objects.get(task_id=task_id)
+39
View File
@@ -32,6 +32,7 @@ from documents.signals.handlers import update_llm_suggestions_cache
from documents.tests.utils import DirectoriesMixin
from documents.tests.utils import read_streaming_response
from paperless.models import ApplicationConfiguration
from paperless_ai.exceptions import LLMProviderError
from paperless_ai.exceptions import LLMTimeoutError
@@ -140,6 +141,9 @@ class TestViews(DirectoriesMixin, TestCase):
codename__contains="sharelink",
)
self.user.user_permissions.add(*sharelink_permissions)
self.user.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
self.user.save()
self.client.force_login(self.user)
@@ -201,6 +205,9 @@ class TestViews(DirectoriesMixin, TestCase):
codename__contains="sharelink",
)
self.user.user_permissions.add(*sharelink_permissions)
self.user.user_permissions.add(
Permission.objects.get(codename="view_document"),
)
self.client.force_login(self.user)
create_response = self.client.post(
@@ -737,6 +744,38 @@ class TestAISuggestions(DirectoriesMixin, TestCase):
get_llm_suggestion_cache(self.document.pk, backend="openai-like"),
)
@patch("documents.views.get_ai_document_classification")
@override_settings(
AI_ENABLED=True,
LLM_BACKEND="openai-like",
)
def test_ai_suggestions_with_llm_provider_error(
self,
mock_get_ai_classification,
) -> None:
mock_get_ai_classification.side_effect = LLMProviderError(
"confidential provider response",
)
self.client.force_login(user=self.user)
response = self.client.get(
f"/api/documents/{self.document.pk}/ai_suggestions/",
)
self.assertEqual(response.status_code, status.HTTP_502_BAD_GATEWAY)
self.assertEqual(
response.json(),
{
"ai": [
"AI backend rejected the request. Check logs for details.",
],
},
)
self.assertNotIn("confidential provider response", response.content.decode())
self.assertIsNone(
get_llm_suggestion_cache(self.document.pk, backend="openai-like"),
)
@patch("documents.views.get_ai_document_classification")
@override_settings(
AI_ENABLED=True,
+35 -2
View File
@@ -23,6 +23,7 @@ from guardian.shortcuts import get_users_with_perms
from httpx import ConnectError
from httpx import HTTPError
from httpx import HTTPStatusError
from pytest_django.fixtures import Settings
from pytest_httpx import HTTPXMock
from rest_framework.test import APIClient
from rest_framework.test import APITestCase
@@ -38,7 +39,6 @@ from paperless_ai.exceptions import LLMTimeoutError
if TYPE_CHECKING:
from django.db.models import QuerySet
from pytest_django.fixtures import SettingsWrapper
from documents import tasks
from documents.data_models import ConsumableDocument
@@ -5356,7 +5356,7 @@ class TestDateWorkflowLocalization(
def test_document_consumption_workflow_localization(
self,
tmp_path: Path,
settings: SettingsWrapper,
settings: Settings,
title_template: str,
expected_title: str,
) -> None:
@@ -5711,6 +5711,39 @@ class TestApplyAISuggestionsWorkflowAction(
self.assertEqual(changed, [])
self.assertIn("AI is not enabled", "".join(cm.output))
def test_document_without_content_does_nothing(self) -> None:
"""
GIVEN:
- A document whose OCR content is empty or whitespace-only
WHEN:
- AI suggestions are applied by a workflow
THEN:
- The classifier is not called and the document is left unchanged
"""
action = self.make_action(ai_overwrite_existing=True)
for content in ("", " \n\t"):
with self.subTest(content=content):
self.doc.content = content
self.doc.save(update_fields=["content"])
with (
mock.patch(
"documents.workflows.ai.get_ai_document_classification",
) as get_classification,
self.assertLogs(
"paperless.workflows.ai",
level="WARNING",
) as cm,
):
changed = apply_ai_suggestions_to_document(action, self.doc)
self.assertEqual(changed, [])
get_classification.assert_not_called()
self.assertIn("has no content", "".join(cm.output))
self.doc.refresh_from_db()
self.assertEqual(self.doc.title, "original.pdf")
def test_invalid_configuration_leaves_document_untouched(self) -> None:
"""
GIVEN: