mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-09-16 22:58:00 +00:00
Merge branch 'dev' into feature/centralized-share-links
This commit is contained in:
@@ -10,7 +10,7 @@ import pytest
|
||||
from django.contrib.auth import get_user_model
|
||||
from django.contrib.contenttypes.models import ContentType
|
||||
from guardian.shortcuts import clear_ct_cache
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
from pytest_django.fixtures import Settings
|
||||
from rest_framework.test import APIClient
|
||||
|
||||
from documents.tests.factories import DocumentFactory
|
||||
@@ -100,7 +100,7 @@ def sample_doc(
|
||||
@pytest.fixture()
|
||||
def _search_index(
|
||||
tmp_path: Path,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> Generator[None, None, None]:
|
||||
"""Create a temp index directory and point INDEX_DIR at it.
|
||||
|
||||
@@ -118,7 +118,7 @@ def _search_index(
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def settings_timezone(settings: SettingsWrapper) -> zoneinfo.ZoneInfo:
|
||||
def settings_timezone(settings: Settings) -> zoneinfo.ZoneInfo:
|
||||
return zoneinfo.ZoneInfo(settings.TIME_ZONE)
|
||||
|
||||
|
||||
|
||||
@@ -70,7 +70,7 @@ def clear_lru_cache() -> Generator[None, None, None]:
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_date_parser_settings(settings: pytest_django.fixtures.SettingsWrapper) -> Any:
|
||||
def mock_date_parser_settings(settings: pytest_django.fixtures.Settings) -> Any:
|
||||
"""
|
||||
Override Django settings for the duration of date parser tests.
|
||||
"""
|
||||
|
||||
@@ -6,7 +6,7 @@ from pathlib import Path
|
||||
|
||||
import pytest
|
||||
import pytest_mock
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
from pytest_django.fixtures import Settings
|
||||
|
||||
from documents.export.sinks import DirectoryExportSink
|
||||
from documents.export.sinks import ExportSink
|
||||
@@ -242,7 +242,7 @@ class TestZipExportSink:
|
||||
self,
|
||||
tmp_path: Path,
|
||||
source_file: Path,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> None:
|
||||
scratch_dir = tmp_path / "scratch"
|
||||
settings.SCRATCH_DIR = scratch_dir
|
||||
@@ -261,7 +261,7 @@ class TestZipExportSink:
|
||||
def test_abort_after_manifest_written_cleans_up_pending_tmp(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> None:
|
||||
scratch_dir = tmp_path / "scratch"
|
||||
settings.SCRATCH_DIR = scratch_dir
|
||||
|
||||
@@ -1,25 +1,21 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import tempfile
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
from documents.search._backend import reset_backend
|
||||
from documents.search._schema import build_schema
|
||||
from documents.search._tokenizer import register_tokenizers
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Generator
|
||||
from pathlib import Path
|
||||
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
from pytest_django.fixtures import Settings
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def index_dir(tmp_path: Path, settings: SettingsWrapper) -> Path:
|
||||
def index_dir(tmp_path: Path, settings: Settings) -> Path:
|
||||
path = tmp_path / "index"
|
||||
path.mkdir()
|
||||
settings.INDEX_DIR = path
|
||||
@@ -35,11 +31,3 @@ def backend() -> Generator[TantivyBackend, None, None]:
|
||||
finally:
|
||||
b.close()
|
||||
reset_backend()
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def index() -> tantivy.Index:
|
||||
"""A real Tantivy index for parse-acceptance tests (module scope for speed)."""
|
||||
idx = tantivy.Index(build_schema(), path=tempfile.mkdtemp())
|
||||
register_tokenizers(idx, "english")
|
||||
return idx
|
||||
|
||||
@@ -0,0 +1,541 @@
|
||||
"""Result-level acceptance corpus: real documents indexed via build_schema(),
|
||||
real queries run through parse_user_query(), matched-document-ID sets
|
||||
asserted, not intermediate ASTs or query strings. This is paperless-ngx's
|
||||
analogue of whoosh-compat's own tests/emitter/test_acceptance_e2e.py.
|
||||
|
||||
Supersedes test_query.py's TestParseUserQuery result-level cases.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC
|
||||
from datetime import datetime
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import time_machine
|
||||
from django.contrib.auth.models import User
|
||||
|
||||
from documents.models import CustomField
|
||||
from documents.models import CustomFieldInstance
|
||||
from documents.models import Document
|
||||
from documents.models import DocumentType
|
||||
from documents.models import Note
|
||||
from documents.models import StoragePath
|
||||
from documents.search._query import parse_user_query
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
FROZEN_NOW = datetime(2026, 6, 15, 12, 0, tzinfo=UTC)
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
"""Create a Document and index it in one step, for the common case
|
||||
where nothing needs to happen between the two (no related Note/
|
||||
CustomFieldInstance to attach first)."""
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def indexed_documents(backend: TantivyBackend) -> dict[str, int]:
|
||||
"""Index a small fixture set, return {label: doc_id} for corpus queries."""
|
||||
docs = {
|
||||
"invoice_2020": _index(
|
||||
backend,
|
||||
title="Invoice 2020",
|
||||
content="invoice total due",
|
||||
checksum="acc-invoice-2020",
|
||||
archive_serial_number=100,
|
||||
),
|
||||
"invoice_2021": _index(
|
||||
backend,
|
||||
title="Invoice 2021",
|
||||
content="invoice total due",
|
||||
checksum="acc-invoice-2021",
|
||||
archive_serial_number=101,
|
||||
),
|
||||
"invoice_2023": _index(
|
||||
backend,
|
||||
title="Invoice 2023",
|
||||
content="invoice total due",
|
||||
checksum="acc-invoice-2023",
|
||||
archive_serial_number=102,
|
||||
),
|
||||
"receipt_2022": _index(
|
||||
backend,
|
||||
title="Receipt 2022",
|
||||
content="receipt total due",
|
||||
checksum="acc-receipt-2022",
|
||||
archive_serial_number=103,
|
||||
),
|
||||
}
|
||||
return {label: doc.pk for label, doc in docs.items()}
|
||||
|
||||
|
||||
class TestIssue13568BracketWildcard:
|
||||
"""paperless-ngx#13568: title:202[0-3]* must keep its character class,
|
||||
not fold to a prefix query that silently drops it."""
|
||||
|
||||
def test_bracket_class_wildcard_matches_only_in_range_years(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_documents: dict[str, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Four indexed documents titled Invoice 2020/2021/2023 and
|
||||
Receipt 2022
|
||||
WHEN:
|
||||
- "title:202[0-1]*" is searched ([0-1], not [0-3], is
|
||||
deliberate: the fixture's trailing digits are 0/1/2/3, so a
|
||||
[0-3] class would match all four and pass even if the
|
||||
character class were silently dropped and folded to an
|
||||
unconstrained "202*" prefix; [0-1] partitions the fixture
|
||||
into a genuine in-range/out-of-range split)
|
||||
THEN:
|
||||
- Only the 2020 and 2021 documents match, proving the bracket
|
||||
character class survived (issue #13568's original bug)
|
||||
"""
|
||||
matched = _matched_ids(backend, "title:202[0-1]*")
|
||||
expected = {
|
||||
indexed_documents["invoice_2020"],
|
||||
indexed_documents["invoice_2021"],
|
||||
}
|
||||
assert matched == expected, (
|
||||
"title:202[0-1]* must match 2020/2021 titles and exclude 2022/2023 "
|
||||
"- if this matches everything, the wildcard's character class was "
|
||||
"silently dropped (issue #13568's original bug)"
|
||||
)
|
||||
|
||||
|
||||
class TestFieldBoosts:
|
||||
def test_title_boost_ranks_title_match_above_content_only_match(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- One document whose title contains the query word and another
|
||||
whose content (not title) contains it
|
||||
WHEN:
|
||||
- The query word is searched unfielded
|
||||
THEN:
|
||||
- The title match ranks first, proving our title field boost
|
||||
actually affects ranking
|
||||
"""
|
||||
title_match = _index(
|
||||
backend,
|
||||
title="urgent",
|
||||
content="nothing else relevant",
|
||||
checksum="acc-boost-title",
|
||||
)
|
||||
_index(
|
||||
backend,
|
||||
title="nothing",
|
||||
content="urgent matter here",
|
||||
checksum="acc-boost-content",
|
||||
)
|
||||
query = parse_user_query(backend._index, "urgent", UTC)
|
||||
searcher = backend._index.searcher()
|
||||
results = searcher.search(query, limit=10)
|
||||
ranked_ids = [
|
||||
searcher.doc(addr).to_dict()["id"][0] for _score, addr in results.hits
|
||||
]
|
||||
assert ranked_ids[0] == title_match.pk
|
||||
|
||||
|
||||
class TestJsonSubpaths:
|
||||
def test_notes_user_matches_document_with_that_note_author(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with a Note authored by "alice" and a second,
|
||||
unrelated document with no note
|
||||
WHEN:
|
||||
- "notes.user:alice" is searched
|
||||
THEN:
|
||||
- Only the document with alice's note matches
|
||||
"""
|
||||
alice = User.objects.create_user(username="alice")
|
||||
doc_with_note = Document.objects.create(
|
||||
title="Has note",
|
||||
content="x",
|
||||
checksum="acc-note-with",
|
||||
)
|
||||
Note.objects.create(document=doc_with_note, user=alice, note="reminder")
|
||||
backend.add_or_update(doc_with_note)
|
||||
_index(backend, title="No note", content="x", checksum="acc-note-without")
|
||||
matched = _matched_ids(backend, "notes.user:alice")
|
||||
assert matched == {doc_with_note.pk}
|
||||
|
||||
def test_custom_fields_name_and_value_combine(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with a "Contract Number" custom field valued
|
||||
"policy", and a second document with a differently-named
|
||||
custom field also valued "policy"
|
||||
WHEN:
|
||||
- 'custom_fields.name:"Contract Number" custom_fields.value:policy'
|
||||
is searched
|
||||
THEN:
|
||||
- Only the document whose field name AND value both match is
|
||||
returned
|
||||
"""
|
||||
field = CustomField.objects.create(
|
||||
name="Contract Number",
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
other_field = CustomField.objects.create(
|
||||
name="Other Field",
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
matching = Document.objects.create(
|
||||
title="Matching",
|
||||
content="x",
|
||||
checksum="acc-cf-matching",
|
||||
)
|
||||
CustomFieldInstance.objects.create(
|
||||
document=matching,
|
||||
field=field,
|
||||
value_text="policy",
|
||||
)
|
||||
backend.add_or_update(matching)
|
||||
non_matching = Document.objects.create(
|
||||
title="Non-matching",
|
||||
content="x",
|
||||
checksum="acc-cf-nonmatching",
|
||||
)
|
||||
CustomFieldInstance.objects.create(
|
||||
document=non_matching,
|
||||
field=other_field,
|
||||
value_text="policy",
|
||||
)
|
||||
backend.add_or_update(non_matching)
|
||||
matched = _matched_ids(
|
||||
backend,
|
||||
'custom_fields.name:"Contract Number" custom_fields.value:policy',
|
||||
)
|
||||
assert matched == {matching.pk}
|
||||
|
||||
|
||||
class TestUnregisteredIdFieldFoldsToLiteralText:
|
||||
"""tag_id, owner_id, etc. are intentionally excluded from the
|
||||
FieldRegistry - always internal index columns, never meant to be
|
||||
query-addressable. Prove an unregistered field folds to a literal
|
||||
text search that matches nothing, rather than erroring."""
|
||||
|
||||
def test_tag_id_query_matches_nothing(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_documents: dict[str, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A real indexed corpus and "tag_id", a field intentionally
|
||||
excluded from the FieldRegistry (an internal index column,
|
||||
never meant to be query-addressable)
|
||||
WHEN:
|
||||
- "tag_id:5" is searched
|
||||
THEN:
|
||||
- It folds to a literal text search and matches nothing,
|
||||
rather than erroring
|
||||
"""
|
||||
matched = _matched_ids(backend, "tag_id:5")
|
||||
assert matched == set()
|
||||
|
||||
|
||||
class TestFuzzyBlendSurvivesWhooshGrammar:
|
||||
"""A query mixing whoosh-only grammar (a date keyword) with a typo'd
|
||||
free-text word must still fuzzy-match the intended document when
|
||||
ADVANCED_FUZZY_SEARCH_THRESHOLD is enabled. The fuzzy clause is built
|
||||
from the parsed query's free-text tokens (whoosh_compat's
|
||||
free_text_tokens), never from the raw query string, so whoosh grammar
|
||||
that tantivy's own parser rejects cannot knock the fuzzy clause out."""
|
||||
|
||||
def test_typo_fuzzy_matches_alongside_date_keyword(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
settings,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- ADVANCED_FUZZY_SEARCH_THRESHOLD enabled, and a document
|
||||
indexed with content "receipt total due"
|
||||
WHEN:
|
||||
- The query blends whoosh-only grammar tantivy's own parser
|
||||
rejects ("added:today") with a one-transposition misspelling
|
||||
of a word in the indexed content
|
||||
THEN:
|
||||
- The document still matches, because the fuzzy clause is
|
||||
built from the parsed query's free-text tokens
|
||||
(whoosh_compat's free_text_tokens), never from the raw
|
||||
query string, so grammar tantivy's parser cannot handle
|
||||
cannot knock the fuzzy clause out
|
||||
"""
|
||||
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.5
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
doc = _index(
|
||||
backend,
|
||||
title="Receipt March",
|
||||
content="receipt total due",
|
||||
checksum="fuzzy-blend-1",
|
||||
archive_serial_number=900,
|
||||
)
|
||||
# Sanity: the exact spelling matches through the exact clause.
|
||||
assert doc.pk in _matched_ids(backend, "added:today receipt")
|
||||
# The regression: the misspelling (one transposition) only
|
||||
# matches via the fuzzy clause, and "added:today" is
|
||||
# whoosh-only grammar tantivy's parser rejects, so raw-string
|
||||
# fuzzy parsing skips the clause entirely and this returns
|
||||
# nothing. The typo is deliberate; keep codespell away from it.
|
||||
typo_query = "added:today reciept" # codespell:ignore reciept
|
||||
assert doc.pk in _matched_ids(backend, typo_query)
|
||||
|
||||
def test_negated_words_do_not_fuzzy_match(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
settings,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- ADVANCED_FUZZY_SEARCH_THRESHOLD enabled, and a document
|
||||
containing the NOT'd word ("receipt") but not the positive
|
||||
word ("total"), so nothing matches the exact clause -- the
|
||||
shape a naive fuzzy string built from ALL words (including
|
||||
the NOT'd one) would make this document the sole hit,
|
||||
normalize its score to 1.0, and survive any threshold (a
|
||||
shape with an exact-matching sibling document would NOT
|
||||
discriminate: normalization would rank the resurfaced
|
||||
document far below the exact match and the threshold would
|
||||
cut it even for a naive implementation)
|
||||
WHEN:
|
||||
- "added:today total NOT receipt" is searched
|
||||
THEN:
|
||||
- The document does not match; a term the user excluded must
|
||||
not resurface through the fuzzy clause
|
||||
"""
|
||||
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.5
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
_index(
|
||||
backend,
|
||||
title="Receipt Archive",
|
||||
content="receipt archived stack",
|
||||
checksum="fuzzy-blend-2",
|
||||
archive_serial_number=901,
|
||||
)
|
||||
assert _matched_ids(backend, "added:today total NOT receipt") == set()
|
||||
|
||||
|
||||
class TestUnquotedDateKeywordPhrases:
|
||||
"""The unquoted spelling (added:previous month) is honored natively by
|
||||
whoosh-compat's own grammar for this closed phrase vocabulary, no
|
||||
app-level rewrite is involved. Pins that the historically supported
|
||||
spelling keeps working now that paperless no longer pre-quotes it."""
|
||||
|
||||
@pytest.fixture
|
||||
def period_documents(self, backend: TantivyBackend) -> dict[str, int]:
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
in_may = _index(
|
||||
backend,
|
||||
title="May Doc",
|
||||
content="statement",
|
||||
checksum="kw-may",
|
||||
archive_serial_number=910,
|
||||
added=datetime(2026, 5, 20, 12, 0, tzinfo=UTC),
|
||||
)
|
||||
in_june = _index(
|
||||
backend,
|
||||
title="June Doc",
|
||||
content="statement",
|
||||
checksum="kw-june",
|
||||
archive_serial_number=911,
|
||||
added=datetime(2026, 6, 10, 12, 0, tzinfo=UTC),
|
||||
)
|
||||
return {"in_may": in_may.pk, "in_june": in_june.pk}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
pytest.param("added:previous month", id="unquoted"),
|
||||
pytest.param('added:"previous month"', id="quoted"),
|
||||
pytest.param("added:Previous Month", id="unquoted-mixed-case"),
|
||||
],
|
||||
)
|
||||
def test_unquoted_matches_the_same_documents_as_quoted(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
period_documents: dict[str, int],
|
||||
query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two documents added in different months, time frozen so
|
||||
only one falls in "previous month"
|
||||
WHEN:
|
||||
- The same date-keyword phrase is spelled unquoted, quoted,
|
||||
and unquoted with mixed case
|
||||
THEN:
|
||||
- All three spellings match the same document; paperless no
|
||||
longer pre-quotes this phrase before parsing, relying on
|
||||
whoosh-compat's own grammar to accept it unquoted natively
|
||||
"""
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
assert _matched_ids(backend, query) == {period_documents["in_may"]}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
pytest.param("added:this month", id="this-month"),
|
||||
pytest.param("added:this year", id="this-year"),
|
||||
pytest.param("added:previous week", id="previous-week"),
|
||||
pytest.param("added:previous quarter", id="previous-quarter"),
|
||||
pytest.param("added:previous year", id="previous-year"),
|
||||
pytest.param("created:previous month", id="created-field"),
|
||||
pytest.param("modified:previous month", id="modified-field"),
|
||||
],
|
||||
)
|
||||
def test_every_phrase_and_date_field_parses_without_error(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
period_documents: dict[str, int],
|
||||
query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Our real schema and every date-keyword phrase in the
|
||||
vocabulary, against every date field we expose (added,
|
||||
created, modified)
|
||||
WHEN:
|
||||
- Each combination is searched
|
||||
THEN:
|
||||
- It parses and searches cleanly against our schema (no
|
||||
SearchQueryError, so no HTTP 400); exact window semantics
|
||||
are whoosh-compat's own and are pinned in its own suite
|
||||
"""
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
_matched_ids(backend, query)
|
||||
|
||||
def test_text_field_keyword_words_are_ordinary_text(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
period_documents: dict[str, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- period_documents (indexed by added-date) and a third
|
||||
document whose title literally contains the words
|
||||
"previous month"
|
||||
WHEN:
|
||||
- "title:previous month" is searched
|
||||
THEN:
|
||||
- Only the document whose title contains those words matches;
|
||||
"previous month" after a TEXT field (or unfielded) is
|
||||
ordinary text, not a date phrase, so the date-window
|
||||
documents do not match
|
||||
"""
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
wordy = _index(
|
||||
backend,
|
||||
title="Notes from the previous month",
|
||||
content="meeting notes",
|
||||
checksum="kw-text",
|
||||
archive_serial_number=912,
|
||||
)
|
||||
assert _matched_ids(backend, "title:previous month") == {wordy.pk}
|
||||
|
||||
|
||||
class TestFieldAliases:
|
||||
"""type:/path: are registry aliases for document_type:/storage_path:.
|
||||
The only other alias coverage is parse-shape; these prove resolution
|
||||
end-to-end against a real index."""
|
||||
|
||||
def test_type_alias_and_canonical_name_match_the_same_document(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with document_type "invoice", and a decoy
|
||||
document with no type whose content merely mentions
|
||||
"invoice" (document_type is itself a default search field,
|
||||
so if alias resolution ever broke and "type:invoice"
|
||||
demoted to unfielded text, the token would STILL match the
|
||||
typed document through the field value; the decoy carrying
|
||||
the query word in content is what makes a demoted search
|
||||
distinguishable, since it would then match both documents
|
||||
and fail the exact-set assertion -- the title avoids
|
||||
stemming to "type": English stems Typed -> type)
|
||||
WHEN:
|
||||
- "type:invoice" and "document_type:invoice" are each
|
||||
searched
|
||||
THEN:
|
||||
- Both resolve to the same document, proving the "type" alias
|
||||
and its canonical field name agree end-to-end against a
|
||||
real index
|
||||
"""
|
||||
invoice_type = DocumentType.objects.create(name="invoice")
|
||||
typed = _index(
|
||||
backend,
|
||||
title="First",
|
||||
content="quarterly statement",
|
||||
checksum="alias-type-1",
|
||||
document_type=invoice_type,
|
||||
)
|
||||
_index(
|
||||
backend,
|
||||
title="Second",
|
||||
content="invoice mentioned in body",
|
||||
checksum="alias-type-2",
|
||||
)
|
||||
assert _matched_ids(backend, "type:invoice") == {typed.pk}
|
||||
assert _matched_ids(backend, "document_type:invoice") == {typed.pk}
|
||||
|
||||
def test_path_alias_and_canonical_name_match_the_same_document(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document stored under storage_path "archive", and a decoy
|
||||
document with no storage_path whose content merely mentions
|
||||
"archive" (storage_path is NOT a default search field
|
||||
today, so a demoted "path:archive" already matches nothing;
|
||||
the content decoy keeps this test discriminating even if
|
||||
storage_path ever joins the defaults)
|
||||
WHEN:
|
||||
- "path:archive" and "storage_path:archive" are each searched
|
||||
THEN:
|
||||
- Both resolve to the same document, proving the "path" alias
|
||||
and its canonical field name agree end-to-end against a
|
||||
real index
|
||||
"""
|
||||
archive = StoragePath.objects.create(name="archive", path="archive/{title}")
|
||||
stored = _index(
|
||||
backend,
|
||||
title="Stored",
|
||||
content="quarterly statement",
|
||||
checksum="alias-path-1",
|
||||
storage_path=archive,
|
||||
)
|
||||
_index(
|
||||
backend,
|
||||
title="Loose",
|
||||
content="archive mentioned in body",
|
||||
checksum="alias-path-2",
|
||||
)
|
||||
assert _matched_ids(backend, "path:archive") == {stored.pk}
|
||||
assert _matched_ids(backend, "storage_path:archive") == {stored.pk}
|
||||
@@ -4,6 +4,8 @@ from pathlib import Path
|
||||
import pytest
|
||||
from django.contrib.auth.models import Group
|
||||
from django.contrib.auth.models import User
|
||||
from django.db import connection
|
||||
from django.test.utils import CaptureQueriesContext
|
||||
from guardian.shortcuts import assign_perm
|
||||
from pytest_mock import MockerFixture
|
||||
|
||||
@@ -102,6 +104,191 @@ class TestWriteBatch:
|
||||
assert len(backend.search_ids("indexable", user=None)) == 1
|
||||
|
||||
|
||||
class TestAddOrUpdateIds:
|
||||
"""Test WriteBatch.add_or_update_ids(), the bulk id-based upsert path.
|
||||
|
||||
Unlike add_or_update() called once per document, this resolves viewer
|
||||
permissions and effective (versioned) content in bulk against the ids as
|
||||
a whole, so it must produce identical indexed output to the per-document
|
||||
path while issuing a constant number of queries regardless of batch size.
|
||||
"""
|
||||
|
||||
def test_missing_id_is_skipped_not_errored(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
doc = Document.objects.create(
|
||||
title="doc",
|
||||
content="present",
|
||||
checksum="EXIST1",
|
||||
pk=1,
|
||||
)
|
||||
missing_pk = 999
|
||||
|
||||
with backend.batch_update() as batch:
|
||||
batch.add_or_update_ids([doc.pk, missing_pk])
|
||||
|
||||
assert backend.search_ids("present", user=None) == [doc.pk]
|
||||
|
||||
def test_query_count_does_not_scale_with_batch_size(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""Each query count must stay far below N, not merely match between
|
||||
two runs -- an exact-equality assertion between two measurements is
|
||||
at the mercy of incidental process-level caches (e.g. Django's
|
||||
ContentType.objects.get_for_model) warming on whichever run happens
|
||||
first, which makes counts differ by a query for reasons unrelated to
|
||||
batch size. A generous fixed bound sidesteps that: the old
|
||||
per-document path issued roughly 8 queries per document, so 50
|
||||
documents under a bound this low proves the fix regardless of cache
|
||||
state.
|
||||
"""
|
||||
max_queries_for_any_batch_size = 15
|
||||
|
||||
small_docs = [
|
||||
Document.objects.create(
|
||||
title="doc",
|
||||
content=f"unique{i}",
|
||||
checksum=f"SMALL{i}",
|
||||
pk=i,
|
||||
)
|
||||
for i in range(1, 3)
|
||||
]
|
||||
with CaptureQueriesContext(connection) as ctx_small:
|
||||
with backend.batch_update() as batch:
|
||||
batch.add_or_update_ids([d.pk for d in small_docs])
|
||||
assert len(ctx_small.captured_queries) <= max_queries_for_any_batch_size
|
||||
|
||||
large_docs = [
|
||||
Document.objects.create(
|
||||
title="doc",
|
||||
content=f"unique{i}",
|
||||
checksum=f"LARGE{i}",
|
||||
pk=i,
|
||||
)
|
||||
for i in range(100, 150)
|
||||
]
|
||||
with CaptureQueriesContext(connection) as ctx_large:
|
||||
with backend.batch_update() as batch:
|
||||
batch.add_or_update_ids([d.pk for d in large_docs])
|
||||
assert len(ctx_large.captured_queries) <= max_queries_for_any_batch_size
|
||||
|
||||
for doc in large_docs:
|
||||
assert backend.search_ids(f"unique{doc.pk}", user=None) == [doc.pk]
|
||||
|
||||
def test_resolves_direct_user_grant_in_bulk(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
owner = UserFactory()
|
||||
user = UserFactory()
|
||||
doc = Document.objects.create(
|
||||
title="doc",
|
||||
checksum="PERM1",
|
||||
pk=1,
|
||||
owner=owner,
|
||||
)
|
||||
assign_perm("view_document", user, doc)
|
||||
|
||||
with backend.batch_update() as batch:
|
||||
batch.add_or_update_ids([doc.pk])
|
||||
|
||||
assert backend.search_ids("doc", user=user) == [doc.pk]
|
||||
other = UserFactory()
|
||||
assert backend.search_ids("doc", user=other) == []
|
||||
|
||||
def test_resolves_group_grant_in_bulk(self, backend: TantivyBackend) -> None:
|
||||
owner = UserFactory()
|
||||
group = Group.objects.create(name="reviewers")
|
||||
user = UserFactory()
|
||||
user.groups.add(group)
|
||||
doc = Document.objects.create(
|
||||
title="doc",
|
||||
checksum="GPERM1",
|
||||
pk=1,
|
||||
owner=owner,
|
||||
)
|
||||
assign_perm("view_document", group, doc)
|
||||
|
||||
with backend.batch_update() as batch:
|
||||
batch.add_or_update_ids([doc.pk])
|
||||
|
||||
assert backend.search_ids("doc", user=user) == [doc.pk]
|
||||
other = UserFactory()
|
||||
assert backend.search_ids("doc", user=other) == []
|
||||
|
||||
def test_indexes_notes_and_custom_fields(self, backend: TantivyBackend) -> None:
|
||||
note_author = UserFactory(username="noter")
|
||||
field = CustomField.objects.create(
|
||||
name="Invoice Number",
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
doc = Document.objects.create(title="doc", checksum="RICH1", pk=1)
|
||||
Note.objects.create(document=doc, note="Reviewed", user=note_author)
|
||||
CustomFieldInstance.objects.create(
|
||||
document=doc,
|
||||
field=field,
|
||||
value_text="INV-42",
|
||||
)
|
||||
|
||||
with backend.batch_update() as batch:
|
||||
batch.add_or_update_ids([doc.pk])
|
||||
|
||||
assert backend.search_ids("notes.user:noter", user=None) == [doc.pk]
|
||||
assert backend.search_ids("custom_fields.value:INV-42", user=None) == [
|
||||
doc.pk,
|
||||
]
|
||||
|
||||
def test_uses_effective_content_for_versioned_documents(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
root = Document.objects.create(
|
||||
title="Statement",
|
||||
content="stale text",
|
||||
checksum="ROOT1",
|
||||
pk=1,
|
||||
)
|
||||
Document.objects.create(
|
||||
title="Statement",
|
||||
content="latest version text",
|
||||
checksum="VER1",
|
||||
pk=2,
|
||||
root_document=root,
|
||||
version_index=1,
|
||||
)
|
||||
|
||||
with backend.batch_update() as batch:
|
||||
batch.add_or_update_ids([root.pk])
|
||||
|
||||
assert backend.search_ids("latest", user=None) == [root.pk]
|
||||
assert backend.search_ids("stale", user=None) == []
|
||||
|
||||
def test_reindexes_documents_already_in_the_index(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""add_or_update_ids must upsert, matching add_or_update's behaviour."""
|
||||
doc = Document.objects.create(
|
||||
title="doc",
|
||||
content="original",
|
||||
checksum="UP1",
|
||||
pk=1,
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
assert backend.search_ids("original", user=None) == [doc.pk]
|
||||
|
||||
doc.content = "updated"
|
||||
doc.save()
|
||||
|
||||
with backend.batch_update() as batch:
|
||||
batch.add_or_update_ids([doc.pk])
|
||||
|
||||
assert backend.search_ids("original", user=None) == []
|
||||
assert backend.search_ids("updated", user=None) == [doc.pk]
|
||||
|
||||
|
||||
class TestSearch:
|
||||
"""Test search query parsing and matching via search_ids."""
|
||||
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
"""``checksum`` wildcard patterns stay literal end to end, once user queries
|
||||
route through whoosh-compat.
|
||||
|
||||
The registry-level fact (the pattern normalizer folds a KEYWORD pattern
|
||||
rather than stemming it) is pinned on its own in
|
||||
``test_keyword_pattern_literal.py``. This proves it actually reaches a real
|
||||
query: ``checksum:ceded*`` must match only the document whose checksum
|
||||
starts with "ceded", not the one whose checksum stems to the same run.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
CEDEF00D = "cedef00ddeadbeef0123456789abcdef01234567"
|
||||
CEDEDEAD = "cededeadbeef567801234567" + "89abcdef01234567"
|
||||
|
||||
|
||||
class TestChecksumPrefixQueries:
|
||||
@pytest.fixture
|
||||
def indexed(self, backend: TantivyBackend) -> None:
|
||||
for i, checksum in enumerate((CEDEF00D, CEDEDEAD)):
|
||||
doc = Document.objects.create(
|
||||
title=f"Checksum doc {i}",
|
||||
content="invoices for the quarter",
|
||||
checksum=checksum,
|
||||
archive_serial_number=940 + i,
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
def _ids(self, backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
def test_prefix_matches_only_the_document_that_starts_with_it(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed: None,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two documents indexed with checksums that share a stem when
|
||||
run through the English stemmer ("cedef00d..." and
|
||||
"cededead...") but only one literally starts with "ceded"
|
||||
WHEN:
|
||||
- "checksum:ceded*" is searched
|
||||
THEN:
|
||||
- Only the document whose checksum literally starts with
|
||||
"ceded" matches; the pattern normalizer folds a KEYWORD
|
||||
pattern rather than stemming it, so this reaches a real
|
||||
query end to end
|
||||
"""
|
||||
matched = self._ids(backend, "checksum:ceded*")
|
||||
expected = Document.objects.get(checksum=CEDEDEAD).pk
|
||||
assert matched == {expected}
|
||||
|
||||
def test_text_prefix_still_reaches_the_stemmed_index(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed: None,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two documents indexed with content "invoices for the
|
||||
quarter"
|
||||
WHEN:
|
||||
- "invoice*" is searched against the TEXT content field
|
||||
THEN:
|
||||
- Both documents match, confirming the checksum field's
|
||||
literal-pattern behavior is specific to KEYWORD fields and
|
||||
does not affect TEXT field wildcard matching against
|
||||
stemmed terms
|
||||
"""
|
||||
assert len(self._ids(backend, "invoice*")) == 2
|
||||
@@ -0,0 +1,212 @@
|
||||
"""The CJK bigram clause blended into QUERY-mode searches.
|
||||
|
||||
The clause exists so CJK runs are matchable at all (the default analyzers
|
||||
keep a whitespace-free CJK run as one indivisible token), but it must not
|
||||
widen the query beyond what the user asked for: a CJK term the query
|
||||
excludes, or restricts to one field, must not come back through it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestCjkParseFailureDegradesGracefully:
|
||||
def test_a_cjk_run_tantivy_cannot_parse_drops_the_clause_only(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A CJK run and an index-like object whose parse_query is
|
||||
forced to raise
|
||||
WHEN:
|
||||
- _parse_cjk_text is called
|
||||
THEN:
|
||||
- It returns None instead of propagating, so a CJK run tantivy
|
||||
cannot parse only drops the bigram clause rather than
|
||||
failing the whole query. Broad on purpose (bare except
|
||||
Exception), unlike the fuzzy blend's narrower ValueError
|
||||
guard: a CJK run is not filtered to a guaranteed-safe token
|
||||
set the way the fuzzy blend's word string is, so the exact
|
||||
failure mode tantivy could raise here is not pinned down
|
||||
"""
|
||||
from documents.search._query import _parse_cjk_text
|
||||
|
||||
class _RaisingIndex:
|
||||
def parse_query(self, *args: object, **kwargs: object) -> object:
|
||||
raise RuntimeError("synthetic parse failure")
|
||||
|
||||
assert _parse_cjk_text(_RaisingIndex(), "東京", ["bigram_content"]) is None
|
||||
|
||||
def test_no_cjk_text_at_all_returns_none_without_parsing(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A raw query string with no CJK characters at all
|
||||
WHEN:
|
||||
- _build_cjk_query (the simple TEXT/TITLE-mode builder) is
|
||||
called directly
|
||||
THEN:
|
||||
- It returns None without ever attempting to parse anything.
|
||||
The only real caller already guards this with _has_cjk(),
|
||||
so this is defensive: it keeps the function safe to call on
|
||||
its own, not a path a real search currently reaches
|
||||
"""
|
||||
from documents.search._query import _build_cjk_query
|
||||
|
||||
assert _build_cjk_query(None, "invoice total due", ["bigram_content"]) is None
|
||||
|
||||
|
||||
class TestCjkClauseFollowsTheParsedQuery:
|
||||
def test_negated_cjk_term_is_excluded(self, backend: TantivyBackend) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two documents both matching "invoice", one whose content
|
||||
also contains 漢字
|
||||
WHEN:
|
||||
- "invoice NOT 漢字" is searched
|
||||
THEN:
|
||||
- Only the document without 漢字 matches; 'invoice NOT 漢字'
|
||||
must not return the document containing 漢字
|
||||
"""
|
||||
with_cjk = _index(
|
||||
backend,
|
||||
title="Invoice A",
|
||||
content="invoice total 漢字",
|
||||
checksum="cjk-neg-1",
|
||||
)
|
||||
without_cjk = _index(
|
||||
backend,
|
||||
title="Invoice B",
|
||||
content="invoice total only",
|
||||
checksum="cjk-neg-2",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "invoice") == {with_cjk.pk, without_cjk.pk}
|
||||
assert _matched_ids(backend, "invoice NOT 漢字") == {without_cjk.pk}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("threshold", "expected"),
|
||||
[
|
||||
pytest.param(None, {"titled"}, id="fuzzy_off"),
|
||||
pytest.param(0.0, {"titled", "content_only"}, id="fuzzy_on"),
|
||||
],
|
||||
)
|
||||
def test_fielded_cjk_term_searches_only_that_field(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
settings: SettingsWrapper,
|
||||
threshold: float | None,
|
||||
expected: set[str],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- One document with 東京 in its title, another with 東京 only
|
||||
in its content, and ADVANCED_FUZZY_SEARCH_THRESHOLD either
|
||||
off or on
|
||||
WHEN:
|
||||
- "title:東京" is searched
|
||||
THEN:
|
||||
- With fuzzy off, only the titled document matches: the CJK
|
||||
clause honours the field, so 'title:東京' must not match a
|
||||
document whose 東京 is only in the content. With fuzzy on,
|
||||
the content-only document is also readmitted, because the
|
||||
fuzzy clause contributes every free-text term UNFIELDED by
|
||||
design (see _try_parse_fuzzy_query) on its own
|
||||
0.1-boosted terms -- a documented trade-off, pinned here so
|
||||
it stays deliberate
|
||||
"""
|
||||
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = threshold
|
||||
content_only = _index(
|
||||
backend,
|
||||
title="Tokyo report",
|
||||
content="東京都の人口は約1400万人です",
|
||||
checksum="cjk-field-1",
|
||||
)
|
||||
titled = _index(
|
||||
backend,
|
||||
title="東京都の報告書",
|
||||
content="an english summary",
|
||||
checksum="cjk-field-2",
|
||||
)
|
||||
pks = {"titled": titled.pk, "content_only": content_only.pk}
|
||||
|
||||
assert _matched_ids(backend, "東京") == set(pks.values())
|
||||
assert _matched_ids(backend, "title:東京") == {pks[label] for label in expected}
|
||||
|
||||
def test_cjk_on_a_non_default_field_builds_no_clause(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with 東京 in its content
|
||||
WHEN:
|
||||
- "notes:東京" is searched (a field outside the default
|
||||
search fields)
|
||||
THEN:
|
||||
- Nothing matches; a CJK term restricted to a field outside
|
||||
the default search fields has nothing to contribute to the
|
||||
bigram clause, so it must not fall back to matching 東京 in
|
||||
the content
|
||||
"""
|
||||
_index(
|
||||
backend,
|
||||
title="Tokyo report",
|
||||
content="東京都の人口は約1400万人です",
|
||||
checksum="cjk-notes-1",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "notes:東京") == set()
|
||||
|
||||
def test_bare_cjk_term_still_matches_every_default_field(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- One document with 重要 in its content, another with 重要 in
|
||||
its title
|
||||
WHEN:
|
||||
- "重要" and "重要 OR report" are each searched unfielded
|
||||
THEN:
|
||||
- Both documents match either way; the clause's reason for
|
||||
existing is that an unfielded CJK run matches wherever it
|
||||
is indexed, and does so alongside a latin term
|
||||
"""
|
||||
in_content = _index(
|
||||
backend,
|
||||
title="report",
|
||||
content="本文に重要な情報",
|
||||
checksum="cjk-bare-1",
|
||||
)
|
||||
in_title = _index(
|
||||
backend,
|
||||
title="重要な報告書",
|
||||
content="english only",
|
||||
checksum="cjk-bare-2",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "重要") == {in_content.pk, in_title.pk}
|
||||
assert _matched_ids(backend, "重要 OR report") == {
|
||||
in_content.pk,
|
||||
in_title.pk,
|
||||
}
|
||||
@@ -0,0 +1,86 @@
|
||||
"""Whoosh's compact, separator-free date spelling, resolved end to end.
|
||||
|
||||
whoosh-compat owns both widths of this spelling and asserts both forms'
|
||||
bounds directly in its own test suite: the 8-digit form as a whole calendar
|
||||
day (lower bound, upper bound and exclusivity), and the 14-digit form as a
|
||||
single instant. The 14-digit form is kept here as the single representative
|
||||
because it is the one that exercises paperless's ``added`` DATETIME fast
|
||||
field at full precision: the corpus separates a document at the named
|
||||
instant from one on the same calendar day at another hour and one on the
|
||||
next day at the same hour, so a query that degrades into a whole-day
|
||||
window, or drops the time of day, matches the wrong set rather than passing
|
||||
on a corpus that could not tell the difference.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC
|
||||
from datetime import datetime
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def docs(backend: TantivyBackend) -> dict[str, int]:
|
||||
return {
|
||||
"instant": _index(
|
||||
backend,
|
||||
title="On the instant",
|
||||
content="x",
|
||||
checksum="compact-date-instant",
|
||||
added=datetime(2005, 3, 4, 15, 30, tzinfo=UTC),
|
||||
).pk,
|
||||
"same_day": _index(
|
||||
backend,
|
||||
title="Same day, other hour",
|
||||
content="x",
|
||||
checksum="compact-date-same-day",
|
||||
added=datetime(2005, 3, 4, 9, 0, tzinfo=UTC),
|
||||
).pk,
|
||||
"next_day": _index(
|
||||
backend,
|
||||
title="Next day, same hour",
|
||||
content="x",
|
||||
checksum="compact-date-next-day",
|
||||
added=datetime(2005, 3, 5, 15, 30, tzinfo=UTC),
|
||||
).pk,
|
||||
}
|
||||
|
||||
|
||||
def test_fourteen_digits_is_a_single_instant(
|
||||
backend: TantivyBackend,
|
||||
docs: dict[str, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Three documents indexed on the ``added`` DATETIME fast field:
|
||||
one at 2005-03-04T15:30:00, one on the same calendar day at a
|
||||
different hour, and one on the next day at the same hour
|
||||
WHEN:
|
||||
- Searching with the 14-digit compact date form
|
||||
``added:20050304153000``
|
||||
THEN:
|
||||
- Only the document at that exact instant matches; the same-day
|
||||
document is what tells this apart from the 8-digit day-window
|
||||
form, and the next-day document from a form that ignored the
|
||||
time of day altogether
|
||||
"""
|
||||
assert _matched_ids(backend, "added:20050304153000") == {docs["instant"]}
|
||||
@@ -0,0 +1,149 @@
|
||||
"""_ConjunctiveNegations, the AST visitor that collects the subtrees a
|
||||
query excludes from every document it matches, and _any_of, the clause-list
|
||||
collapsing helper it feeds into.
|
||||
|
||||
Result-level proof that a negation reached through NOT/AND survives the
|
||||
fuzzy/CJK blend lives in test_query_negation.py. These are direct unit
|
||||
tests of the visitor's dispatch for the rarer grammar shapes
|
||||
(AndNot/Boosted/AndMaybe/Require) that file's real-corpus queries don't
|
||||
happen to exercise, plus the empty-clause-list case of _any_of.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import whoosh_compat.ast as wc_ast
|
||||
|
||||
from documents.models import Document
|
||||
from documents.search._query import _any_of
|
||||
from documents.search._query import _ConjunctiveNegations
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _term(text: str) -> wc_ast.Term:
|
||||
return wc_ast.Term(field=None, text=text)
|
||||
|
||||
|
||||
class TestConjunctiveNegationsVisitor:
|
||||
def test_visit_andnot_hoists_the_negative_branch(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- An AndNot(positive=a, negative=b) node
|
||||
WHEN:
|
||||
- _ConjunctiveNegations visits it
|
||||
THEN:
|
||||
- The negative branch is collected as an exclusion, since
|
||||
AndNot requires positive and excludes negative
|
||||
"""
|
||||
negative = _term("b")
|
||||
node = wc_ast.AndNot(positive=_term("a"), negative=negative)
|
||||
assert _ConjunctiveNegations().visit(node) == (negative,)
|
||||
|
||||
def test_visit_andnot_also_collects_negations_already_in_the_positive_branch(
|
||||
self,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- An AndNot node whose positive branch already contains a NOT
|
||||
WHEN:
|
||||
- _ConjunctiveNegations visits it
|
||||
THEN:
|
||||
- Both the positive branch's own negation and the AndNot's
|
||||
negative branch are collected
|
||||
"""
|
||||
excluded_in_positive = _term("excluded")
|
||||
negative = _term("negative")
|
||||
node = wc_ast.AndNot(
|
||||
positive=wc_ast.Not(child=excluded_in_positive),
|
||||
negative=negative,
|
||||
)
|
||||
assert _ConjunctiveNegations().visit(node) == (excluded_in_positive, negative)
|
||||
|
||||
def test_visit_boosted_passes_through_to_the_child(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A Boosted node (e.g. "(invoice NOT secret)^2") wrapping a
|
||||
NOT
|
||||
WHEN:
|
||||
- _ConjunctiveNegations visits it
|
||||
THEN:
|
||||
- The negation inside the boosted child is still collected: a
|
||||
boost must not shield an exclusion from being hoisted
|
||||
"""
|
||||
excluded = _term("secret")
|
||||
node = wc_ast.Boosted(child=wc_ast.Not(child=excluded), boost=2.0)
|
||||
assert _ConjunctiveNegations().visit(node) == (excluded,)
|
||||
|
||||
def test_visit_andmaybe_only_descends_into_required(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- An AndMaybe(required=a, optional=b) node where both required
|
||||
and optional contain their own NOT
|
||||
WHEN:
|
||||
- _ConjunctiveNegations visits it
|
||||
THEN:
|
||||
- Only the negation in the required branch is collected. The
|
||||
optional branch is not a conjunctive constraint on the whole
|
||||
query (documents that fail it still match), so hoisting a
|
||||
negation from it would exclude documents the query does not
|
||||
actually exclude
|
||||
"""
|
||||
excluded_in_required = _term("excluded_in_required")
|
||||
excluded_in_optional = _term("excluded_in_optional")
|
||||
node = wc_ast.AndMaybe(
|
||||
required=wc_ast.Not(child=excluded_in_required),
|
||||
optional=wc_ast.Not(child=excluded_in_optional),
|
||||
)
|
||||
assert _ConjunctiveNegations().visit(node) == (excluded_in_required,)
|
||||
|
||||
def test_visit_require_descends_into_both_branches(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A Require(scored=a, filter_only=b) node where both scored
|
||||
and filter_only contain their own NOT
|
||||
WHEN:
|
||||
- _ConjunctiveNegations visits it
|
||||
THEN:
|
||||
- Both negations are collected: Require constrains the whole
|
||||
query with both branches, one merely scored and the other
|
||||
filter-only, so both are conjunctive
|
||||
"""
|
||||
excluded_in_scored = _term("excluded_in_scored")
|
||||
excluded_in_filter = _term("excluded_in_filter")
|
||||
node = wc_ast.Require(
|
||||
scored=wc_ast.Not(child=excluded_in_scored),
|
||||
filter_only=wc_ast.Not(child=excluded_in_filter),
|
||||
)
|
||||
assert _ConjunctiveNegations().visit(node) == (
|
||||
excluded_in_scored,
|
||||
excluded_in_filter,
|
||||
)
|
||||
|
||||
|
||||
class TestAnyOfEmptyClauseList:
|
||||
def test_no_clauses_returns_a_query_that_matches_nothing(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- No clauses at all
|
||||
WHEN:
|
||||
- _any_of is called with an empty list
|
||||
THEN:
|
||||
- It returns tantivy's empty_query() rather than raising or
|
||||
wrapping zero clauses in a boolean_query, and running it
|
||||
against a real index matches no documents
|
||||
"""
|
||||
doc = Document.objects.create(title="x", content="x", checksum="any-of-empty")
|
||||
backend.add_or_update(doc)
|
||||
|
||||
query = _any_of([])
|
||||
results = backend._index.searcher().search(query, limit=10)
|
||||
assert len(results.hits) == 0
|
||||
@@ -0,0 +1,91 @@
|
||||
"""Pins the correctness gained by deleting the pre-parse
|
||||
_quote_date_keyword_phrases rewrite.
|
||||
|
||||
That rewrite matched date-keyword phrases (e.g. "previous month" after a
|
||||
date field) anywhere in the raw query string, including inside an
|
||||
unrelated quoted string, and inserted quotes mid-phrase there too. Its
|
||||
own docstring gave ``title:"see added:previous month notes"`` as the
|
||||
example of what it corrupted. whoosh-compat's grammar accepts the same
|
||||
phrase vocabulary unquoted natively (see TestUnquotedDateKeywordPhrases
|
||||
in test_acceptance.py), so the rewrite was redundant everywhere it was
|
||||
safe and actively wrong everywhere it was not. This is the one case that
|
||||
tells the two apart: a literal title phrase that happens to contain
|
||||
"added:previous month" as running text.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestQuotedStringContainingDateKeywordText:
|
||||
"""A quoted title phrase containing the literal text
|
||||
"added:previous month" as running words must match on that literal
|
||||
text alone, never spill into an unfielded search for "previous" and
|
||||
"month" across the default search fields the way the deleted rewrite
|
||||
would have decomposed it into."""
|
||||
|
||||
def test_matches_only_the_literal_phrase(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document whose title literally contains "see
|
||||
added:previous month notes", and a decoy document whose
|
||||
title/content carry the individual fragments the deleted
|
||||
_quote_date_keyword_phrases rewrite would have decomposed
|
||||
the phrase into (the decoy would incorrectly match under
|
||||
the deleted rewrite: its title contains the "see added:"
|
||||
and " notes" fragments the corrupted parse required as
|
||||
title phrases, and its content supplies "previous" and
|
||||
"month" as the decomposed word-match clauses the rewrite
|
||||
turned the middle of the phrase into)
|
||||
WHEN:
|
||||
- 'title:"see added:previous month notes"' is searched
|
||||
THEN:
|
||||
- Only the document with the literal phrase matches; it must
|
||||
never spill into an unfielded search for "previous" and
|
||||
"month" across the default search fields
|
||||
"""
|
||||
literal = _index(
|
||||
backend,
|
||||
title="see added:previous month notes",
|
||||
content="quarterly filing",
|
||||
checksum="dkp-literal",
|
||||
archive_serial_number=920,
|
||||
)
|
||||
# Under the deleted rewrite, this decoy would incorrectly match:
|
||||
# its title contains the "see added:" and " notes" fragments the
|
||||
# corrupted parse required as title phrases, and its content
|
||||
# supplies "previous" and "month" as the decomposed word-match
|
||||
# clauses the rewrite turned the middle of the phrase into.
|
||||
decoy = _index(
|
||||
backend,
|
||||
title="see added: quarterly report notes",
|
||||
content="we reviewed the previous statement about month end",
|
||||
checksum="dkp-decoy",
|
||||
archive_serial_number=921,
|
||||
)
|
||||
query = 'title:"see added:previous month notes"'
|
||||
assert _matched_ids(backend, query) == {literal.pk}
|
||||
assert decoy.pk not in _matched_ids(backend, query)
|
||||
@@ -0,0 +1,100 @@
|
||||
"""Date keyword phrases (``today``, etc.) resolved in a non-UTC timezone,
|
||||
end to end.
|
||||
|
||||
paperless's own ``tz=get_current_timezone()`` plumbing
|
||||
(``TantivyBackend._parse_query``) is exercised elsewhere only for
|
||||
relative *ranges* (``added:[-1 week to now]``, in
|
||||
documents/tests/test_api_search.py). This covers a date *keyword*
|
||||
(``today``), whose day boundary depends on the active timezone the same
|
||||
way but goes through whoosh-compat's DateParserPlugin resolution instead
|
||||
of an explicit range.
|
||||
|
||||
Discriminating shape: frozen at 2026-06-15T02:00 UTC, which is
|
||||
2026-06-14T22:00 in America/New_York -- still "today" (06-14) there, but
|
||||
already "today" (06-15) in UTC. Two documents pin both directions of the
|
||||
mistake a hardcoded-UTC bug would make:
|
||||
|
||||
- ``in_ny_today`` (added 2026-06-14T20:00 UTC = 2026-06-14T16:00 NY) is
|
||||
inside New York's "today" window and outside a naive UTC-calendar-day
|
||||
window. A ``tz``-ignoring bug would miss it.
|
||||
- ``in_utc_calendar_day_only`` (added 2026-06-15T10:00 UTC =
|
||||
2026-06-15T06:00 NY) is inside a naive UTC-calendar-day window but
|
||||
outside New York's actual "today" window. A ``tz``-ignoring bug would
|
||||
wrongly match it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC
|
||||
from datetime import datetime
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import time_machine
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
FROZEN_NOW = datetime(2026, 6, 15, 2, 0, tzinfo=UTC)
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestDateKeywordUsesTheActiveTimezone:
|
||||
def test_today_matches_the_new_york_calendar_day_not_the_utc_one(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
settings: SettingsWrapper,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- TIME_ZONE set to America/New_York, time frozen at
|
||||
2026-06-15T02:00 UTC (2026-06-14T22:00 NY -- still "today"
|
||||
there, but already "today" in UTC), and two documents: one
|
||||
added inside New York's "today" window but outside a naive
|
||||
UTC-calendar-day window, the other the reverse (inside a
|
||||
naive UTC-calendar-day window but outside New York's actual
|
||||
"today")
|
||||
WHEN:
|
||||
- "added:today" is searched
|
||||
THEN:
|
||||
- Only the document inside New York's actual "today" window
|
||||
matches, proving our tz=get_current_timezone() plumbing
|
||||
resolves the date keyword in the active timezone rather
|
||||
than a hardcoded UTC calendar day
|
||||
"""
|
||||
settings.TIME_ZONE = "America/New_York"
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
in_ny_today = _index(
|
||||
backend,
|
||||
title="NY today",
|
||||
content="x",
|
||||
checksum="tz-keyword-ny-today",
|
||||
added=datetime(2026, 6, 14, 20, 0, tzinfo=UTC),
|
||||
)
|
||||
# Not captured: the exact-set assertion below already proves
|
||||
# this document (inside a naive UTC-calendar-day window, but
|
||||
# outside New York's actual "today") does not match.
|
||||
_index(
|
||||
backend,
|
||||
title="UTC calendar day only",
|
||||
content="x",
|
||||
checksum="tz-keyword-utc-calendar-day-only",
|
||||
added=datetime(2026, 6, 15, 10, 0, tzinfo=UTC),
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "added:today") == {in_ny_today.pk}
|
||||
@@ -0,0 +1,32 @@
|
||||
"""``_DEFAULT_SEARCH_FIELDS`` must stay a subset of the registered public
|
||||
field names.
|
||||
|
||||
Nothing enforced this before: a rename in PUBLIC_FIELDS not mirrored in
|
||||
``_DEFAULT_SEARCH_FIELDS`` (documents/search/_query.py) would 400 every
|
||||
unfielded search at request time, since ``index.parse_query`` and the
|
||||
fuzzy/CJK clause builders are handed a field name the schema no longer
|
||||
has.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from documents.search._fields import PUBLIC_FIELDS
|
||||
from documents.search._query import _DEFAULT_SEARCH_FIELDS
|
||||
|
||||
|
||||
class TestDefaultSearchFieldsAreRegistered:
|
||||
def test_every_default_search_field_is_a_public_field(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- PUBLIC_FIELDS and _DEFAULT_SEARCH_FIELDS, our own field
|
||||
tables
|
||||
WHEN:
|
||||
- Every name in _DEFAULT_SEARCH_FIELDS is checked against the
|
||||
registered public field names
|
||||
THEN:
|
||||
- Every one is present; a rename in PUBLIC_FIELDS not
|
||||
mirrored here would 400 every unfielded search at request
|
||||
time
|
||||
"""
|
||||
public_field_names = {f.name for f in PUBLIC_FIELDS}
|
||||
assert set(_DEFAULT_SEARCH_FIELDS) <= public_field_names
|
||||
@@ -0,0 +1,475 @@
|
||||
"""Pins the search syntax that ``docs/usage.md`` promises users.
|
||||
|
||||
Every query here is syntax the "Document searches" section of
|
||||
``docs/usage.md`` documents, either spelled as the docs spell it or as a
|
||||
concrete instance of a form the docs describe. Each case indexes real
|
||||
documents and asserts on matched document IDs rather than on the parsed
|
||||
query, because a query that parses cleanly is not necessarily a query that
|
||||
means what the documentation says it means: ``added:now`` parses without a
|
||||
single diagnostic and then matches nothing, because it resolves to an
|
||||
instant rather than to a span.
|
||||
|
||||
The negative cases matter as much as the positive ones. They pin the
|
||||
behaviours the docs explicitly warn about, so that if any of them ever
|
||||
starts working the warning can be removed deliberately rather than being
|
||||
left standing as a lie.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC
|
||||
from datetime import datetime
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import time_machine
|
||||
|
||||
from documents.models import Document
|
||||
from documents.models import Note
|
||||
from documents.models import Tag
|
||||
from documents.search._errors import InvalidDateQuery
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Generator
|
||||
|
||||
from django.contrib.auth.models import User
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
# A Monday, so that "next monday"/"last monday" land a clean week either side.
|
||||
FROZEN_NOW = datetime(2026, 6, 15, 12, 0, tzinfo=UTC)
|
||||
|
||||
# The checksum used in the docs' `checksum:` example.
|
||||
DOC_CHECKSUM = "9f86d081884c7d659a2feaa0c55ad015a3bf4f1b2b0b822cd15d6c15b0f00a08"
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestLogicalExpressions:
|
||||
@pytest.fixture
|
||||
def docs(self, backend: TantivyBackend) -> dict[str, int]:
|
||||
return {
|
||||
"secret": _index(
|
||||
backend,
|
||||
title="Invoice one",
|
||||
content="invoice secret contents",
|
||||
checksum="doc-syntax-secret",
|
||||
).pk,
|
||||
"plain": _index(
|
||||
backend,
|
||||
title="Invoice two",
|
||||
content="invoice ordinary contents",
|
||||
checksum="doc-syntax-plain",
|
||||
).pk,
|
||||
}
|
||||
|
||||
def test_not_excludes_a_term(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
docs: dict[str, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two indexed documents, one containing "secret" and one not
|
||||
WHEN:
|
||||
- "invoice NOT secret" is searched, as docs/usage.md documents
|
||||
THEN:
|
||||
- Only the document without "secret" matches
|
||||
"""
|
||||
assert _matched_ids(backend, "invoice NOT secret") == {docs["plain"]}
|
||||
|
||||
def test_leading_hyphen_requires_the_term_instead_of_excluding_it(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
docs: dict[str, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two indexed documents, one containing "secret" and one not
|
||||
WHEN:
|
||||
- "invoice -secret" is searched (a leading hyphen, not "NOT")
|
||||
THEN:
|
||||
- Only the document containing "secret" matches, because
|
||||
separators are stripped at index time, so "-secret" is
|
||||
indexed as the plain term "secret" and the query becomes an
|
||||
AND rather than an exclusion, exactly as the docs warn
|
||||
"""
|
||||
assert _matched_ids(backend, "invoice -secret") == {docs["secret"]}
|
||||
|
||||
def test_or_inside_parentheses_matches_either_branch(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
docs: dict[str, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two indexed documents, one containing "secret" and one
|
||||
containing "ordinary"
|
||||
WHEN:
|
||||
- "invoice AND (secret OR ordinary)" is searched
|
||||
THEN:
|
||||
- Both documents match
|
||||
"""
|
||||
matched = _matched_ids(backend, "invoice AND (secret OR ordinary)")
|
||||
assert matched == {docs["secret"], docs["plain"]}
|
||||
|
||||
|
||||
class TestPhraseSearch:
|
||||
def test_quoted_phrase_requires_the_words_in_order(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document whose content contains "the quick brown fox jumps"
|
||||
WHEN:
|
||||
- A quoted phrase is searched, in order and out of order
|
||||
THEN:
|
||||
- The in-order phrase matches, and the same words reordered do
|
||||
not
|
||||
"""
|
||||
doc = _index(
|
||||
backend,
|
||||
title="Phrase",
|
||||
content="the quick brown fox jumps",
|
||||
checksum="doc-syntax-phrase",
|
||||
)
|
||||
assert _matched_ids(backend, '"quick brown fox"') == {doc.pk}
|
||||
assert _matched_ids(backend, '"brown quick fox"') == set()
|
||||
|
||||
|
||||
class TestTagCommaList:
|
||||
"""``tag:bills,unpaid`` is published syntax (docs/usage.md), so this checks
|
||||
that the documented spelling still returns what the docs promise: only the
|
||||
document carrying every listed tag.
|
||||
|
||||
It is deliberately not proof of paperless's field configuration, and must
|
||||
not be read as such. Removing ``comma_values`` from the ``tag`` FieldSpec
|
||||
leaves this test passing, because paperless's analyzer splits the literal
|
||||
value "bills,unpaid" into the same two tokens the value-list reading
|
||||
produces, so the two readings select the same documents. The registry fact
|
||||
-- that ``tag`` opts in and no other field does -- is observable only at
|
||||
the registry, and is owned by test_registry.py's
|
||||
``test_tag_is_comma_values``/``test_correspondent_is_not_comma_values``.
|
||||
"""
|
||||
|
||||
def test_comma_list_requires_every_listed_tag(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document carrying both "bills" and "unpaid" tags, and a
|
||||
second document carrying only "bills" (plus "archived")
|
||||
WHEN:
|
||||
- "tag:bills,unpaid" is searched
|
||||
THEN:
|
||||
- Only the document carrying every listed tag matches, and a
|
||||
single-tag "tag:bills" search still matches both documents
|
||||
"""
|
||||
bills = Tag.objects.create(name="bills")
|
||||
unpaid = Tag.objects.create(name="unpaid")
|
||||
archived = Tag.objects.create(name="archived")
|
||||
|
||||
both = Document.objects.create(
|
||||
title="Both tags",
|
||||
content="body",
|
||||
checksum="doc-syntax-tag-both",
|
||||
)
|
||||
both.tags.add(bills, unpaid)
|
||||
backend.add_or_update(both)
|
||||
|
||||
one = Document.objects.create(
|
||||
title="One tag",
|
||||
content="body",
|
||||
checksum="doc-syntax-tag-one",
|
||||
)
|
||||
one.tags.add(bills, archived)
|
||||
backend.add_or_update(one)
|
||||
|
||||
assert _matched_ids(backend, "tag:bills,unpaid") == {both.pk}
|
||||
assert _matched_ids(backend, "tag:bills") == {both.pk, one.pk}
|
||||
|
||||
|
||||
class TestArchiveMetadataFields:
|
||||
@pytest.fixture
|
||||
def doc(self, backend: TantivyBackend, admin_user: User) -> Document:
|
||||
doc = Document.objects.create(
|
||||
title="Metadata",
|
||||
content="body",
|
||||
checksum=DOC_CHECKSUM,
|
||||
archive_serial_number=100,
|
||||
page_count=12,
|
||||
original_filename="invoice.pdf",
|
||||
)
|
||||
Note.objects.create(document=doc, user=admin_user, note="a note")
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
"asn:100",
|
||||
"asn:[50 to 150]",
|
||||
"page_count:12",
|
||||
"page_count:[10 to 20]",
|
||||
"num_notes:1",
|
||||
"num_notes:[1 to 5]",
|
||||
"original_filename:invoice.pdf",
|
||||
f"checksum:{DOC_CHECKSUM}",
|
||||
"checksum:9f86d081*",
|
||||
# A checksum term is stored verbatim, but a checksum *pattern* is
|
||||
# lowercased before it is matched, so an uppercase prefix pattern
|
||||
# still matches even though the uppercase term in the negative
|
||||
# list below does not.
|
||||
"checksum:9F86D081*",
|
||||
],
|
||||
)
|
||||
def test_documented_metadata_query_matches(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
doc: Document,
|
||||
query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with an ASN, page count, a note, an original
|
||||
filename and a known checksum
|
||||
WHEN:
|
||||
- Every documented metadata-field spelling (exact value,
|
||||
range, and, for checksum, a lowercase prefix pattern
|
||||
regardless of the case the pattern itself is typed in) is
|
||||
searched
|
||||
THEN:
|
||||
- Each one matches the document
|
||||
"""
|
||||
assert _matched_ids(backend, query) == {doc.pk}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
# The docs say only a complete, lowercase checksum matches.
|
||||
"checksum:9f86d081",
|
||||
f"checksum:{DOC_CHECKSUM.upper()}",
|
||||
],
|
||||
)
|
||||
def test_partial_or_uppercase_checksum_matches_nothing(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
doc: Document,
|
||||
query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with a known, complete, lowercase checksum
|
||||
WHEN:
|
||||
- An exact-value search is run with a partial or uppercase
|
||||
spelling of that checksum
|
||||
THEN:
|
||||
- Nothing matches, as the docs say only a complete, lowercase
|
||||
checksum matches as an exact value
|
||||
"""
|
||||
assert _matched_ids(backend, query) == set()
|
||||
|
||||
|
||||
class TestDocumentedDateForms:
|
||||
@pytest.fixture(autouse=True)
|
||||
def frozen_now(self) -> Generator[None, None, None]:
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
yield
|
||||
|
||||
@pytest.fixture
|
||||
def dated(self, backend: TantivyBackend) -> dict[str, int]:
|
||||
stamps = {
|
||||
"today": datetime(2026, 6, 15, 9, 0, tzinfo=UTC),
|
||||
"yesterday": datetime(2026, 6, 14, 9, 0, tzinfo=UTC),
|
||||
"tomorrow": datetime(2026, 6, 16, 9, 0, tzinfo=UTC),
|
||||
"next_monday": datetime(2026, 6, 22, 10, 0, tzinfo=UTC),
|
||||
"last_monday": datetime(2026, 6, 8, 10, 0, tzinfo=UTC),
|
||||
"january": datetime(2026, 1, 10, 10, 0, tzinfo=UTC),
|
||||
"old": datetime(2005, 3, 4, 15, 30, tzinfo=UTC),
|
||||
}
|
||||
return {
|
||||
label: _index(
|
||||
backend,
|
||||
title=label,
|
||||
content="dated body",
|
||||
checksum=f"doc-syntax-date-{label}",
|
||||
added=stamp,
|
||||
).pk
|
||||
for label, stamp in stamps.items()
|
||||
}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("query", "label"),
|
||||
[
|
||||
("added:today", "today"),
|
||||
("added:yesterday", "yesterday"),
|
||||
("added:tomorrow", "tomorrow"),
|
||||
('added:"next monday"', "next_monday"),
|
||||
('added:"last monday"', "last_monday"),
|
||||
("added:january", "january"),
|
||||
("added:2005-03-04", "old"),
|
||||
("added:2005-03", "old"),
|
||||
("added:[2005-01-01 to 2005-12-31]", "old"),
|
||||
("added:[2005 to 2009]", "old"),
|
||||
# A full timestamp works, but only quoted when it stands alone,
|
||||
# and only unquoted when it is a range bound. The bare standalone
|
||||
# spelling is pinned as a non-match below.
|
||||
('added:"2005-03-04T15:30:00Z"', "old"),
|
||||
("added:[2005-03-04T09:00:00Z to 2005-03-04T17:00:00Z]", "old"),
|
||||
# A quoted range bound works when the quotes are single ones; the
|
||||
# double-quoted spelling is pinned as an error below.
|
||||
("added:['2005-03-04' to 2005-03-05]", "old"),
|
||||
],
|
||||
)
|
||||
def test_documented_date_form_matches_its_day_or_month(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
dated: dict[str, int],
|
||||
query: str,
|
||||
label: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Documents dated today, yesterday, tomorrow, next/last
|
||||
Monday, in January, and on an old fixed date, indexed
|
||||
against a frozen "now" (a Monday)
|
||||
WHEN:
|
||||
- Every documented date-form spelling is searched: relative
|
||||
keywords, quoted multi-word phrases, a bare year-month, an
|
||||
explicit range, a quoted full timestamp standing alone, an
|
||||
unquoted full timestamp as a range bound, and a
|
||||
single-quoted range bound
|
||||
THEN:
|
||||
- Each form matches exactly the document dated on its day or
|
||||
within its month
|
||||
"""
|
||||
assert _matched_ids(backend, query) == {dated[label]}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
# Zero-width: these resolve to a single instant, not a span, so
|
||||
# nothing in a realistic corpus lands on them. The docs warn
|
||||
# about them rather than presenting them as usable.
|
||||
"added:now",
|
||||
"added:noon",
|
||||
"added:midnight",
|
||||
# Quoting is what rescues the other multi-word date expressions,
|
||||
# so pin that it does not rescue these: the problem is the width
|
||||
# of the resulting range, not the way the value is delimited.
|
||||
# One quoted spelling is enough for that; which keyword sits
|
||||
# inside the quotes is grammar whoosh-compat owns.
|
||||
'added:"now"',
|
||||
# A relative offset, which the warning in the docs names by this
|
||||
# exact spelling. Standing alone it is an instant like the rest of
|
||||
# this list; the same offset used as a range bound is a real
|
||||
# window, pinned by the test below.
|
||||
'added:"-1 week"',
|
||||
],
|
||||
)
|
||||
def test_forms_the_docs_warn_about_match_nothing(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
dated: dict[str, int],
|
||||
query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A realistic dated corpus (see the `dated` fixture)
|
||||
WHEN:
|
||||
- A zero-width date form ("now", "noon", "midnight", a quoted
|
||||
"now") or a standalone relative offset ("-1 week") is
|
||||
searched: each resolves to a single instant rather than a
|
||||
span, and quoting does not rescue them the way it rescues
|
||||
other multi-word date expressions, since the problem is the
|
||||
width of the resulting range, not how the value is
|
||||
delimited
|
||||
THEN:
|
||||
- Nothing matches, exactly as the docs warn, rather than
|
||||
presenting these as usable spellings
|
||||
"""
|
||||
assert _matched_ids(backend, query) == set()
|
||||
|
||||
def test_bare_timestamp_is_rejected_rather_than_matching_nothing(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
dated: dict[str, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A realistic dated corpus, including a document dated at a
|
||||
known full timestamp
|
||||
WHEN:
|
||||
- The bare, unquoted spelling of that full timestamp is
|
||||
searched (the quoted and range-bound spellings pinned above
|
||||
do work and match this fixture's document)
|
||||
THEN:
|
||||
- `InvalidDateQuery` is raised rather than the query silently
|
||||
matching nothing, since this is a user-fixable error the
|
||||
docs tell the user to quote, and the reported value is the
|
||||
whole contiguous fragment the user typed, not just the
|
||||
prefix the date grammar's tokenizer first split on
|
||||
"""
|
||||
with pytest.raises(InvalidDateQuery) as exc_info:
|
||||
_matched_ids(backend, "added:2005-03-04T15:30:00Z")
|
||||
assert exc_info.value.field == "added"
|
||||
assert exc_info.value.value == "2005-03-04T15:30:00Z"
|
||||
|
||||
def test_relative_offset_as_a_range_bound_is_a_real_window(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
dated: dict[str, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A realistic dated corpus, including a document dated two
|
||||
hours before a "last Monday to now" window opens, and
|
||||
documents dated today and yesterday, inside that window
|
||||
WHEN:
|
||||
- "added:['-1 week' to now]" is searched: the same offset
|
||||
that matches nothing standing alone (see the test above),
|
||||
used here as a range bound instead
|
||||
THEN:
|
||||
- The window matches today and yesterday but excludes the
|
||||
document two hours before it opens, showing the bound is
|
||||
the offset itself and not a whole-day rounding of it, as
|
||||
the docs say next to the warning about the standalone form
|
||||
"""
|
||||
assert _matched_ids(backend, "added:['-1 week' to now]") == {
|
||||
dated["today"],
|
||||
dated["yesterday"],
|
||||
}
|
||||
|
||||
def test_double_quoted_range_bound_is_rejected(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
dated: dict[str, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A realistic dated corpus
|
||||
WHEN:
|
||||
- A range bound is double-quoted rather than single-quoted
|
||||
("added:[\"2005-03-04\" to 2005-03-05]")
|
||||
THEN:
|
||||
- `InvalidDateQuery` is raised, pinning which of the two
|
||||
quote characters fails: quoting a range bound is allowed,
|
||||
but only with single quotes, since the double-quoted
|
||||
spelling reaches the date grammar with its quotes still
|
||||
attached and is not a recognizable date
|
||||
"""
|
||||
with pytest.raises(InvalidDateQuery) as exc_info:
|
||||
_matched_ids(backend, 'added:["2005-03-04" to 2005-03-05]')
|
||||
assert exc_info.value.value == '"2005-03-04"'
|
||||
@@ -0,0 +1,364 @@
|
||||
"""Diagnostics route by Cause, and user-facing messages are host-owned.
|
||||
|
||||
whoosh-compat documents ``Diagnostic.message`` as developer output with no
|
||||
stability guarantee, so it must never reach an HTTP response body.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from datetime import UTC
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
from whoosh_compat.errors import Diagnostic
|
||||
from whoosh_compat.errors import DiagnosticKind
|
||||
from whoosh_compat.errors import QueryError
|
||||
from whoosh_compat.errors import cause_for
|
||||
from whoosh_compat.fields import FieldKind
|
||||
from whoosh_compat.fields import FieldRef
|
||||
|
||||
from documents.search._errors import SearchQueryError
|
||||
from documents.search._query import _map_emit_error
|
||||
from documents.search._query import _single_diagnostic_to_error
|
||||
from documents.search._query import parse_user_query
|
||||
from documents.search._schema import build_schema
|
||||
from documents.search._tokenizer import register_tokenizers
|
||||
|
||||
pytestmark = pytest.mark.search
|
||||
|
||||
_LIBRARY_PROSE = "INTERNAL LIBRARY WORDING WITH raw tantivy detail"
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def query_index() -> tantivy.Index:
|
||||
"""An in-memory, unstemmed index; these tests only parse, never index."""
|
||||
idx = tantivy.Index(build_schema(), path=None)
|
||||
register_tokenizers(idx, "")
|
||||
return idx
|
||||
|
||||
|
||||
def _diagnostic(
|
||||
kind: DiagnosticKind,
|
||||
*,
|
||||
field: FieldRef | None = FieldRef("title"),
|
||||
field_kind: FieldKind | None = FieldKind.TEXT,
|
||||
) -> Diagnostic:
|
||||
"""A Diagnostic shaped like the emitter's, with the library's own
|
||||
kind -> cause mapping rather than a hand-picked cause."""
|
||||
return Diagnostic(
|
||||
kind=kind,
|
||||
cause=cause_for(kind),
|
||||
message=_LIBRARY_PROSE,
|
||||
field=field,
|
||||
field_kind=field_kind,
|
||||
)
|
||||
|
||||
|
||||
class TestEmitErrorRouting:
|
||||
"""Every Cause gets a distinguishable treatment, not just "a 400"."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"kind",
|
||||
[
|
||||
DiagnosticKind.BACKEND_REJECTED,
|
||||
DiagnosticKind.AST_INVALID_SHAPE,
|
||||
DiagnosticKind.AST_UNKNOWN_FIELD,
|
||||
],
|
||||
)
|
||||
def test_internal_cause_is_not_converted(self, kind: DiagnosticKind) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A QueryError wrapping a Diagnostic whose Cause is INTERNAL
|
||||
(BACKEND_REJECTED/AST_INVALID_SHAPE/AST_UNKNOWN_FIELD)
|
||||
WHEN:
|
||||
- _map_emit_error processes it
|
||||
THEN:
|
||||
- The original QueryError propagates unchanged, so it surfaces
|
||||
as a 500 monitoring can see, never a 400 blaming the user
|
||||
"""
|
||||
error = QueryError(_diagnostic(kind))
|
||||
with pytest.raises(QueryError) as excinfo:
|
||||
_map_emit_error(error)
|
||||
assert excinfo.value is error
|
||||
|
||||
def test_misconfigured_cause_is_logged_and_reraised(
|
||||
self,
|
||||
caplog: pytest.LogCaptureFixture,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A QueryError for SCHEMA_FIELD_MISSING naming field "asn"
|
||||
WHEN:
|
||||
- _map_emit_error processes it
|
||||
THEN:
|
||||
- Exactly one ERROR log record is emitted naming the field and
|
||||
the diagnostic kind, and the original QueryError propagates
|
||||
unchanged: a registry/schema disagreement is transient (the
|
||||
exact same query succeeds once the index is rebuilt), so it
|
||||
surfaces as a 500 an operator can see rather than a 400
|
||||
telling the client their query is permanently invalid
|
||||
"""
|
||||
kind = DiagnosticKind.SCHEMA_FIELD_MISSING
|
||||
error = QueryError(_diagnostic(kind, field=FieldRef("asn")))
|
||||
with (
|
||||
caplog.at_level(logging.ERROR, logger="paperless.search"),
|
||||
pytest.raises(QueryError) as excinfo,
|
||||
):
|
||||
_map_emit_error(error)
|
||||
assert excinfo.value is error
|
||||
errors = [r for r in caplog.records if r.levelno == logging.ERROR]
|
||||
assert len(errors) == 1
|
||||
assert "asn" in errors[0].getMessage()
|
||||
assert kind.name in errors[0].getMessage()
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"kind",
|
||||
[
|
||||
DiagnosticKind.TEXT_RANGE,
|
||||
DiagnosticKind.PATTERN_TOO_COMPLEX,
|
||||
DiagnosticKind.EXISTS_REQUIRES_FAST,
|
||||
],
|
||||
)
|
||||
def test_unsupported_cause_is_a_400_with_no_operator_log(
|
||||
self,
|
||||
kind: DiagnosticKind,
|
||||
caplog: pytest.LogCaptureFixture,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A QueryError for a query tantivy cannot run
|
||||
(TEXT_RANGE/PATTERN_TOO_COMPLEX/EXISTS_REQUIRES_FAST)
|
||||
WHEN:
|
||||
- _map_emit_error processes it
|
||||
THEN:
|
||||
- It becomes a SearchQueryError with no log record at WARNING
|
||||
or above; a query tantivy cannot run is the user's to fix,
|
||||
not an operator alert. EXISTS_REQUIRES_FAST is nominally
|
||||
MISCONFIGURED but belongs here: it is decided from the
|
||||
registry's own FieldSpec, so it never reports a disagreement
|
||||
anyone could resolve
|
||||
"""
|
||||
with caplog.at_level(logging.WARNING, logger="paperless.search"):
|
||||
error = _map_emit_error(QueryError(_diagnostic(kind)))
|
||||
assert isinstance(error, SearchQueryError)
|
||||
assert caplog.records == []
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"kind",
|
||||
[
|
||||
DiagnosticKind.TEXT_RANGE,
|
||||
DiagnosticKind.PATTERN_TOO_COMPLEX,
|
||||
DiagnosticKind.EXISTS_REQUIRES_FAST,
|
||||
],
|
||||
)
|
||||
def test_user_facing_message_never_echoes_library_prose(
|
||||
self,
|
||||
kind: DiagnosticKind,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A QueryError carrying whoosh-compat's own developer-facing
|
||||
message text (SCHEMA_FIELD_MISSING excluded: it is now
|
||||
re-raised rather than converted, so it never produces a
|
||||
user-facing message at all, see
|
||||
test_misconfigured_cause_is_logged_and_reraised)
|
||||
WHEN:
|
||||
- _map_emit_error processes it
|
||||
THEN:
|
||||
- The resulting error's string never contains that library
|
||||
prose
|
||||
"""
|
||||
error = _map_emit_error(QueryError(_diagnostic(kind)))
|
||||
assert _LIBRARY_PROSE not in str(error)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"kind",
|
||||
[
|
||||
DiagnosticKind.TEXT_RANGE,
|
||||
DiagnosticKind.PATTERN_TOO_COMPLEX,
|
||||
DiagnosticKind.EXISTS_REQUIRES_FAST,
|
||||
],
|
||||
)
|
||||
def test_user_facing_message_names_the_field(
|
||||
self,
|
||||
kind: DiagnosticKind,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A QueryError for a JSON subpath field (custom_fields.value)
|
||||
WHEN:
|
||||
- _map_emit_error processes it
|
||||
THEN:
|
||||
- The resulting error names the field using its canonical
|
||||
dotted form, including the subpath (FieldRef.__str__ yields
|
||||
this dotted name, so every user-reachable emit kind can name
|
||||
it)
|
||||
"""
|
||||
diagnostic = _diagnostic(
|
||||
kind,
|
||||
field=FieldRef("custom_fields", "value"),
|
||||
field_kind=FieldKind.JSON,
|
||||
)
|
||||
error = _map_emit_error(QueryError(diagnostic))
|
||||
assert "custom_fields.value" in str(error)
|
||||
|
||||
|
||||
class TestParseDiagnosticMessages:
|
||||
"""Parse-time diagnostics are host-worded too, off field_kind."""
|
||||
|
||||
def test_too_deep_is_a_400_without_library_prose(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A parse-time Diagnostic for TOO_DEEP with no field
|
||||
WHEN:
|
||||
- _single_diagnostic_to_error processes it
|
||||
THEN:
|
||||
- It becomes a SearchQueryError with no library prose in its
|
||||
message
|
||||
"""
|
||||
error = _single_diagnostic_to_error(
|
||||
_diagnostic(DiagnosticKind.TOO_DEEP, field=None, field_kind=None),
|
||||
)
|
||||
assert isinstance(error, SearchQueryError)
|
||||
assert _LIBRARY_PROSE not in str(error)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("kind", "field_kind"),
|
||||
[
|
||||
(DiagnosticKind.PATTERN_ON_NUMERIC, FieldKind.U64),
|
||||
(DiagnosticKind.PATTERN_ON_BOOLEAN_EXISTS, FieldKind.BOOLEAN_EXISTS),
|
||||
(DiagnosticKind.PATTERN_ON_SUBPATH, FieldKind.JSON),
|
||||
],
|
||||
)
|
||||
def test_pattern_on_kinds_name_the_field_and_its_kind(
|
||||
self,
|
||||
kind: DiagnosticKind,
|
||||
field_kind: FieldKind,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A parse-time Diagnostic for a pattern used against a kind
|
||||
that cannot take one
|
||||
(PATTERN_ON_NUMERIC/PATTERN_ON_BOOLEAN_EXISTS/PATTERN_ON_SUBPATH)
|
||||
WHEN:
|
||||
- _single_diagnostic_to_error processes it
|
||||
THEN:
|
||||
- The message names both the field and its kind, with no
|
||||
library prose
|
||||
"""
|
||||
error = _single_diagnostic_to_error(
|
||||
_diagnostic(kind, field=FieldRef("asn"), field_kind=field_kind),
|
||||
)
|
||||
message = str(error)
|
||||
assert _LIBRARY_PROSE not in message
|
||||
assert "asn" in message
|
||||
assert field_kind.name.lower() in message
|
||||
|
||||
def test_single_char_bracket_range_names_the_field_and_the_value(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A SINGLE_CHAR_BRACKET_RANGE diagnostic for "title" with
|
||||
raw_value "200[1-9]"
|
||||
WHEN:
|
||||
- _single_diagnostic_to_error processes it
|
||||
THEN:
|
||||
- The resulting SearchQueryError names both the field and the
|
||||
offending value, with no library prose
|
||||
"""
|
||||
diagnostic = Diagnostic(
|
||||
kind=DiagnosticKind.SINGLE_CHAR_BRACKET_RANGE,
|
||||
cause=cause_for(DiagnosticKind.SINGLE_CHAR_BRACKET_RANGE),
|
||||
message=_LIBRARY_PROSE,
|
||||
field=FieldRef("title"),
|
||||
field_kind=FieldKind.TEXT,
|
||||
raw_value="200[1-9]",
|
||||
)
|
||||
error = _single_diagnostic_to_error(diagnostic)
|
||||
message = str(error)
|
||||
assert isinstance(error, SearchQueryError)
|
||||
assert _LIBRARY_PROSE not in message
|
||||
assert "title" in message
|
||||
assert "200[1-9]" in message
|
||||
|
||||
|
||||
class TestRealQueriesRouteCorrectly:
|
||||
"""The routing table against diagnostics emit() really produces."""
|
||||
|
||||
def test_text_range_is_a_400_naming_the_field(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A real query index
|
||||
WHEN:
|
||||
- parse_user_query is called with a text-range query
|
||||
("title:[a to b]")
|
||||
THEN:
|
||||
- It raises SearchQueryError naming "title"
|
||||
"""
|
||||
with pytest.raises(SearchQueryError) as excinfo:
|
||||
parse_user_query(query_index, "title:[a to b]", UTC)
|
||||
assert "title" in str(excinfo.value)
|
||||
|
||||
def test_wildcard_on_a_numeric_field_is_a_400_naming_the_field(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A real query index
|
||||
WHEN:
|
||||
- parse_user_query is called with a wildcard on a numeric
|
||||
field ("asn:12*")
|
||||
THEN:
|
||||
- It raises SearchQueryError naming "asn"
|
||||
"""
|
||||
with pytest.raises(SearchQueryError) as excinfo:
|
||||
parse_user_query(query_index, "asn:12*", UTC)
|
||||
assert "asn" in str(excinfo.value)
|
||||
|
||||
def test_single_char_bracket_range_is_a_400_naming_field_and_value(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A real query index
|
||||
WHEN:
|
||||
- parse_user_query is called with "title:200[1-9]"
|
||||
THEN:
|
||||
- It raises SearchQueryError naming both "title" and
|
||||
"200[1-9]"
|
||||
"""
|
||||
with pytest.raises(SearchQueryError) as excinfo:
|
||||
parse_user_query(query_index, "title:200[1-9]", UTC)
|
||||
message = str(excinfo.value)
|
||||
assert "title" in message
|
||||
assert "200[1-9]" in message
|
||||
|
||||
def test_internal_diagnostic_escapes_as_a_query_error(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- tantivy_emit monkeypatched to raise a QueryError with an
|
||||
INTERNAL-cause diagnostic (BACKEND_REJECTED), the one case
|
||||
with no query text of its own involved
|
||||
WHEN:
|
||||
- parse_user_query runs a normal query ("invoice")
|
||||
THEN:
|
||||
- The QueryError propagates unconverted; emit() reporting a
|
||||
defect in itself must not become a user-facing 400
|
||||
"""
|
||||
import documents.search._query as query_mod
|
||||
|
||||
def raise_internal(*args: object, **kwargs: object) -> None:
|
||||
raise QueryError(_diagnostic(DiagnosticKind.BACKEND_REJECTED))
|
||||
|
||||
monkeypatch.setattr(query_mod, "tantivy_emit", raise_internal)
|
||||
with pytest.raises(QueryError):
|
||||
parse_user_query(query_index, "invoice", UTC)
|
||||
@@ -0,0 +1,124 @@
|
||||
"""``field:*`` on a JSON field is user error, not an operator alert.
|
||||
|
||||
whoosh-compat classifies EXISTS_REQUIRES_FAST as MISCONFIGURED, and
|
||||
_map_emit_error used to route every MISCONFIGURED diagnostic to an ERROR log.
|
||||
But the kind is decided from the registry's own FieldSpec (kind plus fast)
|
||||
without consulting the index schema, and field_descriptors() builds the JSON
|
||||
fields non-fast deliberately, so nothing is misconfigured and no operator
|
||||
action can clear the condition. Any authenticated user could otherwise emit
|
||||
ERROR lines in a loop by repeating ``notes:*``.
|
||||
|
||||
SCHEMA_FIELD_MISSING, the other MISCONFIGURED kind, does compare the registry
|
||||
against the live schema, so it stays an ERROR log. But it is not a 400
|
||||
either: the exact same query would succeed once the index is rebuilt, so it
|
||||
is a transient server-side condition, not a permanently bad request, and is
|
||||
re-raised the same way an INTERNAL cause is.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from datetime import UTC
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
from whoosh_compat.errors import Diagnostic
|
||||
from whoosh_compat.errors import DiagnosticKind
|
||||
from whoosh_compat.errors import QueryError
|
||||
from whoosh_compat.errors import cause_for
|
||||
from whoosh_compat.fields import FieldKind
|
||||
from whoosh_compat.fields import FieldRef
|
||||
|
||||
from documents.search._errors import SearchQueryError
|
||||
from documents.search._query import _map_emit_error
|
||||
from documents.search._query import parse_user_query
|
||||
from documents.search._schema import build_schema
|
||||
from documents.search._tokenizer import register_tokenizers
|
||||
|
||||
pytestmark = pytest.mark.search
|
||||
|
||||
# Every spelling of "does this JSON field have a value" a user can type.
|
||||
EXISTS_QUERIES = [
|
||||
"notes:*",
|
||||
"notes.note:*",
|
||||
"notes.user:*",
|
||||
"custom_fields:*",
|
||||
"custom_fields.name:*",
|
||||
"custom_fields.value:*",
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def query_index() -> tantivy.Index:
|
||||
idx = tantivy.Index(build_schema(), path=None)
|
||||
register_tokenizers(idx, "")
|
||||
return idx
|
||||
|
||||
|
||||
class TestJsonExistsIsUserError:
|
||||
@pytest.mark.parametrize("query", EXISTS_QUERIES)
|
||||
def test_query_is_a_400_that_emits_no_error_log(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
caplog: pytest.LogCaptureFixture,
|
||||
query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A real query index, and every spelling of "does this JSON
|
||||
field have a value" (notes:*, notes.note:*, custom_fields:*,
|
||||
etc.)
|
||||
WHEN:
|
||||
- parse_user_query runs the query
|
||||
THEN:
|
||||
- It raises SearchQueryError naming the field, and no
|
||||
ERROR-level log record is emitted; EXISTS_REQUIRES_FAST on a
|
||||
JSON field is by design, not a misconfiguration an operator
|
||||
could act on
|
||||
"""
|
||||
with caplog.at_level(logging.WARNING, logger="paperless.search"):
|
||||
with pytest.raises(SearchQueryError) as excinfo:
|
||||
parse_user_query(query_index, query, UTC)
|
||||
assert query.split(":", maxsplit=1)[0] in str(excinfo.value)
|
||||
assert [r for r in caplog.records if r.levelno >= logging.ERROR] == []
|
||||
|
||||
|
||||
class TestGenuineMisconfigurationStillLogs:
|
||||
def test_schema_field_missing_is_an_error_log_and_reraised(
|
||||
self,
|
||||
caplog: pytest.LogCaptureFixture,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A QueryError for SCHEMA_FIELD_MISSING: the registry naming a
|
||||
field the index schema does not have
|
||||
WHEN:
|
||||
- _map_emit_error processes it
|
||||
THEN:
|
||||
- It logs exactly one ERROR record naming the diagnostic kind
|
||||
(a real mismatch an operator can fix, so it keeps the
|
||||
alert), and the original QueryError propagates unchanged
|
||||
rather than becoming a SearchQueryError: the exact same
|
||||
query would succeed once the index is rebuilt, so this is a
|
||||
transient server-side condition, not a permanently bad
|
||||
request, and surfaces as a 500 rather than a 400
|
||||
"""
|
||||
kind = DiagnosticKind.SCHEMA_FIELD_MISSING
|
||||
error = QueryError(
|
||||
Diagnostic(
|
||||
kind=kind,
|
||||
cause=cause_for(kind),
|
||||
message="field 'asn' is not defined in the index schema",
|
||||
field=FieldRef("asn"),
|
||||
field_kind=FieldKind.U64,
|
||||
),
|
||||
)
|
||||
with (
|
||||
caplog.at_level(logging.ERROR, logger="paperless.search"),
|
||||
pytest.raises(QueryError) as excinfo,
|
||||
):
|
||||
_map_emit_error(error)
|
||||
assert excinfo.value is error
|
||||
records = [r for r in caplog.records if r.levelno == logging.ERROR]
|
||||
assert len(records) == 1
|
||||
assert kind.name in records[0].getMessage()
|
||||
@@ -0,0 +1,261 @@
|
||||
"""The words the fuzzy blend clause hands back to tantivy's parser.
|
||||
|
||||
The clause re-parses a word string through tantivy, which analyzes it
|
||||
again, so the words must be the query's raw text rather than the analyzed
|
||||
text (analysis is not idempotent), and must still be split into plain
|
||||
words so that hyphenated, dotted and quoted terms keep contributing.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def fuzzy_enabled(settings: SettingsWrapper) -> None:
|
||||
"""Enable the fuzzy blend clause. The threshold doubles as a minimum
|
||||
score filter, so it is set to 0.0: every hit passes and the test sees
|
||||
the clause's matching behaviour, not the filter's."""
|
||||
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.0
|
||||
|
||||
|
||||
class TestFuzzyClauseParseFailureDegradesGracefully:
|
||||
def test_a_word_string_tantivy_rejects_drops_the_clause_only(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A parsed query with free-text words, and an index-like
|
||||
object whose parse_query is forced to raise ValueError
|
||||
WHEN:
|
||||
- _try_parse_fuzzy_query is called
|
||||
THEN:
|
||||
- It returns None instead of propagating, so a fuzzy word
|
||||
string tantivy's own parser rejects only drops the fuzzy
|
||||
clause: the exact/CJK clauses still stand rather than the
|
||||
whole query failing. The ValueError guard is insurance (the
|
||||
word string is plain tokens, so tantivy accepting it is
|
||||
expected, not assumed)
|
||||
"""
|
||||
import whoosh_compat as wc
|
||||
|
||||
from documents.search._query import _DEFAULT_SEARCH_FIELDS
|
||||
from documents.search._query import _try_parse_fuzzy_query
|
||||
from documents.search._registry import get_field_registry
|
||||
|
||||
registry = get_field_registry(None)
|
||||
result = wc.parse(
|
||||
"invoice",
|
||||
registry=registry,
|
||||
default_fields=_DEFAULT_SEARCH_FIELDS,
|
||||
)
|
||||
|
||||
class _RaisingIndex:
|
||||
def parse_query(self, *args: object, **kwargs: object) -> object:
|
||||
raise ValueError("synthetic parse failure")
|
||||
|
||||
assert _try_parse_fuzzy_query(_RaisingIndex(), result.ast, registry) is None
|
||||
|
||||
|
||||
class TestFuzzyClauseWords:
|
||||
def test_a_stemmed_word_is_not_stemmed_a_second_time(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Documents whose content contains "universities", a
|
||||
one-transposition typo of it ("universties"), and two
|
||||
unrelated words that share its stem prefix ("univalent",
|
||||
"unicycle")
|
||||
WHEN:
|
||||
- Searching for "universities" with the fuzzy blend enabled
|
||||
THEN:
|
||||
- Only the correctly-spelled document and its typo match; the
|
||||
clause does not widen far enough to reach the unrelated
|
||||
words. 'universities' stems to 'univers'; feeding that back
|
||||
to tantivy would stem it again to 'univ', whose fuzzy prefix
|
||||
reaches unrelated words - the clause must stay wide enough
|
||||
for a typo and no wider
|
||||
"""
|
||||
wanted = _index(
|
||||
backend,
|
||||
title="A",
|
||||
content="universities of europe",
|
||||
checksum="fuzz-stem-1",
|
||||
)
|
||||
typo = _index(
|
||||
backend,
|
||||
title="B",
|
||||
content="universties of europe",
|
||||
checksum="fuzz-stem-2",
|
||||
)
|
||||
_index(
|
||||
backend,
|
||||
title="C",
|
||||
content="univalent chemical bonds",
|
||||
checksum="fuzz-stem-3",
|
||||
)
|
||||
_index(
|
||||
backend,
|
||||
title="D",
|
||||
content="unicycle repair manual",
|
||||
checksum="fuzz-stem-4",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "universities") == {wanted.pk, typo.pk}
|
||||
|
||||
def test_a_hyphenated_term_still_reaches_the_clause(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document whose content contains a near-miss of "COVID-19"
|
||||
("covidx")
|
||||
WHEN:
|
||||
- Searching for "COVID-19" with the fuzzy blend enabled
|
||||
THEN:
|
||||
- The document matches; 'COVID-19' is one raw token, so unless
|
||||
it is split into words, it carries characters the re-parse
|
||||
would read as grammar, is dropped, and the whole query loses
|
||||
its fuzzy clause
|
||||
"""
|
||||
misspelled = _index(
|
||||
backend,
|
||||
title="A",
|
||||
content="covidx testing results",
|
||||
checksum="fuzz-hyphen-1",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "COVID-19") == {misspelled.pk}
|
||||
|
||||
def test_a_phrase_still_reaches_the_clause(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document whose content near-misses a quoted phrase
|
||||
WHEN:
|
||||
- Searching for the quoted phrase '"tax reports"' with the
|
||||
fuzzy blend enabled
|
||||
THEN:
|
||||
- The document matches; a phrase is one raw token carrying a
|
||||
space, and is the whole query's only free text here, so it
|
||||
must still reach the clause
|
||||
"""
|
||||
near_miss = _index(
|
||||
backend,
|
||||
title="A",
|
||||
content="taxation reportage weekly",
|
||||
checksum="fuzz-phrase-1",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, '"tax reports"') == {near_miss.pk}
|
||||
|
||||
|
||||
class TestBooleanKeywordsInRawText:
|
||||
"""Tantivy's boolean keywords are word runs, so they survive the cut
|
||||
into words and its own parser reads them as grammar. Raw query text
|
||||
reaches that parser with its case intact, so a quoted phrase can carry
|
||||
them in."""
|
||||
|
||||
@pytest.fixture
|
||||
def corpus(self, backend: TantivyBackend) -> dict[str, int]:
|
||||
both = _index(
|
||||
backend,
|
||||
title="A",
|
||||
content="taxation reportage weekly",
|
||||
checksum="fuzz-kw-1",
|
||||
)
|
||||
tax_only = _index(
|
||||
backend,
|
||||
title="B",
|
||||
content="taxation only here",
|
||||
checksum="fuzz-kw-2",
|
||||
)
|
||||
report_only = _index(
|
||||
backend,
|
||||
title="C",
|
||||
content="reportage only here",
|
||||
checksum="fuzz-kw-3",
|
||||
)
|
||||
return {
|
||||
"both": both.pk,
|
||||
"tax_only": tax_only.pk,
|
||||
"report_only": report_only.pk,
|
||||
}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
pytest.param('"tax AND reports"', id="and"),
|
||||
pytest.param('"tax OR reports"', id="or"),
|
||||
pytest.param('"tax NOT reports"', id="not"),
|
||||
pytest.param('"tax IN reports"', id="in"),
|
||||
],
|
||||
)
|
||||
def test_a_keyword_inside_a_phrase_stays_an_ordinary_word(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
corpus: dict[str, int],
|
||||
query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Three documents: one with both "taxation" and "reportage",
|
||||
one with only "taxation", one with only "reportage"
|
||||
WHEN:
|
||||
- Searching for a quoted phrase carrying a tantivy boolean
|
||||
keyword as one of its words (e.g. '"tax AND reports"')
|
||||
THEN:
|
||||
- The keyword stays an ordinary word inside the phrase, and
|
||||
the fuzzy clause matches all three documents, the same
|
||||
disjunction as the plain '"tax reports"' phrase: AND must
|
||||
not turn it into a conjunction, NOT must not give it its own
|
||||
exclusion, IN must not fail the parse
|
||||
"""
|
||||
assert _matched_ids(backend, '"tax reports"') == set(corpus.values())
|
||||
assert _matched_ids(backend, query) == set(corpus.values())
|
||||
|
||||
def test_a_trailing_keyword_does_not_drop_the_clause(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
corpus: dict[str, int],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Three documents: one with both "taxation" and "reportage",
|
||||
one with only "taxation", one with only "reportage"
|
||||
WHEN:
|
||||
- Searching for '"tax AND"', a phrase ending in a tantivy
|
||||
syntax error
|
||||
THEN:
|
||||
- The fuzzy clause still matches on "tax"; 'tax AND' alone is
|
||||
a syntax error to tantivy's parser, which would otherwise
|
||||
cost the whole query its fuzzy clause
|
||||
"""
|
||||
assert _matched_ids(backend, '"tax AND"') == {
|
||||
corpus["both"],
|
||||
corpus["tax_only"],
|
||||
}
|
||||
@@ -0,0 +1,249 @@
|
||||
"""Regression coverage for the unguarded TEXT-mode highlight query.
|
||||
|
||||
parse_simple_text_highlight_query re-parses simple-search tokens through
|
||||
Tantivy's query-string parser to build a SnippetGenerator-compatible query.
|
||||
Simple-search tokens keep arbitrary punctuation (quotes, colons, brackets,
|
||||
slashes), so any token carrying Tantivy query grammar raised an unguarded
|
||||
ValueError. The search itself had already succeeded by the time this ran:
|
||||
only the highlight step failed, and with the DocumentViewSet.list
|
||||
exception handler narrowed elsewhere on this branch, that ValueError now
|
||||
reaches the client as a bare 500 rather than a 400.
|
||||
|
||||
Covers three angles:
|
||||
- the query builder itself: quoting each token as its own escaped phrase
|
||||
should let it parse instead of raising, for every failure mode a plain-
|
||||
text query can trigger (syntax error, unknown field, unsupported regex).
|
||||
- highlight_hits: even when a token still can't be expressed as a
|
||||
highlight query, the guard must fall back to a query that still
|
||||
produces usable highlight HTML, not silently empty ones.
|
||||
- the real API endpoint: pinning the previously-500 status to 200.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
from rest_framework import status
|
||||
|
||||
from documents.search._backend import SearchMode
|
||||
from documents.search._query import parse_simple_text_highlight_query
|
||||
from documents.search._schema import build_schema
|
||||
from documents.search._tokenizer import register_tokenizers
|
||||
from documents.tests.factories import DocumentFactory
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from rest_framework.test import APIClient
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
# Each spelling below trips a different Tantivy parser failure mode:
|
||||
# 'a"b' -> Syntax Error (unterminated quote)
|
||||
# foo:bar -> unknown field
|
||||
# (a -> Syntax Error (unbalanced group)
|
||||
# [a -> Syntax Error (unbalanced range)
|
||||
# /a/ -> Unsupported query (regex queries disallowed)
|
||||
_MALFORMED_QUERIES = [
|
||||
pytest.param('a"b', id="unterminated_quote"),
|
||||
pytest.param("foo:bar", id="unknown_field"),
|
||||
pytest.param("(a", id="unbalanced_group"),
|
||||
pytest.param("[a", id="unbalanced_range"),
|
||||
pytest.param("/a/", id="unsupported_regex"),
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def query_index() -> tantivy.Index:
|
||||
"""An in-memory, unstemmed index for parse-only tests."""
|
||||
schema = build_schema()
|
||||
idx = tantivy.Index(schema, path=None)
|
||||
register_tokenizers(idx, "")
|
||||
return idx
|
||||
|
||||
|
||||
class TestParseSimpleTextHighlightQueryDoesNotRaise:
|
||||
"""The query builder itself must tolerate Tantivy syntax in its tokens."""
|
||||
|
||||
@pytest.mark.parametrize("raw_query", _MALFORMED_QUERIES)
|
||||
def test_malformed_token_does_not_raise(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
raw_query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A simple-search query token carrying Tantivy query grammar
|
||||
(unterminated quote, unknown field, unbalanced group/range,
|
||||
or unsupported regex)
|
||||
WHEN:
|
||||
- parse_simple_text_highlight_query builds a highlight query
|
||||
from it
|
||||
THEN:
|
||||
- It returns a tantivy.Query instead of raising, since each
|
||||
token is quoted as its own escaped phrase rather than fed
|
||||
to the parser raw
|
||||
"""
|
||||
assert isinstance(
|
||||
parse_simple_text_highlight_query(query_index, raw_query),
|
||||
tantivy.Query,
|
||||
)
|
||||
|
||||
|
||||
class TestHighlightHitsProducesUsableHighlights:
|
||||
"""highlight_hits must keep producing real <b>-wrapped snippet HTML for
|
||||
these queries, not merely avoid raising."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"raw_query",
|
||||
[*_MALFORMED_QUERIES, pytest.param("plain text", id="plain_text_sanity")],
|
||||
)
|
||||
def test_highlight_still_contains_matched_text(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
raw_query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document whose content contains the raw query text
|
||||
verbatim
|
||||
WHEN:
|
||||
- backend.highlight_hits builds highlights for a TEXT-mode
|
||||
search using that same (possibly Tantivy-grammar-carrying)
|
||||
query text
|
||||
THEN:
|
||||
- The hit still carries a content highlight with real
|
||||
<b>-wrapped matched-term markup, not an empty fallback
|
||||
"""
|
||||
doc = DocumentFactory.create(
|
||||
title="probe",
|
||||
content=f"needle content containing {raw_query} literally here",
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
hits = backend.highlight_hits(
|
||||
raw_query,
|
||||
[doc.pk],
|
||||
search_mode=SearchMode.TEXT,
|
||||
)
|
||||
|
||||
assert len(hits) == 1
|
||||
highlights = hits[0]["highlights"]
|
||||
assert "content" in highlights, (
|
||||
f"Expected a content highlight for {raw_query!r}, got: {highlights!r}"
|
||||
)
|
||||
assert "<b>" in highlights["content"], (
|
||||
f"Highlight for {raw_query!r} carries no matched-term markup: "
|
||||
f"{highlights['content']!r}"
|
||||
)
|
||||
|
||||
|
||||
class TestHighlightGuardDiscriminatesOnValueError:
|
||||
"""The guard added to highlight_hits must catch exactly ValueError, the
|
||||
same shape as the sibling notes_text guard, and let anything else
|
||||
through -- so a real library defect is never mistaken for a harmless
|
||||
syntax error."""
|
||||
|
||||
def test_non_value_error_is_not_swallowed(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- parse_simple_text_highlight_query patched to raise
|
||||
RuntimeError instead of a syntax-related ValueError
|
||||
WHEN:
|
||||
- backend.highlight_hits is called
|
||||
THEN:
|
||||
- The RuntimeError propagates unguarded; the highlight guard
|
||||
must catch exactly ValueError, the same shape as the
|
||||
sibling notes_text guard, never mistaking a real library
|
||||
defect for a harmless syntax error
|
||||
"""
|
||||
import documents.search._backend as backend_mod
|
||||
|
||||
def raise_runtime_error(*args: object, **kwargs: object) -> object:
|
||||
raise RuntimeError("synthetic bug, unrelated to query syntax")
|
||||
|
||||
monkeypatch.setattr(
|
||||
backend_mod,
|
||||
"parse_simple_text_highlight_query",
|
||||
raise_runtime_error,
|
||||
)
|
||||
|
||||
doc = DocumentFactory.create(title="probe", content="anything here")
|
||||
backend.add_or_update(doc)
|
||||
|
||||
with pytest.raises(RuntimeError):
|
||||
backend.highlight_hits(
|
||||
"anything",
|
||||
[doc.pk],
|
||||
search_mode=SearchMode.TEXT,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("_search_index")
|
||||
class TestApiNoLongerReturns500:
|
||||
"""Pins the actual regression: a matching TEXT-mode search whose query
|
||||
string carries Tantivy syntax must return results, not a server error."""
|
||||
|
||||
@pytest.mark.parametrize("raw_query", _MALFORMED_QUERIES)
|
||||
def test_malformed_text_query_returns_200(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
raw_query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A matching document whose content contains the raw query
|
||||
text, indexed via the real search index fixture
|
||||
WHEN:
|
||||
- A TEXT-mode search is issued through the real API with a
|
||||
query string carrying Tantivy syntax
|
||||
THEN:
|
||||
- The response is 200 with the expected result count, not a
|
||||
500 (the regression this file exists to pin)
|
||||
"""
|
||||
from documents.search import get_backend
|
||||
|
||||
doc = DocumentFactory.create(
|
||||
title="probe",
|
||||
content=f"needle content containing {raw_query} literally here",
|
||||
)
|
||||
get_backend().add_or_update(doc)
|
||||
|
||||
response = admin_client.get(f"/api/documents/?text={raw_query}")
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
assert response.data["count"] == 1
|
||||
|
||||
def test_plain_text_query_still_returns_200(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A matching document indexed via the real search index
|
||||
fixture
|
||||
WHEN:
|
||||
- An ordinary TEXT-mode search (no Tantivy syntax) is issued
|
||||
THEN:
|
||||
- The response is 200 with the expected result count; sanity
|
||||
check that the guard does not mask a total failure of the
|
||||
ordinary highlight path
|
||||
"""
|
||||
from documents.search import get_backend
|
||||
|
||||
doc = DocumentFactory.create(
|
||||
title="probe",
|
||||
content="needle content containing plain text literally here",
|
||||
)
|
||||
get_backend().add_or_update(doc)
|
||||
|
||||
response = admin_client.get("/api/documents/?text=plain text")
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
assert response.data["count"] == 1
|
||||
@@ -0,0 +1,206 @@
|
||||
"""Bare notes:/custom_fields: prefix resolution.
|
||||
|
||||
"notes:foo"/"custom_fields:foo" were valid fielded searches before the
|
||||
whoosh-compat migration. The registry only exposes them as JSON subpaths, so
|
||||
each JSON FieldSpec declares a default subpath (SubpathSpec(default=True)):
|
||||
notes: resolves to notes.note:, custom_fields: resolves to
|
||||
custom_fields.value:. This replaced an earlier regex-based rewrite
|
||||
(_rewrite_bare_json_field_prefixes) that ran on the raw query string before
|
||||
parsing and was blind to quoting, so a phrase like
|
||||
content:"payment notes: none" was silently corrupted into a notes-field
|
||||
search and matched nothing. Resolving the default subpath inside the parser
|
||||
instead means quoting is already understood by the time it happens.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
from django.contrib.auth.models import User
|
||||
|
||||
from documents.models import CustomField
|
||||
from documents.models import CustomFieldInstance
|
||||
from documents.models import Document
|
||||
from documents.models import Note
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestBareJsonFieldPrefixes:
|
||||
def test_bare_notes_prefix_searches_note_text(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with a note whose text contains a word, and a
|
||||
decoy document whose content (not notes) contains the same
|
||||
word
|
||||
WHEN:
|
||||
- A bare "notes:" prefix query is run (notes declares "note"
|
||||
as its default subpath)
|
||||
THEN:
|
||||
- Only the document whose note matches is returned; the
|
||||
decoy's content match does not resurface through a demoted
|
||||
text search
|
||||
"""
|
||||
alice = User.objects.create_user(username="alice")
|
||||
with_note = Document.objects.create(
|
||||
title="Has note",
|
||||
content="x",
|
||||
checksum="bare-notes-with",
|
||||
)
|
||||
Note.objects.create(document=with_note, user=alice, note="crocodile")
|
||||
backend.add_or_update(with_note)
|
||||
# This document's CONTENT contains the words a demoted text search
|
||||
# would match; it must NOT match once the prefix addresses notes.
|
||||
_index(
|
||||
backend,
|
||||
title="Notes about things",
|
||||
content="notes crocodile mention",
|
||||
checksum="bare-notes-decoy",
|
||||
)
|
||||
assert _matched_ids(backend, "notes:crocodile") == {with_note.pk}
|
||||
|
||||
def test_bare_custom_fields_prefix_searches_values(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with a custom field instance whose value
|
||||
contains a word, and a decoy document whose content (not a
|
||||
custom field value) contains the same word
|
||||
WHEN:
|
||||
- A bare "custom_fields:" prefix query is run (custom_fields
|
||||
declares "value" as its default subpath)
|
||||
THEN:
|
||||
- Only the document whose custom field value matches is
|
||||
returned
|
||||
"""
|
||||
field = CustomField.objects.create(
|
||||
name="Policy Number",
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
with_value = Document.objects.create(
|
||||
title="Has field",
|
||||
content="x",
|
||||
checksum="bare-cf-with",
|
||||
)
|
||||
CustomFieldInstance.objects.create(
|
||||
document=with_value,
|
||||
field=field,
|
||||
value_text="crocodile",
|
||||
)
|
||||
backend.add_or_update(with_value)
|
||||
_index(
|
||||
backend,
|
||||
title="Custom things",
|
||||
content="custom fields crocodile",
|
||||
checksum="bare-cf-decoy",
|
||||
)
|
||||
assert _matched_ids(backend, "custom_fields:crocodile") == {with_value.pk}
|
||||
|
||||
def test_subpath_spellings_are_untouched(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with a note carrying both an author and note text
|
||||
WHEN:
|
||||
- The explicit subpath spellings "notes.user:" and
|
||||
"notes.note:" are queried
|
||||
THEN:
|
||||
- Both resolve to their intended subpath and match the
|
||||
document; the default-subpath resolution for the bare
|
||||
prefix does not interfere with explicit subpath addressing
|
||||
"""
|
||||
bob = User.objects.create_user(username="bob")
|
||||
doc = Document.objects.create(
|
||||
title="Bob note",
|
||||
content="x",
|
||||
checksum="bare-subpath",
|
||||
)
|
||||
Note.objects.create(document=doc, user=bob, note="remark")
|
||||
backend.add_or_update(doc)
|
||||
assert _matched_ids(backend, "notes.user:bob") == {doc.pk}
|
||||
assert _matched_ids(backend, "notes.note:remark") == {doc.pk}
|
||||
|
||||
|
||||
class TestQuotedPhraseContainingNotesColonIsNotCorrupted:
|
||||
"""The regex rewrite this migration removes was blind to quoting: it
|
||||
matched "notes:" anywhere in the raw query string, including inside an
|
||||
already-quoted phrase on an unrelated field, silently turning
|
||||
content:"payment notes: none" into a notes-field search that matched
|
||||
nothing. Resolving the default subpath during parsing (which is
|
||||
quote-aware) fixes this."""
|
||||
|
||||
def test_quoted_phrase_with_notes_colon_matches_by_content(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document whose content literally contains the text
|
||||
"payment notes: none" inside a quoted phrase
|
||||
WHEN:
|
||||
- A query quoting that exact phrase against the content
|
||||
field is run
|
||||
THEN:
|
||||
- It matches by content, rather than the "notes:" substring
|
||||
inside the quotes being corrupted into a notes-field search
|
||||
that matches nothing (the bug the deleted regex rewrite
|
||||
caused, since it was blind to quoting)
|
||||
"""
|
||||
target = _index(
|
||||
backend,
|
||||
title="Statement",
|
||||
content="payment notes: none",
|
||||
checksum="quoted-phrase-notes-colon",
|
||||
)
|
||||
assert _matched_ids(
|
||||
backend,
|
||||
'content:"payment notes: none"',
|
||||
) == {target.pk}
|
||||
|
||||
def test_quoted_phrase_matches_the_same_document_unquoted(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document whose content contains the same words as the
|
||||
previous test's phrase, but without the colon
|
||||
WHEN:
|
||||
- A query quoting that phrase against the content field is
|
||||
run
|
||||
THEN:
|
||||
- It matches by content, proving the earlier fix is about
|
||||
quote-awareness specifically, not about the words
|
||||
themselves being unsearchable
|
||||
"""
|
||||
target = _index(
|
||||
backend,
|
||||
title="Statement",
|
||||
content="payment notes none",
|
||||
checksum="quoted-phrase-no-colon",
|
||||
)
|
||||
assert _matched_ids(
|
||||
backend,
|
||||
'content:"payment notes none"',
|
||||
) == {target.pk}
|
||||
@@ -0,0 +1,92 @@
|
||||
"""Every declared JSON subpath must actually be written to the index.
|
||||
|
||||
PUBLIC_FIELDS declares each JSON field's subpaths (e.g. ``notes`` ->
|
||||
{"user", "note"}), but nothing coupled that declaration to what
|
||||
``_backend.py``'s document builder actually writes into the JSON blob at
|
||||
index time. A subpath declared but never written would be
|
||||
queryable-but-always-empty -- syntactically valid, silently matching
|
||||
nothing -- with no test failure anywhere.
|
||||
|
||||
This indexes one real document carrying values for every JSON field
|
||||
(a Note, a CustomFieldInstance) and inspects the document's own stored
|
||||
JSON payload, rather than running field-specific queries: that way a
|
||||
future JSON field's subpaths are covered automatically, without a new
|
||||
per-subpath query having to be added by hand each time.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
from django.contrib.auth.models import User
|
||||
from whoosh_compat import FieldKind
|
||||
|
||||
from documents.models import CustomField
|
||||
from documents.models import CustomFieldInstance
|
||||
from documents.models import Document
|
||||
from documents.models import Note
|
||||
from documents.search._fields import PUBLIC_FIELDS
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
class TestJsonSubpathsAreWrittenAtIndexTime:
|
||||
def test_every_declared_json_subpath_appears_in_the_stored_document(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with a Note and a CustomFieldInstance attached
|
||||
WHEN:
|
||||
- The document is indexed via TantivyBackend.add_or_update
|
||||
THEN:
|
||||
- Every subpath PUBLIC_FIELDS declares for notes/custom_fields
|
||||
is present as a key in the document's stored JSON payload
|
||||
"""
|
||||
user = User.objects.create_user(username="completeness-user")
|
||||
field = CustomField.objects.create(
|
||||
name="Completeness Field",
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
doc = Document.objects.create(
|
||||
title="Completeness doc",
|
||||
content="x",
|
||||
checksum="json-subpath-completeness",
|
||||
)
|
||||
Note.objects.create(document=doc, user=user, note="a note")
|
||||
CustomFieldInstance.objects.create(
|
||||
document=doc,
|
||||
field=field,
|
||||
value_text="a value",
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
index = backend._index
|
||||
searcher = index.searcher()
|
||||
hits = searcher.search(
|
||||
tantivy.Query.term_query(index.schema, "id", doc.pk),
|
||||
limit=1,
|
||||
).hits
|
||||
assert hits, "the document was not indexed"
|
||||
stored = searcher.doc(hits[0][1]).to_dict()
|
||||
|
||||
json_fields = [f for f in PUBLIC_FIELDS if f.kind is FieldKind.JSON]
|
||||
assert json_fields, "no JSON fields declared - fixture is stale"
|
||||
for field_spec in json_fields:
|
||||
stored_values = stored.get(field_spec.name)
|
||||
assert stored_values, (
|
||||
f"{field_spec.name} was not written to the index at all"
|
||||
)
|
||||
written_keys = stored_values[0].keys()
|
||||
for subpath in field_spec.subpaths:
|
||||
assert subpath in written_keys, (
|
||||
f"{field_spec.name}.{subpath} is declared in PUBLIC_FIELDS "
|
||||
"but _backend.py's document builder never writes it - it "
|
||||
"would be queryable but always empty"
|
||||
)
|
||||
@@ -0,0 +1,62 @@
|
||||
"""Wildcard patterns on KEYWORD fields must stay literal.
|
||||
|
||||
``checksum`` is the only KEYWORD field: it is indexed with the raw tokenizer,
|
||||
so its terms are never lowercased, folded or stemmed. Running its wildcard
|
||||
patterns through the stemming normalizer rewrote hex prefixes ("ceded" ->
|
||||
"cede") and returned documents whose checksum did not start with what the user
|
||||
typed, which for an identity field is a wrong answer.
|
||||
|
||||
This covers only the registry-level normalizer, which is all that exists to
|
||||
prove at this point in the stack: user queries are not yet routed through
|
||||
whoosh-compat (that lands with the query-layer PR), so the same fact proven
|
||||
end to end against real indexed documents lives in
|
||||
``test_checksum_prefix_queries.py``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.search._registry import get_field_registry
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from whoosh_compat import FieldRegistry
|
||||
from whoosh_compat import PatternNormalizer
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _normalizer(registry: FieldRegistry, name: str) -> PatternNormalizer:
|
||||
ref = registry.make_ref(name)
|
||||
assert ref is not None
|
||||
resolved = registry.resolve(ref)
|
||||
assert resolved is not None
|
||||
assert resolved.spec.pattern_normalizer is not None
|
||||
return resolved.spec.pattern_normalizer
|
||||
|
||||
|
||||
class TestKeywordPatternNormalizer:
|
||||
@pytest.mark.parametrize(
|
||||
"run",
|
||||
[
|
||||
pytest.param("ceded", id="stems_to_cede"),
|
||||
pytest.param("added", id="stems_to_ad"),
|
||||
pytest.param("cafed", id="stems_to_cafe"),
|
||||
],
|
||||
)
|
||||
def test_keyword_runs_are_folded_not_stemmed(self, run: str) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The "checksum" field's registered pattern normalizer
|
||||
(KEYWORD kind, "en" registry)
|
||||
WHEN:
|
||||
- A wildcard pattern run is normalized
|
||||
THEN:
|
||||
- The run is returned unchanged, never widened to a stem (which
|
||||
would return checksums that do not start with what the user
|
||||
typed)
|
||||
"""
|
||||
normalize = _normalizer(get_field_registry("en"), "checksum")
|
||||
assert normalize(run) == run
|
||||
@@ -0,0 +1,156 @@
|
||||
"""The pattern normalizer's stem-alternates contract, and its consistency
|
||||
with the index-side analyzer.
|
||||
|
||||
Query patterns are normalized but were not stemmed, while index terms are
|
||||
stemmed, so the natural spelling of a prefix search matched nothing:
|
||||
``invoice*`` found no document although ``invoic*`` did. v2's index was
|
||||
UNSTEMMED (whoosh ``TEXT()`` defaults to ``StandardAnalyzer``), so this
|
||||
regressed against both baselines.
|
||||
|
||||
These are pure unit tests against ``_make_pattern_normalizer`` and
|
||||
``stem_pattern_text`` directly, no query routing involved. The end-to-end
|
||||
proof that a real wildcard query actually reaches a stemmed index term
|
||||
lives in ``test_pattern_stemming.py``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.search._registry import _make_pattern_normalizer
|
||||
from documents.search._tokenizer import ascii_fold
|
||||
from documents.search._tokenizer import paperless_text_analyzer
|
||||
from documents.search._tokenizer import stem_pattern_text
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from whoosh_compat import PatternNormalizer
|
||||
|
||||
|
||||
class TestStemsMatchTheIndexAnalyzer:
|
||||
"""stem_pattern_text rebuilds paperless_text_analyzer's stemming tail rather
|
||||
than sharing it, so a filter added to the index analyzer alone would silently
|
||||
stop patterns from reaching the terms it produces.
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"language",
|
||||
["en", "de", "fr", "es", "sv", None, "klingon"],
|
||||
)
|
||||
@pytest.mark.parametrize(
|
||||
"word",
|
||||
["Copies", "copyright", "Companies", "Invoices", "laufen", "casas", "Straße"],
|
||||
)
|
||||
def test_stem_equals_the_index_term(self, word: str, language: str | None) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A word, across several representative index languages
|
||||
("en", "de", "fr", "es", "sv"), no language, and an
|
||||
unsupported language ("klingon")
|
||||
WHEN:
|
||||
- `stem_pattern_text` (the pattern-side stemmer) processes the
|
||||
folded word, and `paperless_text_analyzer` (the index-side
|
||||
analyzer) independently processes the same word
|
||||
THEN:
|
||||
- The two produce the identical term. `stem_pattern_text`
|
||||
rebuilds `paperless_text_analyzer`'s stemming tail rather
|
||||
than sharing it, so a filter added to the index analyzer
|
||||
alone would silently stop patterns from reaching the terms
|
||||
it produces; this pins the two staying in sync
|
||||
"""
|
||||
indexed = paperless_text_analyzer(language).analyze(word)[0]
|
||||
assert stem_pattern_text(ascii_fold(word.lower()), language) == indexed
|
||||
|
||||
|
||||
def _forms(normalize: PatternNormalizer, text: str) -> tuple[str, ...]:
|
||||
"""The distinct forms a term may match, in order, the way the emitter reads
|
||||
the normalizer's answer (see whoosh_compat.PatternNormalizer)."""
|
||||
result = normalize(text)
|
||||
if isinstance(result, str):
|
||||
return (result,)
|
||||
return tuple(dict.fromkeys(result))
|
||||
|
||||
|
||||
class TestPatternNormalizer:
|
||||
@pytest.mark.parametrize(
|
||||
("text", "expected"),
|
||||
[
|
||||
("Invoice", ("invoice", "invoic")),
|
||||
("companies", ("companies", "compani")),
|
||||
# y -> i is a substitution, so both forms are needed: the index
|
||||
# holds "librari" for "library" and "library" for "librarian".
|
||||
("library", ("library", "librari")),
|
||||
# A run the stemmer leaves alone collapses back to one form, so it
|
||||
# costs exactly the one regex branch it did before.
|
||||
("invoic", ("invoic",)),
|
||||
("Universit", ("universit",)),
|
||||
("Café", ("cafe",)),
|
||||
],
|
||||
)
|
||||
def test_offers_the_typed_run_and_its_stem(
|
||||
self,
|
||||
text: str,
|
||||
expected: tuple[str, ...],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The "en" pattern normalizer
|
||||
WHEN:
|
||||
- It processes a literal run (e.g. "Invoice", "library",
|
||||
"Café")
|
||||
THEN:
|
||||
- It returns the folded run and, where it differs, the
|
||||
stemmed form, as distinct alternatives; a run the stemmer
|
||||
leaves alone (e.g. "invoic") collapses back to the single
|
||||
folded form. "library" needs both forms since y -> i is a
|
||||
substitution: the index holds "librari" for "library" and
|
||||
"library" for "librarian"
|
||||
"""
|
||||
assert _forms(_make_pattern_normalizer("en"), text) == expected
|
||||
|
||||
def test_run_that_yields_no_token_falls_back_to_the_typed_run(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The "en" pattern normalizer
|
||||
WHEN:
|
||||
- It processes a run past the analyzer's remove_long limit
|
||||
THEN:
|
||||
- The run analyzes to zero tokens, so there is no stem to
|
||||
offer, and only the folded run remains
|
||||
"""
|
||||
over_long = "invoices" * 20
|
||||
assert _forms(_make_pattern_normalizer("en"), over_long) == (over_long,)
|
||||
|
||||
@pytest.mark.parametrize("language", [None, "klingon"])
|
||||
def test_unstemmed_language_folds_only(self, language: str | None) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A pattern normalizer with no language configured, or one
|
||||
this build has no stemmer for ("klingon")
|
||||
WHEN:
|
||||
- It processes "Invoices"
|
||||
THEN:
|
||||
- Only the folded form ("invoices") is offered, since with no
|
||||
stemmer configured the index holds surface forms and the
|
||||
pattern must keep them too
|
||||
"""
|
||||
assert _forms(_make_pattern_normalizer(language), "Invoices") == ("invoices",)
|
||||
|
||||
@pytest.mark.parametrize("char", ["a", "Z", "é"])
|
||||
def test_a_single_character_collapses_to_one_folded_form(self, char: str) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The "en" pattern normalizer
|
||||
WHEN:
|
||||
- It processes a single character
|
||||
THEN:
|
||||
- Exactly one, one-character form is returned. A bracket
|
||||
class body is normalized one character at a time and the
|
||||
answer is used only when it is a single one-character
|
||||
form, so a stemmer that changed a lone character would
|
||||
silently disable folding inside classes
|
||||
"""
|
||||
forms = _forms(_make_pattern_normalizer("en"), char)
|
||||
assert len(forms) == 1
|
||||
assert len(forms[0]) == 1
|
||||
@@ -0,0 +1,221 @@
|
||||
"""Wildcard patterns must match a stemmed index, end to end.
|
||||
|
||||
Query patterns are normalized but were not stemmed, while index terms are
|
||||
stemmed, so the natural spelling of a prefix search matched nothing:
|
||||
``invoice*`` found no document although ``invoic*`` did. v2's index was
|
||||
UNSTEMMED (whoosh ``TEXT()`` defaults to ``StandardAnalyzer``), so this
|
||||
regressed against both baselines.
|
||||
|
||||
These are end-to-end tests against a real indexed document and a real
|
||||
query, proving the pattern normalizer's stem-alternates contract actually
|
||||
reaches a stemmed index term. The pure unit tests against the normalizer
|
||||
function itself live in ``test_pattern_normalizer.py``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
CONTENT = (
|
||||
"invoice total due for electricity from both companies, "
|
||||
"payments made to the university library, copies attached"
|
||||
)
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def indexed_doc(backend: TantivyBackend) -> Document:
|
||||
doc = Document.objects.create(
|
||||
title="Invoice 2020 productname",
|
||||
content=CONTENT,
|
||||
checksum="pattern-stemming-1",
|
||||
archive_serial_number=900,
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestPrefixStemming:
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
"invoice*",
|
||||
"electricity*",
|
||||
"companies*",
|
||||
"payments*",
|
||||
"library*",
|
||||
"title:Invoice*",
|
||||
],
|
||||
)
|
||||
def test_full_word_prefix_matches_its_stem(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document indexed with content containing "invoice",
|
||||
"electricity", "companies", "payments", "library" and title
|
||||
"Invoice 2020 productname"
|
||||
WHEN:
|
||||
- A prefix wildcard on the full, unstemmed word is queried
|
||||
(e.g. "invoice*", "title:Invoice*")
|
||||
THEN:
|
||||
- The document matches, since the pattern normalizer offers
|
||||
the word's stem as an alternative alongside the typed run,
|
||||
reaching the stemmed index term
|
||||
"""
|
||||
assert _matched_ids(backend, query) == {indexed_doc.id}
|
||||
|
||||
@pytest.mark.parametrize("query", ["invoic*", "electr*", "payment*"])
|
||||
def test_already_stemmed_prefix_still_matches(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The same indexed document
|
||||
WHEN:
|
||||
- A prefix wildcard is typed already in its stemmed spelling
|
||||
(e.g. "invoic*")
|
||||
THEN:
|
||||
- The document still matches, since the typed-run alternative
|
||||
is itself a prefix of the stored stemmed term
|
||||
"""
|
||||
assert _matched_ids(backend, query) == {indexed_doc.id}
|
||||
|
||||
@pytest.mark.parametrize("query", ["univers*", "librar*"])
|
||||
def test_partial_prefix_reaches_the_stemmed_term(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The same indexed document
|
||||
WHEN:
|
||||
- A prefix shorter than a whole word is queried ("univers*",
|
||||
"librar*")
|
||||
THEN:
|
||||
- It still matches, and neither case needs the two-alternative
|
||||
path to do it: measured under "en", the stemmer leaves
|
||||
"librar" alone, so it has one form, and that form is a
|
||||
prefix of the "librari" the index holds for "library";
|
||||
"univers" stems to the *shorter* "univ", and the run as
|
||||
typed and its stem are both prefixes of the "univers" the
|
||||
index holds for "university". The case where the two forms
|
||||
genuinely diverge, and only one of them matches, is
|
||||
test_stem_substitution_reaches_both_the_inflection_and_the_compound
|
||||
"""
|
||||
assert _matched_ids(backend, query) == {indexed_doc.id}
|
||||
|
||||
def test_full_word_reaches_the_stem_but_a_fragment_of_it_does_not(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The same indexed document, storing "university" as "univers"
|
||||
WHEN:
|
||||
- "universities*", "universit*" and "univers*" are each queried
|
||||
THEN:
|
||||
- "universities*" matches, since the stem of "universities" is
|
||||
that same "univers"; "universit*" matches nothing, since
|
||||
"universit" is a prefix of neither its own stem nor the
|
||||
stored term. The alternatives widen recall without turning
|
||||
a wildcard into a prefix search over the original text.
|
||||
usage.md tells a reader whose `universit*` finds nothing to
|
||||
shorten it to `univers*`, which matches
|
||||
"""
|
||||
assert _matched_ids(backend, "universities*") == {indexed_doc.id}
|
||||
assert _matched_ids(backend, "universit*") == set()
|
||||
assert _matched_ids(backend, "univers*") == {indexed_doc.id}
|
||||
|
||||
def test_pattern_past_the_stem_boundary_is_documented_not_fixed(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The same indexed document, with "productname" indexed as
|
||||
"productnam"
|
||||
WHEN:
|
||||
- "produ*name" (a pattern straddling the stem boundary) is
|
||||
queried
|
||||
THEN:
|
||||
- It matches nothing; produ*name cannot match a stemmed
|
||||
index, and usage.md must not advertise it. Pinned so the
|
||||
limitation is deliberate, not accidental
|
||||
"""
|
||||
assert _matched_ids(backend, "produ*name") == set()
|
||||
|
||||
def test_stem_substitution_reaches_both_the_inflection_and_the_compound(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The indexed document (containing "copies") plus a second
|
||||
document titled "Copyright notice" with content "copyright
|
||||
notice for the work"
|
||||
WHEN:
|
||||
- "copy*" and "copyright*" are each queried
|
||||
THEN:
|
||||
- "copy*" matches both documents, and "copyright*" matches
|
||||
only the compound one. English stemming substitutes as well
|
||||
as truncates: "copy" and "copies" both index as "copi",
|
||||
while "copyright" keeps its literal "y". Neither form is a
|
||||
prefix of the other, so no single normalized string reaches
|
||||
both; the run is therefore emitted as a disjunction of the
|
||||
folded and stemmed forms, and "copy*" reaches the base
|
||||
word, its inflections and the compound alike
|
||||
"""
|
||||
compound = Document.objects.create(
|
||||
title="Copyright notice",
|
||||
content="copyright notice for the work",
|
||||
checksum="pattern-stemming-2",
|
||||
archive_serial_number=901,
|
||||
)
|
||||
backend.add_or_update(compound)
|
||||
|
||||
assert _matched_ids(backend, "copy*") == {indexed_doc.id, compound.id}
|
||||
assert _matched_ids(backend, "copyright*") == {compound.id}
|
||||
|
||||
|
||||
class TestBracketClassStillFolds:
|
||||
def test_class_body_matches_case_insensitively(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The indexed document, titled "Invoice 2020 productname"
|
||||
WHEN:
|
||||
- A bracket-class pattern mixing case is queried
|
||||
("title:[IP]nvoice*")
|
||||
THEN:
|
||||
- It matches: the class body is folded per character, which
|
||||
the alternatives contract preserves only because a lone
|
||||
character stems to itself
|
||||
"""
|
||||
assert _matched_ids(backend, "title:[IP]nvoice*") == {indexed_doc.id}
|
||||
@@ -0,0 +1,198 @@
|
||||
"""Permission filtering must hold against the real indexed document shape.
|
||||
|
||||
Only three of the index's unsigned ``*_id`` columns are load-bearing:
|
||||
``owner_id``, ``viewer_id`` and ``viewer_group_id``, all read by
|
||||
build_permission_filter. The rest (correspondent/document_type/storage_path/tag
|
||||
ids) were written on every document and read by nothing, and were dropped.
|
||||
|
||||
These tests index real Documents through the backend's own document builder and
|
||||
assert result-level visibility per user, so a mistake about which columns are
|
||||
load-bearing shows up as documents leaking across users rather than as a passing
|
||||
unit test over a hand-built index.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
from django.contrib.auth.models import Group
|
||||
from django.contrib.auth.models import User
|
||||
from guardian.shortcuts import assign_perm
|
||||
|
||||
from documents.models import Correspondent
|
||||
from documents.models import Document
|
||||
from documents.models import DocumentType
|
||||
from documents.models import StoragePath
|
||||
from documents.models import Tag
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def owner() -> User:
|
||||
return User.objects.create_user(username="owner")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def stranger() -> User:
|
||||
return User.objects.create_user(username="stranger")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def viewer() -> User:
|
||||
return User.objects.create_user(username="viewer")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def group_member() -> User:
|
||||
user = User.objects.create_user(username="group_member")
|
||||
user.groups.add(Group.objects.create(name="accounting"))
|
||||
return user
|
||||
|
||||
|
||||
class TestPermissionFilteringOnIndexedDocuments:
|
||||
def test_unowned_document_is_visible_to_everyone(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
stranger: User,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with no owner, indexed via the backend's real
|
||||
document builder
|
||||
WHEN:
|
||||
- A stranger (no relation to the document) searches
|
||||
THEN:
|
||||
- The document is visible to them
|
||||
"""
|
||||
doc = Document.objects.create(
|
||||
title="Public Invoice",
|
||||
content="invoice total due",
|
||||
checksum="perm-unowned",
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
assert backend.search_ids("invoice", user=stranger) == [doc.pk]
|
||||
|
||||
def test_owned_document_is_visible_only_to_its_owner(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
owner: User,
|
||||
stranger: User,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document owned by one user, indexed via the backend's
|
||||
real document builder
|
||||
WHEN:
|
||||
- The owner and an unrelated stranger each search
|
||||
THEN:
|
||||
- The owner sees the document; the stranger does not
|
||||
"""
|
||||
doc = Document.objects.create(
|
||||
title="Private Invoice",
|
||||
content="invoice total due",
|
||||
checksum="perm-owned",
|
||||
owner=owner,
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
assert backend.search_ids("invoice", user=owner) == [doc.pk]
|
||||
assert backend.search_ids("invoice", user=stranger) == []
|
||||
|
||||
def test_explicitly_shared_document_is_visible_to_the_viewer(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
owner: User,
|
||||
viewer: User,
|
||||
stranger: User,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document owned by one user and explicitly shared with a
|
||||
second user via guardian's view_document permission,
|
||||
indexed via the backend's real document builder
|
||||
WHEN:
|
||||
- The shared viewer and an unrelated stranger each search
|
||||
THEN:
|
||||
- The viewer sees the document; the stranger does not
|
||||
"""
|
||||
doc = Document.objects.create(
|
||||
title="Shared Invoice",
|
||||
content="invoice total due",
|
||||
checksum="perm-shared-user",
|
||||
owner=owner,
|
||||
)
|
||||
assign_perm("view_document", viewer, doc)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
assert backend.search_ids("invoice", user=viewer) == [doc.pk]
|
||||
assert backend.search_ids("invoice", user=stranger) == []
|
||||
|
||||
def test_group_shared_document_is_visible_to_group_members(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
owner: User,
|
||||
group_member: User,
|
||||
stranger: User,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document owned by one user and shared with a group via
|
||||
guardian's view_document permission, indexed via the
|
||||
backend's real document builder
|
||||
WHEN:
|
||||
- A member of that group and an unrelated stranger each
|
||||
search
|
||||
THEN:
|
||||
- The group member sees the document; the stranger does not
|
||||
"""
|
||||
doc = Document.objects.create(
|
||||
title="Group Invoice",
|
||||
content="invoice total due",
|
||||
checksum="perm-shared-group",
|
||||
owner=owner,
|
||||
)
|
||||
assign_perm("view_document", group_member.groups.first(), doc)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
assert backend.search_ids("invoice", user=group_member) == [doc.pk]
|
||||
assert backend.search_ids("invoice", user=stranger) == []
|
||||
|
||||
def test_metadata_does_not_widen_visibility(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
owner: User,
|
||||
stranger: User,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document owned by one user and carrying
|
||||
correspondent/document_type/storage_path/tag metadata,
|
||||
indexed via the backend's real document builder
|
||||
WHEN:
|
||||
- The owner and an unrelated stranger each search
|
||||
THEN:
|
||||
- The owner sees the document; the stranger does not, since
|
||||
the dropped, non-load-bearing metadata *_id columns must
|
||||
not widen visibility beyond the owner_id/viewer_id/
|
||||
viewer_group_id filter
|
||||
"""
|
||||
doc = Document.objects.create(
|
||||
title="Tagged Invoice",
|
||||
content="invoice total due",
|
||||
checksum="perm-metadata",
|
||||
owner=owner,
|
||||
correspondent=Correspondent.objects.create(name="ACME"),
|
||||
document_type=DocumentType.objects.create(name="Bill"),
|
||||
storage_path=StoragePath.objects.create(name="Archive", path="archive/"),
|
||||
)
|
||||
doc.tags.add(Tag.objects.create(name="paid"))
|
||||
backend.add_or_update(doc)
|
||||
|
||||
assert backend.search_ids("invoice", user=owner) == [doc.pk]
|
||||
assert backend.search_ids("invoice", user=stranger) == []
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,198 @@
|
||||
"""Negation must survive the blended query.
|
||||
|
||||
parse_user_query ORs an exact clause with optional fuzzy and CJK clauses.
|
||||
Each of those is built from positive terms only, so unless the query's
|
||||
exclusions are applied to the blend as a whole, a document the exact
|
||||
clause excluded is re-admitted by whichever other clause is enabled.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def fuzzy_enabled(settings: SettingsWrapper) -> None:
|
||||
"""Enable the fuzzy blend clause. The threshold doubles as a minimum
|
||||
score filter, so it is set to 0.0: every hit passes and the test sees
|
||||
the clause's matching behaviour, not the filter's."""
|
||||
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.0
|
||||
|
||||
|
||||
class TestNegationConstrainsEveryClause:
|
||||
@pytest.mark.usefixtures("fuzzy_enabled")
|
||||
def test_fuzzy_clause_does_not_readmit_an_excluded_document(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two documents both matching a positive term, one of which
|
||||
also contains a word the query excludes, with the fuzzy
|
||||
blend clause enabled
|
||||
WHEN:
|
||||
- A query combining the positive term with a NOT exclusion is
|
||||
run
|
||||
THEN:
|
||||
- Only the document without the excluded word is returned;
|
||||
the fuzzy clause (built from positive terms only) does not
|
||||
readmit the document the exact clause excluded
|
||||
"""
|
||||
secret = _index(
|
||||
backend,
|
||||
title="Invoice A",
|
||||
content="invoice total secret",
|
||||
checksum="neg-fuzzy-1",
|
||||
)
|
||||
public = _index(
|
||||
backend,
|
||||
title="Invoice B",
|
||||
content="invoice total public",
|
||||
checksum="neg-fuzzy-2",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "invoice") == {secret.pk, public.pk}
|
||||
assert _matched_ids(backend, "invoice NOT secret") == {public.pk}
|
||||
|
||||
def test_cjk_clause_does_not_readmit_an_excluded_document(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two documents both containing a CJK run, one of which also
|
||||
contains a word the query excludes
|
||||
WHEN:
|
||||
- A query combining the CJK term with a NOT exclusion is run
|
||||
THEN:
|
||||
- Only the document without the excluded word is returned;
|
||||
the CJK clause legitimately carries the CJK run, so
|
||||
rebuilding it from the AST cannot help here, only applying
|
||||
the exclusion above the blend keeps the excluded document
|
||||
out
|
||||
"""
|
||||
secret = _index(
|
||||
backend,
|
||||
title="Tokyo A",
|
||||
content="東京都の秘密です secret",
|
||||
checksum="neg-cjk-1",
|
||||
)
|
||||
public = _index(
|
||||
backend,
|
||||
title="Tokyo B",
|
||||
content="東京都の報告書です public",
|
||||
checksum="neg-cjk-2",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "東京") == {secret.pk, public.pk}
|
||||
assert _matched_ids(backend, "東京 NOT secret") == {public.pk}
|
||||
|
||||
@pytest.mark.usefixtures("fuzzy_enabled")
|
||||
def test_disjunctive_negation_still_admits_the_other_branch(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document matching a positive term and also containing a
|
||||
word a disjunctive NOT branch excludes, plus an unrelated
|
||||
document
|
||||
WHEN:
|
||||
- A query of the shape "term OR NOT excluded_word" is run
|
||||
THEN:
|
||||
- Both documents are returned; "invoice OR NOT secret"
|
||||
excludes nothing on its own, so a document matching the
|
||||
left branch stays in even though it contains the excluded
|
||||
word
|
||||
"""
|
||||
secret_invoice = _index(
|
||||
backend,
|
||||
title="Invoice A",
|
||||
content="invoice total secret",
|
||||
checksum="neg-or-1",
|
||||
)
|
||||
unrelated = _index(
|
||||
backend,
|
||||
title="Recipe",
|
||||
content="flour and water",
|
||||
checksum="neg-or-2",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "invoice OR NOT secret") == {
|
||||
secret_invoice.pk,
|
||||
unrelated.pk,
|
||||
}
|
||||
|
||||
def test_a_negation_under_or_does_not_constrain_the_cjk_clause(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two CJK documents, one of which also contains a word an OR
|
||||
branch's own NOT excludes, plus an unrelated latin document
|
||||
WHEN:
|
||||
- The exclusion is under a disjunctive OR branch, versus in
|
||||
conjunctive position
|
||||
THEN:
|
||||
- Under OR, the excluded document still matches through the
|
||||
CJK clause (an exclusion that is one branch's own condition
|
||||
cannot be restated above the blend without dropping
|
||||
documents the other branch matches, so it is left where it
|
||||
is and the CJK clause stays unconstrained by it -- this
|
||||
shows through here in a way it does not for latin text,
|
||||
since the exact clause cannot match a CJK run at all, so
|
||||
the CJK clause is the only thing matching the CJK
|
||||
documents, and the excluded one comes with it)
|
||||
- Under conjunctive "AND NOT", the same exclusion is hoisted
|
||||
and does constrain the CJK clause, pinning the deliberate
|
||||
limit of the hoist
|
||||
"""
|
||||
secret = _index(
|
||||
backend,
|
||||
title="Tokyo A",
|
||||
content="東京都の秘密です secret",
|
||||
checksum="neg-or-cjk-1",
|
||||
)
|
||||
public = _index(
|
||||
backend,
|
||||
title="Tokyo B",
|
||||
content="東京都の報告書です public",
|
||||
checksum="neg-or-cjk-2",
|
||||
)
|
||||
bill = _index(
|
||||
backend,
|
||||
title="Bill",
|
||||
content="bill payment received",
|
||||
checksum="neg-or-cjk-3",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "(東京 AND NOT secret) OR bill") == {
|
||||
bill.pk,
|
||||
public.pk,
|
||||
secret.pk,
|
||||
}
|
||||
# The same exclusion in conjunctive position is hoisted, and does
|
||||
# constrain the CJK clause.
|
||||
assert _matched_ids(backend, "東京 AND NOT secret") == {public.pk}
|
||||
@@ -0,0 +1,224 @@
|
||||
from collections.abc import Sequence
|
||||
|
||||
import pytest
|
||||
from whoosh_compat import FieldKind
|
||||
from whoosh_compat import FieldRegistry
|
||||
from whoosh_compat.fields import ResolvedField
|
||||
|
||||
from documents.search._fields import PUBLIC_FIELDS
|
||||
from documents.search._registry import get_field_registry
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def registry() -> FieldRegistry:
|
||||
return get_field_registry(None)
|
||||
|
||||
|
||||
def _resolve(registry: FieldRegistry, name: str) -> ResolvedField:
|
||||
ref = registry.make_ref(name)
|
||||
assert ref is not None, f"{name} is not a valid field ref"
|
||||
resolved = registry.resolve(ref)
|
||||
assert resolved is not None, f"{name} did not resolve"
|
||||
return resolved
|
||||
|
||||
|
||||
def _distinct_forms(result: str | Sequence[str]) -> tuple[str, ...]:
|
||||
"""The forms a term may match, in order, the way whoosh-compat's emitter
|
||||
reads a pattern_normalizer's answer: a bare str is one form, a sequence is
|
||||
several, deduplicated."""
|
||||
if isinstance(result, str):
|
||||
return (result,)
|
||||
return tuple(dict.fromkeys(result))
|
||||
|
||||
|
||||
class TestFieldRegistry:
|
||||
def test_no_queryable_field_name_ends_in_id(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- PUBLIC_FIELDS, the canonical query-syntax field table
|
||||
WHEN:
|
||||
- Every declared field name is inspected
|
||||
THEN:
|
||||
- None of them end in "_id" (internal id columns, written for
|
||||
permission filtering and joins, must never reach the query
|
||||
surface; checked against PUBLIC_FIELDS rather than the
|
||||
registry so a leak is caught where it is declared)
|
||||
"""
|
||||
leaked = [f.name for f in PUBLIC_FIELDS if f.name.endswith("_id")]
|
||||
assert not leaked, f"internal id fields reached the query surface: {leaked}"
|
||||
|
||||
def test_type_alias_resolves_to_document_type(
|
||||
self,
|
||||
registry: FieldRegistry,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The field registry
|
||||
WHEN:
|
||||
- The alias "type" is resolved
|
||||
THEN:
|
||||
- It resolves to the canonical "document_type" field
|
||||
"""
|
||||
assert _resolve(registry, "type").spec.name == "document_type"
|
||||
|
||||
def test_path_alias_resolves_to_storage_path(self, registry: FieldRegistry) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The field registry
|
||||
WHEN:
|
||||
- The alias "path" is resolved
|
||||
THEN:
|
||||
- It resolves to the canonical "storage_path" field
|
||||
"""
|
||||
assert _resolve(registry, "path").spec.name == "storage_path"
|
||||
|
||||
def test_notes_json_subpaths_resolve(self, registry: FieldRegistry) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The field registry
|
||||
WHEN:
|
||||
- "notes.user" is resolved
|
||||
THEN:
|
||||
- It resolves to the "notes" field with json_path "user"
|
||||
"""
|
||||
resolved = _resolve(registry, "notes.user")
|
||||
assert resolved.spec.name == "notes"
|
||||
assert resolved.json_path == "user"
|
||||
assert resolved.is_subpath is True
|
||||
|
||||
def test_custom_fields_json_subpaths_resolve(self, registry: FieldRegistry) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The field registry
|
||||
WHEN:
|
||||
- "custom_fields.name" and "custom_fields.value" are resolved
|
||||
THEN:
|
||||
- Both resolve without error
|
||||
"""
|
||||
for raw in ("custom_fields.name", "custom_fields.value"):
|
||||
_resolve(registry, raw)
|
||||
|
||||
def test_tag_is_comma_values(self, registry: FieldRegistry) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The field registry
|
||||
WHEN:
|
||||
- The "tag" field is resolved
|
||||
THEN:
|
||||
- It is marked comma_values=True
|
||||
"""
|
||||
assert _resolve(registry, "tag").spec.comma_values is True
|
||||
|
||||
def test_correspondent_is_not_comma_values(self, registry: FieldRegistry) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The field registry
|
||||
WHEN:
|
||||
- The "correspondent" field is resolved
|
||||
THEN:
|
||||
- It is not marked comma_values ("tag" is the only field that
|
||||
opts in; end to end the two readings of
|
||||
"correspondent:foo,bar" agree anyway, since the analyzer
|
||||
splits the literal value on the comma regardless, so this is
|
||||
only observable at the registry level)
|
||||
"""
|
||||
assert _resolve(registry, "correspondent").spec.comma_values is False
|
||||
|
||||
def test_created_is_date_kind(self, registry: FieldRegistry) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The field registry
|
||||
WHEN:
|
||||
- The "created" field is resolved
|
||||
THEN:
|
||||
- Its kind is DATE and date_only is True
|
||||
"""
|
||||
resolved = _resolve(registry, "created")
|
||||
assert resolved.spec.kind is FieldKind.DATE
|
||||
assert resolved.spec.date_only is True
|
||||
|
||||
def test_analyzer_lowercases_and_ascii_folds(self, registry: FieldRegistry) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The field registry with no language configured (no stemmer
|
||||
in the analyzer chain)
|
||||
WHEN:
|
||||
- The "title" field's analyzer processes "Café"
|
||||
THEN:
|
||||
- It is lowercased and ASCII-folded to the single token "cafe"
|
||||
"""
|
||||
resolved = _resolve(registry, "title")
|
||||
assert resolved.spec.analyzer is not None
|
||||
assert resolved.spec.analyzer("Café") == ["cafe"]
|
||||
|
||||
def test_checksum_analyzer_is_identity_single_token(
|
||||
self,
|
||||
registry: FieldRegistry,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The field registry
|
||||
WHEN:
|
||||
- The "checksum" field's analyzer (raw tokenizer, no
|
||||
splitting) processes "ABC-123"
|
||||
THEN:
|
||||
- It is returned unchanged as a single token
|
||||
"""
|
||||
resolved = _resolve(registry, "checksum")
|
||||
assert resolved.spec.analyzer is not None
|
||||
assert resolved.spec.analyzer("ABC-123") == ["ABC-123"]
|
||||
|
||||
def test_pattern_normalizer_follows_the_registry_language(
|
||||
self,
|
||||
registry: FieldRegistry,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A registry with no language, and a registry built for "en"
|
||||
WHEN:
|
||||
- The "title" field's pattern normalizer processes "Running"
|
||||
THEN:
|
||||
- With no language, only the folded run is offered
|
||||
("running"), since the index holds surface forms
|
||||
- With "en", the stem is offered too ("run"), since indexed
|
||||
terms are stemmed and the pattern has to reach them
|
||||
"""
|
||||
resolved = _resolve(registry, "title")
|
||||
assert resolved.spec.pattern_normalizer is not None
|
||||
assert _distinct_forms(resolved.spec.pattern_normalizer("Running")) == (
|
||||
"running",
|
||||
)
|
||||
|
||||
resolved_en = _resolve(get_field_registry("en"), "title")
|
||||
assert resolved_en.spec.pattern_normalizer is not None
|
||||
assert _distinct_forms(resolved_en.spec.pattern_normalizer("Running")) == (
|
||||
"running",
|
||||
"run",
|
||||
)
|
||||
|
||||
def test_registry_is_cached_per_language(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Two calls to get_field_registry("en")
|
||||
WHEN:
|
||||
- Both calls are made
|
||||
THEN:
|
||||
- They return the same registry instance
|
||||
"""
|
||||
a = get_field_registry("en")
|
||||
b = get_field_registry("en")
|
||||
assert a is b
|
||||
|
||||
def test_registry_rebuilds_on_language_change(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A call to get_field_registry("en") and a call to
|
||||
get_field_registry("de")
|
||||
WHEN:
|
||||
- Both calls are made
|
||||
THEN:
|
||||
- They return different registry instances
|
||||
"""
|
||||
a = get_field_registry("en")
|
||||
b = get_field_registry("de")
|
||||
assert a is not b
|
||||
@@ -5,13 +5,19 @@ from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.search._fields import PUBLIC_FIELDS
|
||||
from documents.search._schema import SCHEMA_VERSION
|
||||
from documents.search._schema import build_schema
|
||||
from documents.search._schema import field_descriptors
|
||||
from documents.search._schema import needs_rebuild
|
||||
from documents.search._schema import schema_fingerprint
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
import tantivy
|
||||
from pytest_django.fixtures import Settings
|
||||
|
||||
|
||||
pytestmark = pytest.mark.search
|
||||
|
||||
@@ -25,18 +31,24 @@ class TestNeedsRebuild:
|
||||
def test_returns_false_when_version_and_language_match(
|
||||
self,
|
||||
index_dir: Path,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> None:
|
||||
settings.SEARCH_LANGUAGE = "en"
|
||||
(index_dir / ".index_settings.json").write_text(
|
||||
json.dumps({"schema_version": SCHEMA_VERSION, "language": "en"}),
|
||||
json.dumps(
|
||||
{
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"language": "en",
|
||||
"schema_fingerprint": schema_fingerprint(),
|
||||
},
|
||||
),
|
||||
)
|
||||
assert needs_rebuild(index_dir) is False
|
||||
|
||||
def test_returns_true_on_schema_version_mismatch(
|
||||
self,
|
||||
index_dir: Path,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> None:
|
||||
settings.SEARCH_LANGUAGE = None
|
||||
(index_dir / ".index_settings.json").write_text(
|
||||
@@ -47,7 +59,7 @@ class TestNeedsRebuild:
|
||||
def test_returns_true_when_version_is_not_an_integer(
|
||||
self,
|
||||
index_dir: Path,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> None:
|
||||
settings.SEARCH_LANGUAGE = None
|
||||
(index_dir / ".index_settings.json").write_text(
|
||||
@@ -58,7 +70,7 @@ class TestNeedsRebuild:
|
||||
def test_returns_true_when_language_key_missing(
|
||||
self,
|
||||
index_dir: Path,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> None:
|
||||
settings.SEARCH_LANGUAGE = "en"
|
||||
(index_dir / ".index_settings.json").write_text(
|
||||
@@ -69,10 +81,68 @@ class TestNeedsRebuild:
|
||||
def test_returns_true_when_language_differs(
|
||||
self,
|
||||
index_dir: Path,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> None:
|
||||
settings.SEARCH_LANGUAGE = "de"
|
||||
(index_dir / ".index_settings.json").write_text(
|
||||
json.dumps({"schema_version": SCHEMA_VERSION, "language": "en"}),
|
||||
)
|
||||
assert needs_rebuild(index_dir) is True
|
||||
|
||||
|
||||
def _schema_fields(schema: tantivy.Schema) -> dict[str, dict]:
|
||||
"""{name: field-state} for every field declared on a tantivy Schema.
|
||||
|
||||
tantivy-py 0.26 exposes no public introspection API on Schema (no
|
||||
__iter__, get_field, to_json, etc.) -- __reduce__() (used internally for
|
||||
pickling) is the only way to recover the field list, so we lean on it
|
||||
here for test assertions only.
|
||||
"""
|
||||
state = schema.__reduce__()[1][0]
|
||||
return {field["name"]: field for field in state["inner"]}
|
||||
|
||||
|
||||
class TestSchemaMatchesPublicFields:
|
||||
def test_every_public_field_is_in_the_schema(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- PUBLIC_FIELDS and the tantivy schema built by build_schema()
|
||||
WHEN:
|
||||
- Every field declared in PUBLIC_FIELDS is checked against the
|
||||
schema
|
||||
THEN:
|
||||
- Each one is present as a field in the built schema
|
||||
"""
|
||||
schema = build_schema()
|
||||
schema_field_names = set(_schema_fields(schema))
|
||||
for field in PUBLIC_FIELDS:
|
||||
assert field.name in schema_field_names, (
|
||||
f"{field.name} is in PUBLIC_FIELDS but missing from build_schema()"
|
||||
)
|
||||
|
||||
|
||||
class TestFastFlagAgreement:
|
||||
def test_every_public_field_fast_flag_matches_the_built_schema(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- PUBLIC_FIELDS and field_descriptors() (the latter is exactly
|
||||
the input build_schema()'s SchemaBuilder consumes for the
|
||||
`fast` kwarg on every field kind, so it pins the agreement
|
||||
without depending on a private tantivy-py pickled
|
||||
representation)
|
||||
WHEN:
|
||||
- Every PUBLIC_FIELDS entry's fast flag is compared against
|
||||
field_descriptors()' fast flag for the same field
|
||||
THEN:
|
||||
- They agree for every field, catching a fast=True
|
||||
PUBLIC_FIELDS entry the builder silently ignores here
|
||||
instead of at a user's field:* existence query, which
|
||||
whoosh-compat's registry trusts PUBLIC_FIELDS' fast flag to
|
||||
resolve
|
||||
"""
|
||||
descriptor_fast = {d.name: d.fast for d in field_descriptors()}
|
||||
for public_field in PUBLIC_FIELDS:
|
||||
assert descriptor_fast[public_field.name] == public_field.fast, (
|
||||
f"{public_field.name}: PUBLIC_FIELDS says fast={public_field.fast} but"
|
||||
f" field_descriptors() says fast={descriptor_fast[public_field.name]}"
|
||||
)
|
||||
|
||||
@@ -0,0 +1,587 @@
|
||||
"""The schema fingerprint stamped into .index_settings.json.
|
||||
|
||||
tantivy compares schemas by *ordered* field list, and `tantivy.Index(schema,
|
||||
path=...)` (what every write path does) raises on any difference. SCHEMA_VERSION
|
||||
is the manual guard against that, but build_schema() is edited for *parser*
|
||||
reasons - adding an alias, flipping fast=True, adding a subpath - by people not
|
||||
thinking about the on-disk index, and forgetting the bump is exactly how this
|
||||
branch's bug happened.
|
||||
|
||||
The fingerprint is the automatic guard: it hashes the field descriptor list that
|
||||
build_schema() itself iterates, so any change to a field's name, kind, options
|
||||
or *position* forces a rebuild on its own.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
|
||||
from documents.search import _schema
|
||||
from documents.search._schema import SCHEMA_VERSION
|
||||
from documents.search._schema import FieldDescriptor
|
||||
from documents.search._schema import _write_sentinels
|
||||
from documents.search._schema import build_schema
|
||||
from documents.search._schema import field_descriptors
|
||||
from documents.search._schema import needs_rebuild
|
||||
from documents.search._schema import schema_fingerprint
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
|
||||
pytestmark = pytest.mark.search
|
||||
|
||||
# The on-disk field layout of a v2 index, pinned as data. Any edit here is an
|
||||
# index-format change: it must come with a rebuild, which the fingerprint now
|
||||
# forces automatically. Reproduced from build_schema()'s output as it stood
|
||||
# before the descriptor refactor, so it also pins that the refactor changed
|
||||
# nothing.
|
||||
PINNED_DESCRIPTORS: tuple[FieldDescriptor, ...] = (
|
||||
FieldDescriptor("id", "u64", stored=True, indexed=True, fast=True, tokenizer=None),
|
||||
FieldDescriptor(
|
||||
"title",
|
||||
"text",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"content",
|
||||
"text",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"correspondent",
|
||||
"text",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"document_type",
|
||||
"text",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"storage_path",
|
||||
"text",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"original_filename",
|
||||
"text",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"tag",
|
||||
"text",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"checksum",
|
||||
"text",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="raw",
|
||||
),
|
||||
FieldDescriptor("asn", "u64", stored=True, indexed=True, fast=True, tokenizer=None),
|
||||
FieldDescriptor(
|
||||
"page_count",
|
||||
"u64",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer=None,
|
||||
),
|
||||
FieldDescriptor(
|
||||
"num_notes",
|
||||
"u64",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer=None,
|
||||
),
|
||||
FieldDescriptor(
|
||||
"created",
|
||||
"date",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer=None,
|
||||
),
|
||||
FieldDescriptor(
|
||||
"modified",
|
||||
"date",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer=None,
|
||||
),
|
||||
FieldDescriptor(
|
||||
"added",
|
||||
"date",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer=None,
|
||||
),
|
||||
FieldDescriptor(
|
||||
"notes",
|
||||
"json",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"notes_text",
|
||||
"text",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"custom_fields",
|
||||
"json",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"title_sort",
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer="simple_analyzer",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"correspondent_sort",
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer="simple_analyzer",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"type_sort",
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer="simple_analyzer",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"bigram_content",
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="bigram_analyzer",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"bigram_title",
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="bigram_analyzer",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"bigram_correspondent",
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="bigram_analyzer",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"bigram_document_type",
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="bigram_analyzer",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"bigram_tag",
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="bigram_analyzer",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"simple_title",
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="simple_search_analyzer",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"simple_content",
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="simple_search_analyzer",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"autocomplete_word",
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="raw",
|
||||
),
|
||||
FieldDescriptor(
|
||||
"owner_id",
|
||||
"u64",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer=None,
|
||||
),
|
||||
FieldDescriptor(
|
||||
"viewer_id",
|
||||
"u64",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer=None,
|
||||
),
|
||||
FieldDescriptor(
|
||||
"viewer_group_id",
|
||||
"u64",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer=None,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _schema_fields(schema: tantivy.Schema) -> list[dict]:
|
||||
"""The tantivy-level field list, in declaration order.
|
||||
|
||||
tantivy-py 0.26 exposes no public introspection API on Schema, so
|
||||
__reduce__() (its pickling hook) is the only way to recover the field list.
|
||||
It is used here, in a test, precisely because it is the representation the
|
||||
persisted fingerprint must NOT depend on.
|
||||
"""
|
||||
return schema.__reduce__()[1][0]["inner"]
|
||||
|
||||
|
||||
def _sentinels(index_dir: Path, **overrides: object) -> None:
|
||||
data = {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"language": None,
|
||||
"schema_fingerprint": schema_fingerprint(),
|
||||
}
|
||||
data.update(overrides)
|
||||
(index_dir / ".index_settings.json").write_text(json.dumps(data))
|
||||
|
||||
|
||||
class TestDescriptorsDescribeTheBuiltSchema:
|
||||
def test_descriptors_match_the_pinned_field_layout(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- PINNED_DESCRIPTORS, a frozen snapshot of the v2 on-disk field
|
||||
layout, reproduced from build_schema()'s output as it stood
|
||||
before the descriptor refactor
|
||||
WHEN:
|
||||
- field_descriptors() is called
|
||||
THEN:
|
||||
- It matches the pinned layout exactly, in the same order,
|
||||
pinning that the refactor changed nothing
|
||||
"""
|
||||
assert tuple(field_descriptors()) == PINNED_DESCRIPTORS
|
||||
|
||||
def test_built_schema_matches_the_descriptors(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The schema built by build_schema()
|
||||
WHEN:
|
||||
- Its fields are read back via __reduce__() (schema.__reduce__(),
|
||||
tantivy-py's pickling hook)
|
||||
THEN:
|
||||
- Every field's name, kind, stored/fast flags and tokenizer
|
||||
match what field_descriptors() declared as input; the
|
||||
descriptors are not a parallel description, they are the
|
||||
input, so a descriptor edit cannot claim a shape the
|
||||
SchemaBuilder did not actually build
|
||||
"""
|
||||
kinds = {"text": "text", "json": "json_object", "u64": "u64", "date": "date"}
|
||||
built = [
|
||||
(
|
||||
field["name"],
|
||||
field["type"],
|
||||
field["options"]["stored"],
|
||||
bool(field["options"].get("fast")),
|
||||
(field["options"].get("indexing") or {}).get("tokenizer"),
|
||||
)
|
||||
for field in _schema_fields(build_schema())
|
||||
]
|
||||
expected = [
|
||||
(
|
||||
descriptor.name,
|
||||
kinds[descriptor.kind],
|
||||
descriptor.stored,
|
||||
descriptor.fast,
|
||||
descriptor.tokenizer,
|
||||
)
|
||||
for descriptor in field_descriptors()
|
||||
]
|
||||
assert built == expected
|
||||
|
||||
|
||||
class TestFingerprintSensitivity:
|
||||
def test_a_field_option_change_moves_the_fingerprint(
|
||||
self,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The current schema fingerprint
|
||||
WHEN:
|
||||
- A single field descriptor's "fast" option is changed, with
|
||||
no other change
|
||||
THEN:
|
||||
- The fingerprint changes
|
||||
"""
|
||||
before = schema_fingerprint()
|
||||
changed = field_descriptors()
|
||||
changed[1] = changed[1]._replace(fast=True)
|
||||
monkeypatch.setattr(_schema, "field_descriptors", lambda: changed)
|
||||
|
||||
assert schema_fingerprint() != before
|
||||
|
||||
def test_reordering_alone_moves_the_fingerprint(
|
||||
self,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The current schema fingerprint
|
||||
WHEN:
|
||||
- Two field descriptors are swapped, with no other change (the
|
||||
original bug: same fields, different declaration order)
|
||||
THEN:
|
||||
- The fingerprint changes; a set- or dict-based fingerprint
|
||||
would be blind to this, and tantivy would reject every write
|
||||
against the existing index
|
||||
"""
|
||||
before = schema_fingerprint()
|
||||
swapped = field_descriptors()
|
||||
swapped[1], swapped[2] = swapped[2], swapped[1]
|
||||
monkeypatch.setattr(_schema, "field_descriptors", lambda: swapped)
|
||||
|
||||
assert schema_fingerprint() != before
|
||||
|
||||
|
||||
class TestFingerprintIsIndependentOfTantivy:
|
||||
def test_a_tantivy_option_key_addition_would_not_move_it(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The built schema's raw field list, and the same list with a
|
||||
new tantivy-internal option key added (simulating a
|
||||
tantivy-py upgrade)
|
||||
WHEN:
|
||||
- Both raw lists are hashed directly, and schema_fingerprint()
|
||||
is compared against a hash of field_descriptors()
|
||||
THEN:
|
||||
- The raw hashes differ (hashing schema.__reduce__() would
|
||||
force a global reindex on every tantivy-py upgrade), but
|
||||
schema_fingerprint() is unaffected, since it hashes
|
||||
field_descriptors(), never tantivy's own representation
|
||||
"""
|
||||
fields = _schema_fields(build_schema())
|
||||
upgraded = [
|
||||
{**field, "options": {**field["options"], "coerce": True}}
|
||||
for field in fields
|
||||
]
|
||||
assert _hash(upgraded) != _hash(fields)
|
||||
assert schema_fingerprint() == _fingerprint_of(field_descriptors())
|
||||
|
||||
def test_fingerprint_never_touches_the_schema_builder(
|
||||
self,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- tantivy.SchemaBuilder replaced with a stand-in that raises if
|
||||
constructed
|
||||
WHEN:
|
||||
- build_schema() is called (and raises), then
|
||||
schema_fingerprint() is called again
|
||||
THEN:
|
||||
- schema_fingerprint() still matches its earlier value,
|
||||
proving it never consults SchemaBuilder
|
||||
"""
|
||||
before = schema_fingerprint()
|
||||
|
||||
class _RemovedSchemaBuilder:
|
||||
def __init__(self) -> None:
|
||||
raise AssertionError("tantivy.SchemaBuilder was consulted")
|
||||
|
||||
monkeypatch.setattr(tantivy, "SchemaBuilder", _RemovedSchemaBuilder)
|
||||
with pytest.raises(AssertionError):
|
||||
build_schema()
|
||||
|
||||
assert schema_fingerprint() == before
|
||||
|
||||
|
||||
def _hash(payload: object) -> str:
|
||||
return hashlib.blake2b(json.dumps(payload).encode()).hexdigest()
|
||||
|
||||
|
||||
def _fingerprint_of(descriptors: list[FieldDescriptor]) -> str:
|
||||
return _hash([list(descriptor) for descriptor in descriptors])
|
||||
|
||||
|
||||
class TestNeedsRebuildOnFingerprint:
|
||||
def test_matching_fingerprint_does_not_rebuild(
|
||||
self,
|
||||
index_dir: Path,
|
||||
settings: SettingsWrapper,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- An index directory whose sentinel file records the current
|
||||
schema_fingerprint()
|
||||
WHEN:
|
||||
- needs_rebuild() is called
|
||||
THEN:
|
||||
- It returns False
|
||||
"""
|
||||
settings.SEARCH_LANGUAGE = None
|
||||
_sentinels(index_dir)
|
||||
|
||||
assert needs_rebuild(index_dir) is False
|
||||
|
||||
def test_stale_fingerprint_rebuilds_despite_a_matching_version(
|
||||
self,
|
||||
index_dir: Path,
|
||||
settings: SettingsWrapper,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- An index directory whose sentinel matches SCHEMA_VERSION,
|
||||
but field_descriptors() is patched to add a field the
|
||||
fingerprint never saw (schema edited, version not bumped)
|
||||
WHEN:
|
||||
- needs_rebuild() is called
|
||||
THEN:
|
||||
- It returns True; without the fingerprint check,
|
||||
`reindex --if-needed` would report the index up to date and
|
||||
every subsequent write would raise
|
||||
"""
|
||||
settings.SEARCH_LANGUAGE = None
|
||||
_sentinels(index_dir)
|
||||
extended = [
|
||||
*field_descriptors(),
|
||||
FieldDescriptor(
|
||||
"new_field",
|
||||
"u64",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer=None,
|
||||
),
|
||||
]
|
||||
monkeypatch.setattr(_schema, "field_descriptors", lambda: extended)
|
||||
|
||||
assert needs_rebuild(index_dir) is True
|
||||
|
||||
def test_reordered_schema_rebuilds(
|
||||
self,
|
||||
index_dir: Path,
|
||||
settings: SettingsWrapper,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- An index directory whose sentinel matches the current
|
||||
fingerprint, but field_descriptors() is patched to swap two
|
||||
fields' order
|
||||
WHEN:
|
||||
- needs_rebuild() is called
|
||||
THEN:
|
||||
- It returns True
|
||||
"""
|
||||
settings.SEARCH_LANGUAGE = None
|
||||
_sentinels(index_dir)
|
||||
reordered = field_descriptors()
|
||||
reordered[1], reordered[2] = reordered[2], reordered[1]
|
||||
monkeypatch.setattr(_schema, "field_descriptors", lambda: reordered)
|
||||
|
||||
assert needs_rebuild(index_dir) is True
|
||||
|
||||
def test_missing_fingerprint_rebuilds(
|
||||
self,
|
||||
index_dir: Path,
|
||||
settings: SettingsWrapper,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- An index directory whose sentinel has no "schema_fingerprint"
|
||||
key at all
|
||||
WHEN:
|
||||
- needs_rebuild() is called
|
||||
THEN:
|
||||
- It returns True; an index whose schema shape nobody recorded
|
||||
is rebuilt rather than trusted
|
||||
"""
|
||||
settings.SEARCH_LANGUAGE = None
|
||||
(index_dir / ".index_settings.json").write_text(
|
||||
json.dumps({"schema_version": SCHEMA_VERSION, "language": None}),
|
||||
)
|
||||
|
||||
assert needs_rebuild(index_dir) is True
|
||||
|
||||
def test_written_sentinels_satisfy_the_check(
|
||||
self,
|
||||
index_dir: Path,
|
||||
settings: SettingsWrapper,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- An index directory whose sentinels are written by
|
||||
_write_sentinels() itself
|
||||
WHEN:
|
||||
- needs_rebuild() is called
|
||||
THEN:
|
||||
- It returns False
|
||||
"""
|
||||
settings.SEARCH_LANGUAGE = "en"
|
||||
_write_sentinels(index_dir)
|
||||
|
||||
assert needs_rebuild(index_dir) is False
|
||||
@@ -0,0 +1,178 @@
|
||||
"""SCHEMA_VERSION must change whenever build_schema()'s field list or order does.
|
||||
|
||||
tantivy compares schemas by *ordered* field list. ``Index.open()`` loads the
|
||||
schema from the index's own ``meta.json``, so reads against an index built by an
|
||||
older release keep working after a field reorder. Writes do not:
|
||||
``WriteBatch.__enter__`` calls ``tantivy.Index(build_schema(), path=...)``, an
|
||||
open-or-create that raises ``ValueError`` on any schema difference. Nothing
|
||||
catches that ValueError, so consumption, index_document and bulk edit all
|
||||
hard-fail while ``/api/status/`` still reports the index healthy.
|
||||
|
||||
The only thing that saves such an install is ``needs_rebuild()`` noticing the
|
||||
version stamped in ``.index_settings.json`` is stale.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
from django.conf import settings as django_settings
|
||||
|
||||
from documents.search._schema import build_schema
|
||||
from documents.search._schema import needs_rebuild
|
||||
from documents.search._schema import open_or_rebuild_index
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
|
||||
pytestmark = [pytest.mark.search]
|
||||
|
||||
RELEASED_V1_SCHEMA_VERSION = 1
|
||||
|
||||
|
||||
def _build_released_v1_schema() -> tantivy.Schema:
|
||||
"""Frozen copy of build_schema() as shipped in v3.0.x (schema version 1).
|
||||
|
||||
Deliberately duplicated rather than imported: it must keep describing the
|
||||
on-disk layout of already-deployed indexes even as build_schema() evolves.
|
||||
"""
|
||||
sb = tantivy.SchemaBuilder()
|
||||
|
||||
sb.add_unsigned_field("id", stored=True, indexed=True, fast=True)
|
||||
sb.add_text_field("checksum", stored=True, tokenizer_name="raw")
|
||||
|
||||
for field in (
|
||||
"title",
|
||||
"correspondent",
|
||||
"document_type",
|
||||
"storage_path",
|
||||
"original_filename",
|
||||
"content",
|
||||
):
|
||||
sb.add_text_field(field, stored=True, tokenizer_name="paperless_text")
|
||||
|
||||
for field in ("title_sort", "correspondent_sort", "type_sort"):
|
||||
sb.add_text_field(
|
||||
field,
|
||||
stored=False,
|
||||
tokenizer_name="simple_analyzer",
|
||||
fast=True,
|
||||
)
|
||||
|
||||
for field in (
|
||||
"bigram_content",
|
||||
"bigram_title",
|
||||
"bigram_correspondent",
|
||||
"bigram_document_type",
|
||||
"bigram_tag",
|
||||
):
|
||||
sb.add_text_field(field, stored=False, tokenizer_name="bigram_analyzer")
|
||||
|
||||
for field in ("simple_title", "simple_content"):
|
||||
sb.add_text_field(field, stored=False, tokenizer_name="simple_search_analyzer")
|
||||
|
||||
sb.add_text_field("autocomplete_word", stored=False, tokenizer_name="raw")
|
||||
sb.add_text_field("tag", stored=True, tokenizer_name="paperless_text")
|
||||
|
||||
sb.add_json_field("notes", stored=True, tokenizer_name="paperless_text")
|
||||
sb.add_text_field("notes_text", stored=True, tokenizer_name="paperless_text")
|
||||
sb.add_json_field("custom_fields", stored=True, tokenizer_name="paperless_text")
|
||||
|
||||
for field in (
|
||||
"correspondent_id",
|
||||
"document_type_id",
|
||||
"storage_path_id",
|
||||
"tag_id",
|
||||
"owner_id",
|
||||
"viewer_id",
|
||||
"viewer_group_id",
|
||||
):
|
||||
sb.add_unsigned_field(field, stored=False, indexed=True, fast=True)
|
||||
|
||||
for field in ("created", "modified", "added"):
|
||||
sb.add_date_field(field, stored=True, indexed=True, fast=True)
|
||||
|
||||
for field in ("asn", "page_count", "num_notes"):
|
||||
sb.add_unsigned_field(field, stored=True, indexed=True, fast=True)
|
||||
|
||||
return sb.build()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def released_v1_index(tmp_path: Path) -> Path:
|
||||
"""An index directory as a v3.0.x install would leave it on disk."""
|
||||
index_dir = tmp_path / "index"
|
||||
index_dir.mkdir()
|
||||
tantivy.Index(_build_released_v1_schema(), path=str(index_dir))
|
||||
(index_dir / ".index_settings.json").write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"schema_version": RELEASED_V1_SCHEMA_VERSION,
|
||||
"language": django_settings.SEARCH_LANGUAGE,
|
||||
},
|
||||
),
|
||||
)
|
||||
return index_dir
|
||||
|
||||
|
||||
class TestUpgradeFromReleasedV1Index:
|
||||
def test_released_v1_index_is_flagged_for_rebuild(
|
||||
self,
|
||||
released_v1_index: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- An index directory laid out exactly as a v3.0.x (schema
|
||||
version 1) install would leave it
|
||||
WHEN:
|
||||
- needs_rebuild() is called
|
||||
THEN:
|
||||
- It returns True; if this fails,
|
||||
`document_index reindex --if-needed` prints "Search index is
|
||||
up to date" and skips, leaving the mismatched index in place
|
||||
"""
|
||||
assert needs_rebuild(released_v1_index) is True
|
||||
|
||||
def test_opening_a_v1_index_leaves_it_writable(
|
||||
self,
|
||||
released_v1_index: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A v1 index directory
|
||||
WHEN:
|
||||
- open_or_rebuild_index() is called against it
|
||||
THEN:
|
||||
- The directory can be reopened with the current schema
|
||||
without raising; end to end, open_or_rebuild_index must
|
||||
hand back an index the write path can reopen. Before the
|
||||
version bump, needs_rebuild() returned False here, and the
|
||||
stale directory survived untouched, so every subsequent
|
||||
write against it raised tantivy's own schema-mismatch
|
||||
ValueError
|
||||
"""
|
||||
open_or_rebuild_index(released_v1_index)
|
||||
|
||||
tantivy.Index(build_schema(), path=str(released_v1_index))
|
||||
|
||||
def test_rebuilt_index_is_not_rebuilt_again(
|
||||
self,
|
||||
released_v1_index: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A v1 index directory that has just been rebuilt by
|
||||
open_or_rebuild_index()
|
||||
WHEN:
|
||||
- needs_rebuild() is called again
|
||||
THEN:
|
||||
- It returns False; the rebuild must stamp the version it
|
||||
actually wrote, otherwise every startup wipes and reindexes
|
||||
the whole corpus
|
||||
"""
|
||||
open_or_rebuild_index(released_v1_index)
|
||||
|
||||
assert needs_rebuild(released_v1_index) is False
|
||||
@@ -0,0 +1,37 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.search._tokenizer import stem_pattern_text
|
||||
|
||||
pytestmark = pytest.mark.search
|
||||
|
||||
|
||||
class TestStemPatternText:
|
||||
def test_unsupported_language_returns_text_unchanged(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A language code with no Snowball stemmer mapping
|
||||
WHEN:
|
||||
- A pattern run is stemmed for that language
|
||||
THEN:
|
||||
- The run is returned unchanged, since the stemming gate that
|
||||
disables stemming for an unsupported language also disables
|
||||
the pattern-side stemmer
|
||||
"""
|
||||
assert stem_pattern_text("running", "klingon") == "running"
|
||||
|
||||
def test_run_past_remove_long_limit_returns_text_unchanged(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A supported language and a run longer than the remove_long
|
||||
filter's limit (129 characters, matching Document.title's
|
||||
max_length)
|
||||
WHEN:
|
||||
- The over-long run is stemmed
|
||||
THEN:
|
||||
- The remove_long filter drops the token entirely, leaving no
|
||||
stem to substitute, so the run is returned unchanged
|
||||
"""
|
||||
long_run = "a" * 130
|
||||
assert stem_pattern_text(long_run, "en") == long_run
|
||||
@@ -7,8 +7,8 @@ import pytest
|
||||
import tantivy
|
||||
|
||||
from documents.search._tokenizer import _bigram_analyzer
|
||||
from documents.search._tokenizer import _paperless_text
|
||||
from documents.search._tokenizer import _simple_search_analyzer
|
||||
from documents.search._tokenizer import paperless_text_analyzer
|
||||
from documents.search._tokenizer import register_tokenizers
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -25,7 +25,7 @@ class TestTokenizers:
|
||||
sb.add_text_field("content", stored=True, tokenizer_name="paperless_text")
|
||||
schema = sb.build()
|
||||
idx = tantivy.Index(schema, path=None)
|
||||
idx.register_tokenizer("paperless_text", _paperless_text(""))
|
||||
idx.register_tokenizer("paperless_text", paperless_text_analyzer(""))
|
||||
return idx
|
||||
|
||||
@pytest.fixture
|
||||
|
||||
@@ -1,810 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC
|
||||
from datetime import datetime
|
||||
from typing import TYPE_CHECKING
|
||||
from zoneinfo import ZoneInfo
|
||||
|
||||
import pytest
|
||||
import time_machine
|
||||
|
||||
from documents.search._dates import _precision_bounds
|
||||
|
||||
if TYPE_CHECKING:
|
||||
import tantivy
|
||||
from documents.search._query import _FIELD_BOOSTS
|
||||
from documents.search._query import DEFAULT_SEARCH_FIELDS
|
||||
from documents.search._translate import OPEN_HI
|
||||
from documents.search._translate import OPEN_LO
|
||||
from documents.search._translate import Comma
|
||||
from documents.search._translate import FieldRange
|
||||
from documents.search._translate import FieldValue
|
||||
from documents.search._translate import FieldValueList
|
||||
from documents.search._translate import InvalidDateQuery
|
||||
from documents.search._translate import Passthrough
|
||||
from documents.search._translate import resolve_commas
|
||||
from documents.search._translate import scan
|
||||
from documents.search._translate import translate_query
|
||||
from documents.search._translate import translate_range
|
||||
from documents.search._translate import translate_scalar
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestPrecisionBounds:
|
||||
@pytest.mark.parametrize(
|
||||
("digits", "expected"),
|
||||
[
|
||||
("2020", ((2020, 1, 1), (2021, 1, 1))),
|
||||
("202003", ((2020, 3, 1), (2020, 4, 1))),
|
||||
("202012", ((2020, 12, 1), (2021, 1, 1))),
|
||||
("20200115", ((2020, 1, 15), (2020, 1, 16))),
|
||||
("20201231", ((2020, 12, 31), (2021, 1, 1))),
|
||||
],
|
||||
)
|
||||
def test_valid(self, digits, expected):
|
||||
lo, hi = _precision_bounds(digits)
|
||||
assert (lo.year, lo.month, lo.day) == expected[0]
|
||||
assert (hi.year, hi.month, hi.day) == expected[1]
|
||||
|
||||
@pytest.mark.parametrize("digits", ["202023", "20200230", "20201301", "20", "abcd"])
|
||||
def test_invalid_returns_none(self, digits):
|
||||
assert _precision_bounds(digits) is None
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestScan:
|
||||
def test_plain_words_are_passthrough(self):
|
||||
assert scan("bank statement") == [Passthrough("bank statement")]
|
||||
|
||||
def test_field_value(self):
|
||||
assert scan("created:2020") == [FieldValue("created", "2020")]
|
||||
|
||||
def test_field_value_in_boolean(self):
|
||||
toks = scan("created:2020 OR foo")
|
||||
assert toks == [
|
||||
FieldValue("created", "2020"),
|
||||
Passthrough(" OR foo"),
|
||||
]
|
||||
|
||||
def test_field_value_in_parens(self):
|
||||
toks = scan("(created:2020 OR foo)")
|
||||
assert toks == [
|
||||
Passthrough("("),
|
||||
FieldValue("created", "2020"),
|
||||
Passthrough(" OR foo)"),
|
||||
]
|
||||
|
||||
def test_quoted_value(self):
|
||||
assert scan('correspondent:"A B"') == [FieldValue("correspondent", '"A B"')]
|
||||
|
||||
def test_field_range(self):
|
||||
assert scan("created:[2020 TO 2021]") == [
|
||||
FieldRange("created", "[", "2020", "2021", "]"),
|
||||
]
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("query", "expected"),
|
||||
[
|
||||
pytest.param(
|
||||
"created:[2020 to]",
|
||||
FieldRange("created", "[", "2020", "", "]"),
|
||||
id="open_upper",
|
||||
),
|
||||
pytest.param(
|
||||
"created:[to 2020]",
|
||||
FieldRange("created", "[", "", "2020", "]"),
|
||||
id="open_lower",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_open_range(self, query, expected):
|
||||
assert scan(query) == [expected]
|
||||
|
||||
def test_comma_inside_range_not_split(self):
|
||||
# No depth-0 comma here; the whole thing is one range token.
|
||||
toks = scan("created:[2020 TO 2021]")
|
||||
assert len(toks) == 1
|
||||
|
||||
# --- Edge-case / regression tests (scan must never raise) ---
|
||||
|
||||
def test_url_is_passthrough(self):
|
||||
# "http" is not a known field; the whole URL must pass through verbatim.
|
||||
assert scan("http://example.com") == [Passthrough("http://example.com")]
|
||||
|
||||
def test_unterminated_quote_is_passthrough(self):
|
||||
# title is a known field but the quoted value has no closing quote;
|
||||
# _consume_value returns None so the whole string falls into passthrough.
|
||||
assert scan('title:"abc') == [Passthrough('title:"abc')]
|
||||
|
||||
def test_unterminated_bracket_is_passthrough(self):
|
||||
# created is a known field but the range bracket is never closed;
|
||||
# _consume_range returns None so the whole string falls into passthrough.
|
||||
assert scan("created:[2020") == [Passthrough("created:[2020")]
|
||||
|
||||
def test_empty_value_at_end_is_passthrough(self):
|
||||
# created is a known field but there is no value after the colon
|
||||
# (_consume_value returns None for start >= n), so passthrough.
|
||||
assert scan("created:") == [Passthrough("created:")]
|
||||
|
||||
def test_value_containing_colon(self):
|
||||
# The bare-word value reader stops at whitespace/paren, not at colon,
|
||||
# so "2020:30" is consumed as a single value token.
|
||||
assert scan("created:2020:30") == [FieldValue("created", "2020:30")]
|
||||
|
||||
def test_comma_followed_by_unconsumable_value_stops(self):
|
||||
# A comma followed by whitespace is neither a value-list continuation nor a
|
||||
# clause separator: the value stops and the comma stays as passthrough.
|
||||
assert scan("tag:foo, bar") == [
|
||||
FieldValue("tag", "foo"),
|
||||
Passthrough(", bar"),
|
||||
]
|
||||
|
||||
def test_bracket_without_to_is_open_upper_bound(self):
|
||||
# A bracketed value with no TO falls back to (value, "") -> open upper bound.
|
||||
assert scan("created:[2020]") == [
|
||||
FieldRange("created", "[", "2020", "", "]"),
|
||||
]
|
||||
|
||||
def test_known_field_name_midword_is_passthrough(self):
|
||||
# A known field name embedded mid-word is not a field token (the
|
||||
# word-boundary guard); the whole run stays passthrough.
|
||||
assert scan("xtag:foo") == [Passthrough("xtag:foo")]
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestCommaResolution:
|
||||
def test_value_list_multi_value_field(self):
|
||||
toks = resolve_commas(scan("tag:foo,bar"))
|
||||
assert toks == [FieldValueList("tag", ("foo", "bar"))]
|
||||
|
||||
def test_value_list_three(self):
|
||||
toks = resolve_commas(scan("tag_id:1,2,3"))
|
||||
assert toks == [FieldValueList("tag_id", ("1", "2", "3"))]
|
||||
|
||||
def test_text_field_comma_is_literal(self):
|
||||
# correspondent is not multi-value: comma stays inside the value.
|
||||
toks = resolve_commas(scan("correspondent:foo,bar"))
|
||||
assert toks == [FieldValue("correspondent", "foo,bar")]
|
||||
|
||||
def test_clause_separator_before_known_field(self):
|
||||
toks = resolve_commas(scan("tag:foo,type:bar"))
|
||||
assert toks == [FieldValue("tag", "foo"), Comma(), FieldValue("type", "bar")]
|
||||
|
||||
def test_clause_separator_after_range(self):
|
||||
toks = resolve_commas(scan("created:[2020 TO 2021],added:[2022 TO 2023]"))
|
||||
assert toks == [
|
||||
FieldRange("created", "[", "2020", "2021", "]"),
|
||||
Comma(),
|
||||
FieldRange("added", "[", "2022", "2023", "]"),
|
||||
]
|
||||
|
||||
def test_clause_separator_after_quote(self):
|
||||
toks = resolve_commas(scan('correspondent:"A B",created:[2020 TO 2021]'))
|
||||
assert toks == [
|
||||
FieldValue("correspondent", '"A B"'),
|
||||
Comma(),
|
||||
FieldRange("created", "[", "2020", "2021", "]"),
|
||||
]
|
||||
|
||||
def test_url_comma_is_literal_passthrough(self):
|
||||
toks = resolve_commas(scan("http://example.com/a,b"))
|
||||
assert toks == [Passthrough("http://example.com/a,b")]
|
||||
|
||||
def test_non_multi_value_comma_is_literal(self):
|
||||
# title is not in MULTI_VALUE_FIELDS: comma stays inside the value.
|
||||
toks = resolve_commas(scan("title:10,20"))
|
||||
assert toks == [FieldValue("title", "10,20")]
|
||||
|
||||
def test_clause_separator_before_known_date_field(self):
|
||||
# The comma between a bare value and a known date field acts as a
|
||||
# clause separator; both sides survive as distinct tokens.
|
||||
toks = resolve_commas(scan("correspondent:foo,created:[2020 TO 2021]"))
|
||||
assert toks == [
|
||||
FieldValue("correspondent", "foo"),
|
||||
Comma(),
|
||||
FieldRange("created", "[", "2020", "2021", "]"),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestTranslateScalar:
|
||||
@pytest.mark.parametrize(
|
||||
("field", "value", "expected"),
|
||||
[
|
||||
(
|
||||
"created",
|
||||
"2020",
|
||||
"created:[2020-01-01T00:00:00Z TO 2021-01-01T00:00:00Z}",
|
||||
),
|
||||
(
|
||||
"created",
|
||||
"202003",
|
||||
"created:[2020-03-01T00:00:00Z TO 2020-04-01T00:00:00Z}",
|
||||
),
|
||||
(
|
||||
"created",
|
||||
"20200115",
|
||||
"created:[2020-01-15T00:00:00Z TO 2020-01-16T00:00:00Z}",
|
||||
),
|
||||
(
|
||||
"created",
|
||||
"2020-01-15",
|
||||
"created:[2020-01-15T00:00:00Z TO 2020-01-16T00:00:00Z}",
|
||||
),
|
||||
(
|
||||
"created",
|
||||
"2020-03",
|
||||
"created:[2020-03-01T00:00:00Z TO 2020-04-01T00:00:00Z}",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_partial_and_iso_dates(self, field: str, value: str, expected: str) -> None:
|
||||
assert translate_scalar(field, value, UTC) == expected
|
||||
|
||||
def test_invalid_date_raises(self) -> None:
|
||||
with pytest.raises(InvalidDateQuery) as exc_info:
|
||||
translate_scalar("created", "202023", UTC)
|
||||
assert exc_info.value.field == "created"
|
||||
assert exc_info.value.value == "202023"
|
||||
|
||||
def test_keyword_delegates(self) -> None:
|
||||
# keyword path produces a half-open range; just assert it is a created range
|
||||
out = translate_scalar("created", "today", UTC)
|
||||
assert out.startswith("created:[") and out.endswith("}")
|
||||
|
||||
def test_14digit_compact_datetime(self) -> None:
|
||||
out = translate_scalar("created", "20240115120000", UTC)
|
||||
assert "20240115120000" not in out
|
||||
assert out.startswith("created:")
|
||||
assert out == "created:[2024-01-15T12:00:00Z TO 2024-01-15T12:00:00Z]"
|
||||
|
||||
def test_14digit_invalid_month_raises(self) -> None:
|
||||
with pytest.raises(InvalidDateQuery) as exc_info:
|
||||
translate_scalar("created", "20231300120000", UTC)
|
||||
assert exc_info.value.field == "created"
|
||||
assert exc_info.value.value == "20231300120000"
|
||||
|
||||
def test_unrecognized_value_raises(self) -> None:
|
||||
# A value that is not a keyword, digits, ISO date, or compact timestamp
|
||||
# raises rather than producing invalid Tantivy syntax or silently matching
|
||||
# nothing.
|
||||
with pytest.raises(InvalidDateQuery) as exc_info:
|
||||
translate_scalar("created", "garbage", UTC)
|
||||
assert exc_info.value.field == "created"
|
||||
assert exc_info.value.value == "garbage"
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestTranslateRange:
|
||||
@pytest.mark.parametrize(
|
||||
("lo", "hi", "expected"),
|
||||
[
|
||||
("2005", "2009", "created:[2005-01-01T00:00:00Z TO 2010-01-01T00:00:00Z}"),
|
||||
(
|
||||
"202001",
|
||||
"202006",
|
||||
"created:[2020-01-01T00:00:00Z TO 2020-07-01T00:00:00Z}",
|
||||
),
|
||||
(
|
||||
"20200101",
|
||||
"20201231",
|
||||
"created:[2020-01-01T00:00:00Z TO 2021-01-01T00:00:00Z}",
|
||||
),
|
||||
(
|
||||
"2020-01-01",
|
||||
"2020-12-31",
|
||||
"created:[2020-01-01T00:00:00Z TO 2021-01-01T00:00:00Z}",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_absolute_ranges(self, lo, hi, expected):
|
||||
assert translate_range("created", lo, hi, UTC) == expected
|
||||
|
||||
def test_reversed_swaps(self):
|
||||
assert translate_range("created", "2009", "2005", UTC) == (
|
||||
"created:[2005-01-01T00:00:00Z TO 2010-01-01T00:00:00Z}"
|
||||
)
|
||||
|
||||
def test_open_upper(self):
|
||||
out = translate_range("created", "2020", "", UTC)
|
||||
assert out == f"created:[2020-01-01T00:00:00Z TO {OPEN_HI}]"
|
||||
|
||||
def test_open_lower(self):
|
||||
out = translate_range("created", "", "2020", UTC)
|
||||
assert out == f"created:[{OPEN_LO} TO 2021-01-01T00:00:00Z}}"
|
||||
|
||||
def test_invalid_bound_raises(self):
|
||||
with pytest.raises(InvalidDateQuery) as exc_info:
|
||||
translate_range("created", "202023", "2025", UTC)
|
||||
assert exc_info.value.field == "created"
|
||||
assert exc_info.value.value == "202023"
|
||||
|
||||
def test_invalid_high_bound_raises(self):
|
||||
# Low bound parses, high bound does not -> raise on the high bound.
|
||||
with pytest.raises(InvalidDateQuery) as exc_info:
|
||||
translate_range("created", "2020", "garbage", UTC)
|
||||
assert exc_info.value.field == "created"
|
||||
assert exc_info.value.value == "garbage"
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestTranslateQuery:
|
||||
@pytest.mark.parametrize(
|
||||
("raw", "expected"),
|
||||
[
|
||||
(
|
||||
"created:2020",
|
||||
"created:[2020-01-01T00:00:00Z TO 2021-01-01T00:00:00Z}",
|
||||
),
|
||||
("tag:foo,bar", "tag:foo AND tag:bar"),
|
||||
# 'type' is a user-facing alias rewritten to 'document_type' (the real schema field)
|
||||
("tag:foo,type:bar", "tag:foo AND document_type:bar"),
|
||||
(
|
||||
"created:[2020 TO 2021],added:[2022 TO 2023]",
|
||||
(
|
||||
"created:[2020-01-01T00:00:00Z TO 2022-01-01T00:00:00Z}"
|
||||
" AND "
|
||||
"added:[2022-01-01T00:00:00Z TO 2024-01-01T00:00:00Z}"
|
||||
),
|
||||
),
|
||||
# correspondent is not multi-value: comma stays literal inside the value
|
||||
("correspondent:foo,bar", "correspondent:foo,bar"),
|
||||
],
|
||||
)
|
||||
def test_golden(self, raw: str, expected: str) -> None:
|
||||
assert translate_query(raw, UTC) == expected
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"raw",
|
||||
[
|
||||
"created:2020",
|
||||
"created:202003",
|
||||
"created:[20200101 TO 20201231]",
|
||||
"created:[2020-01-01 TO 2020-12-31]",
|
||||
"created:[2020 to]",
|
||||
"created:[to 2020]",
|
||||
"title:x,created:[2020 TO 2021]",
|
||||
"created:2020 OR foo",
|
||||
"(created:2020 OR invoice)",
|
||||
"tag:foo,type:bar",
|
||||
"bank statement",
|
||||
],
|
||||
)
|
||||
def test_parse_acceptance(self, index: tantivy.Index, raw: str) -> None:
|
||||
translated = translate_query(raw, UTC)
|
||||
# Must not raise:
|
||||
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestFieldAliasing:
|
||||
"""Whoosh->Tantivy field-name aliasing (type/path -> document_type/storage_path)."""
|
||||
|
||||
def test_type_alias(self) -> None:
|
||||
assert translate_query("type:invoice", UTC) == "document_type:invoice"
|
||||
|
||||
def test_path_alias(self) -> None:
|
||||
assert translate_query("path:/foo/bar", UTC) == "storage_path:/foo/bar"
|
||||
|
||||
def test_type_id_alias(self) -> None:
|
||||
assert translate_query("type_id:5", UTC) == "document_type_id:5"
|
||||
|
||||
def test_path_id_alias(self) -> None:
|
||||
assert translate_query("path_id:7", UTC) == "storage_path_id:7"
|
||||
|
||||
def test_clause_separator_plus_alias(self) -> None:
|
||||
# Comma between known fields acts as AND separator; alias still applied.
|
||||
assert (
|
||||
translate_query("tag:foo,type:bar", UTC) == "tag:foo AND document_type:bar"
|
||||
)
|
||||
|
||||
def test_type_range_alias(self) -> None:
|
||||
# type is not a date field; range passes through verbatim with alias applied.
|
||||
assert (
|
||||
translate_query("type:[2020 TO 2021]", UTC)
|
||||
== "document_type:[2020 TO 2021]"
|
||||
)
|
||||
|
||||
def test_parse_acceptance_type(self, index: tantivy.Index) -> None:
|
||||
# Translated output must be accepted by the real Tantivy parser.
|
||||
translated = translate_query("type:invoice", UTC)
|
||||
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
|
||||
|
||||
def test_parse_acceptance_path(self, index: tantivy.Index) -> None:
|
||||
translated = translate_query("path:foo", UTC)
|
||||
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
|
||||
|
||||
|
||||
# Freeze time so relative-date tests are deterministic.
|
||||
_FROZEN_NOW = datetime(2026, 3, 28, 12, 0, 0, tzinfo=UTC)
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestRelativeRanges:
|
||||
"""Relative date-range tokens resolved against a frozen clock."""
|
||||
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_minus_7_days_to_now(self) -> None:
|
||||
assert translate_query("added:[-7 days to now]", UTC) == (
|
||||
"added:[2026-03-21T12:00:00Z TO 2026-03-28T12:00:00Z]"
|
||||
)
|
||||
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_minus_1_week_to_now(self) -> None:
|
||||
assert translate_query("added:[-1 week to now]", UTC) == (
|
||||
"added:[2026-03-21T12:00:00Z TO 2026-03-28T12:00:00Z]"
|
||||
)
|
||||
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_minus_1_month_to_now(self) -> None:
|
||||
assert translate_query("created:[-1 month to now]", UTC) == (
|
||||
"created:[2026-02-28T12:00:00Z TO 2026-03-28T12:00:00Z]"
|
||||
)
|
||||
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_minus_1_year_to_now(self) -> None:
|
||||
assert translate_query("modified:[-1 year to now]", UTC) == (
|
||||
"modified:[2025-03-28T12:00:00Z TO 2026-03-28T12:00:00Z]"
|
||||
)
|
||||
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_minus_3_hours_to_now(self) -> None:
|
||||
assert translate_query("added:[-3 hours to now]", UTC) == (
|
||||
"added:[2026-03-28T09:00:00Z TO 2026-03-28T12:00:00Z]"
|
||||
)
|
||||
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_uppercase_units(self) -> None:
|
||||
assert translate_query("added:[-1 WEEK TO NOW]", UTC) == (
|
||||
"added:[2026-03-21T12:00:00Z TO 2026-03-28T12:00:00Z]"
|
||||
)
|
||||
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_now_minus_7d_compact(self) -> None:
|
||||
assert translate_query("added:[now-7d TO now]", UTC) == (
|
||||
"added:[2026-03-21T12:00:00Z TO 2026-03-28T12:00:00Z]"
|
||||
)
|
||||
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_reversed_range_swapped(self) -> None:
|
||||
# now+1h TO now-1h is reversed; translate_range swaps -> lo=now-1h, hi=now+1h
|
||||
assert translate_query("added:[now+1h TO now-1h]", UTC) == (
|
||||
"added:[2026-03-28T11:00:00Z TO 2026-03-28T13:00:00Z]"
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"raw",
|
||||
[
|
||||
"added:[-7 days to now]",
|
||||
"added:[-1 week to now]",
|
||||
"created:[-1 month to now]",
|
||||
"modified:[-1 year to now]",
|
||||
"added:[-3 hours to now]",
|
||||
"added:[now-7d TO now]",
|
||||
"added:[now+1h TO now-1h]",
|
||||
],
|
||||
)
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_parse_acceptance(self, index: tantivy.Index, raw: str) -> None:
|
||||
translated = translate_query(raw, UTC)
|
||||
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestWhooshUnitAbbreviations:
|
||||
"""
|
||||
Whoosh's PlusMinus date grammar accepted abbreviated unit spellings
|
||||
(e.g. "yrs", "mos", "wks", "hrs", "mins", "secs"); saved views/searches
|
||||
created under the old Whoosh backend can contain those tokens (see
|
||||
https://github.com/paperless-ngx/paperless-ngx/issues/13482), so the
|
||||
Tantivy translator must still accept them.
|
||||
"""
|
||||
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_minus_999_yrs(self) -> None:
|
||||
assert translate_query("created:[-999yrs to now]", UTC) == (
|
||||
"created:[1027-03-28T12:00:00Z TO 2026-03-28T12:00:00Z]"
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("token", "expected_lo"),
|
||||
[
|
||||
("-1y", "2025-03-28T12:00:00Z"),
|
||||
("-1yr", "2025-03-28T12:00:00Z"),
|
||||
("-3mos", "2025-12-28T12:00:00Z"),
|
||||
("-3mo", "2025-12-28T12:00:00Z"),
|
||||
("-2wks", "2026-03-14T12:00:00Z"),
|
||||
("-2wk", "2026-03-14T12:00:00Z"),
|
||||
("-5dys", "2026-03-23T12:00:00Z"),
|
||||
("-5dy", "2026-03-23T12:00:00Z"),
|
||||
("-1hrs", "2026-03-28T11:00:00Z"),
|
||||
("-1hr", "2026-03-28T11:00:00Z"),
|
||||
("-10mins", "2026-03-28T11:50:00Z"),
|
||||
("-10min", "2026-03-28T11:50:00Z"),
|
||||
("-30secs", "2026-03-28T11:59:30Z"),
|
||||
("-30sec", "2026-03-28T11:59:30Z"),
|
||||
],
|
||||
)
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_abbreviated_units(self, token: str, expected_lo: str) -> None:
|
||||
assert translate_query(f"added:[{token} to now]", UTC) == (
|
||||
f"added:[{expected_lo} TO 2026-03-28T12:00:00Z]"
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"raw",
|
||||
[
|
||||
"created:[-999yrs to now]",
|
||||
"added:[-1y to now]",
|
||||
"created:[-3mos to now]",
|
||||
"added:[-2wks to now]",
|
||||
"added:[-5dys to now]",
|
||||
"added:[-1hrs to now]",
|
||||
"added:[-10mins to now]",
|
||||
"added:[-30secs to now]",
|
||||
],
|
||||
)
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_parse_acceptance(self, index: tantivy.Index, raw: str) -> None:
|
||||
translated = translate_query(raw, UTC)
|
||||
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestOperatorNormalization:
|
||||
"""Post-render operator normalization in translate_query."""
|
||||
|
||||
def test_spaced_dash_removed(self) -> None:
|
||||
assert (
|
||||
translate_query("H52.1 - Kurzsichtigkeit", UTC) == "H52.1 Kurzsichtigkeit"
|
||||
)
|
||||
|
||||
def test_spaced_dash_simple(self) -> None:
|
||||
assert translate_query("bar - baz", UTC) == "bar baz"
|
||||
|
||||
def test_trailing_operator_stripped(self) -> None:
|
||||
assert translate_query("foo -", UTC) == "foo"
|
||||
|
||||
def test_date_range_preserved(self) -> None:
|
||||
out = translate_query("created:[2020 TO 2021]", UTC)
|
||||
# Must not corrupt the ISO range
|
||||
assert out == "created:[2020-01-01T00:00:00Z TO 2022-01-01T00:00:00Z}"
|
||||
|
||||
def test_date_scalar_with_or(self) -> None:
|
||||
out = translate_query("created:2020 OR foo", UTC)
|
||||
# The created scalar becomes a range; " OR foo" passes through verbatim.
|
||||
assert out.startswith("created:[")
|
||||
assert "OR foo" in out
|
||||
|
||||
def test_parse_acceptance_spaced_dash(self, index: tantivy.Index) -> None:
|
||||
translated = translate_query("H52.1 - Kurzsichtigkeit", UTC)
|
||||
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
|
||||
|
||||
def test_parse_acceptance_trailing_op(self, index: tantivy.Index) -> None:
|
||||
translated = translate_query("foo -", UTC)
|
||||
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestMultiWordDateKeywords:
|
||||
"""scan() must consume multi-word date keywords as a single value."""
|
||||
|
||||
def test_scan_previous_week_as_single_token(self) -> None:
|
||||
# "created:previous week" must produce one FieldValue with value "previous week",
|
||||
# not FieldValue("created","previous") + Passthrough(" week").
|
||||
toks = scan("created:previous week")
|
||||
assert toks == [FieldValue("created", "previous week")]
|
||||
|
||||
def test_scan_this_month_as_single_token(self) -> None:
|
||||
toks = scan("added:this month")
|
||||
assert toks == [FieldValue("added", "this month")]
|
||||
|
||||
def test_scan_previous_month_as_single_token(self) -> None:
|
||||
toks = scan("created:previous month")
|
||||
assert toks == [FieldValue("created", "previous month")]
|
||||
|
||||
def test_scan_this_year_as_single_token(self) -> None:
|
||||
toks = scan("added:this year")
|
||||
assert toks == [FieldValue("added", "this year")]
|
||||
|
||||
def test_scan_previous_year_as_single_token(self) -> None:
|
||||
toks = scan("created:previous year")
|
||||
assert toks == [FieldValue("created", "previous year")]
|
||||
|
||||
def test_scan_previous_quarter_as_single_token(self) -> None:
|
||||
toks = scan("created:previous quarter")
|
||||
assert toks == [FieldValue("created", "previous quarter")]
|
||||
|
||||
def test_quoted_multi_word_keyword_still_works(self) -> None:
|
||||
# The quoted form must continue to work as before.
|
||||
toks = scan('created:"previous week"')
|
||||
assert toks == [FieldValue("created", '"previous week"')]
|
||||
|
||||
def test_non_date_field_not_affected(self) -> None:
|
||||
# "previous" stops at the space for non-date fields; " week" passes through.
|
||||
toks = scan("correspondent:previous week")
|
||||
assert toks == [
|
||||
FieldValue("correspondent", "previous"),
|
||||
Passthrough(" week"),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestKeywordDateResolution:
|
||||
"""Relative date keywords resolve to exact ISO ranges against a frozen clock.
|
||||
|
||||
Frozen at 2026-03-28 12:00 UTC (a Saturday in Q1) so the week, month,
|
||||
quarter and year rollovers are all exercised by a single anchor.
|
||||
"""
|
||||
|
||||
# created is a DateField: bounds are UTC midnight, no timezone offset.
|
||||
@pytest.mark.parametrize(
|
||||
("keyword", "expected"),
|
||||
[
|
||||
pytest.param(
|
||||
"today",
|
||||
"created:[2026-03-28T00:00:00Z TO 2026-03-29T00:00:00Z}",
|
||||
id="today",
|
||||
),
|
||||
pytest.param(
|
||||
"yesterday",
|
||||
"created:[2026-03-27T00:00:00Z TO 2026-03-28T00:00:00Z}",
|
||||
id="yesterday",
|
||||
),
|
||||
pytest.param(
|
||||
"previous week",
|
||||
"created:[2026-03-16T00:00:00Z TO 2026-03-23T00:00:00Z}",
|
||||
id="previous-week",
|
||||
),
|
||||
pytest.param(
|
||||
"this month",
|
||||
"created:[2026-03-01T00:00:00Z TO 2026-04-01T00:00:00Z}",
|
||||
id="this-month",
|
||||
),
|
||||
pytest.param(
|
||||
"previous month",
|
||||
"created:[2026-02-01T00:00:00Z TO 2026-03-01T00:00:00Z}",
|
||||
id="previous-month",
|
||||
),
|
||||
pytest.param(
|
||||
"this year",
|
||||
"created:[2026-01-01T00:00:00Z TO 2027-01-01T00:00:00Z}",
|
||||
id="this-year",
|
||||
),
|
||||
pytest.param(
|
||||
"previous year",
|
||||
"created:[2025-01-01T00:00:00Z TO 2026-01-01T00:00:00Z}",
|
||||
id="previous-year",
|
||||
),
|
||||
pytest.param(
|
||||
"previous quarter",
|
||||
"created:[2025-10-01T00:00:00Z TO 2026-01-01T00:00:00Z}",
|
||||
id="previous-quarter",
|
||||
),
|
||||
],
|
||||
)
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_date_only_field_keyword_ranges(
|
||||
self,
|
||||
keyword: str,
|
||||
expected: str,
|
||||
) -> None:
|
||||
assert translate_query(f"created:{keyword}", UTC) == expected
|
||||
|
||||
# added is a DateTimeField: local-tz midnight converted to UTC. Tokyo
|
||||
# (+09:00, no DST) shifts each midnight boundary back to 15:00Z the day
|
||||
# before, so this also exercises the local-midnight offset path.
|
||||
@pytest.mark.parametrize(
|
||||
("keyword", "expected"),
|
||||
[
|
||||
pytest.param(
|
||||
"today",
|
||||
"added:[2026-03-27T15:00:00Z TO 2026-03-28T15:00:00Z}",
|
||||
id="today",
|
||||
),
|
||||
pytest.param(
|
||||
"yesterday",
|
||||
"added:[2026-03-26T15:00:00Z TO 2026-03-27T15:00:00Z}",
|
||||
id="yesterday",
|
||||
),
|
||||
pytest.param(
|
||||
"previous week",
|
||||
"added:[2026-03-15T15:00:00Z TO 2026-03-22T15:00:00Z}",
|
||||
id="previous-week",
|
||||
),
|
||||
pytest.param(
|
||||
"this month",
|
||||
"added:[2026-02-28T15:00:00Z TO 2026-03-31T15:00:00Z}",
|
||||
id="this-month",
|
||||
),
|
||||
pytest.param(
|
||||
"previous month",
|
||||
"added:[2026-01-31T15:00:00Z TO 2026-02-28T15:00:00Z}",
|
||||
id="previous-month",
|
||||
),
|
||||
pytest.param(
|
||||
"this year",
|
||||
"added:[2025-12-31T15:00:00Z TO 2026-12-31T15:00:00Z}",
|
||||
id="this-year",
|
||||
),
|
||||
pytest.param(
|
||||
"previous year",
|
||||
"added:[2024-12-31T15:00:00Z TO 2025-12-31T15:00:00Z}",
|
||||
id="previous-year",
|
||||
),
|
||||
pytest.param(
|
||||
"previous quarter",
|
||||
"added:[2025-09-30T15:00:00Z TO 2025-12-31T15:00:00Z}",
|
||||
id="previous-quarter",
|
||||
),
|
||||
],
|
||||
)
|
||||
@time_machine.travel(_FROZEN_NOW, tick=False)
|
||||
def test_datetime_field_keyword_ranges_local_tz(
|
||||
self,
|
||||
keyword: str,
|
||||
expected: str,
|
||||
) -> None:
|
||||
assert translate_query(f"added:{keyword}", ZoneInfo("Asia/Tokyo")) == expected
|
||||
|
||||
|
||||
@pytest.mark.search
|
||||
class TestISODatetimeBounds:
|
||||
"""Full ISO datetime tokens in range bounds must be parsed directly."""
|
||||
|
||||
def test_translate_range_iso_bounds_passthrough(self) -> None:
|
||||
# Already-ISO datetime bounds must pass through as-is (exact instant).
|
||||
result = translate_range(
|
||||
"created",
|
||||
"2020-01-01T00:00:00Z",
|
||||
"2021-01-01T00:00:00Z",
|
||||
UTC,
|
||||
)
|
||||
assert result == "created:[2020-01-01T00:00:00Z TO 2021-01-01T00:00:00Z]"
|
||||
|
||||
def test_translate_query_iso_range_preserved(self) -> None:
|
||||
q = "created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
|
||||
assert translate_query(q, UTC) == q
|
||||
|
||||
def test_translate_query_comma_separated_iso_ranges(self) -> None:
|
||||
q = (
|
||||
"created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z],"
|
||||
"added:[2026-05-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
|
||||
)
|
||||
result = translate_query(q, UTC)
|
||||
assert result == (
|
||||
"created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
|
||||
" AND "
|
||||
"added:[2026-05-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
|
||||
)
|
||||
|
||||
def test_translate_query_text_before_comma_separated_date_clause(self) -> None:
|
||||
result = translate_query("schäfersee,created:previous year", UTC)
|
||||
assert result == (
|
||||
"schäfersee AND created:[2025-01-01T00:00:00Z TO 2026-01-01T00:00:00Z}"
|
||||
)
|
||||
|
||||
def test_invalid_iso_datetime_raises(self) -> None:
|
||||
# A token with "T" that is not valid ISO datetime -> raise.
|
||||
with pytest.raises(InvalidDateQuery) as exc_info:
|
||||
translate_range(
|
||||
"created",
|
||||
"2020-01-01T99:00:00Z",
|
||||
"2021-01-01T00:00:00Z",
|
||||
UTC,
|
||||
)
|
||||
assert exc_info.value.field == "created"
|
||||
assert exc_info.value.value == "2020-01-01T99:00:00Z"
|
||||
|
||||
def test_parse_acceptance_iso_bounds(self, index: tantivy.Index) -> None:
|
||||
q = "created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
|
||||
translated = translate_query(q, UTC)
|
||||
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
|
||||
|
||||
def test_parse_acceptance_comma_iso_ranges(self, index: tantivy.Index) -> None:
|
||||
q = (
|
||||
"created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z],"
|
||||
"added:[2026-05-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
|
||||
)
|
||||
translated = translate_query(q, UTC)
|
||||
index.parse_query(translated, DEFAULT_SEARCH_FIELDS, field_boosts=_FIELD_BOOSTS)
|
||||
@@ -35,7 +35,8 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
|
||||
THEN:
|
||||
- Existing config
|
||||
"""
|
||||
response = self.client.get(self.ENDPOINT, format="json")
|
||||
with patch.dict("os.environ", {}, clear=True):
|
||||
response = self.client.get(self.ENDPOINT, format="json")
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
|
||||
@@ -45,6 +46,7 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
|
||||
response.data[0],
|
||||
{
|
||||
"id": 1,
|
||||
"externally_configured_variables": [],
|
||||
"output_type": None,
|
||||
"pages": None,
|
||||
"language": None,
|
||||
@@ -76,7 +78,7 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
|
||||
"remote_ocr_api_key": None,
|
||||
"remote_ocr_endpoint": None,
|
||||
"remote_ocr_mode": None,
|
||||
"ai_enabled": False,
|
||||
"ai_enabled": None,
|
||||
"llm_embedding_backend": None,
|
||||
"llm_embedding_model": None,
|
||||
"llm_embedding_endpoint": None,
|
||||
@@ -91,6 +93,31 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
|
||||
},
|
||||
)
|
||||
|
||||
def test_api_get_config_reports_external_configuration_without_values(self) -> None:
|
||||
with patch.dict(
|
||||
"os.environ",
|
||||
{
|
||||
"PAPERLESS_OCR_LANGUAGE": "eng",
|
||||
"PAPERLESS_REMOTE_OCR_API_KEY": "secret-value",
|
||||
"PAPERLESS_FUTURE_SETTING": "future-value",
|
||||
"UNRELATED_SETTING": "unrelated-value",
|
||||
},
|
||||
clear=True,
|
||||
):
|
||||
response = self.client.get(self.ENDPOINT, format="json")
|
||||
|
||||
self.assertCountEqual(
|
||||
response.data[0]["externally_configured_variables"],
|
||||
[
|
||||
"PAPERLESS_FUTURE_SETTING",
|
||||
"PAPERLESS_OCR_LANGUAGE",
|
||||
"PAPERLESS_REMOTE_OCR_API_KEY",
|
||||
],
|
||||
)
|
||||
self.assertNotContains(response, "secret-value")
|
||||
self.assertNotContains(response, "future-value")
|
||||
self.assertNotContains(response, "UNRELATED_SETTING")
|
||||
|
||||
def test_api_get_ui_settings_with_config(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
@@ -949,6 +976,26 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
|
||||
)
|
||||
mock_update.assert_called_once()
|
||||
|
||||
@override_settings(AI_ENABLED=True, LLM_EMBEDDING_BACKEND=None)
|
||||
def test_external_ai_setting_triggers_index_update(self) -> None:
|
||||
config = ApplicationConfiguration.objects.first()
|
||||
assert config is not None
|
||||
config.ai_enabled = None
|
||||
config.llm_embedding_backend = None
|
||||
config.save()
|
||||
|
||||
with (
|
||||
patch("documents.tasks.llmindex_index.apply_async") as mock_update,
|
||||
patch("paperless.views.llm_index_exists", return_value=False),
|
||||
):
|
||||
self.client.patch(
|
||||
f"{self.ENDPOINT}1/",
|
||||
json.dumps({"llm_embedding_backend": "openai-like"}),
|
||||
content_type="application/json",
|
||||
)
|
||||
|
||||
mock_update.assert_called_once()
|
||||
|
||||
def test_update_llm_embedding_chunk_size_triggers_rebuild(self) -> None:
|
||||
config = ApplicationConfiguration.objects.first()
|
||||
assert config is not None
|
||||
|
||||
@@ -4,6 +4,7 @@ import json
|
||||
import shutil
|
||||
import zipfile
|
||||
|
||||
from django.contrib.auth.models import Permission
|
||||
from django.contrib.auth.models import User
|
||||
from django.test import override_settings
|
||||
from django.utils import timezone
|
||||
@@ -326,6 +327,9 @@ class TestBulkDownload(DirectoriesMixin, SampleDirMixin, APITestCase):
|
||||
|
||||
def test_download_insufficient_permissions(self) -> None:
|
||||
user = User.objects.create_user(username="temp_user")
|
||||
user.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
self.client.force_authenticate(user=user)
|
||||
|
||||
self.doc2.owner = self.user
|
||||
@@ -339,3 +343,29 @@ class TestBulkDownload(DirectoriesMixin, SampleDirMixin, APITestCase):
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
|
||||
self.assertEqual(response.content, b"Insufficient permissions")
|
||||
|
||||
def test_bad_search_query_returns_400(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Bulk download request selects documents via a saved-search
|
||||
query filter
|
||||
WHEN:
|
||||
- The query contains a malformed field value (an invalid date)
|
||||
THEN:
|
||||
- The response is a 400 naming the bad value, exactly like the
|
||||
search list endpoint, never a 500
|
||||
"""
|
||||
response = self.client.post(
|
||||
self.ENDPOINT,
|
||||
json.dumps(
|
||||
{
|
||||
"all": True,
|
||||
"filters": {"query": "added:notadate"},
|
||||
"content": "originals",
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
self.assertIn(b"notadate", response.content)
|
||||
|
||||
@@ -717,6 +717,44 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
|
||||
self.assertEqual(args[0], [self.doc2.id])
|
||||
self.assertEqual(kwargs["storage_path"], self.sp1.id)
|
||||
|
||||
@mock.patch("documents.serialisers.bulk_edit.set_storage_path")
|
||||
def test_api_bulk_edit_with_all_true_resolves_owned_duplicates(self, m) -> None:
|
||||
self.setup_mock(m, "set_storage_path")
|
||||
user = User.objects.create_user(username="duplicate-owner")
|
||||
user.user_permissions.add(
|
||||
Permission.objects.get(codename="change_document"),
|
||||
)
|
||||
first_duplicate = Document.objects.create(
|
||||
checksum="owned-duplicate",
|
||||
title="First duplicate",
|
||||
owner=user,
|
||||
)
|
||||
second_duplicate = Document.objects.create(
|
||||
checksum="owned-duplicate",
|
||||
title="Second duplicate",
|
||||
owner=user,
|
||||
)
|
||||
self.client.force_authenticate(user=user)
|
||||
|
||||
response = self.client.post(
|
||||
"/api/documents/bulk_edit/",
|
||||
json.dumps(
|
||||
{
|
||||
"all": True,
|
||||
"filters": {"has_duplicates": True},
|
||||
"method": "set_storage_path",
|
||||
"parameters": {"storage_path": self.sp1.id},
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
m.assert_called_once()
|
||||
args, kwargs = m.call_args
|
||||
self.assertCountEqual(args[0], [first_duplicate.id, second_duplicate.id])
|
||||
self.assertEqual(kwargs["storage_path"], self.sp1.id)
|
||||
|
||||
@mock.patch("documents.search.get_backend")
|
||||
@mock.patch("documents.serialisers.bulk_edit.set_storage_path")
|
||||
def test_api_bulk_edit_with_all_true_resolves_documents_from_search_filters(
|
||||
@@ -1046,6 +1084,8 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
|
||||
user1 = User.objects.create(username="user1")
|
||||
self.client.force_authenticate(user=user1)
|
||||
|
||||
assign_perm("view_document", user1, self.doc2)
|
||||
|
||||
response = self.client.post(
|
||||
"/api/documents/selection_data/",
|
||||
json.dumps({"documents": [self.doc2.id]}),
|
||||
@@ -1053,7 +1093,18 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
|
||||
)
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
|
||||
self.assertEqual(response.content, b"Insufficient permissions")
|
||||
|
||||
user1.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
user1 = User.objects.get(pk=user1.pk)
|
||||
self.client.force_authenticate(user=user1)
|
||||
response = self.client.post(
|
||||
"/api/documents/selection_data/",
|
||||
json.dumps({"documents": [self.doc2.id]}),
|
||||
content_type="application/json",
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
|
||||
@mock.patch("documents.serialisers.bulk_edit.set_permissions")
|
||||
def test_set_permissions(self, m) -> None:
|
||||
@@ -1598,6 +1649,40 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
|
||||
def test_legacy_bulk_edit_rejects_out_of_bounds_pdf_doc_index(self) -> None:
|
||||
response = self.client.post(
|
||||
"/api/documents/bulk_edit/",
|
||||
json.dumps(
|
||||
{
|
||||
"documents": [self.doc2.id],
|
||||
"method": "edit_pdf",
|
||||
"parameters": {
|
||||
"operations": [{"page": 1, "doc": 2**32}],
|
||||
},
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
self.assertIn(b"doc index is out of bounds", response.content)
|
||||
|
||||
def test_legacy_bulk_edit_rejects_empty_pdf_operations(self) -> None:
|
||||
response = self.client.post(
|
||||
"/api/documents/bulk_edit/",
|
||||
json.dumps(
|
||||
{
|
||||
"documents": [self.doc2.id],
|
||||
"method": "edit_pdf",
|
||||
"parameters": {"operations": []},
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
self.assertIn(b"operations must not be empty", response.content)
|
||||
|
||||
@mock.patch("documents.views.bulk_edit.edit_pdf")
|
||||
def test_edit_pdf(self, m) -> None:
|
||||
self.setup_mock(m, "edit_pdf")
|
||||
@@ -1648,6 +1733,13 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
self.assertIn(b"Expected a list of items", response.content)
|
||||
|
||||
response = self.client.post(
|
||||
"/api/documents/edit_pdf/",
|
||||
{"documents": [self.doc2.id], "operations": []},
|
||||
format="json",
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
|
||||
response = self.client.post(
|
||||
"/api/documents/edit_pdf/",
|
||||
json.dumps(
|
||||
@@ -1700,6 +1792,21 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
self.assertIn(b"doc must be an integer", response.content)
|
||||
|
||||
for doc_index in (-1, 2**32):
|
||||
with self.subTest(doc_index=doc_index):
|
||||
response = self.client.post(
|
||||
"/api/documents/edit_pdf/",
|
||||
json.dumps(
|
||||
{
|
||||
"documents": [self.doc2.id],
|
||||
"operations": [{"page": 1, "doc": doc_index}],
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
self.assertIn(b"doc index is out of bounds", response.content)
|
||||
|
||||
response = self.client.post(
|
||||
"/api/documents/edit_pdf/",
|
||||
json.dumps(
|
||||
@@ -2021,3 +2128,30 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
self.assertEqual(LogEntry.objects.filter(object_pk=self.doc1.id).count(), 2)
|
||||
|
||||
def test_api_bulk_edit_with_bad_search_query_returns_400(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Bulk edit request selects documents via a saved-search query
|
||||
filter
|
||||
WHEN:
|
||||
- The query contains a malformed field value (an invalid date)
|
||||
THEN:
|
||||
- The response is a 400 naming the bad value, exactly like the
|
||||
search list endpoint, never a 500
|
||||
"""
|
||||
response = self.client.post(
|
||||
"/api/documents/bulk_edit/",
|
||||
json.dumps(
|
||||
{
|
||||
"all": True,
|
||||
"filters": {"query": "added:notadate"},
|
||||
"method": "set_storage_path",
|
||||
"parameters": {"storage_path": self.sp1.id},
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
self.assertIn(b"notadate", response.content)
|
||||
|
||||
@@ -38,6 +38,42 @@ class TestChatStreamingViewInputValidation(APITestCase):
|
||||
)
|
||||
assert resp.status_code == status.HTTP_400_BAD_REQUEST
|
||||
|
||||
def test_answer_is_not_compressed(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A client that accepts compressed responses
|
||||
WHEN:
|
||||
- It asks the chat endpoint a question
|
||||
THEN:
|
||||
- The answer is streamed unencoded, chunk for chunk
|
||||
|
||||
The stream compressors buffer, so a compressed answer arrives in one
|
||||
piece. The view cannot opt out by flagging the request: DRF's request
|
||||
wrapper proxies reads but keeps writes to itself, so the flag never
|
||||
reaches the Django request the middleware sees.
|
||||
"""
|
||||
chunks = [f"token{i} " for i in range(40)]
|
||||
with (
|
||||
mock.patch(
|
||||
"documents.views.AIConfig",
|
||||
return_value=self._mock_ai_enabled(),
|
||||
),
|
||||
mock.patch(
|
||||
"documents.views.stream_chat_with_documents",
|
||||
return_value=iter(chunks),
|
||||
),
|
||||
):
|
||||
resp = self.client.post(
|
||||
"/api/documents/chat/",
|
||||
{"q": "What is in my archive?"},
|
||||
format="json",
|
||||
HTTP_ACCEPT_ENCODING="gzip, deflate, br, zstd",
|
||||
)
|
||||
|
||||
assert resp.status_code == status.HTTP_200_OK
|
||||
assert not resp.has_header("Content-Encoding")
|
||||
assert list(resp.streaming_content) == [c.encode() for c in chunks]
|
||||
|
||||
def test_missing_question_is_rejected(self) -> None:
|
||||
with mock.patch(
|
||||
"documents.views.AIConfig",
|
||||
|
||||
@@ -2,14 +2,12 @@ from __future__ import annotations
|
||||
|
||||
import datetime
|
||||
from typing import TYPE_CHECKING
|
||||
from unittest import TestCase
|
||||
from unittest import mock
|
||||
|
||||
from auditlog.models import LogEntry # type: ignore[import-untyped]
|
||||
from django.contrib.auth.models import Permission
|
||||
from django.contrib.auth.models import User
|
||||
from django.contrib.contenttypes.models import ContentType
|
||||
from django.core.exceptions import FieldError
|
||||
from django.core.files.uploadedfile import SimpleUploadedFile
|
||||
from django.test import TestCase as DjangoTestCase
|
||||
from django.utils import timezone
|
||||
@@ -22,6 +20,7 @@ from documents.filters import TitleContentFilter
|
||||
from documents.models import Document
|
||||
from documents.tests.utils import DirectoriesMixin
|
||||
from documents.tests.utils import read_streaming_response
|
||||
from documents.versioning import annotate_effective_content
|
||||
from documents.views import DocumentSelectionMixin
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -598,6 +597,7 @@ class TestDocumentVersioningApi(DirectoriesMixin, APITestCase):
|
||||
self.assertEqual(input_doc.root_document_id, root.id)
|
||||
self.assertEqual(input_doc.source, DocumentSource.ApiUpload)
|
||||
self.assertEqual(overrides.version_label, "New Version")
|
||||
self.assertEqual(overrides.owner_id, self.user.id)
|
||||
self.assertEqual(overrides.actor_id, self.user.id)
|
||||
|
||||
def test_update_version_with_version_pk_normalizes_to_root(self) -> None:
|
||||
@@ -891,32 +891,104 @@ class TestDocumentVersioningApi(DirectoriesMixin, APITestCase):
|
||||
)
|
||||
|
||||
|
||||
class TestVersionAwareFilters(TestCase):
|
||||
def test_title_content_filter_falls_back_to_content(self) -> None:
|
||||
queryset = mock.Mock()
|
||||
fallback_queryset = mock.Mock()
|
||||
queryset.filter.side_effect = [FieldError("missing field"), fallback_queryset]
|
||||
class TestVersionAwareFilters(DjangoTestCase):
|
||||
"""
|
||||
The filters annotate effective_content themselves rather than relying on
|
||||
the caller's queryset carrying it, so they stay version-aware on a plain
|
||||
Document queryset (e.g. the bulk-edit "select all matching" path).
|
||||
"""
|
||||
|
||||
result = TitleContentFilter().filter(queryset, " latest ")
|
||||
def setUp(self) -> None:
|
||||
super().setUp()
|
||||
self.root = Document.objects.create(
|
||||
title="root",
|
||||
checksum="root",
|
||||
mime_type="application/pdf",
|
||||
content="superseded-content",
|
||||
)
|
||||
Document.objects.create(
|
||||
title="version",
|
||||
checksum="version",
|
||||
mime_type="application/pdf",
|
||||
root_document=self.root,
|
||||
version_index=1,
|
||||
content="latest-content",
|
||||
)
|
||||
self.unversioned = Document.objects.create(
|
||||
title="unversioned",
|
||||
checksum="unversioned",
|
||||
mime_type="application/pdf",
|
||||
content="latest-content",
|
||||
)
|
||||
|
||||
self.assertIs(result, fallback_queryset)
|
||||
self.assertEqual(queryset.filter.call_count, 2)
|
||||
|
||||
def test_effective_content_filter_falls_back_to_content_lookup(self) -> None:
|
||||
queryset = mock.Mock()
|
||||
fallback_queryset = mock.Mock()
|
||||
queryset.filter.side_effect = [FieldError("missing field"), fallback_queryset]
|
||||
|
||||
result = EffectiveContentFilter(lookup_expr="icontains").filter(
|
||||
queryset,
|
||||
def test_title_content_filter_matches_latest_version_content(self) -> None:
|
||||
result = TitleContentFilter().filter(
|
||||
Document.objects.filter(root_document__isnull=True),
|
||||
" latest ",
|
||||
)
|
||||
|
||||
self.assertIs(result, fallback_queryset)
|
||||
first_kwargs = queryset.filter.call_args_list[0].kwargs
|
||||
second_kwargs = queryset.filter.call_args_list[1].kwargs
|
||||
self.assertEqual(first_kwargs, {"effective_content__icontains": "latest"})
|
||||
self.assertEqual(second_kwargs, {"content__icontains": "latest"})
|
||||
self.assertCountEqual(
|
||||
[doc.id for doc in result],
|
||||
[self.root.id, self.unversioned.id],
|
||||
)
|
||||
|
||||
def test_effective_content_filter_matches_latest_version_content(self) -> None:
|
||||
result = EffectiveContentFilter(lookup_expr="icontains").filter(
|
||||
Document.objects.filter(root_document__isnull=True),
|
||||
" latest ",
|
||||
)
|
||||
|
||||
self.assertCountEqual(
|
||||
[doc.id for doc in result],
|
||||
[self.root.id, self.unversioned.id],
|
||||
)
|
||||
|
||||
def test_effective_content_filter_ignores_superseded_content(self) -> None:
|
||||
result = EffectiveContentFilter(lookup_expr="icontains").filter(
|
||||
Document.objects.filter(root_document__isnull=True),
|
||||
"superseded",
|
||||
)
|
||||
|
||||
self.assertEqual(list(result), [])
|
||||
|
||||
def test_filters_reuse_an_existing_annotation(self) -> None:
|
||||
"""
|
||||
Annotating twice under the same alias is an error, so an already
|
||||
annotated queryset (the search path) has to be left alone.
|
||||
"""
|
||||
annotated = annotate_effective_content(
|
||||
Document.objects.filter(root_document__isnull=True),
|
||||
)
|
||||
self.assertIs(annotate_effective_content(annotated), annotated)
|
||||
|
||||
result = EffectiveContentFilter(lookup_expr="icontains").filter(
|
||||
annotated,
|
||||
"latest",
|
||||
)
|
||||
|
||||
self.assertCountEqual(
|
||||
[doc.id for doc in result],
|
||||
[self.root.id, self.unversioned.id],
|
||||
)
|
||||
|
||||
def test_bulk_selection_does_not_match_superseded_content(self) -> None:
|
||||
"""
|
||||
Bulk edit's "select all matching" builds its own queryset, so before
|
||||
the filters annotated for themselves it matched the root document's
|
||||
superseded content -- selecting documents the list view, filtered by
|
||||
the same term, does not show.
|
||||
"""
|
||||
user = User.objects.create_superuser(username="bulk_selection")
|
||||
|
||||
selected = DocumentSelectionMixin()._resolve_document_ids(
|
||||
user=user,
|
||||
validated_data={
|
||||
"all": True,
|
||||
"filters": {"content__icontains": "superseded"},
|
||||
},
|
||||
)
|
||||
|
||||
self.assertEqual(selected, [])
|
||||
|
||||
def test_effective_content_filter_returns_input_for_empty_values(self) -> None:
|
||||
queryset = mock.Mock()
|
||||
|
||||
@@ -981,6 +981,128 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
self.assertEqual(len(results), 1)
|
||||
self.assertEqual(results[0]["id"], doc.id)
|
||||
|
||||
def test_has_duplicates_filter(self) -> None:
|
||||
original_match = Document.objects.create(
|
||||
title="original match",
|
||||
checksum="same-original",
|
||||
)
|
||||
second_original_match = Document.objects.create(
|
||||
title="second original match",
|
||||
checksum="same-original",
|
||||
)
|
||||
archive_match = Document.objects.create(
|
||||
title="archive match",
|
||||
checksum="archive-source",
|
||||
archive_checksum="same-archive",
|
||||
)
|
||||
original_to_archive_match = Document.objects.create(
|
||||
title="original to archive match",
|
||||
checksum="same-archive",
|
||||
)
|
||||
first_archive_match = Document.objects.create(
|
||||
title="first archive match",
|
||||
checksum="first-archive-source",
|
||||
archive_checksum="same-archive-only",
|
||||
)
|
||||
second_archive_match = Document.objects.create(
|
||||
title="second archive match",
|
||||
checksum="second-archive-source",
|
||||
archive_checksum="same-archive-only",
|
||||
)
|
||||
first_empty_archive = Document.objects.create(
|
||||
title="first empty archive",
|
||||
checksum="first-empty-archive",
|
||||
archive_checksum="",
|
||||
)
|
||||
second_empty_archive = Document.objects.create(
|
||||
title="second empty archive",
|
||||
checksum="second-empty-archive",
|
||||
archive_checksum="",
|
||||
)
|
||||
unique = Document.objects.create(title="unique", checksum="unique")
|
||||
version_root = Document.objects.create(
|
||||
title="version root",
|
||||
checksum="version-root",
|
||||
)
|
||||
Document.objects.create(
|
||||
title="version",
|
||||
checksum=unique.checksum,
|
||||
root_document=version_root,
|
||||
version_index=1,
|
||||
)
|
||||
trash_match = Document.objects.create(
|
||||
title="trash match",
|
||||
checksum="trash-match",
|
||||
)
|
||||
trashed_duplicate = Document.objects.create(
|
||||
title="trashed duplicate",
|
||||
checksum="trash-match",
|
||||
)
|
||||
trashed_duplicate.delete()
|
||||
|
||||
response = self.client.get("/api/documents/?has_duplicates=true")
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
self.assertCountEqual(
|
||||
[document["id"] for document in response.data["results"]],
|
||||
[
|
||||
original_match.id,
|
||||
second_original_match.id,
|
||||
archive_match.id,
|
||||
original_to_archive_match.id,
|
||||
first_archive_match.id,
|
||||
second_archive_match.id,
|
||||
trash_match.id,
|
||||
],
|
||||
)
|
||||
|
||||
response = self.client.get("/api/documents/?has_duplicates=false")
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
self.assertCountEqual(
|
||||
[document["id"] for document in response.data["results"]],
|
||||
[
|
||||
unique.id,
|
||||
version_root.id,
|
||||
first_empty_archive.id,
|
||||
second_empty_archive.id,
|
||||
],
|
||||
)
|
||||
|
||||
response = self.client.get(f"/api/documents/{first_empty_archive.id}/")
|
||||
self.assertEqual(response.data["duplicate_documents"], [])
|
||||
|
||||
def test_has_duplicates_filter_respects_document_permissions(self) -> None:
|
||||
owner = User.objects.create_user(username="duplicate-owner")
|
||||
requester = User.objects.create_user(username="duplicate-requester")
|
||||
requester.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
visible_document = Document.objects.create(
|
||||
title="visible document",
|
||||
checksum="permission-match",
|
||||
owner=requester,
|
||||
)
|
||||
hidden_duplicate = Document.objects.create(
|
||||
title="hidden duplicate",
|
||||
checksum="permission-match",
|
||||
owner=owner,
|
||||
)
|
||||
self.client.force_authenticate(user=requester)
|
||||
|
||||
response = self.client.get("/api/documents/?has_duplicates=true")
|
||||
self.assertNotIn(
|
||||
visible_document.id,
|
||||
[document["id"] for document in response.data["results"]],
|
||||
)
|
||||
|
||||
assign_perm("view_document", requester, hidden_duplicate)
|
||||
response = self.client.get("/api/documents/?has_duplicates=true")
|
||||
self.assertIn(
|
||||
visible_document.id,
|
||||
[document["id"] for document in response.data["results"]],
|
||||
)
|
||||
|
||||
def test_custom_fields_icontains_filter_no_duplicates(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
@@ -3493,6 +3615,55 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
self.assertEqual(response.content, b"Insufficient permissions to delete notes")
|
||||
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
|
||||
|
||||
def test_notes_require_global_document_permissions(self) -> None:
|
||||
user = User.objects.create_user(username="note_editor")
|
||||
user.user_permissions.add(
|
||||
*Permission.objects.filter(
|
||||
codename__in=["view_note", "add_note", "delete_note"],
|
||||
),
|
||||
)
|
||||
doc = Document.objects.create(
|
||||
title="test",
|
||||
mime_type="application/pdf",
|
||||
content="notes",
|
||||
owner=user,
|
||||
)
|
||||
note = Note.objects.create(note="Existing", document=doc, user=user)
|
||||
self.client.force_authenticate(user)
|
||||
|
||||
response = self.client.get(f"/api/documents/{doc.pk}/notes/")
|
||||
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
|
||||
|
||||
user.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
user = User.objects.get(pk=user.pk)
|
||||
self.client.force_authenticate(user)
|
||||
response = self.client.get(f"/api/documents/{doc.pk}/notes/")
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
|
||||
response = self.client.post(
|
||||
f"/api/documents/{doc.pk}/notes/",
|
||||
data={"note": "New"},
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
|
||||
|
||||
user.user_permissions.add(
|
||||
Permission.objects.get(codename="change_document"),
|
||||
)
|
||||
user = User.objects.get(pk=user.pk)
|
||||
self.client.force_authenticate(user)
|
||||
response = self.client.post(
|
||||
f"/api/documents/{doc.pk}/notes/",
|
||||
data={"note": "New"},
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
|
||||
response = self.client.delete(
|
||||
f"/api/documents/{doc.pk}/notes/?id={note.pk}",
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
|
||||
def test_delete_note(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
@@ -3891,6 +4062,21 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
|
||||
assign_perm("view_document", user1, doc)
|
||||
|
||||
create_resp = self.client.post(
|
||||
"/api/share_links/",
|
||||
data={
|
||||
"document": doc.pk,
|
||||
"file_version": "original",
|
||||
},
|
||||
format="json",
|
||||
)
|
||||
self.assertEqual(create_resp.status_code, status.HTTP_403_FORBIDDEN)
|
||||
|
||||
user1.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
user1 = User.objects.get(pk=user1.pk)
|
||||
self.client.force_authenticate(user1)
|
||||
create_resp = self.client.post(
|
||||
"/api/share_links/",
|
||||
data={
|
||||
|
||||
@@ -2,10 +2,15 @@ import datetime
|
||||
import json
|
||||
from unittest import mock
|
||||
|
||||
from django.contrib.auth.models import Group
|
||||
from django.contrib.auth.models import Permission
|
||||
from django.contrib.auth.models import User
|
||||
from django.db import connection
|
||||
from django.test import override_settings
|
||||
from django.test.utils import CaptureQueriesContext
|
||||
from guardian.shortcuts import assign_perm
|
||||
from guardian.shortcuts import get_groups_with_perms
|
||||
from guardian.shortcuts import get_users_with_perms
|
||||
from rest_framework import status
|
||||
from rest_framework.test import APITestCase
|
||||
|
||||
@@ -452,6 +457,9 @@ class TestApiStoragePaths(DirectoriesMixin, APITestCase):
|
||||
def test_test_storage_path_requires_document_view_permission(self) -> None:
|
||||
owner = User.objects.create_user(username="owner")
|
||||
unprivileged = User.objects.create_user(username="unprivileged")
|
||||
unprivileged.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
document = Document.objects.create(
|
||||
mime_type="application/pdf",
|
||||
owner=owner,
|
||||
@@ -483,6 +491,23 @@ class TestApiStoragePaths(DirectoriesMixin, APITestCase):
|
||||
)
|
||||
assign_perm("view_document", viewer, document)
|
||||
|
||||
self.client.force_authenticate(user=viewer)
|
||||
response = self.client.post(
|
||||
f"{self.ENDPOINT}test/",
|
||||
json.dumps(
|
||||
{
|
||||
"document": document.id,
|
||||
"path": "path/{{ title }}",
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
|
||||
|
||||
viewer.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
viewer = User.objects.get(pk=viewer.pk)
|
||||
self.client.force_authenticate(user=viewer)
|
||||
response = self.client.post(
|
||||
f"{self.ENDPOINT}test/",
|
||||
@@ -525,6 +550,9 @@ class TestApiStoragePaths(DirectoriesMixin, APITestCase):
|
||||
password="password",
|
||||
email="owner@example.com",
|
||||
)
|
||||
owner.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
document = Document.objects.create(
|
||||
mime_type="application/pdf",
|
||||
owner=owner,
|
||||
@@ -600,6 +628,9 @@ class TestApiStoragePaths(DirectoriesMixin, APITestCase):
|
||||
checksum="123",
|
||||
)
|
||||
assign_perm("view_document", viewer, document)
|
||||
viewer.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
|
||||
self.client.force_authenticate(user=viewer)
|
||||
response = self.client.post(
|
||||
@@ -687,6 +718,9 @@ class TestApiStoragePaths(DirectoriesMixin, APITestCase):
|
||||
)
|
||||
document.tags.add(private_tag)
|
||||
assign_perm("view_document", viewer, document)
|
||||
viewer.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
|
||||
self.client.force_authenticate(user=viewer)
|
||||
response = self.client.post(
|
||||
@@ -740,6 +774,9 @@ class TestApiStoragePaths(DirectoriesMixin, APITestCase):
|
||||
value_int=42,
|
||||
)
|
||||
assign_perm("view_document", viewer, document)
|
||||
viewer.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
|
||||
self.client.force_authenticate(user=viewer)
|
||||
response = self.client.post(
|
||||
@@ -842,6 +879,66 @@ class TestBulkEditObjects(APITestCase):
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
self.assertEqual(StoragePath.objects.count(), 0)
|
||||
|
||||
def test_bulk_objects_set_permissions_batched_across_object_count(
|
||||
self,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Many tags are being bulk-edited to set permissions at once
|
||||
WHEN:
|
||||
- bulk_edit_objects API endpoint is called with set_permissions
|
||||
operation over a small batch vs. a much larger one
|
||||
THEN:
|
||||
- Permissions are applied correctly at both scales
|
||||
- Query count does not grow with the number of tags, i.e. each
|
||||
user/group is applied across all tags with one batched call
|
||||
rather than one call per (tag, identity) pair
|
||||
"""
|
||||
group1 = Group.objects.create(name="perm-group")
|
||||
permissions = {
|
||||
"view": {"users": [self.user1.id, self.user2.id], "groups": [group1.id]},
|
||||
"change": {"users": [self.user1.id], "groups": [group1.id]},
|
||||
}
|
||||
|
||||
def run_with_n_tags(n: int) -> int:
|
||||
tags = [Tag.objects.create(name=f"perm-tag-{n}-{i}") for i in range(n)]
|
||||
with CaptureQueriesContext(connection) as ctx:
|
||||
response = self.client.post(
|
||||
"/api/bulk_edit_objects/",
|
||||
json.dumps(
|
||||
{
|
||||
"objects": [t.id for t in tags],
|
||||
"object_type": "tags",
|
||||
"operation": "set_permissions",
|
||||
"permissions": permissions,
|
||||
"merge": False,
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
for tag in tags:
|
||||
self.assertEqual(get_users_with_perms(tag).count(), 2)
|
||||
self.assertEqual(get_groups_with_perms(tag).count(), 1)
|
||||
return len(ctx.captured_queries)
|
||||
|
||||
small_batch_queries = run_with_n_tags(5)
|
||||
large_batch_queries = run_with_n_tags(50)
|
||||
|
||||
# A tolerance rather than equality, matching the N+1 check in
|
||||
# test_views.py: bulk_create's batch_size caps rows per INSERT, so a
|
||||
# large enough selection does legitimately add statements, and the
|
||||
# per-process ContentType cache makes the first run carry an extra
|
||||
# query. Neither can hide a regression to per-object assignment,
|
||||
# which would be ~10x the small-batch count here.
|
||||
self.assertLessEqual(
|
||||
large_batch_queries,
|
||||
small_batch_queries + 5,
|
||||
"Permission assignment appears to scale with object count: "
|
||||
f"{small_batch_queries} queries for 5 tags vs. "
|
||||
f"{large_batch_queries} for 50",
|
||||
)
|
||||
|
||||
def test_bulk_objects_delete_all_filtered(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
|
||||
@@ -786,6 +786,10 @@ class TestDocumentSearchApi(DirectoriesMixin, APITestCase):
|
||||
tick=False,
|
||||
):
|
||||
response = self.client.get("/api/documents/?query=added:previous month")
|
||||
assert response.status_code == 200, (
|
||||
f"expected a successful search response, got {response.status_code}: "
|
||||
f"{response.data!r}"
|
||||
)
|
||||
results = response.data["results"]
|
||||
|
||||
self.assertEqual(len(results), 1)
|
||||
@@ -818,6 +822,26 @@ class TestDocumentSearchApi(DirectoriesMixin, APITestCase):
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
self.assertIn("invalid-date", str(response.data["query"]))
|
||||
|
||||
def test_search_multiple_bad_fields_returns_all_messages(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- One document added
|
||||
WHEN:
|
||||
- Query with multiple bad fields (e.g. invalid date and invalid number)
|
||||
THEN:
|
||||
- 400 Bad Request with error messages for every bad field,
|
||||
so the user can fix them all in one round-trip
|
||||
"""
|
||||
response = self.client.get(
|
||||
"/api/documents/",
|
||||
{"query": "created:notadate AND asn:notanumber"},
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
messages = response.data["query"]
|
||||
self.assertEqual(len(messages), 2)
|
||||
self.assertTrue(any("created" in m for m in messages))
|
||||
self.assertTrue(any("asn" in m for m in messages))
|
||||
|
||||
@override_settings(
|
||||
TIME_ZONE="UTC",
|
||||
)
|
||||
@@ -861,6 +885,29 @@ class TestDocumentSearchApi(DirectoriesMixin, APITestCase):
|
||||
results = response.data["results"]
|
||||
self.assertEqual({r["id"] for r in results}, {1, 2})
|
||||
|
||||
@mock.patch("documents.search._backend.parse_user_query")
|
||||
def test_search_parser_bug_surfaces_as_500_not_400(self, m) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The query parser itself fails (a whoosh-compat bug, per
|
||||
QueryParserError's own contract: not user-fixable input)
|
||||
WHEN:
|
||||
- Any search request runs
|
||||
THEN:
|
||||
- The error surfaces as a 500 monitoring can see, never a 400
|
||||
blaming the user for a library defect
|
||||
"""
|
||||
from whoosh_compat.errors import QueryParserError
|
||||
|
||||
m.side_effect = QueryParserError("synthetic parser bug")
|
||||
|
||||
self.client.raise_request_exception = False
|
||||
response = self.client.get("/api/documents/?query=anything")
|
||||
self.assertEqual(
|
||||
response.status_code,
|
||||
status.HTTP_500_INTERNAL_SERVER_ERROR,
|
||||
)
|
||||
|
||||
@mock.patch("documents.search._backend.TantivyBackend.autocomplete")
|
||||
def test_search_autocomplete_limits(self, m) -> None:
|
||||
"""
|
||||
@@ -1947,6 +1994,29 @@ class TestDocumentSearchApi(DirectoriesMixin, APITestCase):
|
||||
self.assertEqual(len(response.data["documents"]), 1)
|
||||
self.assertEqual(response.data["documents"][0]["id"], title_match.id)
|
||||
|
||||
def test_global_search_returns_latest_version_content(self) -> None:
|
||||
root = Document.objects.create(
|
||||
title="bank statement",
|
||||
content="superseded content",
|
||||
checksum="GSV1",
|
||||
pk=23,
|
||||
)
|
||||
Document.objects.create(
|
||||
title="bank statement v2",
|
||||
content="latest content",
|
||||
checksum="GSV2",
|
||||
pk=24,
|
||||
root_document=root,
|
||||
version_index=1,
|
||||
)
|
||||
|
||||
self.client.force_authenticate(self.user)
|
||||
|
||||
response = self.client.get("/api/search/?query=bank&db_only=true")
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
returned = {doc["id"]: doc["content"] for doc in response.data["documents"]}
|
||||
self.assertEqual(returned.get(root.id), "latest content")
|
||||
|
||||
def test_global_search_filters_owned_mail_objects(self) -> None:
|
||||
user1 = User.objects.create_user("mail-search-user")
|
||||
user2 = User.objects.create_user("other-mail-search-user")
|
||||
@@ -2035,3 +2105,77 @@ class TestDocumentSearchApi(DirectoriesMixin, APITestCase):
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
response = self.client.get("/api/search/?query=no")
|
||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
|
||||
def _assert_query_finds(self, doc: Document, query: str) -> None:
|
||||
get_backend().add_or_update(doc)
|
||||
response = self.client.get("/api/documents/", {"query": query})
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
ids = [r["id"] for r in response.data["results"]]
|
||||
self.assertIn(doc.id, ids)
|
||||
|
||||
def test_search_by_asn(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with an archive serial number, indexed
|
||||
WHEN:
|
||||
- A query filters by "asn:<value>"
|
||||
THEN:
|
||||
- The document is found
|
||||
"""
|
||||
doc = Document.objects.create(
|
||||
title="Has ASN",
|
||||
content="content",
|
||||
checksum="asn-checksum",
|
||||
archive_serial_number=555,
|
||||
)
|
||||
self._assert_query_finds(doc, "asn:555")
|
||||
|
||||
def test_search_by_page_count(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with a page count, indexed
|
||||
WHEN:
|
||||
- A query filters by "page_count:<value>"
|
||||
THEN:
|
||||
- The document is found
|
||||
"""
|
||||
doc = Document.objects.create(
|
||||
title="Multi-page",
|
||||
content="content",
|
||||
checksum="page-count-checksum",
|
||||
page_count=42,
|
||||
)
|
||||
self._assert_query_finds(doc, "page_count:42")
|
||||
|
||||
def test_search_by_original_filename(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with an original filename, indexed
|
||||
WHEN:
|
||||
- A query filters by "original_filename:<value>"
|
||||
THEN:
|
||||
- The document is found
|
||||
"""
|
||||
doc = Document.objects.create(
|
||||
title="Named file",
|
||||
content="content",
|
||||
checksum="filename-checksum",
|
||||
original_filename="quarterly-report.pdf",
|
||||
)
|
||||
self._assert_query_finds(doc, "original_filename:quarterly-report.pdf")
|
||||
|
||||
def test_search_by_checksum(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document with a checksum, indexed
|
||||
WHEN:
|
||||
- A query filters by "checksum:<value>"
|
||||
THEN:
|
||||
- The document is found
|
||||
"""
|
||||
doc = Document.objects.create(
|
||||
title="Checksum doc",
|
||||
content="content",
|
||||
checksum="deadbeef1234",
|
||||
)
|
||||
self._assert_query_finds(doc, "checksum:deadbeef1234")
|
||||
|
||||
@@ -0,0 +1,344 @@
|
||||
"""The search list endpoint's exception handling: what becomes a 400 and
|
||||
what a library defect surfaces as instead.
|
||||
|
||||
Companion to documents/tests/search/test_error_routing.py, which pins the
|
||||
Cause -> SearchQueryError/QueryError routing inside documents/search/_query.py.
|
||||
These tests pin the layer above it: DocumentViewSet.list's own except clauses,
|
||||
which decide what an already-routed error becomes on the wire.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
from rest_framework import status
|
||||
from whoosh_compat.errors import Cause
|
||||
from whoosh_compat.errors import Diagnostic
|
||||
from whoosh_compat.errors import DiagnosticKind
|
||||
from whoosh_compat.errors import QueryError
|
||||
|
||||
from documents.search import SearchQueryError
|
||||
from documents.tests.factories import DocumentFactory
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from rest_framework.test import APIClient
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
pytestmark = [pytest.mark.django_db, pytest.mark.usefixtures("_search_index")]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def indexed_document() -> Document:
|
||||
from documents.search import get_backend
|
||||
|
||||
doc = DocumentFactory.create(title="quarterly invoice", content="acme corp")
|
||||
get_backend().add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestSearchQueryErrorStillBecomesA400:
|
||||
def test_search_query_error_becomes_a_400_naming_the_field(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- parse_user_query() raising a SearchQueryError naming a field
|
||||
WHEN:
|
||||
- The document list endpoint is queried
|
||||
THEN:
|
||||
- The response is a 400 whose body names the field
|
||||
"""
|
||||
import documents.search._backend as backend_mod
|
||||
|
||||
def raise_search_query_error(*args: object, **kwargs: object) -> object:
|
||||
raise SearchQueryError("bad value for field 'added'")
|
||||
|
||||
monkeypatch.setattr(
|
||||
backend_mod,
|
||||
"parse_user_query",
|
||||
raise_search_query_error,
|
||||
)
|
||||
|
||||
response = admin_client.get("/api/documents/?query=anything")
|
||||
|
||||
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
||||
assert "added" in str(response.data["query"])
|
||||
|
||||
|
||||
class TestLibraryDefectsPropagate:
|
||||
"""The exact regression this task exists to fix: an unexpected or
|
||||
INTERNAL-cause library error must not be relabeled a 400."""
|
||||
|
||||
def test_unexpected_exception_is_not_converted_to_a_400(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- parse_user_query() raising an unrelated exception
|
||||
(ZeroDivisionError), not a SearchQueryError
|
||||
WHEN:
|
||||
- The document list endpoint is queried
|
||||
THEN:
|
||||
- The exception propagates unconverted, rather than being
|
||||
relabeled a 400
|
||||
"""
|
||||
import documents.search._backend as backend_mod
|
||||
|
||||
def raise_zero_division(*args: object, **kwargs: object) -> object:
|
||||
raise ZeroDivisionError("synthetic bug, unrelated to search grammar")
|
||||
|
||||
monkeypatch.setattr(
|
||||
backend_mod,
|
||||
"parse_user_query",
|
||||
raise_zero_division,
|
||||
)
|
||||
|
||||
with pytest.raises(ZeroDivisionError):
|
||||
admin_client.get("/api/documents/?query=anything")
|
||||
|
||||
def test_internal_cause_query_error_is_not_converted_to_a_400(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A real query string running through the real parse and
|
||||
routing pipeline (pre-parse rewrites, wc.parse(), and
|
||||
_map_emit_error's own Cause routing all run for real), except
|
||||
the final emit call (tantivy_emit) is forced to report a
|
||||
library-internal defect (Cause.INTERNAL) - the one
|
||||
library-internal failure mode reachable from a real query
|
||||
WHEN:
|
||||
- The document list endpoint is queried
|
||||
THEN:
|
||||
- The QueryError propagates unconverted, rather than being
|
||||
relabeled a 400
|
||||
"""
|
||||
import documents.search._query as query_mod
|
||||
|
||||
def raise_internal(*args: object, **kwargs: object) -> object:
|
||||
raise QueryError(
|
||||
Diagnostic(
|
||||
kind=DiagnosticKind.BACKEND_REJECTED,
|
||||
cause=Cause.INTERNAL,
|
||||
message="synthetic whoosh-compat emitter defect",
|
||||
),
|
||||
)
|
||||
|
||||
monkeypatch.setattr(query_mod, "tantivy_emit", raise_internal)
|
||||
|
||||
with pytest.raises(QueryError):
|
||||
admin_client.get("/api/documents/?query=invoice")
|
||||
|
||||
|
||||
class TestSelectionPathsAgreeWithSearch:
|
||||
"""DocumentSelectionMixin backs bulk edit, bulk download, and a
|
||||
more_like_id selection filter. It catches only SearchQueryError -- the
|
||||
same contract the search list endpoint enforces above -- so all three
|
||||
must map SearchQueryError to a 400 and let anything else surface."""
|
||||
|
||||
def test_bulk_edit_maps_search_query_error_to_a_400(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- parse_user_query() raising a SearchQueryError naming a field
|
||||
WHEN:
|
||||
- The bulk_edit endpoint is called with a query filter
|
||||
THEN:
|
||||
- The response is a 400 whose body names the field
|
||||
"""
|
||||
import documents.search._backend as backend_mod
|
||||
|
||||
def raise_search_query_error(*args: object, **kwargs: object) -> object:
|
||||
raise SearchQueryError("bad value for field 'added'")
|
||||
|
||||
monkeypatch.setattr(
|
||||
backend_mod,
|
||||
"parse_user_query",
|
||||
raise_search_query_error,
|
||||
)
|
||||
|
||||
response = admin_client.post(
|
||||
"/api/documents/bulk_edit/",
|
||||
{
|
||||
"documents": [],
|
||||
"all": True,
|
||||
"filters": {"query": "anything"},
|
||||
"method": "set_document_type",
|
||||
"parameters": {"document_type": None},
|
||||
},
|
||||
format="json",
|
||||
)
|
||||
|
||||
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
||||
assert "added" in str(response.data["query"])
|
||||
|
||||
def test_bulk_edit_lets_an_unexpected_exception_surface(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- parse_user_query() raising an unrelated exception
|
||||
(ZeroDivisionError), not a SearchQueryError
|
||||
WHEN:
|
||||
- The bulk_edit endpoint is called with a query filter
|
||||
THEN:
|
||||
- The exception propagates unconverted, rather than being
|
||||
relabeled a 400
|
||||
"""
|
||||
import documents.search._backend as backend_mod
|
||||
|
||||
def raise_zero_division(*args: object, **kwargs: object) -> object:
|
||||
raise ZeroDivisionError("synthetic bug, unrelated to search grammar")
|
||||
|
||||
monkeypatch.setattr(
|
||||
backend_mod,
|
||||
"parse_user_query",
|
||||
raise_zero_division,
|
||||
)
|
||||
|
||||
with pytest.raises(ZeroDivisionError):
|
||||
admin_client.post(
|
||||
"/api/documents/bulk_edit/",
|
||||
{
|
||||
"documents": [],
|
||||
"all": True,
|
||||
"filters": {"query": "anything"},
|
||||
"method": "set_document_type",
|
||||
"parameters": {"document_type": None},
|
||||
},
|
||||
format="json",
|
||||
)
|
||||
|
||||
def test_bulk_download_maps_search_query_error_to_a_400(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- parse_user_query() raising a SearchQueryError naming a field
|
||||
WHEN:
|
||||
- The bulk_download endpoint is called with a query filter
|
||||
THEN:
|
||||
- The response is a 400 whose body names the field
|
||||
"""
|
||||
import documents.search._backend as backend_mod
|
||||
|
||||
def raise_search_query_error(*args: object, **kwargs: object) -> object:
|
||||
raise SearchQueryError("bad value for field 'added'")
|
||||
|
||||
monkeypatch.setattr(
|
||||
backend_mod,
|
||||
"parse_user_query",
|
||||
raise_search_query_error,
|
||||
)
|
||||
|
||||
response = admin_client.post(
|
||||
"/api/documents/bulk_download/",
|
||||
{
|
||||
"documents": [],
|
||||
"all": True,
|
||||
"filters": {"query": "anything"},
|
||||
},
|
||||
format="json",
|
||||
)
|
||||
|
||||
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
||||
assert "added" in str(response.data["query"])
|
||||
|
||||
def test_more_like_id_selection_filter_maps_search_query_error_to_a_400(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- TantivyBackend.more_like_this_ids() raising a
|
||||
SearchQueryError
|
||||
WHEN:
|
||||
- The bulk_download endpoint is called with a more_like_id
|
||||
filter
|
||||
THEN:
|
||||
- The response is a 400
|
||||
"""
|
||||
import documents.search._backend as backend_mod
|
||||
|
||||
def raise_search_query_error(*args: object, **kwargs: object) -> object:
|
||||
raise SearchQueryError("similar-document lookup is unavailable")
|
||||
|
||||
monkeypatch.setattr(
|
||||
backend_mod.TantivyBackend,
|
||||
"more_like_this_ids",
|
||||
raise_search_query_error,
|
||||
)
|
||||
|
||||
response = admin_client.post(
|
||||
"/api/documents/bulk_download/",
|
||||
{
|
||||
"documents": [],
|
||||
"all": True,
|
||||
"filters": {"more_like_id": indexed_document.pk},
|
||||
},
|
||||
format="json",
|
||||
)
|
||||
|
||||
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
||||
|
||||
def test_more_like_id_selection_filter_lets_an_unexpected_exception_surface(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- TantivyBackend.more_like_this_ids() raising an unrelated
|
||||
exception (ZeroDivisionError), not a SearchQueryError
|
||||
WHEN:
|
||||
- The bulk_download endpoint is called with a more_like_id
|
||||
filter
|
||||
THEN:
|
||||
- The exception propagates unconverted, rather than being
|
||||
relabeled a 400
|
||||
"""
|
||||
import documents.search._backend as backend_mod
|
||||
|
||||
def raise_zero_division(*args: object, **kwargs: object) -> object:
|
||||
raise ZeroDivisionError("synthetic bug, unrelated to similarity lookup")
|
||||
|
||||
monkeypatch.setattr(
|
||||
backend_mod.TantivyBackend,
|
||||
"more_like_this_ids",
|
||||
raise_zero_division,
|
||||
)
|
||||
|
||||
with pytest.raises(ZeroDivisionError):
|
||||
admin_client.post(
|
||||
"/api/documents/bulk_download/",
|
||||
{
|
||||
"documents": [],
|
||||
"all": True,
|
||||
"filters": {"more_like_id": indexed_document.pk},
|
||||
},
|
||||
format="json",
|
||||
)
|
||||
@@ -0,0 +1,287 @@
|
||||
"""The query-length cap in ``_get_tantivy_query_and_mode``.
|
||||
|
||||
whoosh-compat's fieldname tagger is O(n^2) in plain word characters, so an
|
||||
unbounded ``query`` (SearchMode.QUERY) string is a CPU-exhaustion vector
|
||||
against a single request handler. The GET search endpoint is incidentally
|
||||
bounded by the web server's header limit, but the POST selection-filter
|
||||
path (bulk edit, bulk download) is not -- that is the real vector, so it
|
||||
must be pinned here too, not just the GET path.
|
||||
|
||||
The cap is enforced once, in the shared helper both entry points call, so
|
||||
these tests exercise the real endpoints rather than the helper directly:
|
||||
a construct that looks right in isolation has repeatedly behaved
|
||||
differently end to end on this branch.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
from unittest import mock
|
||||
|
||||
import pytest
|
||||
from rest_framework import status
|
||||
|
||||
import documents.search._backend
|
||||
from documents.tests.factories import DocumentFactory
|
||||
from documents.views import _MAX_QUERY_LENGTH
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from rest_framework.test import APIClient
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
pytestmark = [pytest.mark.django_db, pytest.mark.usefixtures("_search_index")]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def indexed_document() -> Document:
|
||||
from documents.search import get_backend
|
||||
|
||||
doc = DocumentFactory.create(title="quarterly invoice", content="acme corp")
|
||||
get_backend().add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestGetSearchEndpointEnforcesTheCap:
|
||||
def test_query_one_over_the_cap_is_a_400(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The GET search endpoint
|
||||
WHEN:
|
||||
- A query one character over `_MAX_QUERY_LENGTH` is submitted
|
||||
THEN:
|
||||
- The response is a 400 naming both the actual length and the
|
||||
cap, and the query is rejected before it ever reaches the
|
||||
parser -- the 400 alone doesn't prove that, since the parser
|
||||
could run first and the view could discard the result
|
||||
"""
|
||||
query = "a" * (_MAX_QUERY_LENGTH + 1)
|
||||
|
||||
with mock.patch(
|
||||
"documents.search._backend.parse_user_query",
|
||||
wraps=documents.search._backend.parse_user_query,
|
||||
) as parse_spy:
|
||||
response = admin_client.get("/api/documents/", {"query": query})
|
||||
|
||||
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
||||
message = str(response.data["query"])
|
||||
assert str(_MAX_QUERY_LENGTH) in message
|
||||
assert str(_MAX_QUERY_LENGTH + 1) in message
|
||||
parse_spy.assert_not_called()
|
||||
|
||||
def test_query_at_exactly_the_cap_is_accepted(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The GET search endpoint
|
||||
WHEN:
|
||||
- A query exactly `_MAX_QUERY_LENGTH` characters long is
|
||||
submitted
|
||||
THEN:
|
||||
- The response is a 200 (the cap is inclusive, not exclusive)
|
||||
"""
|
||||
query = "a" * _MAX_QUERY_LENGTH
|
||||
|
||||
response = admin_client.get("/api/documents/", {"query": query})
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
|
||||
def test_an_ordinary_query_is_unaffected(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The GET search endpoint and an indexed document
|
||||
WHEN:
|
||||
- An ordinary, well-under-the-cap query is submitted
|
||||
THEN:
|
||||
- The cap has no effect on a normal search: the matching
|
||||
document is returned
|
||||
"""
|
||||
response = admin_client.get("/api/documents/", {"query": "invoice"})
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
assert response.data["count"] == 1
|
||||
|
||||
|
||||
class TestPostSelectionPathsEnforceTheCap:
|
||||
"""The bulk-edit and bulk-download selection filters share the same
|
||||
helper the GET search path uses. This is the path that actually
|
||||
matters: it is not bounded by a web server's header-length limit the
|
||||
way the GET path incidentally is."""
|
||||
|
||||
def test_bulk_edit_query_one_over_the_cap_is_a_400(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The bulk-edit selection-filter endpoint
|
||||
WHEN:
|
||||
- Its `filters.query` is one character over `_MAX_QUERY_LENGTH`
|
||||
THEN:
|
||||
- The response is a 400 naming both the actual length and the
|
||||
cap, and the query is rejected before it ever reaches the
|
||||
parser -- this is the path with no web-server header-length
|
||||
limit to fall back on, so this is the invariant that matters
|
||||
"""
|
||||
query = "a" * (_MAX_QUERY_LENGTH + 1)
|
||||
|
||||
with mock.patch(
|
||||
"documents.search._backend.parse_user_query",
|
||||
wraps=documents.search._backend.parse_user_query,
|
||||
) as parse_spy:
|
||||
response = admin_client.post(
|
||||
"/api/documents/bulk_edit/",
|
||||
{
|
||||
"documents": [],
|
||||
"all": True,
|
||||
"filters": {"query": query},
|
||||
"method": "set_document_type",
|
||||
"parameters": {"document_type": None},
|
||||
},
|
||||
format="json",
|
||||
)
|
||||
|
||||
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
||||
message = str(response.data["query"])
|
||||
assert str(_MAX_QUERY_LENGTH) in message
|
||||
assert str(_MAX_QUERY_LENGTH + 1) in message
|
||||
parse_spy.assert_not_called()
|
||||
|
||||
@mock.patch("documents.bulk_edit.bulk_update_documents.apply_async")
|
||||
def test_bulk_edit_query_at_exactly_the_cap_is_accepted(
|
||||
self,
|
||||
bulk_update_task_mock: mock.MagicMock,
|
||||
admin_client: APIClient,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The bulk-edit selection-filter endpoint
|
||||
WHEN:
|
||||
- Its `filters.query` is exactly `_MAX_QUERY_LENGTH` characters
|
||||
long
|
||||
THEN:
|
||||
- The cap check accepts it and the request reaches the real
|
||||
bulk-edit method (its Celery dispatch is mocked out here,
|
||||
same as every other bulk-edit test, since nothing here is
|
||||
testing that method itself)
|
||||
"""
|
||||
query = "a" * _MAX_QUERY_LENGTH
|
||||
|
||||
response = admin_client.post(
|
||||
"/api/documents/bulk_edit/",
|
||||
{
|
||||
"documents": [],
|
||||
"all": True,
|
||||
"filters": {"query": query},
|
||||
"method": "set_document_type",
|
||||
"parameters": {"document_type": None},
|
||||
},
|
||||
format="json",
|
||||
)
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
|
||||
def test_bulk_download_query_one_over_the_cap_is_a_400(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The bulk-download selection-filter endpoint
|
||||
WHEN:
|
||||
- Its `filters.query` is one character over `_MAX_QUERY_LENGTH`
|
||||
THEN:
|
||||
- The response is a 400 naming both the actual length and the
|
||||
cap, and the query is rejected before it ever reaches the
|
||||
parser
|
||||
"""
|
||||
query = "a" * (_MAX_QUERY_LENGTH + 1)
|
||||
|
||||
with mock.patch(
|
||||
"documents.search._backend.parse_user_query",
|
||||
wraps=documents.search._backend.parse_user_query,
|
||||
) as parse_spy:
|
||||
response = admin_client.post(
|
||||
"/api/documents/bulk_download/",
|
||||
{
|
||||
"documents": [],
|
||||
"all": True,
|
||||
"filters": {"query": query},
|
||||
},
|
||||
format="json",
|
||||
)
|
||||
|
||||
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
||||
message = str(response.data["query"])
|
||||
assert str(_MAX_QUERY_LENGTH) in message
|
||||
assert str(_MAX_QUERY_LENGTH + 1) in message
|
||||
parse_spy.assert_not_called()
|
||||
|
||||
|
||||
class TestGlobalSearchEnforcesTheCapToo:
|
||||
"""GlobalSearchView calls the backend directly, not through the shared helper.
|
||||
|
||||
It hardcodes SearchMode.TEXT, which is linear rather than quadratic, so it
|
||||
was never the CPU-exhaustion vector. It is capped anyway so that "every
|
||||
user query string reaching the backend passes a length check" is an
|
||||
invariant rather than a claim with an exception: the view already bounds
|
||||
the query from below, and a later change letting it select a mode would
|
||||
otherwise reopen the hole silently.
|
||||
"""
|
||||
|
||||
def test_query_one_over_the_cap_is_a_400(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- GlobalSearchView, which calls the backend directly in
|
||||
SearchMode.TEXT rather than through the shared cap-checking
|
||||
helper
|
||||
WHEN:
|
||||
- Its query is one character over `_MAX_QUERY_LENGTH`
|
||||
THEN:
|
||||
- The response is still a 400, keeping "every user query
|
||||
string reaching the backend passes a length check" an
|
||||
invariant with no exception, even though TEXT mode is
|
||||
linear and was never itself the CPU-exhaustion vector
|
||||
"""
|
||||
response = admin_client.get(
|
||||
"/api/search/",
|
||||
{"query": "a" * (_MAX_QUERY_LENGTH + 1)},
|
||||
)
|
||||
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
||||
|
||||
def test_query_at_exactly_the_cap_is_accepted(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- GlobalSearchView
|
||||
WHEN:
|
||||
- Its query is exactly `_MAX_QUERY_LENGTH` characters long
|
||||
THEN:
|
||||
- The response is a 200 (the cap is inclusive, not exclusive)
|
||||
"""
|
||||
response = admin_client.get(
|
||||
"/api/search/",
|
||||
{"query": "a" * _MAX_QUERY_LENGTH},
|
||||
)
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
@@ -0,0 +1,88 @@
|
||||
"""An unterminated ``[`` date range bracket at the API level.
|
||||
|
||||
``created:[2020`` (with or without a dangling ``to <value>``) now raises
|
||||
BAD_DATE and the search endpoint returns HTTP 400, where it used to parse
|
||||
past the missing ``]`` and silently pass the malformed range through.
|
||||
A 400 is correct: malformed input should fail loudly rather than silently
|
||||
matching an unintended query. Pinned at the API level -- the layer a user
|
||||
or client actually sees -- rather than only against the parser directly.
|
||||
|
||||
The properly closed decoy proves the bracket is what matters, not
|
||||
whoosh-compat's date grammar generally: ``created:[2020 to 2021]`` parses
|
||||
and searches cleanly.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
from rest_framework import status
|
||||
|
||||
from documents.tests.factories import DocumentFactory
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from rest_framework.test import APIClient
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
pytestmark = [pytest.mark.django_db, pytest.mark.usefixtures("_search_index")]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def indexed_document() -> Document:
|
||||
from documents.search import get_backend
|
||||
|
||||
doc = DocumentFactory.create(title="quarterly invoice", content="acme corp")
|
||||
get_backend().add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestUnterminatedBracketReturnsA400:
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
pytest.param("created:[2020", id="missing_upper_bound_and_bracket"),
|
||||
pytest.param("created:[2020 to 2021", id="missing_closing_bracket"),
|
||||
],
|
||||
)
|
||||
def test_unterminated_bracket_is_a_400(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
indexed_document: Document,
|
||||
query: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The search endpoint
|
||||
WHEN:
|
||||
- A date-range query with a missing closing `]` (with or
|
||||
without a dangling upper bound) is submitted
|
||||
THEN:
|
||||
- The response is a 400 naming the field, rather than parsing
|
||||
past the missing bracket and silently passing the malformed
|
||||
range through
|
||||
"""
|
||||
response = admin_client.get(f"/api/documents/?query={query}")
|
||||
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
||||
assert "created" in str(response.data["query"])
|
||||
|
||||
def test_properly_closed_bracket_still_searches_cleanly(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
indexed_document: Document,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- The search endpoint
|
||||
WHEN:
|
||||
- A properly closed date-range query is submitted
|
||||
THEN:
|
||||
- The response is a 200 (the decoy proving the missing
|
||||
bracket, not whoosh-compat's date grammar generally, is
|
||||
what the 400 above is about)
|
||||
"""
|
||||
response = admin_client.get(
|
||||
"/api/documents/?query=created:[2020 to 2021]",
|
||||
)
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
@@ -69,6 +69,16 @@ class TestTrashAPI(DirectoriesMixin, APITestCase):
|
||||
self.assertEqual(resp.status_code, status.HTTP_200_OK)
|
||||
self.assertEqual(Document.global_objects.count(), 0)
|
||||
|
||||
def test_trash_list_requires_global_document_view_permission(self) -> None:
|
||||
user = User.objects.create_user(username="trash_owner")
|
||||
document = Document.objects.create(title="Owned", owner=user)
|
||||
document.delete()
|
||||
self.client.force_authenticate(user)
|
||||
|
||||
response = self.client.get("/api/trash/")
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
|
||||
|
||||
def test_trash_api_empty_all(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
@@ -207,3 +217,65 @@ class TestTrashAPI(DirectoriesMixin, APITestCase):
|
||||
)
|
||||
self.assertEqual(resp.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
self.assertIn("have not yet been deleted", resp.data["documents"][0])
|
||||
|
||||
def _make_versioned_document(self) -> tuple[Document, list[Document]]:
|
||||
root = Document.objects.create(
|
||||
title="root",
|
||||
content="root-content",
|
||||
checksum="root",
|
||||
mime_type="application/pdf",
|
||||
)
|
||||
versions = [
|
||||
Document.objects.create(
|
||||
title=f"v{index}",
|
||||
content=f"v{index}-content",
|
||||
checksum=f"v{index}",
|
||||
mime_type="application/pdf",
|
||||
root_document=root,
|
||||
version_index=index,
|
||||
)
|
||||
for index in range(1, 3)
|
||||
]
|
||||
return root, versions
|
||||
|
||||
def test_api_trash_restore_document_restores_its_versions(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Existing document with two versions
|
||||
WHEN:
|
||||
- API request to delete the document
|
||||
- API request to restore it from the trash
|
||||
THEN:
|
||||
- Only the document itself is listed in the trash
|
||||
- A version cannot be restored without its root
|
||||
- The document is restored together with all of its versions
|
||||
"""
|
||||
root, versions = self._make_versioned_document()
|
||||
|
||||
self.client.force_login(user=self.user)
|
||||
self.client.delete(f"/api/documents/{root.pk}/")
|
||||
self.assertEqual(Document.deleted_objects.count(), 3)
|
||||
|
||||
resp = self.client.get("/api/trash/")
|
||||
self.assertEqual(resp.status_code, status.HTTP_200_OK)
|
||||
self.assertEqual(resp.data["count"], 1)
|
||||
self.assertEqual(resp.data["results"][0]["id"], root.pk)
|
||||
|
||||
# A version cannot be restored while its root remains in the trash.
|
||||
resp = self.client.post(
|
||||
"/api/trash/",
|
||||
{"action": "restore", "documents": [versions[0].pk]},
|
||||
)
|
||||
self.assertEqual(resp.status_code, status.HTTP_400_BAD_REQUEST)
|
||||
self.assertIn("Restore the root document", resp.data["documents"][0])
|
||||
|
||||
resp = self.client.post(
|
||||
"/api/trash/",
|
||||
{"action": "restore", "documents": [root.pk]},
|
||||
)
|
||||
self.assertEqual(resp.status_code, status.HTTP_200_OK)
|
||||
self.assertEqual(Document.deleted_objects.count(), 0)
|
||||
self.assertCountEqual(
|
||||
Document.objects.filter(root_document=root).values_list("id", flat=True),
|
||||
[version.pk for version in versions],
|
||||
)
|
||||
|
||||
@@ -194,6 +194,48 @@ class TestApiWorkflows(DirectoriesMixin, APITestCase):
|
||||
self.assertEqual(response.status_code, status.HTTP_201_CREATED)
|
||||
self.assertEqual(Workflow.objects.count(), 2)
|
||||
|
||||
def test_api_create_workflow_ignores_nested_action_id(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- An existing workflow action
|
||||
WHEN:
|
||||
- API request to create a workflow includes that action's ID
|
||||
THEN:
|
||||
- A new action is created without changing the existing action
|
||||
"""
|
||||
original_title = self.action.assign_title
|
||||
|
||||
response = self.client.post(
|
||||
self.ENDPOINT,
|
||||
json.dumps(
|
||||
{
|
||||
"name": "Workflow 2",
|
||||
"order": 1,
|
||||
"triggers": [
|
||||
{
|
||||
"sources": [DocumentSource.ApiUpload],
|
||||
"type": WorkflowTrigger.WorkflowTriggerType.CONSUMPTION,
|
||||
"filter_filename": "*",
|
||||
},
|
||||
],
|
||||
"actions": [
|
||||
{
|
||||
"id": self.action.id,
|
||||
"assign_title": "New Action Title",
|
||||
},
|
||||
],
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_201_CREATED)
|
||||
self.action.refresh_from_db()
|
||||
self.assertEqual(self.action.assign_title, original_title)
|
||||
new_action = Workflow.objects.get(name="Workflow 2").actions.get()
|
||||
self.assertNotEqual(new_action.id, self.action.id)
|
||||
self.assertEqual(new_action.assign_title, "New Action Title")
|
||||
|
||||
def test_api_create_workflow_nested(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
|
||||
@@ -5,8 +5,11 @@ from unittest import mock
|
||||
|
||||
import pikepdf
|
||||
from django.contrib.auth.models import Group
|
||||
from django.contrib.auth.models import Permission
|
||||
from django.contrib.auth.models import User
|
||||
from django.db import connection
|
||||
from django.test import TestCase
|
||||
from django.test.utils import CaptureQueriesContext
|
||||
from guardian.shortcuts import assign_perm
|
||||
from guardian.shortcuts import get_groups_with_perms
|
||||
from guardian.shortcuts import get_users_with_perms
|
||||
@@ -19,6 +22,7 @@ from documents.models import Document
|
||||
from documents.models import DocumentType
|
||||
from documents.models import StoragePath
|
||||
from documents.models import Tag
|
||||
from documents.permissions import set_permissions_for_objects
|
||||
from documents.tests.utils import DirectoriesMixin
|
||||
|
||||
|
||||
@@ -392,6 +396,11 @@ class TestBulkEdit(DirectoriesMixin, TestCase):
|
||||
self.assertFalse(Document.objects.filter(id=self.doc1.id).exists())
|
||||
self.assertFalse(Document.objects.filter(id=version.id).exists())
|
||||
|
||||
Document.deleted_objects.get(id=self.doc1.id).restore(strict=False)
|
||||
|
||||
self.assertTrue(Document.objects.filter(id=self.doc1.id).exists())
|
||||
self.assertTrue(Document.objects.filter(id=version.id).exists())
|
||||
|
||||
def test_delete_version_document_keeps_root(self) -> None:
|
||||
version = Document.objects.create(
|
||||
checksum="A-v1",
|
||||
@@ -510,6 +519,178 @@ class TestBulkEdit(DirectoriesMixin, TestCase):
|
||||
)
|
||||
self.assertEqual(groups_with_perms.count(), 2)
|
||||
|
||||
@mock.patch("documents.tasks.bulk_update_documents.apply_async")
|
||||
def test_set_permissions_batched_across_document_count(
|
||||
self,
|
||||
m,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Many documents are being bulk-edited to set permissions at once
|
||||
WHEN:
|
||||
- set_permissions runs over a small batch vs. a much larger one
|
||||
THEN:
|
||||
- Permissions are applied correctly at both scales
|
||||
- Query count does not grow with the number of documents, i.e.
|
||||
each user/group is applied across all documents with one
|
||||
batched call rather than one call per (document, identity)
|
||||
pair
|
||||
"""
|
||||
permissions = {
|
||||
"view": {
|
||||
"users": [self.user1.id, self.user2.id],
|
||||
"groups": [self.group2.id],
|
||||
},
|
||||
"change": {
|
||||
"users": [self.user1.id],
|
||||
"groups": [self.group2.id],
|
||||
},
|
||||
}
|
||||
|
||||
def run_with_n_documents(n: int) -> int:
|
||||
docs = [
|
||||
Document.objects.create(checksum=f"perm-{n}-{i}", title=f"perm-{n}-{i}")
|
||||
for i in range(n)
|
||||
]
|
||||
with CaptureQueriesContext(connection) as ctx:
|
||||
bulk_edit.set_permissions(
|
||||
[doc.id for doc in docs],
|
||||
set_permissions=permissions,
|
||||
owner=self.owner,
|
||||
merge=False,
|
||||
)
|
||||
for doc in docs:
|
||||
self.assertEqual(get_users_with_perms(doc).count(), 2)
|
||||
self.assertEqual(get_groups_with_perms(doc).count(), 1)
|
||||
return len(ctx.captured_queries)
|
||||
|
||||
small_batch_queries = run_with_n_documents(5)
|
||||
large_batch_queries = run_with_n_documents(50)
|
||||
|
||||
# A tolerance rather than equality, matching the N+1 check in
|
||||
# test_views.py: bulk_create's batch_size caps rows per INSERT, so a
|
||||
# large enough selection does legitimately add statements, and the
|
||||
# per-process ContentType cache makes the first run carry an extra
|
||||
# query. Neither can hide a regression to per-document assignment,
|
||||
# which would be ~10x the small-batch count here.
|
||||
self.assertLessEqual(
|
||||
large_batch_queries,
|
||||
small_batch_queries + 5,
|
||||
"Permission assignment appears to scale with document count: "
|
||||
f"{small_batch_queries} queries for 5 documents vs. "
|
||||
f"{large_batch_queries} for 50",
|
||||
)
|
||||
|
||||
@mock.patch("documents.tasks.bulk_update_documents.apply_async")
|
||||
def test_set_permissions_grants_direct_perm_even_if_already_granted_via_group(
|
||||
self,
|
||||
m,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A user already has view access to a document via group
|
||||
membership, with no direct grant of their own
|
||||
WHEN:
|
||||
- set_permissions explicitly grants that same user direct view
|
||||
access via bulk_edit
|
||||
THEN:
|
||||
- A direct permission grant is created for the user, not skipped
|
||||
because they already have equivalent access via the group
|
||||
|
||||
Regression test: guardian's queryset-aware assign_perm() (routed to
|
||||
when the target is a list/queryset) skips creating a direct row for
|
||||
anyone whose ObjectPermissionChecker.has_perm() already returns True
|
||||
-- which includes group-derived access. The single-object assign_perm
|
||||
this bulk path replaces has no such check; it always ensures a
|
||||
direct row via get_or_create. Losing that guarantee would mean
|
||||
revoking the group's grant later silently strips access that was
|
||||
supposed to be explicit.
|
||||
"""
|
||||
self.doc1.owner = self.user1
|
||||
self.doc1.save()
|
||||
self.user1.groups.add(self.group1)
|
||||
assign_perm("view_document", self.group1, self.doc1)
|
||||
|
||||
bulk_edit.set_permissions(
|
||||
[self.doc1.id],
|
||||
set_permissions={
|
||||
"view": {"users": [self.user1.id], "groups": []},
|
||||
},
|
||||
merge=True,
|
||||
)
|
||||
|
||||
direct_users = get_users_with_perms(
|
||||
self.doc1,
|
||||
only_with_perms_in=["view_document"],
|
||||
with_group_users=False,
|
||||
)
|
||||
self.assertIn(self.user1, direct_users)
|
||||
|
||||
def test_set_permissions_for_objects_raises_for_unknown_action(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- An unrecognized permission action name with users to grant it
|
||||
to
|
||||
WHEN:
|
||||
- set_permissions_for_objects is called
|
||||
THEN:
|
||||
- Permission.DoesNotExist is raised, not a silent no-op
|
||||
|
||||
Regression test: the endpoint that calls this
|
||||
(BulkEditObjectPermissionsView) never actually validates action
|
||||
names against the raw client-supplied permissions dict --
|
||||
BulkEditObjectsSerializer._validate_permissions calls
|
||||
validate_set_permissions() only for its side-effecting user/group id
|
||||
checks and discards the filtered dict it returns -- so a bogus
|
||||
action key reaches this function as-is. Resolving the Permission via
|
||||
a bare `.filter()` (which returns empty instead of raising) would
|
||||
silently drop the grant and report success.
|
||||
"""
|
||||
with self.assertRaises(Permission.DoesNotExist):
|
||||
set_permissions_for_objects(
|
||||
{"not_a_real_action": {"users": [self.user1.id], "groups": []}},
|
||||
Document,
|
||||
[self.doc1.pk],
|
||||
)
|
||||
|
||||
def test_set_permissions_for_objects_unknown_action_applies_nothing(
|
||||
self,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A permissions dict with a valid action ordered ahead of an
|
||||
unrecognized one
|
||||
WHEN:
|
||||
- set_permissions_for_objects is called
|
||||
THEN:
|
||||
- Permission.DoesNotExist is raised
|
||||
- The valid action ahead of it is not applied either
|
||||
|
||||
Every action is resolved before any row is written, so a bad action
|
||||
name cannot leave a half-applied change behind. That matters because
|
||||
BulkEditObjectsView turns this exception into a 400: without the
|
||||
up-front resolution the client would be told the request failed
|
||||
while the leading action had already been committed.
|
||||
"""
|
||||
with self.assertRaises(Permission.DoesNotExist):
|
||||
set_permissions_for_objects(
|
||||
{
|
||||
"view": {"users": [self.user1.id], "groups": []},
|
||||
"not_a_real_action": {"users": [self.user1.id], "groups": []},
|
||||
},
|
||||
Document,
|
||||
[self.doc1.pk],
|
||||
)
|
||||
|
||||
self.assertNotIn(
|
||||
self.user1,
|
||||
get_users_with_perms(
|
||||
self.doc1,
|
||||
only_with_perms_in=["view_document"],
|
||||
with_group_users=False,
|
||||
),
|
||||
)
|
||||
|
||||
@mock.patch("documents.models.Document.delete")
|
||||
def test_delete_documents_old_uuid_field(self, m) -> None:
|
||||
m.side_effect = Exception("Data too long for column 'transaction_id' at row 1")
|
||||
@@ -1461,6 +1642,16 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
||||
mock_group.assert_not_called()
|
||||
mock_consume_file.assert_not_called()
|
||||
|
||||
@mock.patch("pikepdf.open")
|
||||
def test_edit_pdf_rejects_invalid_operations(self, mock_open) -> None:
|
||||
for operations in ([], [{"page": 1, "doc": 2**32}]):
|
||||
with self.subTest(operations=operations):
|
||||
with self.assertLogs("paperless.bulk_edit", level="ERROR"):
|
||||
with self.assertRaisesRegex(ValueError, "index is out of bounds"):
|
||||
bulk_edit.edit_pdf([self.doc2.id], operations)
|
||||
|
||||
mock_open.assert_not_called()
|
||||
|
||||
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file.delay")
|
||||
@mock.patch("documents.tasks.consume_file.apply_async")
|
||||
@mock.patch("documents.bulk_edit.tempfile.mkdtemp")
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import pytest
|
||||
from django.core.checks import Error
|
||||
from django.core.checks import Warning
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
from pytest_django.fixtures import Settings
|
||||
from pytest_mock import MockerFixture
|
||||
|
||||
from documents.checks import filename_format_check
|
||||
@@ -47,7 +47,7 @@ class TestFilenameFormatCheck:
|
||||
)
|
||||
def test_warns_on_old_style_format(
|
||||
self,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
filename_format: str,
|
||||
expected_hint: str,
|
||||
) -> None:
|
||||
|
||||
@@ -3,6 +3,7 @@ import warnings
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
from django.conf import settings
|
||||
from django.test import TestCase
|
||||
@@ -11,6 +12,7 @@ from django.test import override_settings
|
||||
from documents.classifier import ClassifierModelCorruptError
|
||||
from documents.classifier import DocumentClassifier
|
||||
from documents.classifier import IncompatibleClassifierVersionError
|
||||
from documents.classifier import _predict_with_threshold
|
||||
from documents.classifier import load_classifier
|
||||
from documents.models import Correspondent
|
||||
from documents.models import Document
|
||||
@@ -625,6 +627,103 @@ class TestClassifier(DirectoriesMixin, TestCase):
|
||||
self.assertEqual(self.classifier.predict_storage_path(doc1.content), sp.pk)
|
||||
self.assertIsNone(self.classifier.predict_storage_path(doc2.content))
|
||||
|
||||
def test_predict_rejects_prediction_below_match_threshold(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Classifiers trained against test data with confident predictions
|
||||
WHEN:
|
||||
- CLASSIFIER_MATCH_THRESHOLD exceeds the model's confidence
|
||||
THEN:
|
||||
- Every predict_* method discards the match in favor of no match
|
||||
"""
|
||||
c1 = Correspondent.objects.create(
|
||||
name="c1",
|
||||
matching_algorithm=Correspondent.MATCH_AUTO,
|
||||
)
|
||||
dt1 = DocumentType.objects.create(
|
||||
name="dt1",
|
||||
matching_algorithm=DocumentType.MATCH_AUTO,
|
||||
)
|
||||
sp1 = StoragePath.objects.create(
|
||||
name="sp1",
|
||||
matching_algorithm=StoragePath.MATCH_AUTO,
|
||||
)
|
||||
|
||||
doc1 = Document.objects.create(
|
||||
title="doc1",
|
||||
content="this is a document from c1",
|
||||
correspondent=c1,
|
||||
document_type=dt1,
|
||||
storage_path=sp1,
|
||||
checksum="A",
|
||||
)
|
||||
Document.objects.create(
|
||||
title="doc2",
|
||||
content="this is a document from no one",
|
||||
checksum="B",
|
||||
)
|
||||
|
||||
self.classifier.train()
|
||||
|
||||
predictors = {
|
||||
"correspondent": self.classifier.predict_correspondent,
|
||||
"document_type": self.classifier.predict_document_type,
|
||||
"storage_path": self.classifier.predict_storage_path,
|
||||
}
|
||||
# No real prediction can reach a confidence this high, so this
|
||||
# isolates the threshold check from the model's actual output.
|
||||
with override_settings(CLASSIFIER_MATCH_THRESHOLD=0.999999):
|
||||
for name, predict in predictors.items():
|
||||
with self.subTest(field=name):
|
||||
self.assertIsNone(predict(doc1.content))
|
||||
|
||||
def test_train_uses_balanced_sample_weight(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A training set with correspondents, document types and storage paths
|
||||
WHEN:
|
||||
- The classifier is trained
|
||||
THEN:
|
||||
- Each MLP classifier is fit with balanced sample weights, so that
|
||||
over-represented classes don't dominate predictions
|
||||
"""
|
||||
c1 = Correspondent.objects.create(
|
||||
name="c1",
|
||||
matching_algorithm=Correspondent.MATCH_AUTO,
|
||||
)
|
||||
dt1 = DocumentType.objects.create(
|
||||
name="dt1",
|
||||
matching_algorithm=DocumentType.MATCH_AUTO,
|
||||
)
|
||||
sp1 = StoragePath.objects.create(
|
||||
name="sp1",
|
||||
matching_algorithm=StoragePath.MATCH_AUTO,
|
||||
)
|
||||
|
||||
Document.objects.create(
|
||||
title="doc1",
|
||||
content="this is a document from c1",
|
||||
correspondent=c1,
|
||||
document_type=dt1,
|
||||
storage_path=sp1,
|
||||
checksum="A",
|
||||
)
|
||||
Document.objects.create(
|
||||
title="doc2",
|
||||
content="this is a document from no one",
|
||||
checksum="B",
|
||||
)
|
||||
|
||||
with mock.patch(
|
||||
"sklearn.utils.class_weight.compute_sample_weight",
|
||||
return_value=None,
|
||||
) as mocked_compute_sample_weight:
|
||||
self.classifier.train()
|
||||
|
||||
self.assertEqual(mocked_compute_sample_weight.call_count, 3)
|
||||
for call in mocked_compute_sample_weight.call_args_list:
|
||||
self.assertEqual(call.args[0], "balanced")
|
||||
|
||||
def test_one_tag_predict(self) -> None:
|
||||
t1 = Tag.objects.create(name="t1", matching_algorithm=Tag.MATCH_AUTO, pk=12)
|
||||
|
||||
@@ -810,6 +909,52 @@ class TestClassifier(DirectoriesMixin, TestCase):
|
||||
load_classifier(raise_exception=True)
|
||||
|
||||
|
||||
class _StubProbaClassifier:
|
||||
"""
|
||||
A fake scikit-learn classifier exposing just enough of the API for
|
||||
`_predict_with_threshold`: `classes_` and `predict_proba`.
|
||||
"""
|
||||
|
||||
def __init__(self, classes: list[int], probabilities: list[float]) -> None:
|
||||
self.classes_ = np.array(classes)
|
||||
self._probabilities = np.array([probabilities])
|
||||
|
||||
def predict_proba(self, X) -> np.ndarray:
|
||||
return self._probabilities
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("classes", "probabilities", "threshold", "expected"),
|
||||
[
|
||||
# confident prediction above the threshold is returned
|
||||
([-1, 3], [0.1, 0.9], 0.6, 3),
|
||||
# prediction below the threshold is discarded
|
||||
([-1, 3], [0.45, 0.55], 0.6, None),
|
||||
# boundary: exactly at the threshold is accepted, not discarded
|
||||
([-1, 3], [0.4, 0.6], 0.6, 3),
|
||||
# the winning class is the "no match" pseudo-class, regardless of its
|
||||
# own confidence
|
||||
([-1, 3], [0.99, 0.01], 0.0, None),
|
||||
# threshold of 0.0 disables the confidence check entirely
|
||||
([-1, 3], [0.45, 0.55], 0.0, 3),
|
||||
],
|
||||
)
|
||||
def test_predict_with_threshold(classes, probabilities, threshold, expected) -> None:
|
||||
classifier = _StubProbaClassifier(classes, probabilities)
|
||||
result = _predict_with_threshold(classifier, X=None, threshold=threshold)
|
||||
assert result == expected
|
||||
|
||||
|
||||
def test_classifier_match_threshold_default() -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- No PAPERLESS_CLASSIFIER_MATCH_THRESHOLD environment variable is set
|
||||
THEN:
|
||||
- The classifier match threshold defaults to 0.6
|
||||
"""
|
||||
assert settings.CLASSIFIER_MATCH_THRESHOLD == 0.6
|
||||
|
||||
|
||||
def test_preprocess_content() -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
|
||||
@@ -0,0 +1,457 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from types import SimpleNamespace
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
from django.db import connection
|
||||
from django.test.utils import CaptureQueriesContext
|
||||
from rest_framework import status
|
||||
|
||||
from documents.models import Document
|
||||
from documents.tests.factories import DocumentFactory
|
||||
from documents.versioning import LATEST_VERSION_CONTENT_PREFETCH_ATTR
|
||||
from documents.versioning import has_prefetched_effective_content
|
||||
from documents.versioning import latest_version_content_prefetch
|
||||
from documents.views import DocumentViewSet
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from rest_framework.test import APIClient
|
||||
|
||||
|
||||
class TestNeedsEffectiveContentAnnotation:
|
||||
"""
|
||||
DocumentViewSet._needs_effective_content_annotation() decides whether
|
||||
the effective_content correlated subquery is worth attaching to the
|
||||
queryset at all -- see TestDocumentListEffectiveContentAnnotation below
|
||||
for why. This only checks that decision's own logic (a plain query-param
|
||||
membership test), not that Django/DRF's filtering machinery works.
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("params", "expected"),
|
||||
[
|
||||
({}, False),
|
||||
({"ordering": "-added"}, False),
|
||||
({"tags__id__in": "1,2"}, False),
|
||||
({"search": ""}, False),
|
||||
({"search": " "}, False),
|
||||
({"content__icontains": ""}, False),
|
||||
({"search": "foo"}, True),
|
||||
({"title_content": "foo"}, True),
|
||||
({"content__istartswith": "foo"}, True),
|
||||
({"content__iendswith": "foo"}, True),
|
||||
({"content__icontains": "foo"}, True),
|
||||
({"content__iexact": "foo"}, True),
|
||||
],
|
||||
)
|
||||
def test_detects_content_filter_params(
|
||||
self,
|
||||
params: dict[str, str],
|
||||
expected: bool, # noqa: FBT001
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A view bound to a request carrying the given query params
|
||||
WHEN:
|
||||
- Checking whether the effective_content annotation is needed
|
||||
THEN:
|
||||
- It is needed only for requests that actually filter on it
|
||||
"""
|
||||
view = DocumentViewSet()
|
||||
view.request = SimpleNamespace(query_params=params)
|
||||
|
||||
assert view._needs_effective_content_annotation() is expected
|
||||
|
||||
|
||||
class TestNeedsEffectiveContentPrefetch:
|
||||
"""
|
||||
DocumentViewSet._needs_effective_content_prefetch() decides whether the
|
||||
single-version content prefetch is worth attaching. It has to read the
|
||||
`fields` param exactly the way get_serializer() does, or a request whose
|
||||
response includes content ends up without the prefetch and pays
|
||||
get_effective_content()'s per-instance fallback instead.
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("params", "expected"),
|
||||
[
|
||||
pytest.param({}, True, id="no-fields-param-keeps-every-field"),
|
||||
pytest.param({"fields": ""}, True, id="blank-fields-keeps-every-field"),
|
||||
pytest.param(
|
||||
{"fields": "id,content"},
|
||||
True,
|
||||
id="content-among-requested-fields",
|
||||
),
|
||||
pytest.param({"fields": "content"}, True, id="content-only"),
|
||||
pytest.param({"fields": "id"}, False, id="content-not-requested"),
|
||||
pytest.param(
|
||||
{"fields": "id,title"},
|
||||
False,
|
||||
id="several-fields-without-content",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_detects_whether_content_can_reach_the_response(
|
||||
self,
|
||||
params: dict[str, str],
|
||||
expected: bool, # noqa: FBT001
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A view bound to a request carrying the given query params
|
||||
WHEN:
|
||||
- Checking whether the content prefetch is needed
|
||||
THEN:
|
||||
- It is needed exactly when get_serializer() would emit content,
|
||||
which treats a blank `fields` the same as an absent one
|
||||
"""
|
||||
view = DocumentViewSet()
|
||||
view.request = SimpleNamespace(query_params=params)
|
||||
|
||||
assert view._needs_effective_content_prefetch() is expected
|
||||
|
||||
|
||||
@pytest.mark.django_db
|
||||
class TestDocumentListEffectiveContentAnnotation:
|
||||
"""
|
||||
DocumentViewSet.get_queryset() only attaches the effective_content
|
||||
correlated subquery when a request actually filters on it. Attaching it
|
||||
unconditionally re-executes it once per candidate row before the page's
|
||||
LIMIT is applied -- fine on SQLite/Postgres, but pathological on
|
||||
MariaDB's default cardinality estimation for the root_document_id
|
||||
self-join once candidate counts get large (see the root_document_id /
|
||||
effective_content perf investigation).
|
||||
"""
|
||||
|
||||
def test_list_without_content_filter_skips_annotation_but_returns_latest_content(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A root document whose latest version has different content
|
||||
WHEN:
|
||||
- Listing documents with no search/content-filter param
|
||||
THEN:
|
||||
- The response still reflects the latest version's content
|
||||
- The database never evaluates effective_content per row
|
||||
"""
|
||||
root = DocumentFactory(content="old-root-content")
|
||||
DocumentFactory(
|
||||
root_document=root,
|
||||
version_index=1,
|
||||
content="new-version-content",
|
||||
)
|
||||
|
||||
with CaptureQueriesContext(connection) as ctx:
|
||||
response = admin_client.get("/api/documents/?fields=id,content")
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
assert response.data["results"] == [
|
||||
{"id": root.id, "content": "new-version-content"},
|
||||
]
|
||||
assert not any(
|
||||
"effective_content" in query["sql"] for query in ctx.captured_queries
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"fields_param",
|
||||
[
|
||||
pytest.param("", id="blank-fields"),
|
||||
pytest.param("id,content", id="content-requested"),
|
||||
],
|
||||
)
|
||||
def test_content_resolves_without_a_query_per_document(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
fields_param: str,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- One versioned root document, then two more
|
||||
WHEN:
|
||||
- Listing documents with a `fields` param that keeps content
|
||||
THEN:
|
||||
- Every root's content resolves to its latest version's
|
||||
- The query count does not grow with the number of documents,
|
||||
i.e. a blank `fields` does not skip the prefetch and fall back
|
||||
to loading each root's deferred version content
|
||||
"""
|
||||
first = DocumentFactory(content="first-root-content")
|
||||
DocumentFactory(
|
||||
root_document=first,
|
||||
version_index=1,
|
||||
content="first-version-content",
|
||||
)
|
||||
|
||||
with CaptureQueriesContext(connection) as one_document:
|
||||
response = admin_client.get(f"/api/documents/?fields={fields_param}")
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
assert [r["content"] for r in response.data["results"]] == [
|
||||
"first-version-content",
|
||||
]
|
||||
|
||||
for index in range(2):
|
||||
root = DocumentFactory(content=f"root-content-{index}")
|
||||
DocumentFactory(
|
||||
root_document=root,
|
||||
version_index=1,
|
||||
content=f"version-content-{index}",
|
||||
)
|
||||
with CaptureQueriesContext(connection) as three_documents:
|
||||
response = admin_client.get(f"/api/documents/?fields={fields_param}")
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
assert sorted(r["content"] for r in response.data["results"]) == [
|
||||
"first-version-content",
|
||||
"version-content-0",
|
||||
"version-content-1",
|
||||
]
|
||||
assert len(_get_document_queries(three_documents)) == len(
|
||||
_get_document_queries(one_document),
|
||||
)
|
||||
|
||||
def test_list_without_content_field_skips_prefetch_and_omits_content(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A versioned root document
|
||||
WHEN:
|
||||
- Listing documents without asking for content
|
||||
THEN:
|
||||
- Content is neither serialized nor resolved
|
||||
- Nothing pays for the prefetch or the per-instance fallback
|
||||
"""
|
||||
root = DocumentFactory(content="root-content")
|
||||
DocumentFactory(
|
||||
root_document=root,
|
||||
version_index=1,
|
||||
content="version-content",
|
||||
)
|
||||
|
||||
with CaptureQueriesContext(connection) as ctx:
|
||||
response = admin_client.get("/api/documents/?fields=id")
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
assert response.data["results"] == [{"id": root.id}]
|
||||
assert _get_effective_content_fallback_queries(ctx) == []
|
||||
# Only the list query itself reads a content column: no extra query
|
||||
# for the skipped prefetch, none for a per-instance fallback
|
||||
content_queries = [
|
||||
query
|
||||
for query in ctx.captured_queries
|
||||
if '"documents_document"."content"' in query["sql"]
|
||||
]
|
||||
assert len(content_queries) == 1
|
||||
|
||||
def test_latest_version_content_prefetch_carries_only_the_newest_version(
|
||||
self,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A root document with two versions
|
||||
WHEN:
|
||||
- Fetching the root through latest_version_content_prefetch()
|
||||
THEN:
|
||||
- The prefetch carries only the single newest version, not every
|
||||
historical version's content (the whole point of not reusing
|
||||
the metadata-only "versions" prefetch for this)
|
||||
"""
|
||||
root = DocumentFactory(content="root-content")
|
||||
DocumentFactory(
|
||||
root_document=root,
|
||||
version_index=1,
|
||||
content="older-version-content",
|
||||
)
|
||||
DocumentFactory(
|
||||
root_document=root,
|
||||
version_index=2,
|
||||
content="newest-version-content",
|
||||
)
|
||||
|
||||
fetched_root = (
|
||||
Document.objects.filter(pk=root.pk)
|
||||
.prefetch_related(
|
||||
latest_version_content_prefetch(),
|
||||
)
|
||||
.get()
|
||||
)
|
||||
|
||||
latest = getattr(fetched_root, LATEST_VERSION_CONTENT_PREFETCH_ATTR)
|
||||
assert [v.content for v in latest] == ["newest-version-content"]
|
||||
|
||||
|
||||
class TestHasPrefetchedEffectiveContent:
|
||||
"""
|
||||
DocumentSerializer.to_representation() only calls get_effective_content()
|
||||
when has_prefetched_effective_content() says it's cheap -- otherwise a
|
||||
caller that never set up an annotation or prefetch (TrashView,
|
||||
GlobalSearchView, which build their own querysets and don't display
|
||||
content at all) would pay for a per-instance query nobody asked for.
|
||||
"""
|
||||
|
||||
def test_false_with_no_annotation_or_prefetch(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document the ORM never annotated or prefetched for
|
||||
WHEN:
|
||||
- Asking whether its effective content is already resolved
|
||||
THEN:
|
||||
- It is not, so the serializer must leave it alone
|
||||
"""
|
||||
document = DocumentFactory.build()
|
||||
|
||||
assert has_prefetched_effective_content(document) is False
|
||||
|
||||
def test_true_with_effective_content_annotation(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document carrying the queryset's effective_content annotation
|
||||
WHEN:
|
||||
- Asking whether its effective content is already resolved
|
||||
THEN:
|
||||
- It is, straight off the annotation
|
||||
"""
|
||||
document = DocumentFactory.build()
|
||||
document.effective_content = "resolved"
|
||||
|
||||
assert has_prefetched_effective_content(document) is True
|
||||
|
||||
def test_true_with_lean_prefetch_attr_even_when_empty(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document the lean content prefetch ran for, finding no versions
|
||||
WHEN:
|
||||
- Asking whether its effective content is already resolved
|
||||
THEN:
|
||||
- It is: an empty prefetch is an answer, not a missing one
|
||||
"""
|
||||
document = DocumentFactory.build()
|
||||
setattr(document, LATEST_VERSION_CONTENT_PREFETCH_ATTR, [])
|
||||
|
||||
assert has_prefetched_effective_content(document) is True
|
||||
|
||||
def test_true_with_metadata_versions_prefetch_cache(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document carrying only the metadata "versions" prefetch
|
||||
WHEN:
|
||||
- Asking whether its effective content is already resolved
|
||||
THEN:
|
||||
- It is, via get_effective_content()'s prefetch-cache branch
|
||||
"""
|
||||
document = DocumentFactory.build()
|
||||
document._prefetched_objects_cache = {"versions": []}
|
||||
|
||||
assert has_prefetched_effective_content(document) is True
|
||||
|
||||
|
||||
def _get_document_queries(
|
||||
ctx: CaptureQueriesContext,
|
||||
) -> list[dict[str, str]]:
|
||||
"""
|
||||
The queries a list request spends on the documents themselves, i.e.
|
||||
everything but the one-time django_content_type lookup guardian's
|
||||
permission filtering makes. That lookup is process-cached, and the
|
||||
autouse fixture in conftest clears the cache before every test, so it
|
||||
lands in whichever request happens to run first and never repeats --
|
||||
counting it makes a request look like it costs one query more than the
|
||||
identical request after it.
|
||||
"""
|
||||
return [q for q in ctx.captured_queries if '"django_content_type"' not in q["sql"]]
|
||||
|
||||
|
||||
def _get_effective_content_fallback_queries(
|
||||
ctx: CaptureQueriesContext,
|
||||
) -> list[dict[str, str]]:
|
||||
"""
|
||||
Document.get_effective_content()'s per-instance fallback (no annotation,
|
||||
no prefetch) is a `.values_list("content", flat=True).first()` query --
|
||||
a SELECT of just the content column. Distinct from get_versions()'s own,
|
||||
unrelated per-instance metadata query (id/checksum/added/etc, no
|
||||
content) run to build the "versions" response field, which isn't part
|
||||
of what this test file covers.
|
||||
"""
|
||||
return [
|
||||
q
|
||||
for q in ctx.captured_queries
|
||||
if q["sql"].startswith('SELECT "documents_document"."content" FROM')
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.django_db
|
||||
class TestTrashAndGlobalSearchEffectiveContentIsNeverPerInstance:
|
||||
"""
|
||||
TrashView and GlobalSearchView serialize Document instances with
|
||||
DocumentSerializer too, but build their querysets independently of
|
||||
DocumentViewSet.get_queryset(). TrashView doesn't display content at all,
|
||||
so it keeps the document's own unresolved content; GlobalSearchView
|
||||
annotates effective_content itself, so it shows the latest version's.
|
||||
Neither should ever fall back to a per-instance query.
|
||||
"""
|
||||
|
||||
def test_trash_list_shows_unresolved_content_with_no_extra_query(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A trashed root document whose own content differs from what a
|
||||
version would have had (also trashed, deletion cascades)
|
||||
WHEN:
|
||||
- Listing trash
|
||||
THEN:
|
||||
- The response shows the document's own content
|
||||
- Nothing ever queries for versions to resolve it
|
||||
"""
|
||||
root = DocumentFactory(content="own-content")
|
||||
DocumentFactory(
|
||||
root_document=root,
|
||||
version_index=1,
|
||||
content="version-content",
|
||||
)
|
||||
root.delete()
|
||||
|
||||
with CaptureQueriesContext(connection) as ctx:
|
||||
response = admin_client.get("/api/trash/")
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
[result] = [r for r in response.data["results"] if r["id"] == root.id]
|
||||
assert result["content"] == "own-content"
|
||||
assert _get_effective_content_fallback_queries(ctx) == []
|
||||
|
||||
def test_global_search_db_only_shows_latest_version_content_with_no_extra_query(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A root document, findable by title, whose own content differs
|
||||
from its latest version's
|
||||
WHEN:
|
||||
- Using the global search endpoint's db_only mode
|
||||
THEN:
|
||||
- The response shows the latest version's content, resolved by
|
||||
GlobalSearchView's own effective_content annotation
|
||||
- There is no per-instance fallback query
|
||||
"""
|
||||
root = DocumentFactory(title="findme", content="own-content")
|
||||
DocumentFactory(
|
||||
root_document=root,
|
||||
version_index=1,
|
||||
content="version-content",
|
||||
)
|
||||
|
||||
with CaptureQueriesContext(connection) as ctx:
|
||||
response = admin_client.get(
|
||||
"/api/search/?query=findme&db_only=true",
|
||||
)
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
[result] = [d for d in response.data["documents"] if d["id"] == root.id]
|
||||
assert result["content"] == "version-content"
|
||||
assert _get_effective_content_fallback_queries(ctx) == []
|
||||
@@ -110,7 +110,7 @@ class TestDocument(TestCase):
|
||||
checksum="checksum",
|
||||
mime_type="application/pdf",
|
||||
)
|
||||
Document.objects.create(
|
||||
version = Document.objects.create(
|
||||
root_document=root,
|
||||
correspondent=root.correspondent,
|
||||
title="Version",
|
||||
@@ -124,6 +124,10 @@ class TestDocument(TestCase):
|
||||
self.assertEqual(Document.objects.count(), 0)
|
||||
self.assertEqual(Document.deleted_objects.count(), 2)
|
||||
|
||||
root.restore(strict=False)
|
||||
|
||||
self.assertTrue(Document.objects.filter(pk=version.pk).exists())
|
||||
|
||||
def test_file_name(self) -> None:
|
||||
doc = Document(
|
||||
mime_type="application/pdf",
|
||||
|
||||
@@ -43,7 +43,7 @@ if TYPE_CHECKING:
|
||||
from collections.abc import Generator
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
from pytest_django.fixtures import Settings
|
||||
from pytest_mock import MockerFixture
|
||||
|
||||
|
||||
@@ -136,6 +136,23 @@ def wait_for_mock_call(
|
||||
return False
|
||||
|
||||
|
||||
def sleep_past_stability(
|
||||
owner: FileStabilityTracker | ConsumerThread,
|
||||
*,
|
||||
windows: float = 1.5,
|
||||
) -> None:
|
||||
"""
|
||||
Block until a tracked file's stability window has certainly elapsed.
|
||||
|
||||
Args:
|
||||
owner: The tracker, or the consumer thread running one, whose
|
||||
configured stability delay sets the wait.
|
||||
windows: How many stability windows to wait, giving slop for a slow
|
||||
or loaded test runner.
|
||||
"""
|
||||
sleep(owner.stability_delay * windows)
|
||||
|
||||
|
||||
class TestTrackedFile:
|
||||
"""Tests for the TrackedFile dataclass."""
|
||||
|
||||
@@ -261,6 +278,56 @@ class TestFileStabilityTracker:
|
||||
assert len(stable) == 0
|
||||
assert stability_tracker.pending_count == 1
|
||||
|
||||
def test_get_stable_files_skips_empty_file(
|
||||
self,
|
||||
stability_tracker: FileStabilityTracker,
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A zero byte file, tracked and past its stability delay
|
||||
WHEN:
|
||||
- Stable files are collected
|
||||
THEN:
|
||||
- The file is not yielded for consumption
|
||||
- The file is dropped from tracking rather than held, so an
|
||||
abandoned placeholder does not keep the watch loop awake
|
||||
"""
|
||||
empty = tmp_path / "scan.pdf"
|
||||
empty.write_bytes(b"")
|
||||
stability_tracker.track(empty, Change.added)
|
||||
sleep_past_stability(stability_tracker)
|
||||
|
||||
stable = list(stability_tracker.get_stable_files())
|
||||
|
||||
assert stable == []
|
||||
assert stability_tracker.pending_count == 0
|
||||
|
||||
def test_empty_file_is_yielded_once_content_arrives(
|
||||
self,
|
||||
stability_tracker: FileStabilityTracker,
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A zero byte file which was dropped from tracking while empty
|
||||
WHEN:
|
||||
- The writer fills the file and a new event re-tracks it
|
||||
THEN:
|
||||
- The file is yielded for consumption once it is stable
|
||||
"""
|
||||
target = tmp_path / "scan.pdf"
|
||||
target.write_bytes(b"")
|
||||
stability_tracker.track(target, Change.added)
|
||||
sleep_past_stability(stability_tracker)
|
||||
assert list(stability_tracker.get_stable_files()) == []
|
||||
|
||||
target.write_bytes(b"%PDF-1.4 content")
|
||||
stability_tracker.track(target, Change.modified)
|
||||
sleep_past_stability(stability_tracker)
|
||||
|
||||
assert list(stability_tracker.get_stable_files()) == [target]
|
||||
|
||||
def test_get_stable_files_deleted_during_check(self, temp_file: Path) -> None:
|
||||
"""Test deleted file is not returned during stability check."""
|
||||
tracker = FileStabilityTracker(stability_delay=0.1)
|
||||
@@ -605,7 +672,7 @@ class TestCommandValidation:
|
||||
|
||||
def test_raises_for_missing_consumption_dir(
|
||||
self,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> None:
|
||||
"""Test command raises error when directory is not provided."""
|
||||
settings.CONSUMPTION_DIR = None
|
||||
@@ -639,7 +706,7 @@ class TestCommandOneshot:
|
||||
scratch_dir: Path,
|
||||
sample_pdf: Path,
|
||||
mock_consume_file_delay: MagicMock,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> None:
|
||||
"""Test oneshot mode processes existing files."""
|
||||
target = consumption_dir / "document.pdf"
|
||||
@@ -659,7 +726,7 @@ class TestCommandOneshot:
|
||||
scratch_dir: Path,
|
||||
sample_pdf: Path,
|
||||
mock_consume_file_delay: MagicMock,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> None:
|
||||
"""Test oneshot mode processes files recursively."""
|
||||
subdir = consumption_dir / "subdir"
|
||||
@@ -681,7 +748,7 @@ class TestCommandOneshot:
|
||||
consumption_dir: Path,
|
||||
scratch_dir: Path,
|
||||
mock_consume_file_delay: MagicMock,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> None:
|
||||
"""Test oneshot mode ignores unsupported file extensions."""
|
||||
target = consumption_dir / "document.xyz"
|
||||
@@ -879,6 +946,51 @@ class TestCommandWatch:
|
||||
|
||||
mock_consume_file_delay.apply_async.assert_called()
|
||||
|
||||
def test_scanner_placeholder_is_not_consumed_while_empty(
|
||||
self,
|
||||
consumption_dir: Path,
|
||||
sample_pdf: Path,
|
||||
mock_consume_file_delay: MagicMock,
|
||||
start_consumer: Callable[..., ConsumerThread],
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A scanner which creates a zero byte placeholder and only writes
|
||||
the page some time later (GH discussion #13969)
|
||||
WHEN:
|
||||
- The placeholder sits untouched well past the stability delay
|
||||
- The scanner then writes the real content
|
||||
THEN:
|
||||
- The empty placeholder is never queued, as it could only fail
|
||||
with "Unsupported mime type inode/x-empty"
|
||||
- The file is queued exactly once, when the content lands
|
||||
"""
|
||||
thread = start_consumer(stability_delay=0.2)
|
||||
|
||||
target = consumption_dir / "scan.pdf"
|
||||
target.write_bytes(b"") # the scanner's placeholder
|
||||
|
||||
# Well past the stability delay: the old behaviour queued it here.
|
||||
sleep_past_stability(thread, windows=5)
|
||||
if thread.exception:
|
||||
raise thread.exception
|
||||
assert mock_consume_file_delay.apply_async.call_count == 0
|
||||
|
||||
shutil.copy(sample_pdf, target) # the scanner finishes the page
|
||||
|
||||
assert wait_for_mock_call(
|
||||
mock_consume_file_delay.apply_async,
|
||||
timeout_s=5.0,
|
||||
)
|
||||
if thread.exception:
|
||||
raise thread.exception
|
||||
|
||||
assert mock_consume_file_delay.apply_async.call_count == 1
|
||||
queued_doc = mock_consume_file_delay.apply_async.call_args.kwargs["kwargs"][
|
||||
"input_doc"
|
||||
]
|
||||
assert queued_doc.original_file.name == "scan.pdf"
|
||||
|
||||
def test_ignores_macos_files(
|
||||
self,
|
||||
consumption_dir: Path,
|
||||
@@ -1256,7 +1368,7 @@ class TestProcessExistingFilesQueued:
|
||||
consumption_dir: Path,
|
||||
sample_pdf: Path,
|
||||
mock_consume_file_delay: MagicMock,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
) -> None:
|
||||
"""The set returned seeds the rescan's queued set, avoiding re-queue."""
|
||||
target = consumption_dir / "document.pdf"
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
from collections.abc import Generator
|
||||
|
||||
import pytest
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
from pytest_django.fixtures import Settings
|
||||
|
||||
from documents.parsers import get_default_file_extension
|
||||
from documents.parsers import get_supported_file_extensions
|
||||
@@ -14,7 +14,7 @@ from paperless.parsers.tika import TikaDocumentParser
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def _tika_registry(settings: SettingsWrapper) -> Generator[None, None, None]:
|
||||
def _tika_registry(settings: Settings) -> Generator[None, None, None]:
|
||||
"""
|
||||
Rebuild the parser registry with Tika enabled for the duration of the
|
||||
test, then reset on exit so other tests see the default (Tika-disabled)
|
||||
|
||||
@@ -309,6 +309,9 @@ class TestEmailDocumentPermissionBoundary:
|
||||
):
|
||||
owner = User.objects.create_user(username="owner")
|
||||
requester = User.objects.create_user(username="requester")
|
||||
requester.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
rest_api_client.force_authenticate(user=requester)
|
||||
hidden = DocumentFactory(owner=owner)
|
||||
|
||||
@@ -364,6 +367,27 @@ class TestBulkEditChangePermissionBoundary:
|
||||
|
||||
@pytest.mark.django_db
|
||||
class TestBulkDownloadPermissionChecksRootDocument:
|
||||
def test_download_requires_global_view_permission(
|
||||
self,
|
||||
rest_api_client,
|
||||
paperless_dirs,
|
||||
_media_settings,
|
||||
):
|
||||
owner = User.objects.create_user(username="owner")
|
||||
requester = User.objects.create_user(username="requester")
|
||||
root = DocumentFactory(owner=owner)
|
||||
root.source_path.write_bytes(b"%PDF-1.4 test")
|
||||
assign_perm("view_document", requester, root)
|
||||
rest_api_client.force_authenticate(user=requester)
|
||||
|
||||
response = rest_api_client.post(
|
||||
"/api/documents/bulk_download/",
|
||||
{"documents": [root.pk]},
|
||||
format="json",
|
||||
)
|
||||
|
||||
assert response.status_code == HTTPStatus.FORBIDDEN
|
||||
|
||||
def test_permission_checked_on_root_not_on_version(
|
||||
self,
|
||||
rest_api_client,
|
||||
@@ -372,6 +396,9 @@ class TestBulkDownloadPermissionChecksRootDocument:
|
||||
):
|
||||
owner = User.objects.create_user(username="owner")
|
||||
requester = User.objects.create_user(username="requester")
|
||||
requester.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
rest_api_client.force_authenticate(user=requester)
|
||||
root = DocumentFactory(owner=owner)
|
||||
# a version of root that the requester has NOT been individually granted
|
||||
@@ -396,6 +423,9 @@ class TestBulkDownloadPermissionChecksRootDocument:
|
||||
# `stranger` case) can't tell the two apart, since they're denied
|
||||
# either way.
|
||||
version_only_grantee = User.objects.create_user(username="version_only_grantee")
|
||||
version_only_grantee.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
assign_perm("view_document", version_only_grantee, version)
|
||||
rest_api_client.force_authenticate(user=version_only_grantee)
|
||||
response = rest_api_client.post(
|
||||
@@ -417,6 +447,9 @@ class TestTrashRestorePermissionBoundary:
|
||||
):
|
||||
owner = User.objects.create_user(username="owner")
|
||||
requester = User.objects.create_user(username="requester")
|
||||
requester.user_permissions.add(
|
||||
Permission.objects.get(codename="delete_document"),
|
||||
)
|
||||
rest_api_client.force_authenticate(user=requester)
|
||||
doc = DocumentFactory(owner=owner)
|
||||
assign_perm("view_document", requester, doc) # view only, NOT delete
|
||||
@@ -435,6 +468,9 @@ class TestTrashRestorePermissionBoundary:
|
||||
):
|
||||
owner = User.objects.create_user(username="owner")
|
||||
requester = User.objects.create_user(username="requester")
|
||||
requester.user_permissions.add(
|
||||
Permission.objects.get(codename="delete_document"),
|
||||
)
|
||||
rest_api_client.force_authenticate(user=requester)
|
||||
doc = DocumentFactory(owner=owner)
|
||||
assign_perm("delete_document", requester, doc)
|
||||
@@ -447,6 +483,22 @@ class TestTrashRestorePermissionBoundary:
|
||||
)
|
||||
assert response.status_code == HTTPStatus.OK
|
||||
|
||||
def test_restore_requires_global_delete_permission(self, rest_api_client):
|
||||
owner = User.objects.create_user(username="owner")
|
||||
requester = User.objects.create_user(username="requester")
|
||||
rest_api_client.force_authenticate(user=requester)
|
||||
doc = DocumentFactory(owner=owner)
|
||||
assign_perm("delete_document", requester, doc)
|
||||
doc.delete()
|
||||
|
||||
response = rest_api_client.post(
|
||||
"/api/trash/",
|
||||
{"documents": [doc.pk], "action": "restore"},
|
||||
format="json",
|
||||
)
|
||||
|
||||
assert response.status_code == HTTPStatus.FORBIDDEN
|
||||
|
||||
|
||||
@pytest.mark.django_db
|
||||
class TestTrashViewExcludesExplicitlyGrantedDocuments:
|
||||
@@ -463,6 +515,9 @@ class TestTrashViewExcludesExplicitlyGrantedDocuments:
|
||||
def test_explicit_grant_does_not_leak_trashed_document(self, rest_api_client):
|
||||
owner = User.objects.create_user(username="trash_owner")
|
||||
grantee = User.objects.create_user(username="trash_grantee")
|
||||
grantee.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
doc = DocumentFactory(owner=owner)
|
||||
doc.delete() # soft delete
|
||||
assign_perm("view_document", grantee, doc)
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import pytest
|
||||
import regex
|
||||
from django.conf import settings
|
||||
from pytest_mock import MockerFixture
|
||||
|
||||
from documents.regex import safe_regex_finditer
|
||||
@@ -9,6 +10,12 @@ from documents.regex import safe_regex_sub
|
||||
from documents.regex import validate_regex_pattern
|
||||
|
||||
|
||||
def test_regex_timeout_uses_configured_setting() -> None:
|
||||
from documents.regex import REGEX_TIMEOUT_SECONDS
|
||||
|
||||
assert REGEX_TIMEOUT_SECONDS == settings.MATCH_REGEX_TIMEOUT_SECONDS
|
||||
|
||||
|
||||
class TestValidateRegexPattern:
|
||||
def test_valid_pattern(self) -> None:
|
||||
validate_regex_pattern(r"\d+")
|
||||
|
||||
@@ -6,8 +6,10 @@ from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
from django.conf import settings
|
||||
from django.contrib.auth.models import Permission
|
||||
from django.contrib.auth.models import User
|
||||
from django.utils import timezone
|
||||
from guardian.shortcuts import assign_perm
|
||||
from rest_framework import serializers
|
||||
from rest_framework import status
|
||||
from rest_framework.test import APITestCase
|
||||
@@ -48,6 +50,37 @@ class ShareLinkBundleAPITests(DirectoriesMixin, APITestCase):
|
||||
delay_mock.assert_called_once()
|
||||
self.assertEqual(delay_mock.call_args.kwargs["kwargs"]["bundle_id"], bundle.pk)
|
||||
|
||||
@mock.patch("documents.views.build_share_link_bundle.apply_async")
|
||||
def test_create_bundle_requires_global_document_view_permission(
|
||||
self,
|
||||
delay_mock,
|
||||
) -> None:
|
||||
owner = User.objects.create_user(username="document_owner")
|
||||
requester = User.objects.create_user(username="bundle_creator")
|
||||
requester.user_permissions.add(
|
||||
Permission.objects.get(codename="add_sharelinkbundle"),
|
||||
)
|
||||
document = DocumentFactory.create(owner=owner)
|
||||
assign_perm("view_document", requester, document)
|
||||
self.client.force_authenticate(requester)
|
||||
payload = {
|
||||
"document_ids": [document.pk],
|
||||
"file_version": ShareLink.FileVersion.ARCHIVE,
|
||||
"expiration_days": 7,
|
||||
}
|
||||
|
||||
response = self.client.post(self.ENDPOINT, payload, format="json")
|
||||
self.assertEqual(response.status_code, status.HTTP_403_FORBIDDEN)
|
||||
|
||||
requester.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
requester = User.objects.get(pk=requester.pk)
|
||||
self.client.force_authenticate(requester)
|
||||
response = self.client.post(self.ENDPOINT, payload, format="json")
|
||||
self.assertEqual(response.status_code, status.HTTP_201_CREATED)
|
||||
delay_mock.assert_called_once()
|
||||
|
||||
def test_create_bundle_rejects_missing_documents(self) -> None:
|
||||
payload = {
|
||||
"document_ids": [9999],
|
||||
|
||||
@@ -2,6 +2,7 @@ from unittest import mock
|
||||
|
||||
from django.contrib.auth.models import Permission
|
||||
from django.contrib.auth.models import User
|
||||
from rest_framework import status
|
||||
from rest_framework.test import APITestCase
|
||||
|
||||
from documents import bulk_edit
|
||||
@@ -108,6 +109,44 @@ class TestTagHierarchy(DirectoriesMixin, APITestCase):
|
||||
self.document.refresh_from_db()
|
||||
assert self.document.tags.count() == 0
|
||||
|
||||
def test_remove_inbox_tags_removes_nested_children(self) -> None:
|
||||
inbox = Tag.objects.create(name="Inbox", is_inbox_tag=True)
|
||||
nested = Tag.objects.create(name="Nested", tn_parent=inbox)
|
||||
self.document.add_nested_tags([nested])
|
||||
|
||||
resp = self.client.patch(
|
||||
f"/api/documents/{self.document.pk}/",
|
||||
{"title": "new title", "remove_inbox_tags": True},
|
||||
format="json",
|
||||
)
|
||||
assert resp.status_code == status.HTTP_200_OK
|
||||
self.document.refresh_from_db()
|
||||
assert self.document.tags.count() == 0
|
||||
|
||||
# A subsequent save must not re-add the inbox tag as an ancestor
|
||||
resp = self.client.patch(
|
||||
f"/api/documents/{self.document.pk}/",
|
||||
{"title": "another title", "tags": [], "remove_inbox_tags": True},
|
||||
format="json",
|
||||
)
|
||||
assert resp.status_code == status.HTTP_200_OK
|
||||
self.document.refresh_from_db()
|
||||
assert self.document.tags.count() == 0
|
||||
|
||||
def test_remove_inbox_tags_keeps_inbox_when_nested_child_added(self) -> None:
|
||||
inbox = Tag.objects.create(name="Inbox", is_inbox_tag=True)
|
||||
nested = Tag.objects.create(name="Nested", tn_parent=inbox)
|
||||
self.document.add_nested_tags([inbox])
|
||||
|
||||
self.client.patch(
|
||||
f"/api/documents/{self.document.pk}/",
|
||||
{"tags": [nested.pk], "remove_inbox_tags": True},
|
||||
format="json",
|
||||
)
|
||||
self.document.refresh_from_db()
|
||||
tags = set(self.document.tags.values_list("pk", flat=True))
|
||||
assert tags == {inbox.pk, nested.pk}
|
||||
|
||||
def test_bulk_edit_respects_hierarchy(self) -> None:
|
||||
bulk_edit.add_tag([self.document.pk], self.child.pk)
|
||||
self.document.refresh_from_db()
|
||||
|
||||
@@ -106,6 +106,17 @@ class TestBeforeTaskPublishHandler:
|
||||
assert task.task_type == PaperlessTask.TaskType.TRAIN_CLASSIFIER
|
||||
assert task.trigger_source == PaperlessTask.TriggerSource.MANUAL
|
||||
|
||||
# A Celery retry republishes with the same task_id; this must not
|
||||
# raise a duplicate-key IntegrityError, and must leave the original
|
||||
# PENDING record alone.
|
||||
send_publish(
|
||||
"documents.tasks.train_classifier",
|
||||
(),
|
||||
{},
|
||||
headers={"id": task_id},
|
||||
)
|
||||
assert PaperlessTask.objects.filter(task_id=task_id).count() == 1
|
||||
|
||||
def test_creates_task_for_sanity_check(self) -> None:
|
||||
task_id = send_publish("documents.tasks.sanity_check", (), {})
|
||||
task = PaperlessTask.objects.get(task_id=task_id)
|
||||
|
||||
@@ -32,6 +32,7 @@ from documents.signals.handlers import update_llm_suggestions_cache
|
||||
from documents.tests.utils import DirectoriesMixin
|
||||
from documents.tests.utils import read_streaming_response
|
||||
from paperless.models import ApplicationConfiguration
|
||||
from paperless_ai.exceptions import LLMProviderError
|
||||
from paperless_ai.exceptions import LLMTimeoutError
|
||||
|
||||
|
||||
@@ -140,6 +141,9 @@ class TestViews(DirectoriesMixin, TestCase):
|
||||
codename__contains="sharelink",
|
||||
)
|
||||
self.user.user_permissions.add(*sharelink_permissions)
|
||||
self.user.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
self.user.save()
|
||||
|
||||
self.client.force_login(self.user)
|
||||
@@ -201,6 +205,9 @@ class TestViews(DirectoriesMixin, TestCase):
|
||||
codename__contains="sharelink",
|
||||
)
|
||||
self.user.user_permissions.add(*sharelink_permissions)
|
||||
self.user.user_permissions.add(
|
||||
Permission.objects.get(codename="view_document"),
|
||||
)
|
||||
self.client.force_login(self.user)
|
||||
|
||||
create_response = self.client.post(
|
||||
@@ -737,6 +744,38 @@ class TestAISuggestions(DirectoriesMixin, TestCase):
|
||||
get_llm_suggestion_cache(self.document.pk, backend="openai-like"),
|
||||
)
|
||||
|
||||
@patch("documents.views.get_ai_document_classification")
|
||||
@override_settings(
|
||||
AI_ENABLED=True,
|
||||
LLM_BACKEND="openai-like",
|
||||
)
|
||||
def test_ai_suggestions_with_llm_provider_error(
|
||||
self,
|
||||
mock_get_ai_classification,
|
||||
) -> None:
|
||||
mock_get_ai_classification.side_effect = LLMProviderError(
|
||||
"confidential provider response",
|
||||
)
|
||||
|
||||
self.client.force_login(user=self.user)
|
||||
response = self.client.get(
|
||||
f"/api/documents/{self.document.pk}/ai_suggestions/",
|
||||
)
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_502_BAD_GATEWAY)
|
||||
self.assertEqual(
|
||||
response.json(),
|
||||
{
|
||||
"ai": [
|
||||
"AI backend rejected the request. Check logs for details.",
|
||||
],
|
||||
},
|
||||
)
|
||||
self.assertNotIn("confidential provider response", response.content.decode())
|
||||
self.assertIsNone(
|
||||
get_llm_suggestion_cache(self.document.pk, backend="openai-like"),
|
||||
)
|
||||
|
||||
@patch("documents.views.get_ai_document_classification")
|
||||
@override_settings(
|
||||
AI_ENABLED=True,
|
||||
|
||||
@@ -23,6 +23,7 @@ from guardian.shortcuts import get_users_with_perms
|
||||
from httpx import ConnectError
|
||||
from httpx import HTTPError
|
||||
from httpx import HTTPStatusError
|
||||
from pytest_django.fixtures import Settings
|
||||
from pytest_httpx import HTTPXMock
|
||||
from rest_framework.test import APIClient
|
||||
from rest_framework.test import APITestCase
|
||||
@@ -38,7 +39,6 @@ from paperless_ai.exceptions import LLMTimeoutError
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from django.db.models import QuerySet
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
|
||||
from documents import tasks
|
||||
from documents.data_models import ConsumableDocument
|
||||
@@ -5356,7 +5356,7 @@ class TestDateWorkflowLocalization(
|
||||
def test_document_consumption_workflow_localization(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
settings: SettingsWrapper,
|
||||
settings: Settings,
|
||||
title_template: str,
|
||||
expected_title: str,
|
||||
) -> None:
|
||||
@@ -5711,6 +5711,39 @@ class TestApplyAISuggestionsWorkflowAction(
|
||||
self.assertEqual(changed, [])
|
||||
self.assertIn("AI is not enabled", "".join(cm.output))
|
||||
|
||||
def test_document_without_content_does_nothing(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document whose OCR content is empty or whitespace-only
|
||||
WHEN:
|
||||
- AI suggestions are applied by a workflow
|
||||
THEN:
|
||||
- The classifier is not called and the document is left unchanged
|
||||
"""
|
||||
action = self.make_action(ai_overwrite_existing=True)
|
||||
|
||||
for content in ("", " \n\t"):
|
||||
with self.subTest(content=content):
|
||||
self.doc.content = content
|
||||
self.doc.save(update_fields=["content"])
|
||||
|
||||
with (
|
||||
mock.patch(
|
||||
"documents.workflows.ai.get_ai_document_classification",
|
||||
) as get_classification,
|
||||
self.assertLogs(
|
||||
"paperless.workflows.ai",
|
||||
level="WARNING",
|
||||
) as cm,
|
||||
):
|
||||
changed = apply_ai_suggestions_to_document(action, self.doc)
|
||||
|
||||
self.assertEqual(changed, [])
|
||||
get_classification.assert_not_called()
|
||||
self.assertIn("has no content", "".join(cm.output))
|
||||
self.doc.refresh_from_db()
|
||||
self.assertEqual(self.doc.title, "original.pdf")
|
||||
|
||||
def test_invalid_configuration_leaves_document_untouched(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
|
||||
Reference in New Issue
Block a user