test(search): collapse the date survivors onto one representative

Follow-up to 98bd31f7d, applying the stricter reading of "one representative
case per integration seam". The trimmed unit-abbreviation and reversed-range
survivors reached the same seams as test_compact_date_forms.py by different
spellings: no mutation kills either without also killing the compact case,
and the compact case additionally fails when queries are resolved in a fixed
non-local timezone, which neither of the other two catches because now-5h and
now+/-1h are timezone-invariant. Verified before deleting: the compact
survivor fails under all three mutations (fixed non-local timezone, `added`
registered as TEXT instead of DATETIME, `added` indexed at day precision).

The spellings themselves are asserted by the library, per unit in
tests/test_relative_date_unit_abbreviations.py and for both range kinds in
tests/test_parser_dates.py. The reversed-range file's other purpose, to invert
as the signal that the library's bound swap landed, is spent: it inverted.

Also give TestTagCommaList a docstring saying what it is. It is a docs-contract
regression test for a published spelling, not proof of paperless's field
configuration: removing comma_values from the tag FieldSpec leaves it passing,
because the analyzer splits the literal value into the same tokens. The
registry fact is owned by test_registry.py.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
stumpylog
2026-08-20 15:49:31 -07:00
co-authored by Claude Opus 5
parent cf1cce5f9d
commit caf87136f6
3 changed files with 14 additions and 157 deletions
@@ -113,6 +113,20 @@ class TestPhraseSearch:
class TestTagCommaList:
"""``tag:bills,unpaid`` is published syntax (docs/usage.md), so this checks
that the documented spelling still returns what the docs promise: only the
document carrying every listed tag.
It is deliberately not proof of paperless's field configuration, and must
not be read as such. Removing ``comma_values`` from the ``tag`` FieldSpec
leaves this test passing, because paperless's analyzer splits the literal
value "bills,unpaid" into the same two tokens the value-list reading
produces, so the two readings select the same documents. The registry fact
-- that ``tag`` opts in and no other field does -- is observable only at
the registry, and is owned by test_registry.py's
``test_tag_is_comma_values``/``test_correspondent_is_not_comma_values``.
"""
def test_comma_list_requires_every_listed_tag(
self,
backend: TantivyBackend,
@@ -1,81 +0,0 @@
"""A reversed relative range, resolved end to end against the index.
whoosh-compat owns the bound swap itself and asserts it directly, for both
the absolute and the relative spelling
(``test_reversed_relative_range_swaps_like_the_absolute_case`` in
``tests/test_parser_dates.py``, DIVERGENCES.md entry 53). Paperless does not
rewrite reversed ranges anywhere; ``documents/search/_query.py`` passes the
query through untouched.
What is kept here is the relative spelling only, because it is the one whose
bounds depend on paperless's own plumbing: ``now±1h`` is resolved against the
timezone paperless hands the parser, so the emitted window is paperless's
result rather than the library's alone. The corpus separates a document
inside the two-hour window from one that is outside it but inside the wider
window a non-swapping reading produces, so a bound that resolves in the wrong
timezone, or a swap that does not happen, matches the wrong set.
"""
from __future__ import annotations
from datetime import UTC
from datetime import datetime
from typing import TYPE_CHECKING
import pytest
import time_machine
from documents.models import Document
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
FROZEN_NOW = datetime(2026, 6, 15, 12, 0, tzinfo=UTC)
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
@pytest.fixture
def docs(backend: TantivyBackend) -> dict[str, int]:
with time_machine.travel(FROZEN_NOW, tick=False):
return {
# Inside the 2h window [11:00, 13:00] that both the forward and
# the reversed spelling resolve to.
"in_window": _index(
backend,
title="Forward window doc",
content="x",
checksum="reversed-relative-forward",
added=FROZEN_NOW,
).pk,
# Outside that window, but inside the wider window a reversed
# range that day-bumped its upper bound instead of swapping
# would reach.
"outside_the_window": _index(
backend,
title="Later same-week doc",
content="x",
checksum="reversed-relative-outside",
added=datetime(2026, 6, 16, 8, 0, tzinfo=UTC),
).pk,
}
def test_reversed_relative_range_matches_the_two_hour_window(
backend: TantivyBackend,
docs: dict[str, int],
) -> None:
with time_machine.travel(FROZEN_NOW, tick=False):
assert _matched_ids(backend, "added:[now+1h to now-1h]") == {
docs["in_window"],
}
@@ -1,76 +0,0 @@
"""One relative unit abbreviation, resolved end to end against the index.
whoosh-compat owns which unit words exist and what offset each resolves to,
and asserts all six (yrs/mos/wks/hrs/mins/secs, quoted and bare) directly in
``tests/test_relative_date_unit_abbreviations.py``. Repeating that matrix here
would only re-prove the library's grammar; "hrs" is kept as a single
representative so that the paperless-side path the library cannot see -- the
``added`` DATETIME fast field, the timezone the query is resolved in and the
range the backend emits -- is still exercised by an abbreviation-spelled
offset rather than only by absolute dates.
The offset is sub-day on purpose: a whole-day or whole-year offset survives a
wrong timezone, five hours does not. The decoy document sits at a different
offset from the same instant, so an offset that resolves to the wrong width
lands on nothing or on the decoy instead of passing by accident.
"""
from __future__ import annotations
from datetime import UTC
from datetime import datetime
from typing import TYPE_CHECKING
import pytest
import time_machine
from documents.models import Document
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
FROZEN_NOW = datetime(2026, 6, 15, 12, 0, tzinfo=UTC)
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
@pytest.fixture
def unit_documents(backend: TantivyBackend) -> dict[str, int]:
"""One document five hours before the frozen instant, one ten minutes
before it, both indexed together so a mis-resolved offset lands on the
wrong document or on none rather than passing coincidentally."""
with time_machine.travel(FROZEN_NOW, tick=False):
return {
"hrs": _index(
backend,
title="Hours doc",
content="x",
checksum="unit-abbrev-hrs",
added=datetime(2026, 6, 15, 7, 0, tzinfo=UTC),
).pk,
"mins": _index(
backend,
title="Minutes doc",
content="x",
checksum="unit-abbrev-mins",
added=datetime(2026, 6, 15, 11, 50, tzinfo=UTC),
).pk,
}
def test_hours_abbreviation_matches_only_its_own_instant(
backend: TantivyBackend,
unit_documents: dict[str, int],
) -> None:
with time_machine.travel(FROZEN_NOW, tick=False):
assert _matched_ids(backend, 'added:"-5hrs"') == {unit_documents["hrs"]}