diff --git a/src/documents/tests/search/test_comma_value_lists.py b/src/documents/tests/search/test_comma_value_lists.py index 3552a97fc..a3701e674 100644 --- a/src/documents/tests/search/test_comma_value_lists.py +++ b/src/documents/tests/search/test_comma_value_lists.py @@ -11,10 +11,19 @@ matches while foo-only/bar-only correspondents do not. The list semantics are AND (a document must carry every listed value), not OR: a foo-only or bar-only document does not match ``tag:foo,bar``. + +A comma is only a value-list delimiter when what follows it is a value. A +comma directly before another field name separates two clauses instead, +on ``comma_values`` and ordinary fields alike -- ``tag:foo,added:2005-03-04`` +is "tagged foo AND added that day", not "tagged both foo and +added:2005-03-04". That reading was asserted before the whoosh-compat +migration and nowhere after it, so it is pinned here. """ from __future__ import annotations +from datetime import UTC +from datetime import datetime from typing import TYPE_CHECKING import pytest @@ -137,3 +146,86 @@ class TestNonCommaValuesFieldTreatsCommaAsLiteralText: # or neither (an AND, since no single correspondent carries both # values). Either way it would NOT match literal_doc alone. assert _matched_ids(backend, "correspondent:foo,bar") == {literal_doc.pk} + + +class TestCommaBeforeAKnownFieldIsAClauseSeparator: + """The discriminating half: the value-list reading and the clause-separator + reading disagree about every query below. + + On ``tag`` (the one ``comma_values`` field) a value-list reading of + ``tag:foo,added:2005-03-04`` demands a tag literally named + "added:2005-03-04", so it matches nothing at all. On ``title`` (not + ``comma_values``) the alternative is the literal reading proven in + TestNonCommaValuesFieldTreatsCommaAsLiteralText, i.e. a title containing + the text "Alpha,tag:foo", which matches nothing either. Both are exact-set + assertions against a corpus that separates the readings, so neither + alternative survives. + """ + + @pytest.fixture + def docs(self, backend: TantivyBackend) -> dict[str, int]: + foo = Tag.objects.create(name="foo") + bar = Tag.objects.create(name="bar") + + both = Document.objects.create( + title="Alpha", + content="x", + checksum="comma-clause-both", + added=datetime(2005, 3, 4, 15, 30, tzinfo=UTC), + ) + both.tags.set([foo]) + backend.add_or_update(both) + + tag_only = Document.objects.create( + title="Alpha", + content="x", + checksum="comma-clause-tag-only", + added=datetime(2010, 7, 1, 15, 30, tzinfo=UTC), + ) + tag_only.tags.set([foo]) + backend.add_or_update(tag_only) + + date_only = Document.objects.create( + title="Beta", + content="x", + checksum="comma-clause-date-only", + added=datetime(2005, 3, 4, 15, 30, tzinfo=UTC), + ) + date_only.tags.set([bar]) + backend.add_or_update(date_only) + + return { + "both": both.pk, + "tag_only": tag_only.pk, + "date_only": date_only.pk, + } + + def test_comma_separates_a_comma_values_field_from_a_date_clause( + self, + backend: TantivyBackend, + docs: dict[str, int], + ) -> None: + assert _matched_ids(backend, "tag:foo,added:2005-03-04") == {docs["both"]} + + def test_comma_separates_an_ordinary_field_from_a_tag_clause( + self, + backend: TantivyBackend, + docs: dict[str, int], + ) -> None: + assert _matched_ids(backend, "title:Alpha,tag:foo") == { + docs["both"], + docs["tag_only"], + } + + def test_each_clause_alone_matches_more_than_the_pair( + self, + backend: TantivyBackend, + docs: dict[str, int], + ) -> None: + """Both clauses must be doing work: if the separator dropped either + side, the pair would match whatever the surviving side matches.""" + assert _matched_ids(backend, "tag:foo") == {docs["both"], docs["tag_only"]} + assert _matched_ids(backend, "added:2005-03-04") == { + docs["both"], + docs["date_only"], + } diff --git a/src/documents/tests/search/test_compact_date_forms.py b/src/documents/tests/search/test_compact_date_forms.py new file mode 100644 index 000000000..ce4cf344a --- /dev/null +++ b/src/documents/tests/search/test_compact_date_forms.py @@ -0,0 +1,107 @@ +"""Whoosh's compact, separator-free date spellings: ``20050304`` and +``20050304153000``. + +Both were asserted before the whoosh-compat migration (the old +``test_8digit_created_date_field_always_uses_utc_midnight`` and +``test_14digit_compact_datetime``) and by the deleted ``_translate.py``'s own +suite. Afterwards the 8-digit form survived only incidentally, in one +pre-existing API test, and the 14-digit form was asserted nowhere -- the +regression class where a spelling silently stops matching with nothing to +notice. + +The two forms differ in width, not just in length: 8 digits is a calendar-day +window, 14 digits a single instant. The corpus separates them, so a form that +degrades into the other one -- or into a non-match -- fails rather than passing +on the one document that would match either way. +""" + +from __future__ import annotations + +from datetime import UTC +from datetime import datetime +from typing import TYPE_CHECKING + +import pytest + +from documents.models import Document + +if TYPE_CHECKING: + from documents.search._backend import TantivyBackend + +pytestmark = [pytest.mark.search, pytest.mark.django_db] + + +def _matched_ids(backend: TantivyBackend, query: str) -> set[int]: + return set(backend.search_ids(query, user=None)) + + +def _index(backend: TantivyBackend, **kwargs: object) -> Document: + doc = Document.objects.create(**kwargs) + backend.add_or_update(doc) + return doc + + +@pytest.fixture +def docs(backend: TantivyBackend) -> dict[str, int]: + return { + "instant": _index( + backend, + title="On the instant", + content="x", + checksum="compact-date-instant", + added=datetime(2005, 3, 4, 15, 30, tzinfo=UTC), + ).pk, + "same_day": _index( + backend, + title="Same day, other hour", + content="x", + checksum="compact-date-same-day", + added=datetime(2005, 3, 4, 9, 0, tzinfo=UTC), + ).pk, + "next_day": _index( + backend, + title="Next day, same hour", + content="x", + checksum="compact-date-next-day", + added=datetime(2005, 3, 5, 15, 30, tzinfo=UTC), + ).pk, + } + + +class TestCompactDateForms: + def test_eight_digits_is_a_calendar_day_window( + self, + backend: TantivyBackend, + docs: dict[str, int], + ) -> None: + assert _matched_ids(backend, "added:20050304") == { + docs["instant"], + docs["same_day"], + } + + def test_eight_digits_agrees_with_the_hyphenated_spelling( + self, + backend: TantivyBackend, + docs: dict[str, int], + ) -> None: + assert _matched_ids(backend, "added:20050304") == _matched_ids( + backend, + "added:2005-03-04", + ) + + def test_fourteen_digits_is_a_single_instant( + self, + backend: TantivyBackend, + docs: dict[str, int], + ) -> None: + # same_day is what tells this apart from the 8-digit form, next_day + # from a form that ignored the time altogether. + assert _matched_ids(backend, "added:20050304153000") == {docs["instant"]} + + def test_fourteen_digits_addresses_the_hour_it_names( + self, + backend: TantivyBackend, + docs: dict[str, int], + ) -> None: + assert _matched_ids(backend, "added:20050304090000") == {docs["same_day"]} + assert _matched_ids(backend, "added:20050305153000") == {docs["next_day"]}