"""CJK bigram matching in QUERY-mode searches. The bigram fields exist so CJK runs are matchable at all (the default analyzers keep a whitespace-free CJK run as one indivisible token), but matching them must not widen the query beyond what the user asked for: a CJK term the query excludes, or restricts to one field, must not come back through them. """ from __future__ import annotations from typing import TYPE_CHECKING import pytest if TYPE_CHECKING: from collections.abc import Callable from pytest_django.fixtures import SettingsWrapper from documents.models import Document pytestmark = [pytest.mark.search, pytest.mark.django_db] class TestCjkParseFailureDegradesGracefully: def test_a_cjk_run_tantivy_cannot_parse_drops_the_clause_only(self) -> None: """ GIVEN: - A CJK run and an index-like object whose parse_query is forced to raise WHEN: - _parse_cjk_text is called THEN: - It returns None instead of propagating, so a CJK run tantivy cannot parse only drops the bigram clause rather than failing the whole query. Broad on purpose (bare except Exception), unlike the fuzzy blend's narrower ValueError guard: a CJK run is not filtered to a guaranteed-safe token set the way the fuzzy blend's word string is, so the exact failure mode tantivy could raise here is not pinned down """ from documents.search._query import _parse_cjk_text class _RaisingIndex: def parse_query(self, *args: object, **kwargs: object) -> object: raise RuntimeError("synthetic parse failure") assert _parse_cjk_text(_RaisingIndex(), "東京", ["bigram_content"]) is None def test_no_cjk_text_at_all_returns_none_without_parsing(self) -> None: """ GIVEN: - A raw query string with no CJK characters at all WHEN: - _build_cjk_query (the simple TEXT/TITLE-mode builder) is called directly THEN: - It returns None without ever attempting to parse anything. The only real caller already guards this with _has_cjk(), so this is defensive: it keeps the function safe to call on its own, not a path a real search currently reaches """ from documents.search._query import _build_cjk_query assert _build_cjk_query(None, "invoice total due", ["bigram_content"]) is None class TestCjkClauseFollowsTheParsedQuery: def test_negated_cjk_term_is_excluded( self, index_document: Callable[..., Document], matched_ids: Callable[[str], set[int]], ) -> None: """ GIVEN: - Two documents both matching "invoice", one whose content also contains 漢字 WHEN: - "invoice NOT 漢字" is searched THEN: - Only the document without 漢字 matches; 'invoice NOT 漢字' must not return the document containing 漢字 """ with_cjk = index_document( title="Invoice A", content="invoice total 漢字", ) without_cjk = index_document( title="Invoice B", content="invoice total only", ) assert matched_ids("invoice") == {with_cjk.pk, without_cjk.pk} assert matched_ids("invoice NOT 漢字") == {without_cjk.pk} @pytest.mark.parametrize( ("threshold", "expected"), [ pytest.param(None, {"titled"}, id="fuzzy_off"), pytest.param(0.0, {"titled"}, id="fuzzy_on"), ], ) def test_fielded_cjk_term_searches_only_that_field( self, index_document: Callable[..., Document], matched_ids: Callable[[str], set[int]], settings: SettingsWrapper, threshold: float | None, expected: set[str], ) -> None: """ GIVEN: - One document with 東京 in its title, another with 東京 only in its content, and ADVANCED_FUZZY_SEARCH_THRESHOLD either off or on WHEN: - "title:東京" is searched THEN: - Only the titled document matches, whether fuzzy is on or off: both the CJK alternative and the fuzzy alternative are widened in place on the fielded leaf, so 'title:東京' still must not match a document whose 東京 is only in the content """ settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = threshold content_only = index_document( title="Tokyo report", content="東京都の人口は約1400万人です", ) titled = index_document( title="東京都の報告書", content="an english summary", ) pks = {"titled": titled.pk, "content_only": content_only.pk} assert matched_ids("東京") == set(pks.values()) assert matched_ids("title:東京") == {pks[label] for label in expected} def test_cjk_on_a_non_default_field_is_not_widened( self, index_document: Callable[..., Document], matched_ids: Callable[[str], set[int]], ) -> None: """ GIVEN: - A document with 東京 in its content WHEN: - "notes:東京" is searched (a field outside the default search fields) THEN: - Nothing matches; a CJK term restricted to a field outside the default search fields has no bigram companion to widen to, so it must not fall back to matching 東京 in the content """ index_document( title="Tokyo report", content="東京都の人口は約1400万人です", ) assert matched_ids("notes:東京") == set() def test_bare_cjk_term_still_matches_every_default_field( self, index_document: Callable[..., Document], matched_ids: Callable[[str], set[int]], ) -> None: """ GIVEN: - One document with 重要 in its content, another with 重要 in its title WHEN: - "重要" and "重要 OR report" are each searched unfielded THEN: - Both documents match either way; bigram matching exists so that an unfielded CJK run matches wherever it is indexed, and does so alongside a latin term """ in_content = index_document( title="report", content="本文に重要な情報", ) in_title = index_document( title="重要な報告書", content="english only", ) assert matched_ids("重要") == {in_content.pk, in_title.pk} assert matched_ids("重要 OR report") == { in_content.pk, in_title.pk, }