Files
paperless-ngx/src/documents/tests/search/test_cjk_clause.py
T
Trenton H 20d309a413 Enhancement: Match fuzzy terms in place inside the parsed query (#14157)
* Feature: match fuzzy terms in place inside the parsed query

Fuzzy matching was a separate clause OR'd in above the query: a flat bag
of the query's words, re-parsed through tantivy's own parser, blended
beside the exact clause. Nothing around a term reached it, so a fielded
term fuzzed across every default field, a filter did not constrain it, and
an exclusion had to be hoisted back over the whole blend to stop the
clause re-admitting what the query had just excluded.

Widen each leaf where it sits instead, through emit()'s rewrite_leaf hook,
so fielding, negation, AND, REQUIRE and positive filters constrain the
fuzzy match exactly as they constrain the exact one. Each of a leaf's
words becomes a Fuzzy leaf on the leaf's own field, boosted to 0.1, beside
the leaf and any CJK alternative it already had.

* Hello?
2026-09-17 14:55:13 -07:00

194 lines
6.8 KiB
Python

"""CJK bigram matching in QUERY-mode searches.
The bigram fields exist so CJK runs are matchable at all (the default
analyzers keep a whitespace-free CJK run as one indivisible token), but
matching them must not widen the query beyond what the user asked for: a
CJK term the query excludes, or restricts to one field, must not come back
through them.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
if TYPE_CHECKING:
from collections.abc import Callable
from pytest_django.fixtures import SettingsWrapper
from documents.models import Document
pytestmark = [pytest.mark.search, pytest.mark.django_db]
class TestCjkParseFailureDegradesGracefully:
def test_a_cjk_run_tantivy_cannot_parse_drops_the_clause_only(self) -> None:
"""
GIVEN:
- A CJK run and an index-like object whose parse_query is
forced to raise
WHEN:
- _parse_cjk_text is called
THEN:
- It returns None instead of propagating, so a CJK run tantivy
cannot parse only drops the bigram clause rather than
failing the whole query. Broad on purpose (bare except
Exception), unlike the fuzzy blend's narrower ValueError
guard: a CJK run is not filtered to a guaranteed-safe token
set the way the fuzzy blend's word string is, so the exact
failure mode tantivy could raise here is not pinned down
"""
from documents.search._query import _parse_cjk_text
class _RaisingIndex:
def parse_query(self, *args: object, **kwargs: object) -> object:
raise RuntimeError("synthetic parse failure")
assert _parse_cjk_text(_RaisingIndex(), "東京", ["bigram_content"]) is None
def test_no_cjk_text_at_all_returns_none_without_parsing(self) -> None:
"""
GIVEN:
- A raw query string with no CJK characters at all
WHEN:
- _build_cjk_query (the simple TEXT/TITLE-mode builder) is
called directly
THEN:
- It returns None without ever attempting to parse anything.
The only real caller already guards this with _has_cjk(),
so this is defensive: it keeps the function safe to call on
its own, not a path a real search currently reaches
"""
from documents.search._query import _build_cjk_query
assert _build_cjk_query(None, "invoice total due", ["bigram_content"]) is None
class TestCjkClauseFollowsTheParsedQuery:
def test_negated_cjk_term_is_excluded(
self,
index_document: Callable[..., Document],
matched_ids: Callable[[str], set[int]],
) -> None:
"""
GIVEN:
- Two documents both matching "invoice", one whose content
also contains 漢字
WHEN:
- "invoice NOT 漢字" is searched
THEN:
- Only the document without 漢字 matches; 'invoice NOT 漢字'
must not return the document containing 漢字
"""
with_cjk = index_document(
title="Invoice A",
content="invoice total 漢字",
)
without_cjk = index_document(
title="Invoice B",
content="invoice total only",
)
assert matched_ids("invoice") == {with_cjk.pk, without_cjk.pk}
assert matched_ids("invoice NOT 漢字") == {without_cjk.pk}
@pytest.mark.parametrize(
("threshold", "expected"),
[
pytest.param(None, {"titled"}, id="fuzzy_off"),
pytest.param(0.0, {"titled"}, id="fuzzy_on"),
],
)
def test_fielded_cjk_term_searches_only_that_field(
self,
index_document: Callable[..., Document],
matched_ids: Callable[[str], set[int]],
settings: SettingsWrapper,
threshold: float | None,
expected: set[str],
) -> None:
"""
GIVEN:
- One document with 東京 in its title, another with 東京 only
in its content, and ADVANCED_FUZZY_SEARCH_THRESHOLD either
off or on
WHEN:
- "title:東京" is searched
THEN:
- Only the titled document matches, whether fuzzy is on or
off: both the CJK alternative and the fuzzy alternative are
widened in place on the fielded leaf, so 'title:東京' still
must not match a document whose 東京 is only in the content
"""
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = threshold
content_only = index_document(
title="Tokyo report",
content="東京都の人口は約1400万人です",
)
titled = index_document(
title="東京都の報告書",
content="an english summary",
)
pks = {"titled": titled.pk, "content_only": content_only.pk}
assert matched_ids("東京") == set(pks.values())
assert matched_ids("title:東京") == {pks[label] for label in expected}
def test_cjk_on_a_non_default_field_is_not_widened(
self,
index_document: Callable[..., Document],
matched_ids: Callable[[str], set[int]],
) -> None:
"""
GIVEN:
- A document with 東京 in its content
WHEN:
- "notes:東京" is searched (a field outside the default
search fields)
THEN:
- Nothing matches; a CJK term restricted to a field outside
the default search fields has no bigram companion to widen
to, so it must not fall back to matching 東京 in the
content
"""
index_document(
title="Tokyo report",
content="東京都の人口は約1400万人です",
)
assert matched_ids("notes:東京") == set()
def test_bare_cjk_term_still_matches_every_default_field(
self,
index_document: Callable[..., Document],
matched_ids: Callable[[str], set[int]],
) -> None:
"""
GIVEN:
- One document with 重要 in its content, another with 重要 in
its title
WHEN:
- "重要" and "重要 OR report" are each searched unfielded
THEN:
- Both documents match either way; bigram matching exists
so that an unfielded CJK run matches wherever it is
indexed, and does so alongside a latin term
"""
in_content = index_document(
title="report",
content="本文に重要な情報",
)
in_title = index_document(
title="重要な報告書",
content="english only",
)
assert matched_ids("重要") == {in_content.pk, in_title.pk}
assert matched_ids("重要 OR report") == {
in_content.pk,
in_title.pk,
}