mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-10-05 07:40:31 +00:00
* Feature: match fuzzy terms in place inside the parsed query Fuzzy matching was a separate clause OR'd in above the query: a flat bag of the query's words, re-parsed through tantivy's own parser, blended beside the exact clause. Nothing around a term reached it, so a fielded term fuzzed across every default field, a filter did not constrain it, and an exclusion had to be hoisted back over the whole blend to stop the clause re-admitting what the query had just excluded. Widen each leaf where it sits instead, through emit()'s rewrite_leaf hook, so fielding, negation, AND, REQUIRE and positive filters constrain the fuzzy match exactly as they constrain the exact one. Each of a leaf's words becomes a Fuzzy leaf on the leaf's own field, boosted to 0.1, beside the leaf and any CJK alternative it already had. * Hello?
194 lines
6.8 KiB
Python
194 lines
6.8 KiB
Python
"""CJK bigram matching in QUERY-mode searches.
|
|
|
|
The bigram fields exist so CJK runs are matchable at all (the default
|
|
analyzers keep a whitespace-free CJK run as one indivisible token), but
|
|
matching them must not widen the query beyond what the user asked for: a
|
|
CJK term the query excludes, or restricts to one field, must not come back
|
|
through them.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import TYPE_CHECKING
|
|
|
|
import pytest
|
|
|
|
if TYPE_CHECKING:
|
|
from collections.abc import Callable
|
|
|
|
from pytest_django.fixtures import SettingsWrapper
|
|
|
|
from documents.models import Document
|
|
|
|
|
|
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
|
|
|
|
|
class TestCjkParseFailureDegradesGracefully:
|
|
def test_a_cjk_run_tantivy_cannot_parse_drops_the_clause_only(self) -> None:
|
|
"""
|
|
GIVEN:
|
|
- A CJK run and an index-like object whose parse_query is
|
|
forced to raise
|
|
WHEN:
|
|
- _parse_cjk_text is called
|
|
THEN:
|
|
- It returns None instead of propagating, so a CJK run tantivy
|
|
cannot parse only drops the bigram clause rather than
|
|
failing the whole query. Broad on purpose (bare except
|
|
Exception), unlike the fuzzy blend's narrower ValueError
|
|
guard: a CJK run is not filtered to a guaranteed-safe token
|
|
set the way the fuzzy blend's word string is, so the exact
|
|
failure mode tantivy could raise here is not pinned down
|
|
"""
|
|
from documents.search._query import _parse_cjk_text
|
|
|
|
class _RaisingIndex:
|
|
def parse_query(self, *args: object, **kwargs: object) -> object:
|
|
raise RuntimeError("synthetic parse failure")
|
|
|
|
assert _parse_cjk_text(_RaisingIndex(), "東京", ["bigram_content"]) is None
|
|
|
|
def test_no_cjk_text_at_all_returns_none_without_parsing(self) -> None:
|
|
"""
|
|
GIVEN:
|
|
- A raw query string with no CJK characters at all
|
|
WHEN:
|
|
- _build_cjk_query (the simple TEXT/TITLE-mode builder) is
|
|
called directly
|
|
THEN:
|
|
- It returns None without ever attempting to parse anything.
|
|
The only real caller already guards this with _has_cjk(),
|
|
so this is defensive: it keeps the function safe to call on
|
|
its own, not a path a real search currently reaches
|
|
"""
|
|
from documents.search._query import _build_cjk_query
|
|
|
|
assert _build_cjk_query(None, "invoice total due", ["bigram_content"]) is None
|
|
|
|
|
|
class TestCjkClauseFollowsTheParsedQuery:
|
|
def test_negated_cjk_term_is_excluded(
|
|
self,
|
|
index_document: Callable[..., Document],
|
|
matched_ids: Callable[[str], set[int]],
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- Two documents both matching "invoice", one whose content
|
|
also contains 漢字
|
|
WHEN:
|
|
- "invoice NOT 漢字" is searched
|
|
THEN:
|
|
- Only the document without 漢字 matches; 'invoice NOT 漢字'
|
|
must not return the document containing 漢字
|
|
"""
|
|
with_cjk = index_document(
|
|
title="Invoice A",
|
|
content="invoice total 漢字",
|
|
)
|
|
without_cjk = index_document(
|
|
title="Invoice B",
|
|
content="invoice total only",
|
|
)
|
|
|
|
assert matched_ids("invoice") == {with_cjk.pk, without_cjk.pk}
|
|
assert matched_ids("invoice NOT 漢字") == {without_cjk.pk}
|
|
|
|
@pytest.mark.parametrize(
|
|
("threshold", "expected"),
|
|
[
|
|
pytest.param(None, {"titled"}, id="fuzzy_off"),
|
|
pytest.param(0.0, {"titled"}, id="fuzzy_on"),
|
|
],
|
|
)
|
|
def test_fielded_cjk_term_searches_only_that_field(
|
|
self,
|
|
index_document: Callable[..., Document],
|
|
matched_ids: Callable[[str], set[int]],
|
|
settings: SettingsWrapper,
|
|
threshold: float | None,
|
|
expected: set[str],
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- One document with 東京 in its title, another with 東京 only
|
|
in its content, and ADVANCED_FUZZY_SEARCH_THRESHOLD either
|
|
off or on
|
|
WHEN:
|
|
- "title:東京" is searched
|
|
THEN:
|
|
- Only the titled document matches, whether fuzzy is on or
|
|
off: both the CJK alternative and the fuzzy alternative are
|
|
widened in place on the fielded leaf, so 'title:東京' still
|
|
must not match a document whose 東京 is only in the content
|
|
"""
|
|
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = threshold
|
|
content_only = index_document(
|
|
title="Tokyo report",
|
|
content="東京都の人口は約1400万人です",
|
|
)
|
|
titled = index_document(
|
|
title="東京都の報告書",
|
|
content="an english summary",
|
|
)
|
|
pks = {"titled": titled.pk, "content_only": content_only.pk}
|
|
|
|
assert matched_ids("東京") == set(pks.values())
|
|
assert matched_ids("title:東京") == {pks[label] for label in expected}
|
|
|
|
def test_cjk_on_a_non_default_field_is_not_widened(
|
|
self,
|
|
index_document: Callable[..., Document],
|
|
matched_ids: Callable[[str], set[int]],
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- A document with 東京 in its content
|
|
WHEN:
|
|
- "notes:東京" is searched (a field outside the default
|
|
search fields)
|
|
THEN:
|
|
- Nothing matches; a CJK term restricted to a field outside
|
|
the default search fields has no bigram companion to widen
|
|
to, so it must not fall back to matching 東京 in the
|
|
content
|
|
"""
|
|
index_document(
|
|
title="Tokyo report",
|
|
content="東京都の人口は約1400万人です",
|
|
)
|
|
|
|
assert matched_ids("notes:東京") == set()
|
|
|
|
def test_bare_cjk_term_still_matches_every_default_field(
|
|
self,
|
|
index_document: Callable[..., Document],
|
|
matched_ids: Callable[[str], set[int]],
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- One document with 重要 in its content, another with 重要 in
|
|
its title
|
|
WHEN:
|
|
- "重要" and "重要 OR report" are each searched unfielded
|
|
THEN:
|
|
- Both documents match either way; bigram matching exists
|
|
so that an unfielded CJK run matches wherever it is
|
|
indexed, and does so alongside a latin term
|
|
"""
|
|
in_content = index_document(
|
|
title="report",
|
|
content="本文に重要な情報",
|
|
)
|
|
in_title = index_document(
|
|
title="重要な報告書",
|
|
content="english only",
|
|
)
|
|
|
|
assert matched_ids("重要") == {in_content.pk, in_title.pk}
|
|
assert matched_ids("重要 OR report") == {
|
|
in_content.pk,
|
|
in_title.pk,
|
|
}
|