mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-08-27 21:23:20 +00:00
EXISTS_REQUIRES_FAST is MISCONFIGURED and reachable from ordinary query text (notes.user:*), so the operator alert added with the Cause routing fired on every such request. An alert that repeats on every user query is one operators learn to filter out, which defeats routing MISCONFIGURED to an operator at all. The condition is a static configuration fact: it stays true until an operator changes the schema and reindexes, so the first log carries the same information as the ten-thousandth. Deduped on (kind, field) in a per-process set; a restart re-logs, re-surfacing the condition after a config change. The 400 is not deduped: every request still gets its response and its message. The key is bounded by the registry, not by query text. emit() only reports MISCONFIGURED for a field it resolved, and FieldRegistry.resolve returns None for any name or JSON subpath the registry does not declare. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
570 lines
22 KiB
Python
570 lines
22 KiB
Python
from __future__ import annotations
|
|
|
|
import logging
|
|
from typing import TYPE_CHECKING
|
|
from typing import Final
|
|
|
|
import regex
|
|
import tantivy
|
|
import whoosh_compat as wc
|
|
from django.conf import settings
|
|
from whoosh_compat.emitters.tantivy_ import emit as tantivy_emit
|
|
from whoosh_compat.errors import Cause
|
|
from whoosh_compat.errors import Diagnostic
|
|
from whoosh_compat.errors import DiagnosticKind
|
|
from whoosh_compat.errors import QueryError
|
|
|
|
from documents.search._errors import InvalidDateQuery
|
|
from documents.search._errors import InvalidNumberQuery
|
|
from documents.search._errors import MultipleSearchQueryErrors
|
|
from documents.search._errors import SearchQueryError
|
|
from documents.search._fields import PUBLIC_FIELDS
|
|
from documents.search._registry import get_field_registry
|
|
from documents.search._tokenizer import simple_search_tokens
|
|
|
|
if TYPE_CHECKING:
|
|
from datetime import tzinfo
|
|
|
|
logger = logging.getLogger("paperless.search")
|
|
|
|
# Maximum seconds any single regex substitution may run.
|
|
# Prevents ReDoS on adversarial user-supplied query strings.
|
|
_REGEX_TIMEOUT: Final[float] = 1.0
|
|
|
|
# Matches CJK/Hangul characters so queries can be routed to bigram fields.
|
|
# Uses Unicode properties to cover all blocks including Extension B+ planes.
|
|
_CJK_RE: Final = regex.compile(r"[\p{Han}\p{Hiragana}\p{Katakana}\p{Hangul}]+")
|
|
|
|
# Multi-word date-keyword phrases whoosh-compat only accepts quoted.
|
|
# Unquoted has always been the honored spelling, so
|
|
# _quote_date_keyword_phrases below inserts the quotes and nothing else.
|
|
# Single-word keywords (today, yesterday) already parse unquoted.
|
|
_DATE_KEYWORD_PHRASES: Final = (
|
|
"previous week",
|
|
"previous month",
|
|
"previous quarter",
|
|
"previous year",
|
|
"this month",
|
|
"this year",
|
|
)
|
|
|
|
# Field names are case-sensitive (matching the parser's own field
|
|
# tagging); the keyword phrase is case-insensitive (matching the date
|
|
# grammar's leniency for the quoted form). Date fields derived from
|
|
# PUBLIC_FIELDS, never hand-listed.
|
|
_DATE_KEYWORD_PHRASE_RE: Final = regex.compile(
|
|
r"\b("
|
|
+ "|".join(
|
|
regex.escape(f.name)
|
|
for f in PUBLIC_FIELDS
|
|
if f.kind in (wc.FieldKind.DATE, wc.FieldKind.DATETIME)
|
|
)
|
|
+ r"):((?i:"
|
|
+ "|".join(_DATE_KEYWORD_PHRASES)
|
|
+ r"))\b",
|
|
)
|
|
|
|
|
|
def _quote_date_keyword_phrases(raw_query: str) -> str:
|
|
"""Quote unquoted multi-word date keyword phrases on date fields.
|
|
|
|
``added:previous month`` becomes ``added:"previous month"``; already-
|
|
quoted spellings, TEXT fields, and standalone words are untouched.
|
|
Only quoting happens here - every date computation stays in
|
|
whoosh-compat's grammar.
|
|
|
|
Not quote-aware: matches anywhere in raw_query, including inside an
|
|
existing quoted phrase (e.g. ``title:"see added:previous month
|
|
notes"`` would get quotes inserted mid-phrase). Accepted as an
|
|
unlikely-in-practice edge case rather than implementing quote-aware
|
|
scanning.
|
|
"""
|
|
return _DATE_KEYWORD_PHRASE_RE.sub(
|
|
r'\1:"\2"',
|
|
raw_query,
|
|
timeout=_REGEX_TIMEOUT,
|
|
)
|
|
|
|
|
|
# notes:/custom_fields: were valid fielded searches before this migration.
|
|
# whoosh-compat's registry only exposes them as JSON subpaths, so a bare
|
|
# prefix would demote to an unfielded text search. Rewrite live to the
|
|
# equivalent subpath (notes: -> notes.note:, custom_fields: ->
|
|
# custom_fields.value:); custom_fields.name: remains available separately.
|
|
# Not preceded by a word character or dot, so subpath spellings and words
|
|
# merely ending in the prefix are untouched.
|
|
_BARE_JSON_PREFIX_RES: Final = (
|
|
(regex.compile(r"(?<![.\w])notes:(?!\.)"), "notes.note:"),
|
|
(regex.compile(r"(?<![.\w])custom_fields:(?!\.)"), "custom_fields.value:"),
|
|
)
|
|
|
|
|
|
def _rewrite_bare_json_field_prefixes(raw_query: str) -> str:
|
|
"""Rewrite bare ``notes:``/``custom_fields:`` prefixes to their
|
|
subpath equivalents. Prefix substitution only, values untouched.
|
|
|
|
Not quote-aware, same accepted trade-off as
|
|
_quote_date_keyword_phrases: a literal ``notes:`` inside an existing
|
|
quoted phrase on an unrelated field would also get rewritten.
|
|
"""
|
|
for pattern, replacement in _BARE_JSON_PREFIX_RES:
|
|
raw_query = pattern.sub(replacement, raw_query, timeout=_REGEX_TIMEOUT)
|
|
return raw_query
|
|
|
|
|
|
def _user_facing_emit_message(d: Diagnostic) -> str:
|
|
"""A user-safe message for an emit-time QueryError's Diagnostic.
|
|
|
|
Built from the Diagnostic's structured fields (kind, field), never from
|
|
d.message: whoosh-compat documents that as developer/log output with no
|
|
stability guarantee, and PATTERN_TOO_COMPLEX embeds the raw backend
|
|
error text in it.
|
|
"""
|
|
field = str(d.field) if d.field is not None else None
|
|
if d.kind is DiagnosticKind.EXISTS_REQUIRES_FAST:
|
|
return f"Existence searches (field:*) are not supported for field {field!r}."
|
|
if d.kind is DiagnosticKind.TEXT_RANGE:
|
|
return f"Range searches are not supported for field {field!r}."
|
|
if d.kind is DiagnosticKind.PATTERN_TOO_COMPLEX:
|
|
return f"The wildcard pattern for field {field!r} is too complex."
|
|
if d.kind is DiagnosticKind.SCHEMA_FIELD_MISSING:
|
|
return f"Field {field!r} is not available in the search index."
|
|
logger.warning("Unmapped emit diagnostic %s: %s", d.kind, d.message)
|
|
return "The search query could not be executed."
|
|
|
|
|
|
# MISCONFIGURED reports a static configuration fact (the registry declares a
|
|
# field the schema does not carry, or one that is not fast), so it stays true
|
|
# until an operator changes the schema and reindexes. The operator-facing log
|
|
# therefore fires once per (kind, field) per process: an alert that repeats on
|
|
# every user query is one operators learn to filter out. A restart re-logs,
|
|
# which re-surfaces the condition after a config change.
|
|
#
|
|
# Bounded by the registry, not by query text: emit() only reports MISCONFIGURED
|
|
# for a field it resolved, and FieldRegistry.resolve returns None for any name
|
|
# or JSON subpath the registry does not declare (those become AST_UNKNOWN_FIELD,
|
|
# an INTERNAL cause that never reaches here), so a user cannot grow this set.
|
|
_logged_misconfigurations: set[tuple[DiagnosticKind, str]] = set()
|
|
|
|
|
|
def _log_misconfiguration_once(d: Diagnostic) -> None:
|
|
"""Log a registry/schema mismatch the first time this process sees it for
|
|
a given field. Never gates the 400: every request still gets its answer."""
|
|
key = (d.kind, str(d.field))
|
|
if key in _logged_misconfigurations:
|
|
return
|
|
_logged_misconfigurations.add(key)
|
|
logger.error(
|
|
"Search index misconfiguration for field %s (%s): %s",
|
|
d.field,
|
|
d.kind.name,
|
|
d.message,
|
|
)
|
|
|
|
|
|
def _map_emit_error(e: QueryError) -> SearchQueryError:
|
|
"""Route an emit-time QueryError by its Diagnostic's Cause.
|
|
|
|
INVALID_INPUT/UNSUPPORTED are user-input errors, exactly like a parse
|
|
diagnostic, and map to a 400. INTERNAL means a defect in whoosh-compat
|
|
or in our own AST handling, never the user's query, so the QueryError is
|
|
re-raised to surface the same way views.py already lets QueryParserError
|
|
surface. MISCONFIGURED is deliberately both: the registry and the index
|
|
schema disagree, which only an operator can fix, so it is logged as an
|
|
error (once per field per process, see _log_misconfiguration_once), but a
|
|
request is still waiting and the query cannot run either way, so it also
|
|
returns a 400.
|
|
"""
|
|
d = e.diagnostic
|
|
if d.cause is Cause.INTERNAL:
|
|
raise e
|
|
if d.cause is Cause.MISCONFIGURED:
|
|
_log_misconfiguration_once(d)
|
|
return SearchQueryError(_user_facing_emit_message(d))
|
|
|
|
|
|
def _has_cjk(text: str) -> bool:
|
|
"""Return True if text contains any CJK characters."""
|
|
return bool(_CJK_RE.search(text))
|
|
|
|
|
|
def extract_cjk_text(text: str) -> str:
|
|
"""Join the CJK runs in ``text`` for indexing into bigram (char-ngram) fields.
|
|
|
|
Mirrors the query side (``_build_cjk_query``): only CJK runs are ever searched
|
|
against the bigram fields, so only CJK runs are worth indexing there. Latin
|
|
text fed to a character-bigram field is never matched and only bloats the
|
|
index and slows indexing/merge. Returns "" when there is no CJK text.
|
|
"""
|
|
return " ".join(_CJK_RE.findall(text))
|
|
|
|
|
|
def _build_cjk_query(
|
|
index: tantivy.Index,
|
|
raw_query: str,
|
|
fields: list[str],
|
|
) -> tantivy.Query | None:
|
|
"""Build a bigram-field query from the CJK runs in ``raw_query``.
|
|
|
|
Only the CJK character runs are extracted and parsed; ASCII field prefixes,
|
|
boolean operators and date keywords are discarded. This keeps the CJK clause
|
|
plain-text and consistent across query/simple modes (no leaked ``field:``
|
|
semantics, no parse failures from spaced ``-``/``+``), and avoids feeding
|
|
Latin tokens into the character-bigram matcher (which would produce spurious
|
|
matches against unrelated Latin text). Returns None when there is no CJK
|
|
text or the parse fails.
|
|
"""
|
|
cjk_text = extract_cjk_text(raw_query)
|
|
if not cjk_text:
|
|
return None
|
|
try:
|
|
return index.parse_query(cjk_text, fields)
|
|
except Exception:
|
|
# Broad on purpose, unlike _try_parse_fuzzy_query's narrower
|
|
# ValueError: cjk_text isn't filtered to a guaranteed-safe token
|
|
# set the way the fuzzy blend's word string is, so the exact
|
|
# failure mode tantivy could raise here isn't pinned down.
|
|
logger.debug(
|
|
"Skipping CJK search clause: could not parse CJK text: %r",
|
|
cjk_text,
|
|
)
|
|
return None
|
|
|
|
|
|
# A joined fuzzy word string must stay plain words: any token that could
|
|
# read as tantivy query grammar (a colon, bracket, quote, operator...) is
|
|
# dropped rather than escaped. Today's default-field analyzers only emit
|
|
# word characters, so this never fires; it guards a future field whose
|
|
# analyzer passes punctuation through (an identity/keyword analyzer).
|
|
_WORD_TOKEN_RE = regex.compile(r"\w+")
|
|
|
|
|
|
def _try_parse_fuzzy_query(
|
|
index: tantivy.Index,
|
|
ast: wc.ast.Node,
|
|
registry: wc.FieldRegistry,
|
|
) -> tantivy.Query | None:
|
|
"""Build the fuzzy blend clause from the parsed query's free-text
|
|
words, or None if it has none.
|
|
|
|
The clause is built by handing tantivy's own query parser a plain
|
|
word string (there's no clean AST-level fuzzy equivalent to
|
|
whoosh-compat's parse tree, and fuzzy matching was always an
|
|
approximate, secondary, 0.1-boosted clause). The words come from
|
|
whoosh_compat's ``free_text_tokens`` over the already-parsed AST,
|
|
never from the raw query string: raw whoosh grammar (date keywords,
|
|
``[2005 to 2009]`` ranges, bracket-class wildcards) is not tantivy
|
|
syntax, and feeding it here used to knock the fuzzy clause out for
|
|
the whole query the moment any such construct appeared alongside a
|
|
typo'd word. The helper also keeps excluded terms out: a ``NOT``'d
|
|
word must not resurface through the fuzzy clause.
|
|
|
|
Chosen trade-off: a term explicitly fielded on one of the default
|
|
search fields (``correspondent:acme``) contributes its text to the
|
|
word string UNFIELDED, so the fuzzy clause searches it across all
|
|
default fields rather than just the one the user named. That is
|
|
recall-only widening on a secondary 0.1-boosted clause the score
|
|
threshold already disciplines, accepted in exchange for never feeding
|
|
field syntax to tantivy's parser.
|
|
|
|
The ValueError guard stays as insurance (the word string is plain
|
|
tokens, so tantivy accepting it is expected, not assumed): on a parse
|
|
failure the fuzzy clause is skipped and the exact/CJK clauses stand,
|
|
rather than the whole query failing.
|
|
"""
|
|
tokens = wc.free_text_tokens(ast, registry=registry, fields=_DEFAULT_SEARCH_FIELDS)
|
|
words = [t for t in tokens if _WORD_TOKEN_RE.fullmatch(t)]
|
|
if not words:
|
|
return None
|
|
fuzzy_text = " ".join(words)
|
|
try:
|
|
return index.parse_query(
|
|
fuzzy_text,
|
|
_DEFAULT_SEARCH_FIELDS,
|
|
field_boosts=_FIELD_BOOSTS,
|
|
fuzzy_fields={f: (True, 1, True) for f in _DEFAULT_SEARCH_FIELDS},
|
|
)
|
|
except ValueError:
|
|
logger.debug(
|
|
"Skipping fuzzy search clause: token string is not valid "
|
|
"tantivy query syntax: %r",
|
|
fuzzy_text,
|
|
)
|
|
return None
|
|
|
|
|
|
_DEFAULT_SEARCH_FIELDS: Final[list[str]] = [
|
|
"title",
|
|
"content",
|
|
"correspondent",
|
|
"document_type",
|
|
"tag",
|
|
]
|
|
_SIMPLE_SEARCH_FIELDS: Final[list[str]] = ["simple_title", "simple_content"]
|
|
_TITLE_SEARCH_FIELDS: Final[list[str]] = ["simple_title"]
|
|
_CJK_ALL_FIELDS: Final[list[str]] = [
|
|
"bigram_content",
|
|
"bigram_title",
|
|
"bigram_correspondent",
|
|
"bigram_document_type",
|
|
"bigram_tag",
|
|
]
|
|
_CJK_CONTENT_FIELDS: Final[list[str]] = ["bigram_content"]
|
|
_CJK_TITLE_FIELDS: Final[list[str]] = ["bigram_title"]
|
|
_FIELD_BOOSTS = {"title": 2.0}
|
|
_SIMPLE_FIELD_BOOSTS = {"simple_title": 2.0}
|
|
|
|
|
|
def _any_of(clauses: list[tuple[tantivy.Occur, tantivy.Query]]) -> tantivy.Query:
|
|
"""Collapse a clause list: none -> empty, one -> itself (no wasted
|
|
single-clause boolean_query wrapping), many -> boolean_query(clauses)."""
|
|
if not clauses:
|
|
return tantivy.Query.empty_query()
|
|
if len(clauses) == 1:
|
|
return clauses[0][1]
|
|
return tantivy.Query.boolean_query(clauses)
|
|
|
|
|
|
def _build_simple_token_query(
|
|
index: tantivy.Index,
|
|
fields: list[str],
|
|
token: str,
|
|
*,
|
|
allow_infix: bool,
|
|
) -> tantivy.Query:
|
|
escaped = regex.escape(token)
|
|
# The simple analyzer keeps punctuation inside whitespace-delimited terms.
|
|
# Boundary-constrained query tokens may therefore begin either at the indexed
|
|
# term boundary or after punctuation within a term (for example,
|
|
# ``medical-history``). This avoids matching a numeric token such as ``6``
|
|
# in the middle of ``16``.
|
|
pattern = (
|
|
f".*{escaped}.*"
|
|
if allow_infix
|
|
else (
|
|
f"({escaped}.*|"
|
|
rf".*[\x20-\x2f\x3a-\x40\x5b-\x60\x7b-\x7e]{escaped}.*)"
|
|
)
|
|
)
|
|
field_queries: list[tuple[tantivy.Occur, tantivy.Query]] = []
|
|
for field in fields:
|
|
query = tantivy.Query.regex_query(index.schema, field, pattern)
|
|
boost = _SIMPLE_FIELD_BOOSTS.get(field, 1.0)
|
|
if boost > 1.0:
|
|
query = tantivy.Query.boost_query(query, boost)
|
|
field_queries.append((tantivy.Occur.Should, query))
|
|
|
|
return _any_of(field_queries)
|
|
|
|
|
|
def parse_user_query(
|
|
index: tantivy.Index,
|
|
raw_query: str,
|
|
tz: tzinfo,
|
|
) -> tantivy.Query:
|
|
"""
|
|
Parse user query through whoosh-compat, then blend in fuzzy/CJK clauses.
|
|
|
|
1. Two small pre-parse rewrites keep historically honored spellings
|
|
working: unquoted multi-word date keyword phrases on date fields
|
|
are quoted (_quote_date_keyword_phrases), and bare
|
|
notes:/custom_fields: prefixes become their subpath equivalents
|
|
(_rewrite_bare_json_field_prefixes). Then wc.parse() against the
|
|
shared FieldRegistry (whoosh grammar -> AST).
|
|
2. Any diagnostics (bad dates/numbers) map to SearchQueryError subclasses
|
|
and raise — the view returns HTTP 400 with every offending field
|
|
listed, not just the first.
|
|
3. emit() turns the AST into a tantivy.Query directly (no string
|
|
round-trip). A QueryError is routed by its Diagnostic's Cause
|
|
(_map_emit_error): a construct that parses but can't execute against
|
|
tantivy (e.g. a text-field range) is a 400, a registry/schema
|
|
mismatch is logged and a 400, and an INTERNAL defect is re-raised.
|
|
4. Optional fuzzy blend (ADVANCED_FUZZY_SEARCH_THRESHOLD) builds a
|
|
plain word string from the parsed AST's free-text tokens
|
|
(whoosh_compat.free_text_tokens) and feeds THAT to
|
|
index.parse_query — never raw_query, whose whoosh grammar (date
|
|
keywords, bracket-class wildcards, etc.) tantivy's parser rejects,
|
|
which used to silently knock the fuzzy clause out of any mixed
|
|
query (see _try_parse_fuzzy_query).
|
|
5. Optional CJK bigram clause — unchanged from before this migration,
|
|
never went through the pre-whoosh-compat translation layer either.
|
|
"""
|
|
registry = get_field_registry(settings.SEARCH_LANGUAGE)
|
|
raw_query = _quote_date_keyword_phrases(raw_query)
|
|
raw_query = _rewrite_bare_json_field_prefixes(raw_query)
|
|
result = wc.parse(
|
|
raw_query,
|
|
registry=registry,
|
|
default_fields=_DEFAULT_SEARCH_FIELDS,
|
|
field_boosts=_FIELD_BOOSTS,
|
|
tz=tz,
|
|
)
|
|
if result.diagnostics:
|
|
raise _diagnostics_to_error(result.diagnostics)
|
|
|
|
try:
|
|
exact = tantivy_emit(result.ast, index=index, registry=registry)
|
|
except QueryError as e:
|
|
raise _map_emit_error(e) from e
|
|
|
|
cjk_query = (
|
|
_build_cjk_query(index, raw_query, _CJK_ALL_FIELDS)
|
|
if _has_cjk(raw_query)
|
|
else None
|
|
)
|
|
|
|
clauses: list[tuple[tantivy.Occur, tantivy.Query]] = [
|
|
(tantivy.Occur.Should, exact),
|
|
]
|
|
|
|
threshold = settings.ADVANCED_FUZZY_SEARCH_THRESHOLD
|
|
if threshold is not None:
|
|
fuzzy = _try_parse_fuzzy_query(index, result.ast, registry)
|
|
if fuzzy is not None:
|
|
clauses.append(
|
|
(tantivy.Occur.Should, tantivy.Query.boost_query(fuzzy, 0.1)),
|
|
)
|
|
|
|
if cjk_query is not None:
|
|
clauses.append((tantivy.Occur.Should, cjk_query))
|
|
|
|
return _any_of(clauses)
|
|
|
|
|
|
# The three whoosh-compat kinds for a wildcard on a field that cannot
|
|
# carry one. d.field_kind supplies the discriminator, so naming the field's
|
|
# type needs no second trip through the registry.
|
|
_PATTERN_ON_KINDS: Final = frozenset(
|
|
{
|
|
DiagnosticKind.PATTERN_ON_NUMERIC,
|
|
DiagnosticKind.PATTERN_ON_BOOLEAN_EXISTS,
|
|
DiagnosticKind.PATTERN_ON_SUBPATH,
|
|
},
|
|
)
|
|
|
|
|
|
def _diagnostics_to_error(diagnostics: tuple[Diagnostic, ...]) -> SearchQueryError:
|
|
errors = [_single_diagnostic_to_error(d) for d in diagnostics]
|
|
return errors[0] if len(errors) == 1 else MultipleSearchQueryErrors(errors)
|
|
|
|
|
|
def _single_diagnostic_to_error(d: Diagnostic) -> SearchQueryError:
|
|
# d.field is a FieldRef, not a str: str(d.field) gives the canonical
|
|
# dotted name (an aliased query, e.g. type:, reports document_type).
|
|
field_name = str(d.field) if d.field is not None else None
|
|
if d.kind is DiagnosticKind.BAD_DATE:
|
|
return InvalidDateQuery(field_name, d.raw_value)
|
|
if d.kind is DiagnosticKind.BAD_NUMBER:
|
|
return InvalidNumberQuery(field_name, d.raw_value)
|
|
if d.kind is DiagnosticKind.TOO_DEEP:
|
|
return SearchQueryError("The search query is nested too deeply.")
|
|
if d.kind in _PATTERN_ON_KINDS:
|
|
kind_label = f" ({d.field_kind.name.lower()})" if d.field_kind else ""
|
|
return SearchQueryError(
|
|
f"Wildcard patterns are not supported for field "
|
|
f"{field_name!r}{kind_label}.",
|
|
)
|
|
logger.warning("Unmapped parse diagnostic %s: %s", d.kind, d.message)
|
|
return SearchQueryError("The search query could not be executed.")
|
|
|
|
|
|
def parse_simple_query(
|
|
index: tantivy.Index,
|
|
raw_query: str,
|
|
fields: list[str],
|
|
cjk_fields: list[str] | None = None,
|
|
) -> tantivy.Query:
|
|
"""
|
|
Parse a plain-text query using Tantivy over a restricted field set.
|
|
|
|
Query string is escaped and normalized to be treated as "simple" text query.
|
|
When cjk_fields is provided and the query contains CJK characters, an
|
|
additional Should clause searches those bigram-tokenized fields, which match
|
|
CJK substrings the simple analyzer can't (long whitespace-free runs are
|
|
dropped by remove_long).
|
|
"""
|
|
tokens = simple_search_tokens(raw_query)
|
|
|
|
clauses: list[tuple[tantivy.Occur, tantivy.Query]] = []
|
|
if tokens:
|
|
# Match every query token, regardless of its position in the document.
|
|
# Each token may occur in any of the requested fields, so text mode also
|
|
# finds documents whose matches are split between title and content.
|
|
token_queries = [
|
|
(
|
|
tantivy.Occur.Must,
|
|
_build_simple_token_query(
|
|
index,
|
|
fields,
|
|
token,
|
|
# Preserve historical infix matching for single-token
|
|
# searches. In multi-token searches, constrain numeric
|
|
# tokens to boundaries to avoid partial-number overlap.
|
|
# This depends on token content, not query order.
|
|
allow_infix=len(tokens) == 1 or not token.isdecimal(),
|
|
),
|
|
)
|
|
for token in tokens
|
|
]
|
|
clauses.append((tantivy.Occur.Should, _any_of(token_queries)))
|
|
|
|
if cjk_fields and _has_cjk(raw_query):
|
|
cjk_q = _build_cjk_query(index, raw_query, cjk_fields)
|
|
if cjk_q is not None:
|
|
clauses.append((tantivy.Occur.Should, cjk_q))
|
|
|
|
return _any_of(clauses)
|
|
|
|
|
|
def parse_simple_text_highlight_query(
|
|
index: tantivy.Index,
|
|
raw_query: str,
|
|
) -> tantivy.Query:
|
|
"""Build a snippet-friendly query for simple text searches.
|
|
|
|
Simple search matching uses regex queries but for compatibility with Tantivy
|
|
SnippetGenerator we build a plain term query over the content field instead.
|
|
"""
|
|
|
|
# Strip Tantivy operator chars before tokenizing: this is a plain-text
|
|
# highlight query, not a structured boolean query, so +/- are separators.
|
|
tokens = simple_search_tokens(
|
|
regex.sub(r"[-+]", " ", raw_query, timeout=_REGEX_TIMEOUT),
|
|
)
|
|
if not tokens:
|
|
return tantivy.Query.empty_query()
|
|
|
|
return index.parse_query(" ".join(tokens), ["content"])
|
|
|
|
|
|
def parse_simple_text_query(
|
|
index: tantivy.Index,
|
|
raw_query: str,
|
|
) -> tantivy.Query:
|
|
"""
|
|
Parse a plain-text query over title/content for simple search inputs.
|
|
"""
|
|
|
|
return parse_simple_query(
|
|
index,
|
|
raw_query,
|
|
_SIMPLE_SEARCH_FIELDS,
|
|
cjk_fields=_CJK_CONTENT_FIELDS,
|
|
)
|
|
|
|
|
|
def parse_simple_title_query(
|
|
index: tantivy.Index,
|
|
raw_query: str,
|
|
) -> tantivy.Query:
|
|
"""
|
|
Parse a plain-text query over the title field only.
|
|
"""
|
|
|
|
return parse_simple_query(
|
|
index,
|
|
raw_query,
|
|
_TITLE_SEARCH_FIELDS,
|
|
cjk_fields=_CJK_TITLE_FIELDS,
|
|
)
|