mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-10-03 14:50:31 +00:00
The model factories lived in the documents test package, but three other apps needed them. The AI, mail and testing suites all reached across an app boundary to import from documents.tests.factories, which made a private test package into a shared dependency. The factories now live in the shared testing package, where cross-app use is the intended use.
239 lines
8.5 KiB
Python
239 lines
8.5 KiB
Python
"""Regression coverage for the unguarded TEXT-mode highlight query.
|
|
|
|
parse_simple_text_highlight_query re-parses simple-search tokens through
|
|
Tantivy's query-string parser to build a SnippetGenerator-compatible query.
|
|
Simple-search tokens keep arbitrary punctuation (quotes, colons, brackets,
|
|
slashes), so any token carrying Tantivy query grammar raised an unguarded
|
|
ValueError. The search itself had already succeeded by the time this ran:
|
|
only the highlight step failed, and with the DocumentViewSet.list
|
|
exception handler narrowed elsewhere on this branch, that ValueError now
|
|
reaches the client as a bare 500 rather than a 400.
|
|
|
|
Covers three angles:
|
|
- the query builder itself: quoting each token as its own escaped phrase
|
|
should let it parse instead of raising, for every failure mode a plain-
|
|
text query can trigger (syntax error, unknown field, unsupported regex).
|
|
- highlight_hits: even when a token still can't be expressed as a
|
|
highlight query, the guard must fall back to a query that still
|
|
produces usable highlight HTML, not silently empty ones.
|
|
- the real API endpoint: pinning the previously-500 status to 200.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import TYPE_CHECKING
|
|
|
|
import pytest
|
|
import tantivy
|
|
from rest_framework import status
|
|
|
|
from documents.search._backend import SearchMode
|
|
from documents.search._query import parse_simple_text_highlight_query
|
|
from paperless_testing.factories import DocumentFactory
|
|
|
|
if TYPE_CHECKING:
|
|
from rest_framework.test import APIClient
|
|
|
|
from documents.search._backend import TantivyBackend
|
|
|
|
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
|
|
|
# Each spelling below trips a different Tantivy parser failure mode:
|
|
# 'a"b' -> Syntax Error (unterminated quote)
|
|
# foo:bar -> unknown field
|
|
# (a -> Syntax Error (unbalanced group)
|
|
# [a -> Syntax Error (unbalanced range)
|
|
# /a/ -> Unsupported query (regex queries disallowed)
|
|
_MALFORMED_QUERIES = [
|
|
pytest.param('a"b', id="unterminated_quote"),
|
|
pytest.param("foo:bar", id="unknown_field"),
|
|
pytest.param("(a", id="unbalanced_group"),
|
|
pytest.param("[a", id="unbalanced_range"),
|
|
pytest.param("/a/", id="unsupported_regex"),
|
|
]
|
|
|
|
|
|
class TestParseSimpleTextHighlightQueryDoesNotRaise:
|
|
"""The query builder itself must tolerate Tantivy syntax in its tokens."""
|
|
|
|
@pytest.mark.parametrize("raw_query", _MALFORMED_QUERIES)
|
|
def test_malformed_token_does_not_raise(
|
|
self,
|
|
query_index: tantivy.Index,
|
|
raw_query: str,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- A simple-search query token carrying Tantivy query grammar
|
|
(unterminated quote, unknown field, unbalanced group/range,
|
|
or unsupported regex)
|
|
WHEN:
|
|
- parse_simple_text_highlight_query builds a highlight query
|
|
from it
|
|
THEN:
|
|
- It returns a tantivy.Query instead of raising, since each
|
|
token is quoted as its own escaped phrase rather than fed
|
|
to the parser raw
|
|
"""
|
|
assert isinstance(
|
|
parse_simple_text_highlight_query(query_index, raw_query),
|
|
tantivy.Query,
|
|
)
|
|
|
|
|
|
class TestHighlightHitsProducesUsableHighlights:
|
|
"""highlight_hits must keep producing real <b>-wrapped snippet HTML for
|
|
these queries, not merely avoid raising."""
|
|
|
|
@pytest.mark.parametrize(
|
|
"raw_query",
|
|
[*_MALFORMED_QUERIES, pytest.param("plain text", id="plain_text_sanity")],
|
|
)
|
|
def test_highlight_still_contains_matched_text(
|
|
self,
|
|
backend: TantivyBackend,
|
|
raw_query: str,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- A document whose content contains the raw query text
|
|
verbatim
|
|
WHEN:
|
|
- backend.highlight_hits builds highlights for a TEXT-mode
|
|
search using that same (possibly Tantivy-grammar-carrying)
|
|
query text
|
|
THEN:
|
|
- The hit still carries a content highlight with real
|
|
<b>-wrapped matched-term markup, not an empty fallback
|
|
"""
|
|
doc = DocumentFactory.create(
|
|
title="probe",
|
|
content=f"needle content containing {raw_query} literally here",
|
|
)
|
|
backend.add_or_update(doc)
|
|
|
|
hits = backend.highlight_hits(
|
|
raw_query,
|
|
[doc.pk],
|
|
search_mode=SearchMode.TEXT,
|
|
)
|
|
|
|
assert len(hits) == 1
|
|
highlights = hits[0]["highlights"]
|
|
assert "content" in highlights, (
|
|
f"Expected a content highlight for {raw_query!r}, got: {highlights!r}"
|
|
)
|
|
assert "<b>" in highlights["content"], (
|
|
f"Highlight for {raw_query!r} carries no matched-term markup: "
|
|
f"{highlights['content']!r}"
|
|
)
|
|
|
|
|
|
class TestHighlightGuardDiscriminatesOnValueError:
|
|
"""The guard added to highlight_hits must catch exactly ValueError, the
|
|
same shape as the sibling notes_text guard, and let anything else
|
|
through -- so a real library defect is never mistaken for a harmless
|
|
syntax error."""
|
|
|
|
def test_non_value_error_is_not_swallowed(
|
|
self,
|
|
backend: TantivyBackend,
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- parse_simple_text_highlight_query patched to raise
|
|
RuntimeError instead of a syntax-related ValueError
|
|
WHEN:
|
|
- backend.highlight_hits is called
|
|
THEN:
|
|
- The RuntimeError propagates unguarded; the highlight guard
|
|
must catch exactly ValueError, the same shape as the
|
|
sibling notes_text guard, never mistaking a real library
|
|
defect for a harmless syntax error
|
|
"""
|
|
import documents.search._backend as backend_mod
|
|
|
|
def raise_runtime_error(*args: object, **kwargs: object) -> object:
|
|
raise RuntimeError("synthetic bug, unrelated to query syntax")
|
|
|
|
monkeypatch.setattr(
|
|
backend_mod,
|
|
"parse_simple_text_highlight_query",
|
|
raise_runtime_error,
|
|
)
|
|
|
|
doc = DocumentFactory.create(title="probe", content="anything here")
|
|
backend.add_or_update(doc)
|
|
|
|
with pytest.raises(RuntimeError):
|
|
backend.highlight_hits(
|
|
"anything",
|
|
[doc.pk],
|
|
search_mode=SearchMode.TEXT,
|
|
)
|
|
|
|
|
|
@pytest.mark.usefixtures("_search_index")
|
|
class TestApiNoLongerReturns500:
|
|
"""Pins the actual regression: a matching TEXT-mode search whose query
|
|
string carries Tantivy syntax must return results, not a server error."""
|
|
|
|
@pytest.mark.parametrize("raw_query", _MALFORMED_QUERIES)
|
|
def test_malformed_text_query_returns_200(
|
|
self,
|
|
admin_client: APIClient,
|
|
raw_query: str,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- A matching document whose content contains the raw query
|
|
text, indexed via the real search index fixture
|
|
WHEN:
|
|
- A TEXT-mode search is issued through the real API with a
|
|
query string carrying Tantivy syntax
|
|
THEN:
|
|
- The response is 200 with the expected result count, not a
|
|
500 (the regression this file exists to pin)
|
|
"""
|
|
from documents.search import get_backend
|
|
|
|
doc = DocumentFactory.create(
|
|
title="probe",
|
|
content=f"needle content containing {raw_query} literally here",
|
|
)
|
|
get_backend().add_or_update(doc)
|
|
|
|
response = admin_client.get(f"/api/documents/?text={raw_query}")
|
|
|
|
assert response.status_code == status.HTTP_200_OK
|
|
assert response.data["count"] == 1
|
|
|
|
def test_plain_text_query_still_returns_200(
|
|
self,
|
|
admin_client: APIClient,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- A matching document indexed via the real search index
|
|
fixture
|
|
WHEN:
|
|
- An ordinary TEXT-mode search (no Tantivy syntax) is issued
|
|
THEN:
|
|
- The response is 200 with the expected result count; sanity
|
|
check that the guard does not mask a total failure of the
|
|
ordinary highlight path
|
|
"""
|
|
from documents.search import get_backend
|
|
|
|
doc = DocumentFactory.create(
|
|
title="probe",
|
|
content="needle content containing plain text literally here",
|
|
)
|
|
get_backend().add_or_update(doc)
|
|
|
|
response = admin_client.get("/api/documents/?text=plain text")
|
|
|
|
assert response.status_code == status.HTTP_200_OK
|
|
assert response.data["count"] == 1
|