mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-10-03 06:40:31 +00:00
documents/views.py suffers the same problem as serialisers.py: a single 5,395-line file covering every REST resource in the app. Restructures it into a package, one module per domain area, matching the same layout used for documents/serialisers/. paperless/urls.py and paperless_mail/views.py are updated to import from the new submodules. Test files that imported or mock.patch'd documents.views or documents.serialisers symbols directly are updated to point at the correct submodule for both splits. The mypy and pyrefly baselines are updated to reference the new file paths, and a stale comment in chat_qa.j2 pointing at the old location of _get_llm_output_language is corrected.
278 lines
9.7 KiB
Python
278 lines
9.7 KiB
Python
"""The query-length cap in ``_get_tantivy_query_and_mode``.
|
|
|
|
whoosh-compat's fieldname tagger is O(n^2) in plain word characters, so an
|
|
unbounded ``query`` (SearchMode.QUERY) string is a CPU-exhaustion vector
|
|
against a single request handler. The GET search endpoint is incidentally
|
|
bounded by the web server's header limit, but the POST selection-filter
|
|
path (bulk edit, bulk download) is not -- that is the real vector, so it
|
|
must be pinned here too, not just the GET path.
|
|
|
|
The cap is enforced once, in the shared helper both entry points call, so
|
|
these tests exercise the real endpoints rather than the helper directly:
|
|
a construct that looks right in isolation has repeatedly behaved
|
|
differently end to end on this branch.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import TYPE_CHECKING
|
|
from unittest import mock
|
|
|
|
import pytest
|
|
from rest_framework import status
|
|
|
|
import documents.search._backend
|
|
from documents.views.base import _MAX_QUERY_LENGTH
|
|
|
|
if TYPE_CHECKING:
|
|
from rest_framework.test import APIClient
|
|
|
|
from documents.models import Document
|
|
|
|
pytestmark = [pytest.mark.django_db, pytest.mark.usefixtures("_search_index")]
|
|
|
|
|
|
class TestGetSearchEndpointEnforcesTheCap:
|
|
def test_query_one_over_the_cap_is_a_400(
|
|
self,
|
|
admin_client: APIClient,
|
|
searchable_document: Document,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- The GET search endpoint
|
|
WHEN:
|
|
- A query one character over `_MAX_QUERY_LENGTH` is submitted
|
|
THEN:
|
|
- The response is a 400 naming both the actual length and the
|
|
cap, and the query is rejected before it ever reaches the
|
|
parser -- the 400 alone doesn't prove that, since the parser
|
|
could run first and the view could discard the result
|
|
"""
|
|
query = "a" * (_MAX_QUERY_LENGTH + 1)
|
|
|
|
with mock.patch(
|
|
"documents.search._backend.parse_user_query",
|
|
wraps=documents.search._backend.parse_user_query,
|
|
) as parse_spy:
|
|
response = admin_client.get("/api/documents/", {"query": query})
|
|
|
|
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
|
message = str(response.data["query"])
|
|
assert str(_MAX_QUERY_LENGTH) in message
|
|
assert str(_MAX_QUERY_LENGTH + 1) in message
|
|
parse_spy.assert_not_called()
|
|
|
|
def test_query_at_exactly_the_cap_is_accepted(
|
|
self,
|
|
admin_client: APIClient,
|
|
searchable_document: Document,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- The GET search endpoint
|
|
WHEN:
|
|
- A query exactly `_MAX_QUERY_LENGTH` characters long is
|
|
submitted
|
|
THEN:
|
|
- The response is a 200 (the cap is inclusive, not exclusive)
|
|
"""
|
|
query = "a" * _MAX_QUERY_LENGTH
|
|
|
|
response = admin_client.get("/api/documents/", {"query": query})
|
|
|
|
assert response.status_code == status.HTTP_200_OK
|
|
|
|
def test_an_ordinary_query_is_unaffected(
|
|
self,
|
|
admin_client: APIClient,
|
|
searchable_document: Document,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- The GET search endpoint and an indexed document
|
|
WHEN:
|
|
- An ordinary, well-under-the-cap query is submitted
|
|
THEN:
|
|
- The cap has no effect on a normal search: the matching
|
|
document is returned
|
|
"""
|
|
response = admin_client.get("/api/documents/", {"query": "invoice"})
|
|
|
|
assert response.status_code == status.HTTP_200_OK
|
|
assert response.data["count"] == 1
|
|
|
|
|
|
class TestPostSelectionPathsEnforceTheCap:
|
|
"""The bulk-edit and bulk-download selection filters share the same
|
|
helper the GET search path uses. This is the path that actually
|
|
matters: it is not bounded by a web server's header-length limit the
|
|
way the GET path incidentally is."""
|
|
|
|
def test_bulk_edit_query_one_over_the_cap_is_a_400(
|
|
self,
|
|
admin_client: APIClient,
|
|
searchable_document: Document,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- The bulk-edit selection-filter endpoint
|
|
WHEN:
|
|
- Its `filters.query` is one character over `_MAX_QUERY_LENGTH`
|
|
THEN:
|
|
- The response is a 400 naming both the actual length and the
|
|
cap, and the query is rejected before it ever reaches the
|
|
parser -- this is the path with no web-server header-length
|
|
limit to fall back on, so this is the invariant that matters
|
|
"""
|
|
query = "a" * (_MAX_QUERY_LENGTH + 1)
|
|
|
|
with mock.patch(
|
|
"documents.search._backend.parse_user_query",
|
|
wraps=documents.search._backend.parse_user_query,
|
|
) as parse_spy:
|
|
response = admin_client.post(
|
|
"/api/documents/bulk_edit/",
|
|
{
|
|
"documents": [],
|
|
"all": True,
|
|
"filters": {"query": query},
|
|
"method": "set_document_type",
|
|
"parameters": {"document_type": None},
|
|
},
|
|
format="json",
|
|
)
|
|
|
|
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
|
message = str(response.data["query"])
|
|
assert str(_MAX_QUERY_LENGTH) in message
|
|
assert str(_MAX_QUERY_LENGTH + 1) in message
|
|
parse_spy.assert_not_called()
|
|
|
|
@mock.patch("documents.bulk_edit.bulk_update_documents.apply_async")
|
|
def test_bulk_edit_query_at_exactly_the_cap_is_accepted(
|
|
self,
|
|
bulk_update_task_mock: mock.MagicMock,
|
|
admin_client: APIClient,
|
|
searchable_document: Document,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- The bulk-edit selection-filter endpoint
|
|
WHEN:
|
|
- Its `filters.query` is exactly `_MAX_QUERY_LENGTH` characters
|
|
long
|
|
THEN:
|
|
- The cap check accepts it and the request reaches the real
|
|
bulk-edit method (its Celery dispatch is mocked out here,
|
|
same as every other bulk-edit test, since nothing here is
|
|
testing that method itself)
|
|
"""
|
|
query = "a" * _MAX_QUERY_LENGTH
|
|
|
|
response = admin_client.post(
|
|
"/api/documents/bulk_edit/",
|
|
{
|
|
"documents": [],
|
|
"all": True,
|
|
"filters": {"query": query},
|
|
"method": "set_document_type",
|
|
"parameters": {"document_type": None},
|
|
},
|
|
format="json",
|
|
)
|
|
|
|
assert response.status_code == status.HTTP_200_OK
|
|
|
|
def test_bulk_download_query_one_over_the_cap_is_a_400(
|
|
self,
|
|
admin_client: APIClient,
|
|
searchable_document: Document,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- The bulk-download selection-filter endpoint
|
|
WHEN:
|
|
- Its `filters.query` is one character over `_MAX_QUERY_LENGTH`
|
|
THEN:
|
|
- The response is a 400 naming both the actual length and the
|
|
cap, and the query is rejected before it ever reaches the
|
|
parser
|
|
"""
|
|
query = "a" * (_MAX_QUERY_LENGTH + 1)
|
|
|
|
with mock.patch(
|
|
"documents.search._backend.parse_user_query",
|
|
wraps=documents.search._backend.parse_user_query,
|
|
) as parse_spy:
|
|
response = admin_client.post(
|
|
"/api/documents/bulk_download/",
|
|
{
|
|
"documents": [],
|
|
"all": True,
|
|
"filters": {"query": query},
|
|
},
|
|
format="json",
|
|
)
|
|
|
|
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
|
message = str(response.data["query"])
|
|
assert str(_MAX_QUERY_LENGTH) in message
|
|
assert str(_MAX_QUERY_LENGTH + 1) in message
|
|
parse_spy.assert_not_called()
|
|
|
|
|
|
class TestGlobalSearchEnforcesTheCapToo:
|
|
"""GlobalSearchView calls the backend directly, not through the shared helper.
|
|
|
|
It hardcodes SearchMode.TEXT, which is linear rather than quadratic, so it
|
|
was never the CPU-exhaustion vector. It is capped anyway so that "every
|
|
user query string reaching the backend passes a length check" is an
|
|
invariant rather than a claim with an exception: the view already bounds
|
|
the query from below, and a later change letting it select a mode would
|
|
otherwise reopen the hole silently.
|
|
"""
|
|
|
|
def test_query_one_over_the_cap_is_a_400(
|
|
self,
|
|
admin_client: APIClient,
|
|
searchable_document: Document,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- GlobalSearchView, which calls the backend directly in
|
|
SearchMode.TEXT rather than through the shared cap-checking
|
|
helper
|
|
WHEN:
|
|
- Its query is one character over `_MAX_QUERY_LENGTH`
|
|
THEN:
|
|
- The response is still a 400, keeping "every user query
|
|
string reaching the backend passes a length check" an
|
|
invariant with no exception, even though TEXT mode is
|
|
linear and was never itself the CPU-exhaustion vector
|
|
"""
|
|
response = admin_client.get(
|
|
"/api/search/",
|
|
{"query": "a" * (_MAX_QUERY_LENGTH + 1)},
|
|
)
|
|
assert response.status_code == status.HTTP_400_BAD_REQUEST
|
|
|
|
def test_query_at_exactly_the_cap_is_accepted(
|
|
self,
|
|
admin_client: APIClient,
|
|
searchable_document: Document,
|
|
) -> None:
|
|
"""
|
|
GIVEN:
|
|
- GlobalSearchView
|
|
WHEN:
|
|
- Its query is exactly `_MAX_QUERY_LENGTH` characters long
|
|
THEN:
|
|
- The response is a 200 (the cap is inclusive, not exclusive)
|
|
"""
|
|
response = admin_client.get(
|
|
"/api/search/",
|
|
{"query": "a" * _MAX_QUERY_LENGTH},
|
|
)
|
|
assert response.status_code == status.HTTP_200_OK
|