mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-08-01 08:32:18 +00:00
Fix: selection_data re-derives the filtered document set 5 times over (#13229)
* Fix: selection_data re-derives the filtered document set 5 times over _get_selection_data_for_queryset() (powers ?include_selection_data=true on the document list and search endpoints) computed document_count for Correspondent/DocumentType/StoragePath/Tag/CustomField by embedding the caller's full filtered queryset -- filters plus the permission check -- as a subquery inside 5 separate Count(filter=Q(documents__in=queryset), ...) calls. Each one re-evaluates that whole queryset from scratch. Resolve the document ids once into a concrete list and reuse it across all five annotations instead. For Tag/CustomField specifically (M2M via a through-model table), also route through annotate_document_count_by_ids() -- extracted from the tag/custom-field document_count fix (#13203) -- to avoid the same Count(filter=Q(id__in=...), distinct=True)-on-an-M2M-relation anti-pattern diagnosed there. On a 400k-document/1,000-tag corpus, the correspondents portion alone previously didn't finish within several minutes (killed twice while investigating, including one run that left a zombie query still consuming a CPU 30+ minutes later). All five queries together now complete in ~30-35s. Root-caused from a real report (paperless-ngx#13201) via a different, already-fixed query (#13205) -- this one hasn't been reported in the wild yet, found by auditing the same call path. Depends on #13203 for annotate_document_count_by_ids(). Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com> * Address review feedback: drop unnecessary ordering before collecting ids queryset passed into _get_selection_data_for_queryset() carries the default/user-specified ordering, which is irrelevant once we're only collecting a flat id list. Clearing it removes a pointless sort. No measurable change in benchmarking at 400k documents -- the id-list materialization/IN-clause cost still dominates -- but it's a free, strictly-correct cleanup, not just noise-neutral in the other direction. Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com> --------- Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
c9443e890f
commit
fa01396dfd
@@ -1347,6 +1347,63 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
self.assertEqual(selected_type["document_count"], 1)
|
||||
self.assertEqual(selected_storage_path["document_count"], 1)
|
||||
|
||||
def test_selection_data_document_counts_per_tag(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Multiple tags with different numbers of matching documents
|
||||
within the filtered set, including one with no matches
|
||||
WHEN:
|
||||
- Requesting the document list with include_selection_data=true
|
||||
THEN:
|
||||
- Each tag's document_count reflects only documents in the
|
||||
filtered set, not the instance-wide count
|
||||
"""
|
||||
tag_a = Tag.objects.create(name="a")
|
||||
tag_b = Tag.objects.create(name="b")
|
||||
tag_unused = Tag.objects.create(name="unused")
|
||||
custom_field = CustomField.objects.create(
|
||||
name="cf1",
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
|
||||
doc1 = Document.objects.create(checksum="1", correspondent=None)
|
||||
doc1.tags.add(tag_a)
|
||||
doc2 = Document.objects.create(checksum="2")
|
||||
doc2.tags.add(tag_a, tag_b)
|
||||
doc3 = Document.objects.create(checksum="3")
|
||||
doc3.tags.add(tag_b)
|
||||
CustomFieldInstance.objects.create(
|
||||
document=doc1,
|
||||
field=custom_field,
|
||||
value_text="x",
|
||||
)
|
||||
|
||||
# Excluded from the filtered set entirely.
|
||||
excluded = Document.objects.create(checksum="4")
|
||||
excluded.tags.add(tag_a, tag_b, tag_unused)
|
||||
|
||||
response = self.client.get(
|
||||
f"/api/documents/?id__in={doc1.id},{doc2.id},{doc3.id}"
|
||||
"&include_selection_data=true",
|
||||
)
|
||||
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
selection_data = response.data["selection_data"]
|
||||
|
||||
counts_by_tag = {
|
||||
item["id"]: item["document_count"]
|
||||
for item in selection_data["selected_tags"]
|
||||
}
|
||||
self.assertEqual(counts_by_tag[tag_a.id], 2)
|
||||
self.assertEqual(counts_by_tag[tag_b.id], 2)
|
||||
self.assertEqual(counts_by_tag[tag_unused.id], 0)
|
||||
|
||||
counts_by_field = {
|
||||
item["id"]: item["document_count"]
|
||||
for item in selection_data["selected_custom_fields"]
|
||||
}
|
||||
self.assertEqual(counts_by_field[custom_field.id], 1)
|
||||
|
||||
def test_statistics(self) -> None:
|
||||
doc1 = Document.objects.create(
|
||||
title="none1",
|
||||
|
||||
Reference in New Issue
Block a user