mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-10-11 10:37:12 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
642511c7b9 | ||
|
|
210d97a522 | ||
|
|
8a96190359 | ||
|
|
ece8769f59 |
No files matched your search
@@ -67,18 +67,22 @@ class Command(PaperlessCommand):
|
|||||||
if options.get("recreate"):
|
if options.get("recreate"):
|
||||||
wipe_index(settings.INDEX_DIR)
|
wipe_index(settings.INDEX_DIR)
|
||||||
|
|
||||||
documents = Document.objects.select_related(
|
documents = (
|
||||||
"correspondent",
|
Document.objects.filter(root_document__isnull=True)
|
||||||
"document_type",
|
.select_related(
|
||||||
"storage_path",
|
"correspondent",
|
||||||
"owner",
|
"document_type",
|
||||||
).prefetch_related(
|
"storage_path",
|
||||||
"tags",
|
"owner",
|
||||||
"notes__user",
|
)
|
||||||
"custom_fields__field",
|
.prefetch_related(
|
||||||
"versions",
|
"tags",
|
||||||
"barcodes",
|
"notes__user",
|
||||||
"versions__barcodes",
|
"custom_fields__field",
|
||||||
|
"versions",
|
||||||
|
"barcodes",
|
||||||
|
"versions__barcodes",
|
||||||
|
)
|
||||||
)
|
)
|
||||||
total = documents.count()
|
total = documents.count()
|
||||||
rebuild_kwargs = {}
|
rebuild_kwargs = {}
|
||||||
|
|||||||
@@ -8,9 +8,11 @@ from django.contrib.contenttypes.models import ContentType
|
|||||||
from django.db.models import Case
|
from django.db.models import Case
|
||||||
from django.db.models import CharField
|
from django.db.models import CharField
|
||||||
from django.db.models import Count
|
from django.db.models import Count
|
||||||
|
from django.db.models import Exists
|
||||||
from django.db.models import F
|
from django.db.models import F
|
||||||
from django.db.models import IntegerField
|
from django.db.models import IntegerField
|
||||||
from django.db.models import Model
|
from django.db.models import Model
|
||||||
|
from django.db.models import OuterRef
|
||||||
from django.db.models import Q
|
from django.db.models import Q
|
||||||
from django.db.models import QuerySet
|
from django.db.models import QuerySet
|
||||||
from django.db.models import Value
|
from django.db.models import Value
|
||||||
@@ -367,18 +369,14 @@ def permitted_object_ids(
|
|||||||
visible exactly when its parent is, judged by the parent's owner and
|
visible exactly when its parent is, judged by the parent's owner and
|
||||||
grants, so the row's own owner and grants are ignored.
|
grants, so the row's own owner and grants are ignored.
|
||||||
|
|
||||||
Guardian stores ``object_pk`` as a string, so the row key is cast to a
|
Guardian stores ``object_pk`` as a string, so each grant is an ``EXISTS``
|
||||||
string and tested against the user's and groups' grants with a single
|
keyed on the row's id cast to a string, which can use guardian's
|
||||||
uncorrelated ``IN``. Postgres and SQLite build that set once. MariaDB
|
(user, permission, object_pk) unique index. Casting every ``object_pk`` to
|
||||||
evaluates it as an index probe per row, which is cheap because the
|
an integer for an ``id IN (...)`` is not indexable, and MariaDB cannot
|
||||||
lookups use guardian's unique indexes. Casting every ``object_pk`` to an
|
materialize it inside the owner ``OR``, so it re-scans the user's grants
|
||||||
integer instead cannot use an index, and MariaDB cannot materialize it
|
for every row. The user's groups are matched with an ``IN`` subquery
|
||||||
inside the owner ``OR``, so it re-scans the user's grants for every row.
|
rather than a join through the membership table, which SQLite plans badly
|
||||||
A correlated ``EXISTS`` per grant fixes MariaDB too, but Postgres and
|
once the grant tables grow.
|
||||||
SQLite re-run it for every row and end up slower than the original. The
|
|
||||||
user's groups are matched with an ``IN`` subquery rather than a join
|
|
||||||
through the membership table, which SQLite plans badly once the grant
|
|
||||||
tables grow.
|
|
||||||
"""
|
"""
|
||||||
has_soft_delete = hasattr(model, "global_objects")
|
has_soft_delete = hasattr(model, "global_objects")
|
||||||
manager = (
|
manager = (
|
||||||
@@ -424,26 +422,25 @@ def permitted_object_ids(
|
|||||||
"permission__content_type": content_type,
|
"permission__content_type": content_type,
|
||||||
}
|
}
|
||||||
|
|
||||||
# Both grant sets are compared to the row key as strings, exactly as
|
key_as_text = Cast(OuterRef(key_field), CharField(max_length=64))
|
||||||
# guardian stores them, and are uncorrelated, so each engine can build the
|
granted_to_user = Exists(
|
||||||
# set once instead of probing per row.
|
UserObjectPermission.objects.filter(
|
||||||
user_keys = UserObjectPermission.objects.filter(
|
user=user,
|
||||||
user=user,
|
object_pk=key_as_text,
|
||||||
**perm_filter,
|
**perm_filter,
|
||||||
).values_list("object_pk", flat=True)
|
),
|
||||||
group_keys = GroupObjectPermission.objects.filter(
|
|
||||||
group_id__in=user.groups.values("id"),
|
|
||||||
**perm_filter,
|
|
||||||
).values_list("object_pk", flat=True)
|
|
||||||
permitted_keys = user_keys.union(group_keys, all=True)
|
|
||||||
|
|
||||||
return (
|
|
||||||
base_qs.annotate(permitted_key=Cast(key_field, CharField(max_length=64)))
|
|
||||||
.filter(
|
|
||||||
Q(**{owner_field: user.pk}) | unowned | Q(permitted_key__in=permitted_keys),
|
|
||||||
)
|
|
||||||
.values_list("id", flat=True)
|
|
||||||
)
|
)
|
||||||
|
granted_to_group = Exists(
|
||||||
|
GroupObjectPermission.objects.filter(
|
||||||
|
group_id__in=user.groups.values("id"),
|
||||||
|
object_pk=key_as_text,
|
||||||
|
**perm_filter,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
return base_qs.filter(
|
||||||
|
Q(**{owner_field: user.pk}) | unowned | granted_to_user | granted_to_group,
|
||||||
|
).values_list("id", flat=True)
|
||||||
|
|
||||||
|
|
||||||
ModelT = TypeVar("ModelT", bound=Model)
|
ModelT = TypeVar("ModelT", bound=Model)
|
||||||
|
|||||||
@@ -284,9 +284,14 @@ class WriteBatch:
|
|||||||
and adding the new version. This ensures stale document data (e.g., after
|
and adding the new version. This ensures stale document data (e.g., after
|
||||||
permission changes) doesn't persist in the index.
|
permission changes) doesn't persist in the index.
|
||||||
|
|
||||||
|
Only root documents are indexed, with their effective content, so a
|
||||||
|
version is indexed as its root document.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
document: Django Document instance to index
|
document: Django Document instance to index
|
||||||
"""
|
"""
|
||||||
|
if document.root_document_id is not None:
|
||||||
|
document = document.root_document
|
||||||
self.remove(document.pk)
|
self.remove(document.pk)
|
||||||
doc = self._backend._build_tantivy_doc(document)
|
doc = self._backend._build_tantivy_doc(document)
|
||||||
self._writer.add_document(doc)
|
self._writer.add_document(doc)
|
||||||
@@ -311,20 +316,22 @@ class WriteBatch:
|
|||||||
An id with no matching document (e.g. deleted between the caller
|
An id with no matching document (e.g. deleted between the caller
|
||||||
collecting ids and the batch running) is silently skipped, matching
|
collecting ids and the batch running) is silently skipped, matching
|
||||||
``add_or_update()``'s existing single-document deferred-task behavior
|
``add_or_update()``'s existing single-document deferred-task behavior
|
||||||
rather than erroring or leaving a stale index entry.
|
rather than erroring or leaving a stale index entry. The id of a
|
||||||
|
version stands for its root document.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
ids: Primary keys of Document instances to index
|
ids: Primary keys of Document instances to index
|
||||||
"""
|
"""
|
||||||
from documents.models import Document
|
from documents.models import Document
|
||||||
from documents.versioning import annotate_effective_content
|
from documents.versioning import annotate_effective_content
|
||||||
|
from documents.versioning import root_document_ids
|
||||||
|
|
||||||
ids = list(ids)
|
ids = list(ids)
|
||||||
if not ids:
|
if not ids:
|
||||||
return
|
return
|
||||||
|
|
||||||
queryset = annotate_effective_content(
|
queryset = annotate_effective_content(
|
||||||
Document.objects.filter(pk__in=ids)
|
Document.objects.filter(pk__in=root_document_ids(ids))
|
||||||
.select_related("correspondent", "document_type", "storage_path", "owner")
|
.select_related("correspondent", "document_type", "storage_path", "owner")
|
||||||
.prefetch_related(
|
.prefetch_related(
|
||||||
"tags",
|
"tags",
|
||||||
|
|||||||
@@ -490,23 +490,20 @@ def update_document_content_maybe_archive_file(
|
|||||||
shutil.move(thumbnail, document.thumbnail_path)
|
shutil.move(thumbnail, document.thumbnail_path)
|
||||||
|
|
||||||
document.refresh_from_db()
|
document.refresh_from_db()
|
||||||
root_document = (
|
|
||||||
document.root_document if document.root_document_id else document
|
|
||||||
)
|
|
||||||
logger.info(
|
logger.info(
|
||||||
f"Updating index for document {root_document.pk} ({document.archive_checksum})",
|
f"Updating index for document {document_id} ({document.archive_checksum})",
|
||||||
)
|
)
|
||||||
from documents.search import get_backend
|
from documents.search import get_backend
|
||||||
|
|
||||||
get_backend().add_or_update(root_document)
|
get_backend().add_or_update(document)
|
||||||
|
|
||||||
ai_config = AIConfig()
|
ai_config = AIConfig()
|
||||||
if ai_config.llm_index_enabled:
|
if ai_config.llm_index_enabled:
|
||||||
llm_index_add_or_update_document(root_document)
|
llm_index_add_or_update_document(document)
|
||||||
|
|
||||||
clear_document_caches(document.pk)
|
clear_document_caches(document.pk)
|
||||||
if root_document.pk != document.pk:
|
if document.root_document_id is not None:
|
||||||
clear_document_caches(root_document.pk)
|
clear_document_caches(document.root_document_id)
|
||||||
|
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception(
|
logger.exception(
|
||||||
|
|||||||
@@ -293,6 +293,80 @@ class TestAddOrUpdateIds:
|
|||||||
assert backend.search_ids("updated", user=None) == [doc.pk]
|
assert backend.search_ids("updated", user=None) == [doc.pk]
|
||||||
|
|
||||||
|
|
||||||
|
class TestVersionsAreIndexedAsTheirRoot:
|
||||||
|
"""Only root documents are indexed, with their effective content, so
|
||||||
|
every write path that is handed a version indexes its root instead."""
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _root_with_version() -> tuple[Document, Document]:
|
||||||
|
root = DocumentFactory(title="Statement", content="stale text")
|
||||||
|
version = DocumentFactory(
|
||||||
|
title="Statement",
|
||||||
|
content="latest text",
|
||||||
|
root_document=root,
|
||||||
|
version_index=1,
|
||||||
|
)
|
||||||
|
return root, version
|
||||||
|
|
||||||
|
def test_add_or_update_indexes_the_root_of_a_version(
|
||||||
|
self,
|
||||||
|
backend: TantivyBackend,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with a version
|
||||||
|
WHEN:
|
||||||
|
- The version is passed to add_or_update
|
||||||
|
THEN:
|
||||||
|
- The root is indexed with the version's text, and the version is not
|
||||||
|
"""
|
||||||
|
root, version = self._root_with_version()
|
||||||
|
|
||||||
|
backend.add_or_update(version)
|
||||||
|
|
||||||
|
assert backend.search_ids("latest", user=None) == [root.pk]
|
||||||
|
assert backend.search_ids("stale", user=None) == []
|
||||||
|
|
||||||
|
def test_add_or_update_ids_indexes_each_root_once(
|
||||||
|
self,
|
||||||
|
backend: TantivyBackend,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with a version
|
||||||
|
WHEN:
|
||||||
|
- Both ids are passed to add_or_update_ids
|
||||||
|
THEN:
|
||||||
|
- The root is indexed once and the version is not indexed
|
||||||
|
"""
|
||||||
|
root, version = self._root_with_version()
|
||||||
|
|
||||||
|
with backend.batch_update() as batch:
|
||||||
|
batch.add_or_update_ids([version.pk, root.pk])
|
||||||
|
|
||||||
|
assert backend.search_ids("Statement", user=None) == [root.pk]
|
||||||
|
assert backend.search_ids("latest", user=None) == [root.pk]
|
||||||
|
|
||||||
|
def test_add_or_update_ids_resolves_a_lone_version(
|
||||||
|
self,
|
||||||
|
backend: TantivyBackend,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with a version
|
||||||
|
WHEN:
|
||||||
|
- Only the version's id is passed to add_or_update_ids
|
||||||
|
THEN:
|
||||||
|
- The root is indexed
|
||||||
|
"""
|
||||||
|
root, version = self._root_with_version()
|
||||||
|
|
||||||
|
with backend.batch_update() as batch:
|
||||||
|
batch.add_or_update_ids([version.pk])
|
||||||
|
|
||||||
|
assert backend.search_ids("latest", user=None) == [root.pk]
|
||||||
|
|
||||||
|
|
||||||
class TestSearch:
|
class TestSearch:
|
||||||
"""Test search query parsing and matching via search_ids."""
|
"""Test search query parsing and matching via search_ids."""
|
||||||
|
|
||||||
|
|||||||
@@ -1196,6 +1196,59 @@ class TestDocumentSearchApi(DirectoriesMixin, APITestCase):
|
|||||||
self.assertIn(d3.id, result_ids)
|
self.assertIn(d3.id, result_ids)
|
||||||
self.assertNotIn(d4.id, result_ids)
|
self.assertNotIn(d4.id, result_ids)
|
||||||
|
|
||||||
|
def test_search_more_like_version_uses_its_root(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A document similar in content to a root document, and one that is not
|
||||||
|
- A version of the root document, which is never indexed
|
||||||
|
WHEN:
|
||||||
|
- API request for more like the version
|
||||||
|
THEN:
|
||||||
|
- The documents similar to the root are returned, not the version
|
||||||
|
"""
|
||||||
|
indexed = {}
|
||||||
|
for name, title, content, day in (
|
||||||
|
("root", "bank statement 1", "things i paid for in august", (2019, 3, 4)),
|
||||||
|
(
|
||||||
|
"similar",
|
||||||
|
"bank statement 3",
|
||||||
|
"things i paid for in september",
|
||||||
|
(2020, 7, 9),
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"other",
|
||||||
|
"Quarterly Report",
|
||||||
|
"quarterly revenue profit margin",
|
||||||
|
(2021, 11, 30),
|
||||||
|
),
|
||||||
|
):
|
||||||
|
with time_machine.travel(
|
||||||
|
timezone.make_aware(datetime.datetime(*day)),
|
||||||
|
tick=False,
|
||||||
|
):
|
||||||
|
indexed[name] = DocumentFactory(
|
||||||
|
title=title,
|
||||||
|
content=content,
|
||||||
|
created=datetime.date(*day),
|
||||||
|
added=timezone.make_aware(datetime.datetime(*day)),
|
||||||
|
)
|
||||||
|
version = DocumentFactory(
|
||||||
|
root_document=indexed["root"],
|
||||||
|
version_index=1,
|
||||||
|
content="things i paid for in august",
|
||||||
|
)
|
||||||
|
backend = get_backend()
|
||||||
|
for document in indexed.values():
|
||||||
|
backend.add_or_update(document)
|
||||||
|
|
||||||
|
response = self.client.get(f"/api/documents/?more_like_id={version.id}")
|
||||||
|
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||||
|
result_ids = [r["id"] for r in response.data["results"]]
|
||||||
|
self.assertIn(indexed["similar"].id, result_ids)
|
||||||
|
self.assertNotIn(indexed["other"].id, result_ids)
|
||||||
|
self.assertNotIn(version.id, result_ids)
|
||||||
|
|
||||||
def test_more_like_requires_id_of_existing_document(self) -> None:
|
def test_more_like_requires_id_of_existing_document(self) -> None:
|
||||||
"""
|
"""
|
||||||
GIVEN:
|
GIVEN:
|
||||||
|
|||||||
@@ -22,6 +22,7 @@ from documents.models import Document
|
|||||||
from documents.tasks import update_document_content_maybe_archive_file
|
from documents.tasks import update_document_content_maybe_archive_file
|
||||||
from paperless_testing.assertions import FileSystemAssertsMixin
|
from paperless_testing.assertions import FileSystemAssertsMixin
|
||||||
from paperless_testing.dirs import DirectoriesMixin
|
from paperless_testing.dirs import DirectoriesMixin
|
||||||
|
from paperless_testing.factories import DocumentFactory
|
||||||
|
|
||||||
sample_file: Path = Path(__file__).parent / "samples" / "simple.pdf"
|
sample_file: Path = Path(__file__).parent / "samples" / "simple.pdf"
|
||||||
|
|
||||||
@@ -116,6 +117,27 @@ class TestMakeIndex:
|
|||||||
call_command("document_index", "reindex", skip_checks=True)
|
call_command("document_index", "reindex", skip_checks=True)
|
||||||
mock_get_backend.return_value.rebuild.assert_called_once()
|
mock_get_backend.return_value.rebuild.assert_called_once()
|
||||||
|
|
||||||
|
def test_reindex_skips_versions(self, mocker: MockerFixture) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with a version
|
||||||
|
WHEN:
|
||||||
|
- The reindex command runs
|
||||||
|
THEN:
|
||||||
|
- Only the root document is handed to the rebuild, since a version
|
||||||
|
is indexed as its root
|
||||||
|
"""
|
||||||
|
root = DocumentFactory()
|
||||||
|
DocumentFactory(root_document=root, version_index=1)
|
||||||
|
mock_get_backend = mocker.patch(
|
||||||
|
"documents.management.commands.document_index.get_backend",
|
||||||
|
)
|
||||||
|
|
||||||
|
call_command("document_index", "reindex", skip_checks=True)
|
||||||
|
|
||||||
|
documents = mock_get_backend.return_value.rebuild.call_args.args[0]
|
||||||
|
assert list(documents.values_list("pk", flat=True)) == [root.pk]
|
||||||
|
|
||||||
def test_optimize(self) -> None:
|
def test_optimize(self) -> None:
|
||||||
"""Optimize command must execute without error (Tantivy handles optimization automatically)."""
|
"""Optimize command must execute without error (Tantivy handles optimization automatically)."""
|
||||||
call_command("document_index", "optimize", skip_checks=True)
|
call_command("document_index", "optimize", skip_checks=True)
|
||||||
|
|||||||
@@ -310,7 +310,7 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
|
|||||||
|
|
||||||
@mock.patch("documents.tasks.clear_document_caches")
|
@mock.patch("documents.tasks.clear_document_caches")
|
||||||
@mock.patch("documents.search.get_backend")
|
@mock.patch("documents.search.get_backend")
|
||||||
def test_update_content_version_indexes_root(
|
def test_update_content_version_clears_caches_for_root(
|
||||||
self,
|
self,
|
||||||
mock_get_backend: mock.Mock,
|
mock_get_backend: mock.Mock,
|
||||||
mock_clear_caches: mock.Mock,
|
mock_clear_caches: mock.Mock,
|
||||||
@@ -321,8 +321,8 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
|
|||||||
WHEN:
|
WHEN:
|
||||||
- Update content task is called for the version
|
- Update content task is called for the version
|
||||||
THEN:
|
THEN:
|
||||||
- The version's content is updated
|
- The version's content is updated, not the root's
|
||||||
- The root document is indexed rather than the version
|
- The document is indexed
|
||||||
- Caches are cleared for both
|
- Caches are cleared for both
|
||||||
"""
|
"""
|
||||||
root, version = self._create_root_with_version()
|
root, version = self._create_root_with_version()
|
||||||
@@ -334,8 +334,7 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
|
|||||||
"my document",
|
"my document",
|
||||||
)
|
)
|
||||||
self.assertEqual(Document.objects.get(pk=root.pk).content, "root content")
|
self.assertEqual(Document.objects.get(pk=root.pk).content, "root content")
|
||||||
indexed = mock_get_backend.return_value.add_or_update.call_args.args[0]
|
mock_get_backend.return_value.add_or_update.assert_called_once()
|
||||||
self.assertEqual(indexed.pk, root.pk)
|
|
||||||
mock_clear_caches.assert_has_calls(
|
mock_clear_caches.assert_has_calls(
|
||||||
[mock.call(version.pk), mock.call(root.pk)],
|
[mock.call(version.pk), mock.call(root.pk)],
|
||||||
)
|
)
|
||||||
@@ -343,7 +342,7 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
|
|||||||
@override_settings(AI_ENABLED=True, LLM_EMBEDDING_BACKEND="huggingface")
|
@override_settings(AI_ENABLED=True, LLM_EMBEDDING_BACKEND="huggingface")
|
||||||
@mock.patch("documents.tasks.llm_index_add_or_update_document")
|
@mock.patch("documents.tasks.llm_index_add_or_update_document")
|
||||||
@mock.patch("documents.search.get_backend")
|
@mock.patch("documents.search.get_backend")
|
||||||
def test_update_content_version_updates_llm_index_for_root(
|
def test_update_content_version_updates_llm_index(
|
||||||
self,
|
self,
|
||||||
mock_get_backend: mock.Mock,
|
mock_get_backend: mock.Mock,
|
||||||
mock_llm_index: mock.Mock,
|
mock_llm_index: mock.Mock,
|
||||||
@@ -355,14 +354,13 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
|
|||||||
WHEN:
|
WHEN:
|
||||||
- Update content task is called for the version
|
- Update content task is called for the version
|
||||||
THEN:
|
THEN:
|
||||||
- The LLM index is updated for the root document, not the version
|
- The LLM index is updated
|
||||||
"""
|
"""
|
||||||
root, version = self._create_root_with_version()
|
_, version = self._create_root_with_version()
|
||||||
|
|
||||||
tasks.update_document_content_maybe_archive_file(version.pk)
|
tasks.update_document_content_maybe_archive_file(version.pk)
|
||||||
|
|
||||||
mock_llm_index.assert_called_once()
|
mock_llm_index.assert_called_once()
|
||||||
self.assertEqual(mock_llm_index.call_args.args[0].pk, root.pk)
|
|
||||||
|
|
||||||
|
|
||||||
class TestUpdateContentRemoteOCR(DirectoriesMixin, TestCase):
|
class TestUpdateContentRemoteOCR(DirectoriesMixin, TestCase):
|
||||||
|
|||||||
@@ -17,9 +17,26 @@ from django.db.models.functions import RowNumber
|
|||||||
from documents.models import Document
|
from documents.models import Document
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Iterable
|
||||||
|
|
||||||
from rest_framework.request import Request
|
from rest_framework.request import Request
|
||||||
|
|
||||||
|
|
||||||
|
def root_document_ids(ids: Iterable[int]) -> QuerySet[int]:
|
||||||
|
"""
|
||||||
|
The ids of the root documents of the given documents: a root stands for
|
||||||
|
itself and a version for its root. Only the indexes' bookkeeping needs
|
||||||
|
this, since they hold root documents only.
|
||||||
|
"""
|
||||||
|
return (
|
||||||
|
Document.objects.filter(pk__in=ids)
|
||||||
|
.annotate(root_id=Coalesce("root_document_id", "id"))
|
||||||
|
.order_by()
|
||||||
|
.values_list("root_id", flat=True)
|
||||||
|
.distinct()
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def versions_newest_first(documents: QuerySet[Document]) -> QuerySet[Document]:
|
def versions_newest_first(documents: QuerySet[Document]) -> QuerySet[Document]:
|
||||||
"""
|
"""
|
||||||
Sorts versions so the newest one comes first using version_index and not on id,
|
Sorts versions so the newest one comes first using version_index and not on id,
|
||||||
|
|||||||
@@ -329,9 +329,10 @@ def _get_tantivy_query_and_mode(params):
|
|||||||
def _get_more_like_id(query_params: dict[str, Any], user: User | None) -> int:
|
def _get_more_like_id(query_params: dict[str, Any], user: User | None) -> int:
|
||||||
try:
|
try:
|
||||||
more_like_doc_id = int(query_params["more_like_id"])
|
more_like_doc_id = int(query_params["more_like_id"])
|
||||||
more_like_doc = Document.objects.select_related("owner").get(
|
more_like_doc = Document.objects.select_related(
|
||||||
pk=more_like_doc_id,
|
"owner",
|
||||||
)
|
"root_document__owner",
|
||||||
|
).get(pk=more_like_doc_id)
|
||||||
except (TypeError, ValueError, Document.DoesNotExist):
|
except (TypeError, ValueError, Document.DoesNotExist):
|
||||||
raise PermissionDenied(_("Invalid more_like_id"))
|
raise PermissionDenied(_("Invalid more_like_id"))
|
||||||
|
|
||||||
@@ -342,7 +343,8 @@ def _get_more_like_id(query_params: dict[str, Any], user: User | None) -> int:
|
|||||||
):
|
):
|
||||||
raise PermissionDenied(_("Insufficient permissions."))
|
raise PermissionDenied(_("Insufficient permissions."))
|
||||||
|
|
||||||
return more_like_doc_id
|
# Only root documents are indexed, a version stands for its root
|
||||||
|
return more_like_doc.root_document_id or more_like_doc.pk
|
||||||
|
|
||||||
|
|
||||||
class SearchParams(NamedTuple):
|
class SearchParams(NamedTuple):
|
||||||
|
|||||||
@@ -4,8 +4,10 @@ from django.conf import settings
|
|||||||
from django.contrib.auth.models import User
|
from django.contrib.auth.models import User
|
||||||
|
|
||||||
from documents.models import Document
|
from documents.models import Document
|
||||||
from documents.permissions import permitted_document_ids
|
from documents.permissions import permitted_object_ids
|
||||||
|
from documents.permissions import restrict_queryset_to_visible
|
||||||
from documents.permissions import user_is_unrestricted
|
from documents.permissions import user_is_unrestricted
|
||||||
|
from documents.versioning import annotate_effective_content
|
||||||
from paperless.config import AIConfig
|
from paperless.config import AIConfig
|
||||||
from paperless_ai.base_model import ClassificationSuggestions
|
from paperless_ai.base_model import ClassificationSuggestions
|
||||||
from paperless_ai.base_model import TaxonomyChoiceDict
|
from paperless_ai.base_model import TaxonomyChoiceDict
|
||||||
@@ -53,8 +55,8 @@ def _fulltext_similar_documents(
|
|||||||
active superuser - see user_is_unrestricted) is normalized to ``None``
|
active superuser - see user_is_unrestricted) is normalized to ``None``
|
||||||
before calling, since the backend's permission filter has no superuser
|
before calling, since the backend's permission filter has no superuser
|
||||||
short-circuit of its own. Results are re-checked with
|
short-circuit of its own. Results are re-checked with
|
||||||
permitted_document_ids() since Tantivy's indexed permission fields lag the
|
restrict_queryset_to_visible() since Tantivy's indexed permission fields
|
||||||
DB via async reindexing and judge a version by its own owner.
|
lag the DB via async reindexing.
|
||||||
"""
|
"""
|
||||||
from documents.search import get_backend
|
from documents.search import get_backend
|
||||||
|
|
||||||
@@ -68,9 +70,10 @@ def _fulltext_similar_documents(
|
|||||||
)
|
)
|
||||||
if not unrestricted:
|
if not unrestricted:
|
||||||
allowed_ids = set(
|
allowed_ids = set(
|
||||||
Document.objects.filter(
|
restrict_queryset_to_visible(
|
||||||
pk__in=similar_ids,
|
Document.objects.filter(pk__in=similar_ids),
|
||||||
id__in=permitted_document_ids(user),
|
user,
|
||||||
|
"view_document",
|
||||||
).values_list("pk", flat=True),
|
).values_list("pk", flat=True),
|
||||||
)
|
)
|
||||||
similar_ids = [doc_id for doc_id in similar_ids if doc_id in allowed_ids]
|
similar_ids = [doc_id for doc_id in similar_ids if doc_id in allowed_ids]
|
||||||
@@ -111,7 +114,7 @@ def build_prompt_without_rag(
|
|||||||
) -> str:
|
) -> str:
|
||||||
filename = document.filename or ""
|
filename = document.filename or ""
|
||||||
content = truncate_content(
|
content = truncate_content(
|
||||||
document.content[:4000] or "",
|
(document.get_effective_content() or "")[:4000],
|
||||||
chunk_size=config.llm_embedding_chunk_size,
|
chunk_size=config.llm_embedding_chunk_size,
|
||||||
context_size=config.llm_context_size,
|
context_size=config.llm_context_size,
|
||||||
)
|
)
|
||||||
@@ -198,13 +201,13 @@ def get_taxonomy_context(
|
|||||||
# quadratic scan in the vector store at best, and past ~32,763
|
# quadratic scan in the vector store at best, and past ~32,763
|
||||||
# documents a hard sqlite3.OperationalError (SQLite's
|
# documents a hard sqlite3.OperationalError (SQLite's
|
||||||
# bound-parameter limit) at worst.
|
# bound-parameter limit) at worst.
|
||||||
# permitted_document_ids() has its own superuser shortcut that would
|
# permitted_object_ids() has its own superuser shortcut that would
|
||||||
# return every Document's id anyway, so this changes nothing about
|
# return every Document's id anyway, so this changes nothing about
|
||||||
# which documents are considered -- only how we get there.
|
# which documents are considered -- only how we get there.
|
||||||
visible_document_ids = (
|
visible_document_ids = (
|
||||||
None
|
None
|
||||||
if user_is_unrestricted(user)
|
if user_is_unrestricted(user)
|
||||||
else list(permitted_document_ids(user))
|
else list(permitted_object_ids(user, Document, "view_document"))
|
||||||
)
|
)
|
||||||
nodes = retrieve_similar_nodes(
|
nodes = retrieve_similar_nodes(
|
||||||
document,
|
document,
|
||||||
@@ -225,7 +228,9 @@ def get_taxonomy_context(
|
|||||||
|
|
||||||
# similar_documents is already ordered by descending weight; don't lose it.
|
# similar_documents is already ordered by descending weight; don't lose it.
|
||||||
similar_document_ids = [s["document_id"] for s in similar_documents]
|
similar_document_ids = [s["document_id"] for s in similar_documents]
|
||||||
similar_documents_by_id = Document.objects.in_bulk(similar_document_ids)
|
similar_documents_by_id = annotate_effective_content(
|
||||||
|
Document.objects.all(),
|
||||||
|
).in_bulk(similar_document_ids)
|
||||||
similar_docs = [
|
similar_docs = [
|
||||||
similar_documents_by_id[document_id]
|
similar_documents_by_id[document_id]
|
||||||
for document_id in similar_document_ids
|
for document_id in similar_document_ids
|
||||||
@@ -233,7 +238,7 @@ def get_taxonomy_context(
|
|||||||
][:max_docs]
|
][:max_docs]
|
||||||
context_blocks = []
|
context_blocks = []
|
||||||
for similar in similar_docs:
|
for similar in similar_docs:
|
||||||
text = similar.content[:1000] or ""
|
text = (similar.get_effective_content() or "")[:1000]
|
||||||
title = similar.title or similar.filename or "Untitled"
|
title = similar.title or similar.filename or "Untitled"
|
||||||
context_blocks.append(f"TITLE: {title}\n{text}")
|
context_blocks.append(f"TITLE: {title}\n{text}")
|
||||||
except Exception:
|
except Exception:
|
||||||
|
|||||||
@@ -135,6 +135,6 @@ def build_llm_index_text(doc: Document) -> str:
|
|||||||
lines.append(f"Custom Field - {instance.field.name}: {instance}")
|
lines.append(f"Custom Field - {instance.field.name}: {instance}")
|
||||||
|
|
||||||
lines.append("\nContent:\n")
|
lines.append("\nContent:\n")
|
||||||
lines.append(doc.content or "")
|
lines.append(doc.get_effective_content() or "")
|
||||||
|
|
||||||
return _normalize_llm_index_text("\n".join(lines))
|
return _normalize_llm_index_text("\n".join(lines))
|
||||||
@@ -17,6 +17,8 @@ from documents.models import PaperlessTask
|
|||||||
from documents.utils import IterWrapper
|
from documents.utils import IterWrapper
|
||||||
from documents.utils import QuerySetStream
|
from documents.utils import QuerySetStream
|
||||||
from documents.utils import identity
|
from documents.utils import identity
|
||||||
|
from documents.versioning import annotate_effective_content
|
||||||
|
from documents.versioning import root_document_ids
|
||||||
from paperless.config import AIConfig
|
from paperless.config import AIConfig
|
||||||
from paperless_ai.db import db_connection_released
|
from paperless_ai.db import db_connection_released
|
||||||
from paperless_ai.embedding import build_llm_index_text
|
from paperless_ai.embedding import build_llm_index_text
|
||||||
@@ -443,11 +445,11 @@ def update_llm_index(
|
|||||||
"Skipping LLM index update: migration check deferred; "
|
"Skipping LLM index update: migration check deferred; "
|
||||||
"will retry next run."
|
"will retry next run."
|
||||||
)
|
)
|
||||||
documents = Document.objects.select_related(
|
documents = annotate_effective_content(
|
||||||
"correspondent",
|
Document.objects.filter(root_document__isnull=True)
|
||||||
"document_type",
|
.select_related("correspondent", "document_type", "storage_path")
|
||||||
"storage_path",
|
.prefetch_related("tags", "notes", "custom_fields__field"),
|
||||||
).prefetch_related("tags", "notes", "custom_fields__field")
|
)
|
||||||
no_documents = not documents.exists()
|
no_documents = not documents.exists()
|
||||||
|
|
||||||
# Fast exit before touching config: nothing to index and no existing index.
|
# Fast exit before touching config: nothing to index and no existing index.
|
||||||
@@ -483,7 +485,7 @@ def update_llm_index(
|
|||||||
msg = "LLM index rebuilt successfully."
|
msg = "LLM index rebuilt successfully."
|
||||||
else:
|
else:
|
||||||
scoped_documents = (
|
scoped_documents = (
|
||||||
documents.filter(id__in=document_ids)
|
documents.filter(id__in=root_document_ids(document_ids))
|
||||||
if document_ids is not None
|
if document_ids is not None
|
||||||
else documents
|
else documents
|
||||||
)
|
)
|
||||||
@@ -510,7 +512,12 @@ def update_llm_index(
|
|||||||
|
|
||||||
|
|
||||||
def llm_index_add_or_update_document(document: Document):
|
def llm_index_add_or_update_document(document: Document):
|
||||||
"""Add or atomically replace a document's chunks in the index."""
|
"""
|
||||||
|
Add or atomically replace a document's chunks in the index. Only root
|
||||||
|
documents are indexed, so a version is indexed as its root document.
|
||||||
|
"""
|
||||||
|
if document.root_document_id is not None:
|
||||||
|
document = document.root_document
|
||||||
config = AIConfig()
|
config = AIConfig()
|
||||||
new_nodes = build_document_node(
|
new_nodes = build_document_node(
|
||||||
document,
|
document,
|
||||||
@@ -688,7 +695,7 @@ def retrieve_similar_nodes(
|
|||||||
)
|
)
|
||||||
|
|
||||||
query_text = truncate_embedding_query(
|
query_text = truncate_embedding_query(
|
||||||
(document.title or "") + "\n" + (document.content or ""),
|
(document.title or "") + "\n" + (document.get_effective_content() or ""),
|
||||||
chunk_size=config.llm_embedding_chunk_size,
|
chunk_size=config.llm_embedding_chunk_size,
|
||||||
)
|
)
|
||||||
# Hold the shared read lock for the whole retrieval so the connection is
|
# Hold the shared read lock for the whole retrieval so the connection is
|
||||||
|
|||||||
@@ -55,6 +55,7 @@ def mock_document():
|
|||||||
doc.storage_path = None
|
doc.storage_path = None
|
||||||
doc.archive_serial_number = "12345"
|
doc.archive_serial_number = "12345"
|
||||||
doc.content = "This is the document content."
|
doc.content = "This is the document content."
|
||||||
|
doc.get_effective_content.return_value = "This is the document content."
|
||||||
|
|
||||||
cf1 = MagicMock(__str__=lambda x: "Value1")
|
cf1 = MagicMock(__str__=lambda x: "Value1")
|
||||||
cf1.field = MagicMock()
|
cf1.field = MagicMock()
|
||||||
@@ -434,6 +435,55 @@ def test_get_taxonomy_context_preserves_similarity_order_and_distinct_documents(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.django_db
|
||||||
|
class TestClassifierEffectiveContent:
|
||||||
|
"""A root document's text for the LLM is its newest version's content."""
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _root_with_version() -> Document:
|
||||||
|
root = DocumentFactory(title="Statement", content="stale text")
|
||||||
|
DocumentFactory(root_document=root, version_index=1, content="latest text")
|
||||||
|
return root
|
||||||
|
|
||||||
|
def test_prompt_uses_the_newest_versions_content(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with a version
|
||||||
|
WHEN:
|
||||||
|
- The classification prompt is built for the root
|
||||||
|
THEN:
|
||||||
|
- It contains the newest version's content
|
||||||
|
"""
|
||||||
|
prompt = build_prompt_without_rag(self._root_with_version(), AIConfig())
|
||||||
|
|
||||||
|
assert "latest text" in prompt
|
||||||
|
assert "stale text" not in prompt
|
||||||
|
|
||||||
|
@override_settings(LLM_EMBEDDING_BACKEND="huggingface")
|
||||||
|
def test_similar_document_context_uses_the_newest_versions_content(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A similar root document with a version
|
||||||
|
WHEN:
|
||||||
|
- The similar-document context is built
|
||||||
|
THEN:
|
||||||
|
- It contains the newest version's content
|
||||||
|
"""
|
||||||
|
similar = self._root_with_version()
|
||||||
|
document = DocumentFactory(content="Some content")
|
||||||
|
fake_nodes = [
|
||||||
|
SimpleNamespace(metadata={"document_id": str(similar.pk)}, score=0.9),
|
||||||
|
]
|
||||||
|
|
||||||
|
with patch(
|
||||||
|
"paperless_ai.ai_classifier.retrieve_similar_nodes",
|
||||||
|
return_value=fake_nodes,
|
||||||
|
):
|
||||||
|
_candidates, context = get_taxonomy_context(document, user=None)
|
||||||
|
|
||||||
|
assert context == "TITLE: Statement\nlatest text"
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
@pytest.mark.django_db
|
||||||
@override_settings(LLM_EMBEDDING_BACKEND="huggingface")
|
@override_settings(LLM_EMBEDDING_BACKEND="huggingface")
|
||||||
def test_get_taxonomy_context_no_similar_docs():
|
def test_get_taxonomy_context_no_similar_docs():
|
||||||
@@ -552,7 +602,7 @@ class TestGetTaxonomyContextVisibility:
|
|||||||
return_value=[],
|
return_value=[],
|
||||||
)
|
)
|
||||||
mock_permitted = mocker.patch(
|
mock_permitted = mocker.patch(
|
||||||
"paperless_ai.ai_classifier.permitted_document_ids",
|
"paperless_ai.ai_classifier.permitted_object_ids",
|
||||||
)
|
)
|
||||||
user = UserFactory.create(is_superuser=True)
|
user = UserFactory.create(is_superuser=True)
|
||||||
|
|
||||||
@@ -582,7 +632,7 @@ class TestGetTaxonomyContextVisibility:
|
|||||||
return_value=[],
|
return_value=[],
|
||||||
)
|
)
|
||||||
mock_permitted = mocker.patch(
|
mock_permitted = mocker.patch(
|
||||||
"paperless_ai.ai_classifier.permitted_document_ids",
|
"paperless_ai.ai_classifier.permitted_object_ids",
|
||||||
)
|
)
|
||||||
|
|
||||||
get_taxonomy_context(document, None)
|
get_taxonomy_context(document, None)
|
||||||
@@ -611,55 +661,16 @@ class TestGetTaxonomyContextVisibility:
|
|||||||
return_value=[],
|
return_value=[],
|
||||||
)
|
)
|
||||||
mock_permitted = mocker.patch(
|
mock_permitted = mocker.patch(
|
||||||
"paperless_ai.ai_classifier.permitted_document_ids",
|
"paperless_ai.ai_classifier.permitted_object_ids",
|
||||||
return_value=[1, 2, 3],
|
return_value=[1, 2, 3],
|
||||||
)
|
)
|
||||||
user = UserFactory.create(is_superuser=False)
|
user = UserFactory.create(is_superuser=False)
|
||||||
|
|
||||||
get_taxonomy_context(document, user)
|
get_taxonomy_context(document, user)
|
||||||
|
|
||||||
mock_permitted.assert_called_once_with(user)
|
mock_permitted.assert_called_once_with(user, Document, "view_document")
|
||||||
assert mock_retrieve.call_args.kwargs["document_ids"] == [1, 2, 3]
|
assert mock_retrieve.call_args.kwargs["document_ids"] == [1, 2, 3]
|
||||||
|
|
||||||
@pytest.mark.django_db
|
|
||||||
@override_settings(LLM_EMBEDDING_BACKEND="huggingface")
|
|
||||||
def test_version_of_private_root_is_not_visible(
|
|
||||||
self,
|
|
||||||
mocker: pytest_mock.MockerFixture,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A private root document owned by someone else
|
|
||||||
- A version of it whose own owner is unset, as when the root
|
|
||||||
changed hands after the version was created
|
|
||||||
WHEN:
|
|
||||||
- get_taxonomy_context() is called for a non-superuser
|
|
||||||
THEN:
|
|
||||||
- Neither the root nor the version is in the visible ids passed to
|
|
||||||
retrieve_similar_nodes(), since a version follows its root
|
|
||||||
"""
|
|
||||||
owner = UserFactory.create()
|
|
||||||
viewer = UserFactory.create(is_superuser=False)
|
|
||||||
root = DocumentFactory.create(content="private", owner=owner)
|
|
||||||
version = DocumentFactory.create(
|
|
||||||
content="private",
|
|
||||||
owner=None,
|
|
||||||
root_document=root,
|
|
||||||
version_index=1,
|
|
||||||
)
|
|
||||||
source = DocumentFactory.create(content="Some content", owner=viewer)
|
|
||||||
mock_retrieve = mocker.patch(
|
|
||||||
"paperless_ai.ai_classifier.retrieve_similar_nodes",
|
|
||||||
return_value=[],
|
|
||||||
)
|
|
||||||
|
|
||||||
get_taxonomy_context(source, viewer)
|
|
||||||
|
|
||||||
visible = mock_retrieve.call_args.kwargs["document_ids"]
|
|
||||||
assert source.pk in visible
|
|
||||||
assert root.pk not in visible
|
|
||||||
assert version.pk not in visible
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
@pytest.mark.django_db
|
||||||
class TestFulltextSimilarDocuments:
|
class TestFulltextSimilarDocuments:
|
||||||
@@ -842,7 +853,7 @@ class TestFulltextSimilarDocuments:
|
|||||||
- _fulltext_similar_documents() is called with that user
|
- _fulltext_similar_documents() is called with that user
|
||||||
THEN:
|
THEN:
|
||||||
- Only the still-permitted document is returned - the DB
|
- Only the still-permitted document is returned - the DB
|
||||||
re-check via permitted_document_ids() must catch the
|
re-check via restrict_queryset_to_visible() must catch the
|
||||||
document Tantivy's stale index still thinks is visible
|
document Tantivy's stale index still thinks is visible
|
||||||
"""
|
"""
|
||||||
owner = UserFactory.create()
|
owner = UserFactory.create()
|
||||||
@@ -873,45 +884,6 @@ class TestFulltextSimilarDocuments:
|
|||||||
|
|
||||||
assert [s["document_id"] for s in result] == [permitted.pk]
|
assert [s["document_id"] for s in result] == [permitted.pk]
|
||||||
|
|
||||||
def test_excludes_version_of_private_root_for_regular_user(
|
|
||||||
self,
|
|
||||||
fulltext_backend: TantivyBackend,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A regular user and a private root owned by someone else
|
|
||||||
- A version of that root with no owner of its own, which the
|
|
||||||
Tantivy index therefore treats as visible to everyone
|
|
||||||
WHEN:
|
|
||||||
- _fulltext_similar_documents() is called with that user
|
|
||||||
THEN:
|
|
||||||
- The version is not returned, since the DB re-check judges it by
|
|
||||||
its root
|
|
||||||
"""
|
|
||||||
owner = UserFactory.create()
|
|
||||||
viewer = UserFactory.create(is_superuser=False)
|
|
||||||
source = DocumentFactory.create(
|
|
||||||
content="shared content phrase",
|
|
||||||
owner=viewer,
|
|
||||||
)
|
|
||||||
root = DocumentFactory.create(
|
|
||||||
content="shared content phrase",
|
|
||||||
owner=owner,
|
|
||||||
)
|
|
||||||
version = DocumentFactory.create(
|
|
||||||
content="shared content phrase",
|
|
||||||
owner=None,
|
|
||||||
root_document=root,
|
|
||||||
version_index=1,
|
|
||||||
)
|
|
||||||
fulltext_backend.add_or_update(source)
|
|
||||||
fulltext_backend.add_or_update(root)
|
|
||||||
fulltext_backend.add_or_update(version)
|
|
||||||
|
|
||||||
result = _fulltext_similar_documents(source, user=viewer, top_k=5)
|
|
||||||
|
|
||||||
assert version.pk not in [s["document_id"] for s in result]
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
@pytest.mark.django_db
|
||||||
@override_settings(LLM_EMBEDDING_BACKEND="huggingface")
|
@override_settings(LLM_EMBEDDING_BACKEND="huggingface")
|
||||||
|
|||||||
@@ -388,6 +388,108 @@ def test_update_llm_index_partial_update(
|
|||||||
assert after[str(doc2.pk)] == before[str(doc2.pk)]
|
assert after[str(doc2.pk)] == before[str(doc2.pk)]
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.django_db
|
||||||
|
class TestLlmIndexVersions:
|
||||||
|
"""The LLM index holds root documents only: a version is indexed as its root."""
|
||||||
|
|
||||||
|
def test_add_or_update_document_indexes_a_version_as_its_root(
|
||||||
|
self,
|
||||||
|
temp_llm_index_dir: Path,
|
||||||
|
mock_embed_model: FakeEmbedding,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with a version
|
||||||
|
WHEN:
|
||||||
|
- The version is passed to llm_index_add_or_update_document
|
||||||
|
THEN:
|
||||||
|
- Only the root document is in the index
|
||||||
|
"""
|
||||||
|
root = DocumentFactory(content="root content")
|
||||||
|
version = DocumentFactory(root_document=root, version_index=1)
|
||||||
|
|
||||||
|
indexing.llm_index_add_or_update_document(version)
|
||||||
|
|
||||||
|
with indexing.get_vector_store() as store:
|
||||||
|
indexed = store.get_modified_times()
|
||||||
|
|
||||||
|
assert set(indexed) == {str(root.pk)}
|
||||||
|
|
||||||
|
def test_rebuild_skips_versions(
|
||||||
|
self,
|
||||||
|
temp_llm_index_dir: Path,
|
||||||
|
mock_embed_model: FakeEmbedding,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with a version
|
||||||
|
WHEN:
|
||||||
|
- The LLM index is rebuilt
|
||||||
|
THEN:
|
||||||
|
- Only the root document is in the index
|
||||||
|
"""
|
||||||
|
root = DocumentFactory()
|
||||||
|
DocumentFactory(root_document=root, version_index=1)
|
||||||
|
|
||||||
|
indexing.update_llm_index(rebuild=True)
|
||||||
|
|
||||||
|
with indexing.get_vector_store() as store:
|
||||||
|
indexed = store.get_modified_times()
|
||||||
|
|
||||||
|
assert set(indexed) == {str(root.pk)}
|
||||||
|
|
||||||
|
def test_rebuild_indexes_the_newest_versions_content(
|
||||||
|
self,
|
||||||
|
temp_llm_index_dir: Path,
|
||||||
|
mock_embed_model: FakeEmbedding,
|
||||||
|
mocker: pytest_mock.MockerFixture,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with a version
|
||||||
|
WHEN:
|
||||||
|
- The LLM index is rebuilt
|
||||||
|
THEN:
|
||||||
|
- The root's text for the index is the version's content, answered
|
||||||
|
from the query rather than a query per document
|
||||||
|
"""
|
||||||
|
root = DocumentFactory(content="stale text")
|
||||||
|
DocumentFactory(root_document=root, version_index=1, content="latest text")
|
||||||
|
spy = mocker.spy(indexing, "build_document_node")
|
||||||
|
|
||||||
|
indexing.update_llm_index(rebuild=True)
|
||||||
|
|
||||||
|
indexed = spy.call_args.args[0]
|
||||||
|
assert indexed.pk == root.pk
|
||||||
|
assert indexed.effective_content == "latest text"
|
||||||
|
|
||||||
|
def test_incremental_update_by_version_id_refreshes_the_root(
|
||||||
|
self,
|
||||||
|
temp_llm_index_dir: Path,
|
||||||
|
mock_embed_model: FakeEmbedding,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An indexed root document with a version whose root was modified since
|
||||||
|
WHEN:
|
||||||
|
- An incremental update is scoped to the version's id
|
||||||
|
THEN:
|
||||||
|
- The root's entry is refreshed and no entry exists for the version
|
||||||
|
"""
|
||||||
|
root = DocumentFactory()
|
||||||
|
version = DocumentFactory(root_document=root, version_index=1)
|
||||||
|
indexing.update_llm_index(rebuild=True)
|
||||||
|
Document.objects.filter(pk=root.pk).update(modified=timezone.now())
|
||||||
|
root.refresh_from_db()
|
||||||
|
|
||||||
|
indexing.update_llm_index(document_ids=[version.pk])
|
||||||
|
|
||||||
|
with indexing.get_vector_store() as store:
|
||||||
|
indexed = store.get_modified_times()
|
||||||
|
|
||||||
|
assert indexed == {str(root.pk): root.modified.isoformat()}
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
@pytest.mark.django_db
|
||||||
def test_add_or_update_document_updates_existing_entry(
|
def test_add_or_update_document_updates_existing_entry(
|
||||||
temp_llm_index_dir: Path,
|
temp_llm_index_dir: Path,
|
||||||
@@ -637,6 +739,7 @@ class TestLlmIndexAddOrUpdateDocumentEmptyContent:
|
|||||||
|
|
||||||
doc = MagicMock(spec=Document)
|
doc = MagicMock(spec=Document)
|
||||||
doc.id = 42
|
doc.id = 42
|
||||||
|
doc.root_document_id = None
|
||||||
# Must not raise
|
# Must not raise
|
||||||
indexing.llm_index_add_or_update_document(doc)
|
indexing.llm_index_add_or_update_document(doc)
|
||||||
|
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ from paperless_ai.embedding import _normalize_llm_index_text
|
|||||||
from paperless_ai.embedding import build_llm_index_text
|
from paperless_ai.embedding import build_llm_index_text
|
||||||
from paperless_ai.embedding import get_configured_model_name
|
from paperless_ai.embedding import get_configured_model_name
|
||||||
from paperless_ai.embedding import get_embedding_model
|
from paperless_ai.embedding import get_embedding_model
|
||||||
|
from paperless_testing.factories import DocumentFactory
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture
|
@pytest.fixture
|
||||||
@@ -46,6 +47,7 @@ def mock_document():
|
|||||||
doc.correspondent.name = "Test Correspondent"
|
doc.correspondent.name = "Test Correspondent"
|
||||||
doc.archive_serial_number = "12345"
|
doc.archive_serial_number = "12345"
|
||||||
doc.content = "This is the document content."
|
doc.content = "This is the document content."
|
||||||
|
doc.get_effective_content.return_value = "This is the document content."
|
||||||
|
|
||||||
cf1 = MagicMock(__str__=lambda x: "Value1")
|
cf1 = MagicMock(__str__=lambda x: "Value1")
|
||||||
cf1.field = MagicMock()
|
cf1.field = MagicMock()
|
||||||
@@ -280,7 +282,7 @@ def test_build_llm_index_text(mock_document):
|
|||||||
|
|
||||||
|
|
||||||
def test_build_llm_index_text_normalizes_ocr_punctuation_runs(mock_document):
|
def test_build_llm_index_text_normalizes_ocr_punctuation_runs(mock_document):
|
||||||
mock_document.content = (
|
mock_document.get_effective_content.return_value = (
|
||||||
"Introduction ................................................ 7\n"
|
"Introduction ................................................ 7\n"
|
||||||
"Hardware Limitation ________________________________________ 9\n"
|
"Hardware Limitation ________________________________________ 9\n"
|
||||||
"Keep short punctuation like INV-100 and ellipses..."
|
"Keep short punctuation like INV-100 and ellipses..."
|
||||||
@@ -294,6 +296,43 @@ def test_build_llm_index_text_normalizes_ocr_punctuation_runs(mock_document):
|
|||||||
assert "ellipses..." in result
|
assert "ellipses..." in result
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.django_db
|
||||||
|
class TestBuildLlmIndexTextVersions:
|
||||||
|
"""A root document is indexed with its effective content, like in the search index."""
|
||||||
|
|
||||||
|
def test_root_uses_the_newest_versions_content(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with two versions
|
||||||
|
WHEN:
|
||||||
|
- The LLM index text is built for the root
|
||||||
|
THEN:
|
||||||
|
- It contains the newest version's content and not the others'
|
||||||
|
"""
|
||||||
|
root = DocumentFactory(content="stale text")
|
||||||
|
DocumentFactory(root_document=root, version_index=1, content="older text")
|
||||||
|
DocumentFactory(root_document=root, version_index=2, content="latest text")
|
||||||
|
|
||||||
|
text = build_llm_index_text(root)
|
||||||
|
|
||||||
|
assert "latest text" in text
|
||||||
|
assert "stale text" not in text
|
||||||
|
assert "older text" not in text
|
||||||
|
|
||||||
|
def test_root_without_versions_uses_its_own_content(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document without versions
|
||||||
|
WHEN:
|
||||||
|
- The LLM index text is built for it
|
||||||
|
THEN:
|
||||||
|
- It contains the document's own content
|
||||||
|
"""
|
||||||
|
root = DocumentFactory(content="own text")
|
||||||
|
|
||||||
|
assert "own text" in build_llm_index_text(root)
|
||||||
|
|
||||||
|
|
||||||
def test_normalize_llm_index_text_collapses_ocr_leaders_without_joining_lines():
|
def test_normalize_llm_index_text_collapses_ocr_leaders_without_joining_lines():
|
||||||
assert _normalize_llm_index_text("A........B\nC____D----E") == "A B\nC D E"
|
assert _normalize_llm_index_text("A........B\nC____D----E") == "A B\nC D E"
|
||||||
|
|
||||||
|
|||||||
Reference in new issue
Block a user