mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-10-09 09:37:12 +00:00
The search index and the LLM index hold root documents only, a root being indexed with its newest version's content. Every write path therefore had to remember to hand them the root. Several did not: adding or deleting a note on a version, restoring a trashed document together with its versions, and reprocessing a version all wrote the version into the index under its own id, where it could be returned as a separate search result. The paths that walk the whole library did the same: the document_index reindex command put every version in the search index, a full LLM index rebuild embedded every version, and an incremental LLM update scoped to a version id indexed it under its own id. Asking for documents like a version looked the version's own id up in the index and silently found nothing. WriteBatch.add_or_update, WriteBatch.add_or_update_ids and llm_index_add_or_update_document now resolve a version to its root themselves. The reindex command and update_llm_index only walk root documents, and an incremental LLM update is scoped to the roots of the given ids through versioning.root_document_ids, which add_or_update_ids shares. more_like_id returns the root's id for a version. The reprocess task no longer picks the root for the indexes and only still clears the caches of both documents.
109 lines
3.8 KiB
Python
109 lines
3.8 KiB
Python
import logging
|
|
|
|
from django.conf import settings
|
|
from django.db import transaction
|
|
|
|
from documents.management.commands.base import PaperlessCommand
|
|
from documents.models import Document
|
|
from documents.search import get_backend
|
|
from documents.search import needs_rebuild
|
|
from documents.search import reset_backend
|
|
from documents.search import wipe_index
|
|
|
|
logger = logging.getLogger("paperless.management.document_index")
|
|
|
|
|
|
class Command(PaperlessCommand):
|
|
"""
|
|
Django management command for search index operations.
|
|
|
|
Provides subcommands for reindexing documents and optimizing the search index.
|
|
Supports conditional reindexing based on schema version and language changes.
|
|
"""
|
|
|
|
help = "Manages the document index."
|
|
|
|
supports_progress_bar = True
|
|
supports_multiprocessing = False
|
|
|
|
def add_arguments(self, parser):
|
|
super().add_arguments(parser)
|
|
parser.add_argument("command", choices=["reindex", "optimize"])
|
|
parser.add_argument(
|
|
"--recreate",
|
|
action="store_true",
|
|
default=False,
|
|
help="Wipe and recreate the index from scratch (only used with reindex).",
|
|
)
|
|
parser.add_argument(
|
|
"--if-needed",
|
|
action="store_true",
|
|
default=False,
|
|
help=(
|
|
"Skip reindex if the index is already up to date. "
|
|
"Checks schema version and search language sentinels. "
|
|
"Safe to run on every startup or upgrade."
|
|
),
|
|
)
|
|
parser.add_argument(
|
|
"--heap-size-mb",
|
|
type=int,
|
|
default=None,
|
|
help=(
|
|
"Tantivy writer memory budget in MB for a full reindex "
|
|
"(split across writer threads). Defaults to 512MB; lower "
|
|
"this on memory-constrained hosts. Larger values buffer "
|
|
"more documents before flushing a segment, deferring "
|
|
"merge work rather than avoiding it."
|
|
),
|
|
)
|
|
|
|
def handle(self, *args, **options):
|
|
with transaction.atomic():
|
|
if options["command"] == "reindex":
|
|
if options.get("if_needed") and not needs_rebuild(settings.INDEX_DIR):
|
|
self.stdout.write("Search index is up to date.")
|
|
return
|
|
if options.get("recreate"):
|
|
wipe_index(settings.INDEX_DIR)
|
|
|
|
documents = (
|
|
Document.objects.filter(root_document__isnull=True)
|
|
.select_related(
|
|
"correspondent",
|
|
"document_type",
|
|
"storage_path",
|
|
"owner",
|
|
)
|
|
.prefetch_related(
|
|
"tags",
|
|
"notes__user",
|
|
"custom_fields__field",
|
|
"versions",
|
|
"barcodes",
|
|
"versions__barcodes",
|
|
)
|
|
)
|
|
total = documents.count()
|
|
rebuild_kwargs = {}
|
|
if options.get("heap_size_mb") is not None:
|
|
rebuild_kwargs["writer_heap_bytes"] = (
|
|
options["heap_size_mb"] * 1_000_000
|
|
)
|
|
get_backend().rebuild(
|
|
documents,
|
|
iter_wrapper=lambda pairs: self.track(
|
|
pairs,
|
|
description="Indexing documents...",
|
|
total=total,
|
|
),
|
|
**rebuild_kwargs,
|
|
)
|
|
reset_backend()
|
|
|
|
elif options["command"] == "optimize":
|
|
logger.info(
|
|
"document_index optimize is a no-op — Tantivy manages "
|
|
"segment merging automatically.",
|
|
)
|