Files
paperless-ngx/src/documents/management/commands/document_index.py
T
stumpylog 210d97a522 Fix: index document versions as their root document
The search index and the LLM index hold root documents only, a root being
indexed with its newest version's content. Every write path therefore had to
remember to hand them the root. Several did not: adding or deleting a note on
a version, restoring a trashed document together with its versions, and
reprocessing a version all wrote the version into the index under its own id,
where it could be returned as a separate search result. The paths that walk the
whole library did the same: the document_index reindex command put every
version in the search index, a full LLM index rebuild embedded every version,
and an incremental LLM update scoped to a version id indexed it under its own
id. Asking for documents like a version looked the version's own id up in the
index and silently found nothing.

WriteBatch.add_or_update, WriteBatch.add_or_update_ids and
llm_index_add_or_update_document now resolve a version to its root themselves.
The reindex command and update_llm_index only walk root documents, and an
incremental LLM update is scoped to the roots of the given ids through
versioning.root_document_ids, which add_or_update_ids shares. more_like_id
returns the root's id for a version. The reprocess task no longer picks the root
for the indexes and only still clears the caches of both documents.
2026-10-08 15:22:02 -07:00

109 lines
3.8 KiB
Python

import logging
from django.conf import settings
from django.db import transaction
from documents.management.commands.base import PaperlessCommand
from documents.models import Document
from documents.search import get_backend
from documents.search import needs_rebuild
from documents.search import reset_backend
from documents.search import wipe_index
logger = logging.getLogger("paperless.management.document_index")
class Command(PaperlessCommand):
"""
Django management command for search index operations.
Provides subcommands for reindexing documents and optimizing the search index.
Supports conditional reindexing based on schema version and language changes.
"""
help = "Manages the document index."
supports_progress_bar = True
supports_multiprocessing = False
def add_arguments(self, parser):
super().add_arguments(parser)
parser.add_argument("command", choices=["reindex", "optimize"])
parser.add_argument(
"--recreate",
action="store_true",
default=False,
help="Wipe and recreate the index from scratch (only used with reindex).",
)
parser.add_argument(
"--if-needed",
action="store_true",
default=False,
help=(
"Skip reindex if the index is already up to date. "
"Checks schema version and search language sentinels. "
"Safe to run on every startup or upgrade."
),
)
parser.add_argument(
"--heap-size-mb",
type=int,
default=None,
help=(
"Tantivy writer memory budget in MB for a full reindex "
"(split across writer threads). Defaults to 512MB; lower "
"this on memory-constrained hosts. Larger values buffer "
"more documents before flushing a segment, deferring "
"merge work rather than avoiding it."
),
)
def handle(self, *args, **options):
with transaction.atomic():
if options["command"] == "reindex":
if options.get("if_needed") and not needs_rebuild(settings.INDEX_DIR):
self.stdout.write("Search index is up to date.")
return
if options.get("recreate"):
wipe_index(settings.INDEX_DIR)
documents = (
Document.objects.filter(root_document__isnull=True)
.select_related(
"correspondent",
"document_type",
"storage_path",
"owner",
)
.prefetch_related(
"tags",
"notes__user",
"custom_fields__field",
"versions",
"barcodes",
"versions__barcodes",
)
)
total = documents.count()
rebuild_kwargs = {}
if options.get("heap_size_mb") is not None:
rebuild_kwargs["writer_heap_bytes"] = (
options["heap_size_mb"] * 1_000_000
)
get_backend().rebuild(
documents,
iter_wrapper=lambda pairs: self.track(
pairs,
description="Indexing documents...",
total=total,
),
**rebuild_kwargs,
)
reset_backend()
elif options["command"] == "optimize":
logger.info(
"document_index optimize is a no-op — Tantivy manages "
"segment merging automatically.",
)