diff --git a/docker/rootfs/etc/s6-overlay/s6-rc.d/init-complete/dependencies.d/init-llmindex-migrate b/docker/rootfs/etc/s6-overlay/s6-rc.d/init-complete/dependencies.d/init-llmindex-migrate new file mode 100644 index 000000000..e69de29bb diff --git a/docker/rootfs/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/dependencies.d/init-migrations b/docker/rootfs/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/dependencies.d/init-migrations new file mode 100644 index 000000000..e69de29bb diff --git a/docker/rootfs/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/run b/docker/rootfs/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/run new file mode 100755 index 000000000..50fa691c2 --- /dev/null +++ b/docker/rootfs/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/run @@ -0,0 +1,12 @@ +#!/command/with-contenv /usr/bin/bash +# shellcheck shell=bash + +declare -r log_prefix="[init-llmindex-migrate]" + +echo "${log_prefix} Checking for pending LLM index migrations..." +cd "${PAPERLESS_SRC_DIR}" +if [[ -n "${USER_IS_NON_ROOT}" ]]; then + python3 manage.py document_llmindex migrate +else + s6-setuidgid paperless python3 manage.py document_llmindex migrate +fi diff --git a/docker/rootfs/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/type b/docker/rootfs/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/type new file mode 100644 index 000000000..bdd22a185 --- /dev/null +++ b/docker/rootfs/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/type @@ -0,0 +1 @@ +oneshot diff --git a/docker/rootfs/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/up b/docker/rootfs/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/up new file mode 100644 index 000000000..c2016d47a --- /dev/null +++ b/docker/rootfs/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/up @@ -0,0 +1 @@ +/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/run diff --git a/docs/administration.md b/docs/administration.md index 8e27df982..fbc636935 100644 --- a/docs/administration.md +++ b/docs/administration.md @@ -212,6 +212,16 @@ following: This is a no-op if the index is already up to date, so it is safe to run on every upgrade. +5. Migrate the LLM index if needed. + + ```shell-session + cd src + python3 manage.py document_llmindex migrate + ``` + + This is a no-op if the index schema is already current, so it is safe + to run on every upgrade. + ### Database Upgrades Paperless-ngx is compatible with Django-supported versions of PostgreSQL and MariaDB and it is generally @@ -532,7 +542,7 @@ index is updated automatically on the schedule set by can manage it manually: ``` -document_llmindex {rebuild,update,compact} +document_llmindex {rebuild,update,compact,migrate} ``` Specify `rebuild` to build the index from scratch from all documents in the database. Use diff --git a/src/documents/management/commands/document_llmindex.py b/src/documents/management/commands/document_llmindex.py index 7b34ca9a8..216ab631d 100644 --- a/src/documents/management/commands/document_llmindex.py +++ b/src/documents/management/commands/document_llmindex.py @@ -3,6 +3,7 @@ from typing import Any from documents.management.commands.base import PaperlessCommand from documents.tasks import llmindex_index from paperless_ai.indexing import llm_index_compact +from paperless_ai.indexing import llm_index_migrate class Command(PaperlessCommand): @@ -13,12 +14,18 @@ class Command(PaperlessCommand): def add_arguments(self, parser: Any) -> None: super().add_arguments(parser) - parser.add_argument("command", choices=["rebuild", "update", "compact"]) + parser.add_argument( + "command", + choices=["rebuild", "update", "compact", "migrate"], + ) def handle(self, *args: Any, **options: Any) -> None: if options["command"] == "compact": llm_index_compact() return + if options["command"] == "migrate": + llm_index_migrate() + return llmindex_index( rebuild=options["command"] == "rebuild", iter_wrapper=lambda docs: self.track( diff --git a/src/documents/tests/management/test_management_document_llmindex.py b/src/documents/tests/management/test_management_document_llmindex.py index b8a05dd85..3b75338d2 100644 --- a/src/documents/tests/management/test_management_document_llmindex.py +++ b/src/documents/tests/management/test_management_document_llmindex.py @@ -9,6 +9,7 @@ if TYPE_CHECKING: _COMPACT = "documents.management.commands.document_llmindex.llm_index_compact" _INDEX = "documents.management.commands.document_llmindex.llmindex_index" +_MIGRATE = "documents.management.commands.document_llmindex.llm_index_migrate" class TestDocumentLlmindexCommand: @@ -17,6 +18,11 @@ class TestDocumentLlmindexCommand: call_command("document_llmindex", "compact") mock_compact.assert_called_once_with() + def test_migrate_calls_llm_index_migrate(self, mocker: MockerFixture) -> None: + mock_migrate = mocker.patch(_MIGRATE) + call_command("document_llmindex", "migrate") + mock_migrate.assert_called_once_with() + def test_rebuild_calls_llmindex_index_with_rebuild_true( self, mocker: MockerFixture, diff --git a/src/paperless_ai/indexing.py b/src/paperless_ai/indexing.py index ff08d2aeb..cd9a5deee 100644 --- a/src/paperless_ai/indexing.py +++ b/src/paperless_ai/indexing.py @@ -175,6 +175,24 @@ def _exclude_readers(): lock.close() +def _with_exclusive_access(operation: str, fn): + """Run ``fn()`` with exclusive index access (see ``_exclude_readers()``), + for compaction/migration file swaps that must not run while readers are + active. Returns ``fn()``'s result, or None (after logging) if active + readers do not drain within ``LLM_INDEX_COMPACTION_LOCK_TIMEOUT`` -- + callers skip the operation this run; it retries next time. + """ + try: + with _exclude_readers(): + return fn() + except Timeout: + logger.info( + "Skipping LLM index %s: index readers are active; will retry next run.", + operation, + ) + return None + + @contextmanager def write_store(embed_model_name: str | None = None): """Acquire the write lock and yield the vector store. @@ -199,6 +217,21 @@ def write_store(embed_model_name: str | None = None): yield store +def _check_and_run_migrations(store: "PaperlessSqliteVecVectorStore") -> bool: + """Run any pending structural migrations, returning True if a pending + re-embed migration needs the caller to force a rebuild -- never + triggered automatically here. Safe to call before any write, including + delete()/upsert_document(): has_pending_migration() (see its docstring) + keeps this a no-op, with no exclusive access taken, once the store is + current. + """ + if not store.has_pending_migration(): + return False + return bool( + _with_exclusive_access("migration check", store.check_and_run_migrations), + ) + + def _safe_related_name(document: Document, field: str) -> str | None: """ Returns the ``name`` of a related object (correspondent, document_type, @@ -370,15 +403,7 @@ def update_llm_index( happens, since a rebuild always covers the whole library regardless. """ with write_store() as store: - try: - with _exclude_readers(): - needs_reembed = store.check_and_run_migrations() - except Timeout: - logger.info( - "Skipping LLM index migration check: index readers are active; " - "will retry next run.", - ) - needs_reembed = False + needs_reembed = _check_and_run_migrations(store) if needs_reembed: logger.warning( "LLM index migration requires re-embedding; forcing rebuild.", @@ -443,14 +468,7 @@ def update_llm_index( else "No changes detected in LLM index." ) - try: - with _exclude_readers(): - store.compact() - except Timeout: - logger.info( - "Skipping LLM index compaction: index readers are active; " - "will retry next run.", - ) + _with_exclusive_access("compaction", store.compact) return msg @@ -465,25 +483,60 @@ def llm_index_add_or_update_document(document: Document): _embed_nodes(new_nodes, get_embedding_model(config)) with write_store(embed_model_name=get_configured_model_name(config)) as store: + needs_reembed = _check_and_run_migrations(store) + if needs_reembed: + logger.warning( + "Skipping incremental LLM index update for document %s: the " + "index requires re-embedding first. Run 'document_llmindex " + "rebuild' to resolve.", + document.id, + ) + return store.upsert_document(str(document.id), new_nodes) +def llm_index_migrate() -> None: + """Apply any pending LLM index schema migrations, with no reindex. + + Intended to run unconditionally on every startup (see the + init-llmindex-migrate container step and the bare-metal upgrade docs): + has_pending_migration() short-circuits to a metadata-only read once the + store is current, so a healthy install pays almost nothing here. Only + ever applies structural migrations -- a pending re-embed migration is + left for the explicit, deliberate rebuild path (``document_llmindex + update``/``rebuild``) to resolve, since re-embedding can be slow and, + for a metered embedding backend, cost money. + """ + if not AIConfig().llm_index_enabled: + return + with write_store() as store: + needs_reembed = _check_and_run_migrations(store) + if needs_reembed: + logger.warning( + "LLM index requires re-embedding, which this automatic migration " + "check will not do on its own -- it can be slow and, for a " + "metered embedding backend, cost money. Run " + "'document_llmindex rebuild' manually when ready.", + ) + + def llm_index_compact() -> None: """Compact the index immediately, rebuilding the table to reclaim space.""" with write_store() as store: - try: - with _exclude_readers(): - store.compact(force=True) - except Timeout: - logger.info( - "Skipping LLM index compaction: index readers are active; " - "will retry next run.", - ) + _with_exclusive_access("compaction", lambda: store.compact(force=True)) def llm_index_remove_document(document: Document): """Remove a document's chunks from the LLM index.""" with write_store() as store: + if _check_and_run_migrations(store): + logger.warning( + "Skipping removal of document %s from the LLM index: the " + "index requires re-embedding first. Run 'document_llmindex " + "rebuild' to resolve.", + document.id, + ) + return store.delete(str(document.id)) diff --git a/src/paperless_ai/migrations/__init__.py b/src/paperless_ai/migrations/__init__.py new file mode 100644 index 000000000..af63f3400 --- /dev/null +++ b/src/paperless_ai/migrations/__init__.py @@ -0,0 +1,60 @@ +"""Schema migrations for the sqlite-vec vector store. + +Each migration lives in its own module here, named ``mNNNN_description.py`` +(e.g. ``m0001_v1_to_v2.py`` -- a leading digit isn't a valid Python +identifier, hence the ``m`` prefix, unlike Django's own numbered migrations, +which load via a dynamic ``importlib.import_module()`` call rather than a +static import statement), and registers itself into ``MIGRATIONS`` at import +time. ``vector_store.py`` imports those modules at the bottom of the file, +purely for that registration side effect, after ``PaperlessSqliteVecVectorStore`` +is fully defined -- migrations need it to implement ``apply()`` (see +``Migration`` below). + +To add a new migration: add a new ``mNNNN_description.py`` module here that +imports ``PaperlessSqliteVecVectorStore`` from ``paperless_ai.vector_store``, +defines its ``apply()``, and appends a ``Migration`` to ``MIGRATIONS``; then +import that module at the bottom of ``vector_store.py`` and bump +``SCHEMA_VERSION`` there. A migration must freeze its own historical DDL for +any side table its target version depends on (``DROP TABLE IF EXISTS`` + +its own literal ``CREATE TABLE``/``CREATE INDEX`` statements) rather than +delegating to any "current schema" helper -- see ``m0001_v1_to_v2.py`` for +why and the worked example. +""" + +import sqlite3 +from collections.abc import Callable +from dataclasses import dataclass +from dataclasses import field +from typing import Literal + + +@dataclass +class Migration: + """A schema migration for the sqlite-vec vector store. + + kind="structural": rows are copied into a new-schema file with no + re-embedding needed. Supply ``apply(src_conn, dst_conn, dim)``, which + must create every table its target schema needs in ``dst_conn`` and copy + ``src_conn``'s rows and relevant ``index_meta`` keys into it. + ``schema_version`` is written by the migration runner after ``apply`` + returns, not by ``apply`` itself. + + kind="re-embed": the new schema requires fresh embeddings. + ``check_and_run_migrations()`` returns True when it encounters one of + these so the caller can force a full rebuild (which recreates the table + at the current SCHEMA_VERSION). + """ + + from_version: int + to_version: int + kind: Literal["structural", "re-embed"] + description: str + apply: Callable[[sqlite3.Connection, sqlite3.Connection, int], None] | None = field( + default=None, + repr=False, + ) + + +# Registry of all schema migrations in order, populated by each migration +# module's import-time registration (see the module docstring above). +MIGRATIONS: list[Migration] = [] diff --git a/src/paperless_ai/tests/test_ai_indexing.py b/src/paperless_ai/tests/test_ai_indexing.py index 7ae5477cb..c018b9a80 100644 --- a/src/paperless_ai/tests/test_ai_indexing.py +++ b/src/paperless_ai/tests/test_ai_indexing.py @@ -1,3 +1,4 @@ +import logging from pathlib import Path from unittest.mock import MagicMock from unittest.mock import patch @@ -776,6 +777,7 @@ class TestLlmIndexLocking: mocker: pytest_mock.MockerFixture, ) -> None: mock_store = MagicMock() + mock_store.has_pending_migration.return_value = False mocker.patch( "paperless_ai.indexing.write_store", return_value=mocker.MagicMock( @@ -796,12 +798,45 @@ class TestLlmIndexLocking: mock_store.upsert_document.assert_called_once() + def test_add_or_update_document_skips_write_when_reembed_pending( + self, + temp_llm_index_dir: Path, + mock_embed_model: FakeEmbedding, + mocker: pytest_mock.MockerFixture, + ) -> None: + """A pending re-embed migration must block the incremental write, + not let it proceed against a schema that just changed underneath it. + """ + mock_store = MagicMock() + mock_store.has_pending_migration.return_value = True + mock_store.check_and_run_migrations.return_value = True + mocker.patch( + "paperless_ai.indexing.write_store", + return_value=mocker.MagicMock( + __enter__=mocker.MagicMock(return_value=mock_store), + __exit__=mocker.MagicMock(return_value=False), + ), + ) + mock_node = MagicMock() + mock_node.get_content.return_value = "fake node text" + mocker.patch( + "paperless_ai.indexing.build_document_node", + return_value=[mock_node], + ) + + doc = MagicMock(spec=Document) + doc.id = 1 + indexing.llm_index_add_or_update_document(doc) + + mock_store.upsert_document.assert_not_called() + def test_remove_document_uses_write_store( self, temp_llm_index_dir: Path, mocker: pytest_mock.MockerFixture, ) -> None: mock_store = MagicMock() + mock_store.has_pending_migration.return_value = False mocker.patch( "paperless_ai.indexing.write_store", return_value=mocker.MagicMock( @@ -816,6 +851,31 @@ class TestLlmIndexLocking: mock_store.delete.assert_called_once_with("1") + def test_remove_document_skips_write_when_reembed_pending( + self, + temp_llm_index_dir: Path, + mocker: pytest_mock.MockerFixture, + ) -> None: + """A pending re-embed migration must block the delete too, for the + same consistency reason as the incremental-update path. + """ + mock_store = MagicMock() + mock_store.has_pending_migration.return_value = True + mock_store.check_and_run_migrations.return_value = True + mocker.patch( + "paperless_ai.indexing.write_store", + return_value=mocker.MagicMock( + __enter__=mocker.MagicMock(return_value=mock_store), + __exit__=mocker.MagicMock(return_value=False), + ), + ) + + doc = MagicMock(spec=Document) + doc.id = 1 + indexing.llm_index_remove_document(doc) + + mock_store.delete.assert_not_called() + def test_update_llm_index_rebuild_uses_write_store( self, temp_llm_index_dir: Path, @@ -888,6 +948,76 @@ class TestVectorStoreIndexing: assert rows >= 1 +class TestLlmIndexMigrate: + def test_noop_when_ai_disabled(self, mocker: pytest_mock.MockerFixture) -> None: + """ + GIVEN: + - AI/LLM index support is disabled in configuration + WHEN: + - llm_index_migrate() is called + THEN: + - No store is opened and no migration check runs + """ + mocker.patch( + "paperless_ai.indexing.AIConfig", + return_value=mocker.Mock(llm_index_enabled=False), + ) + write_store_mock = mocker.patch("paperless_ai.indexing.write_store") + indexing.llm_index_migrate() + write_store_mock.assert_not_called() + + def test_runs_pending_migration_when_enabled( + self, + mocker: pytest_mock.MockerFixture, + ) -> None: + """ + GIVEN: + - AI/LLM index support is enabled + WHEN: + - llm_index_migrate() is called + THEN: + - The store is opened for write and a migration check runs + """ + mocker.patch( + "paperless_ai.indexing.AIConfig", + return_value=mocker.Mock(llm_index_enabled=True), + ) + store_mock = mocker.MagicMock() + store_mock.has_pending_migration.return_value = False + write_store_cm = mocker.patch("paperless_ai.indexing.write_store") + write_store_cm.return_value.__enter__.return_value = store_mock + indexing.llm_index_migrate() + store_mock.has_pending_migration.assert_called_once() + + def test_logs_warning_when_reembed_needed( + self, + mocker: pytest_mock.MockerFixture, + caplog: pytest.LogCaptureFixture, + ) -> None: + """ + GIVEN: + - AI/LLM index support is enabled + - A pending migration requires re-embedding + WHEN: + - llm_index_migrate() is called + THEN: + - A warning directs the operator to run a manual rebuild, since + this automatic check must never re-embed on its own + """ + mocker.patch( + "paperless_ai.indexing.AIConfig", + return_value=mocker.Mock(llm_index_enabled=True), + ) + store_mock = mocker.MagicMock() + store_mock.has_pending_migration.return_value = True + store_mock.check_and_run_migrations.return_value = True + write_store_cm = mocker.patch("paperless_ai.indexing.write_store") + write_store_cm.return_value.__enter__.return_value = store_mock + with caplog.at_level(logging.WARNING, logger="paperless_ai.indexing"): + indexing.llm_index_migrate() + assert "requires re-embedding" in caplog.text + + @pytest.mark.django_db class TestQuerySimilarDocuments: def test_query_similar_documents_respects_allowed_ids( diff --git a/src/paperless_ai/tests/test_vector_store.py b/src/paperless_ai/tests/test_vector_store.py index 75cc0d17e..908627d27 100644 --- a/src/paperless_ai/tests/test_vector_store.py +++ b/src/paperless_ai/tests/test_vector_store.py @@ -9,11 +9,11 @@ from llama_index.core.vector_stores.types import MetadataFilter from llama_index.core.vector_stores.types import MetadataFilters from llama_index.core.vector_stores.types import VectorStoreQuery +from paperless_ai.migrations import MIGRATIONS +from paperless_ai.migrations import Migration from paperless_ai.vector_store import DB_FILENAME from paperless_ai.vector_store import DEFAULT_TABLE_NAME -from paperless_ai.vector_store import MIGRATIONS from paperless_ai.vector_store import SCHEMA_VERSION -from paperless_ai.vector_store import Migration from paperless_ai.vector_store import PaperlessSqliteVecVectorStore from paperless_ai.vector_store import _build_where @@ -646,3 +646,50 @@ class TestMigrations: assert result is True assert self._schema_version(store) == 2 + + def test_has_pending_migration_false_when_no_table( + self, + store: PaperlessSqliteVecVectorStore, + ) -> None: + """ + GIVEN: + - A vector store with no table created yet + WHEN: + - has_pending_migration() is checked + THEN: + - False is returned (nothing to migrate before anything exists) + """ + assert store.has_pending_migration() is False + + def test_has_pending_migration_false_at_current_version( + self, + store: PaperlessSqliteVecVectorStore, + ) -> None: + """ + GIVEN: + - A store at the current SCHEMA_VERSION + WHEN: + - has_pending_migration() is checked + THEN: + - False is returned + """ + store.add([make_node("a1", "1")]) + assert store.has_pending_migration() is False + + def test_has_pending_migration_true_when_behind( + self, + store: PaperlessSqliteVecVectorStore, + ) -> None: + """ + GIVEN: + - A store whose schema_version has been forced behind SCHEMA_VERSION + WHEN: + - has_pending_migration() is checked + THEN: + - True is returned + """ + store.add([make_node("a1", "1")]) + store.client.execute( + "UPDATE index_meta SET value = '0' WHERE key = 'schema_version'", + ) + assert store.has_pending_migration() is True diff --git a/src/paperless_ai/vector_store.py b/src/paperless_ai/vector_store.py index f6f333576..4c4d0b7a2 100644 --- a/src/paperless_ai/vector_store.py +++ b/src/paperless_ai/vector_store.py @@ -2,16 +2,12 @@ import json import logging import sqlite3 import struct -from collections.abc import Callable from collections.abc import Iterator from collections.abc import Sequence from contextlib import contextmanager -from dataclasses import dataclass -from dataclasses import field from pathlib import Path from types import TracebackType from typing import Any -from typing import Literal import sqlite_vec from llama_index.core.bridge.pydantic import PrivateAttr @@ -26,6 +22,9 @@ from llama_index.core.vector_stores.types import VectorStoreQueryResult from llama_index.core.vector_stores.utils import metadata_dict_to_node from llama_index.core.vector_stores.utils import node_to_metadata_dict +from paperless_ai.migrations import MIGRATIONS +from paperless_ai.migrations import Migration + logger = logging.getLogger("paperless_ai.vector_store") DB_FILENAME = "llmindex.db" @@ -53,38 +52,6 @@ COMPACT_BATCH_SIZE = 500 _FILTER_COLUMNS = frozenset({"document_id", "modified"}) -@dataclass -class Migration: - """A schema migration for the sqlite-vec vector store. - - kind="structural": rows are copied into a new-schema file with no - re-embedding needed. Supply ``apply(src_conn, dst_conn, dim)`` which - must create the vec0 table in ``dst_conn``, copy all rows from - ``src_conn``, and write ``dim`` / ``embed_model`` / ``total_inserts`` to - ``dst_conn``'s ``index_meta``. ``schema_version`` is written by the - migration runner after ``apply`` returns. - - kind="re-embed": the new schema requires fresh embeddings. - ``check_and_run_migrations()`` returns True when it encounters one of - these so the caller can force a full rebuild (which recreates the table - at the current SCHEMA_VERSION). - """ - - from_version: int - to_version: int - kind: Literal["structural", "re-embed"] - description: str - apply: Callable[[sqlite3.Connection, sqlite3.Connection, int], None] | None = field( - default=None, - repr=False, - ) - - -# Registry of all schema migrations in order. Empty at v1 -- this is the -# baseline. Add entries here (and bump SCHEMA_VERSION) when the schema changes. -MIGRATIONS: list[Migration] = [] - - def _pack(embedding: Sequence[float]) -> bytes: return struct.pack(f"{len(embedding)}f", *embedding) @@ -551,6 +518,31 @@ class PaperlessSqliteVecVectorStore(BasePydanticVectorStore): Path(compact_path).replace(db_path) self._conn = self._open_connection(db_path) + def _stored_schema_version(self) -> int | None: + """The schema_version recorded in index_meta, or None if no table + exists. A missing key (a store predating version tracking) is + treated as SCHEMA_VERSION -- i.e. already current -- since no + migration in MIGRATIONS targets a version before tracking began. + """ + if not self.table_exists(): + return None + raw = self._meta_get("schema_version") + return int(raw) if raw is not None else SCHEMA_VERSION + + def has_pending_migration(self) -> bool: + """Cheaply check whether a migration is pending, with no exclusive + access needed -- just a metadata read under the connection callers + already hold via the write FileLock. + + Callers should only pay for check_and_run_migrations()'s exclusive + access (a structural migration's file swap must not run while + readers are active) when this returns True, so that the common + case -- already at SCHEMA_VERSION -- never contends with readers + or a concurrent compaction. + """ + current = self._stored_schema_version() + return current is not None and current < SCHEMA_VERSION + def check_and_run_migrations(self) -> bool: """Apply any pending schema migrations to the store. @@ -559,15 +551,13 @@ class PaperlessSqliteVecVectorStore(BasePydanticVectorStore): this method returns True when one is encountered so the caller can force a full rebuild (which recreates the table at SCHEMA_VERSION). - Must be called under the write FileLock. No-op when the table does - not exist or is already at SCHEMA_VERSION. + Must be called under the write FileLock, with readers excluded (see + has_pending_migration() for a cheap pre-check that avoids paying for + that exclusion in the common case). No-op when the table does not + exist or is already at SCHEMA_VERSION. """ - if not self.table_exists(): - return False - - raw = self._meta_get("schema_version") - current = int(raw) if raw is not None else SCHEMA_VERSION - if current >= SCHEMA_VERSION: + current = self._stored_schema_version() + if current is None or current >= SCHEMA_VERSION: return False pending = sorted( @@ -579,7 +569,7 @@ class PaperlessSqliteVecVectorStore(BasePydanticVectorStore): if migration.kind == "re-embed": logger.warning( "LLM index schema v%d -> v%d requires re-embedding (%s); " - "forcing full rebuild.", + "the caller must force a rebuild.", migration.from_version, migration.to_version, migration.description,