Feature: run pending LLM index migrations automatically on startup

Add document_llmindex migrate, a cheap check-only path (no reindex) safe
to run unconditionally: has_pending_migration() short-circuits to a
metadata-only read once the store is current, so a healthy install pays
almost nothing. If a pending migration would require re-embedding, it
only logs a warning and leaves the index as-is -- re-embedding can be
slow and, for a metered embedding backend, cost money, so it stays a
deliberate manual action (document_llmindex rebuild), never automatic.

Wire it into the Docker image as a new init-llmindex-migrate oneshot
(modeled on init-search-index, gated on init-migrations), and document
the equivalent manual step for bare-metal upgrades.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
stumpylog
2026-07-28 10:48:24 -07:00
co-authored by Claude Sonnet 5
parent b21d50645d
commit df8a6cc715
11 changed files with 181 additions and 4 deletions
@@ -0,0 +1,12 @@
#!/command/with-contenv /usr/bin/bash
# shellcheck shell=bash
declare -r log_prefix="[init-llmindex-migrate]"
echo "${log_prefix} Checking LLM index schema..."
cd "${PAPERLESS_SRC_DIR}"
if [[ -n "${USER_IS_NON_ROOT}" ]]; then
python3 manage.py document_llmindex migrate
else
s6-setuidgid paperless python3 manage.py document_llmindex migrate
fi
@@ -0,0 +1 @@
oneshot
@@ -0,0 +1 @@
/etc/s6-overlay/s6-rc.d/init-llmindex-migrate/run
+23 -2
View File
@@ -104,7 +104,8 @@ $ docker compose down
docker compose up
```
Running `docker compose up` will also apply any new database migrations.
Running `docker compose up` will also apply any new database migrations,
and, if the LLM index is enabled, any pending LLM index schema migrations.
If you see everything working, press CTRL+C once to gracefully stop
paperless. Then you can start paperless-ngx with `-d` to have it run in
the background.
@@ -212,6 +213,17 @@ following:
This is a no-op if the index is already up to date, so it is safe to
run on every upgrade.
5. If the LLM index is enabled, apply any pending LLM index schema
migrations.
```shell-session
cd src
python3 manage.py document_llmindex migrate
```
This is a no-op if the index is already up to date (or the LLM index
is disabled), so it is safe to run on every upgrade.
### Database Upgrades
Paperless-ngx is compatible with Django-supported versions of PostgreSQL and MariaDB and it is generally
@@ -532,7 +544,7 @@ index is updated automatically on the schedule set by
can manage it manually:
```
document_llmindex {rebuild,update,compact}
document_llmindex {rebuild,update,compact,migrate}
```
Specify `rebuild` to build the index from scratch from all documents in the database. Use
@@ -544,6 +556,15 @@ scheduled task runs.
Specify `compact` to reclaim space and optimize the on-disk vector store.
Specify `migrate` to apply any pending index schema migrations without a full reindex.
This is a no-op if the index is already up to date, so it is safe to run on every
startup or upgrade; it is what the container's startup sequence and the
[bare-metal upgrade steps](#bare-metal-updating) run automatically. If a pending
migration would require re-embedding every document, `migrate` only logs a warning and
leaves the index as-is -- re-embedding can be slow and, for a metered embedding backend,
cost money, so it is never triggered automatically. Run `rebuild` yourself when you are
ready.
!!! note
These commands have no effect unless AI is enabled and an embedding backend is
@@ -3,6 +3,7 @@ from typing import Any
from documents.management.commands.base import PaperlessCommand
from documents.tasks import llmindex_index
from paperless_ai.indexing import llm_index_compact
from paperless_ai.indexing import llm_index_migrate
class Command(PaperlessCommand):
@@ -13,12 +14,18 @@ class Command(PaperlessCommand):
def add_arguments(self, parser: Any) -> None:
super().add_arguments(parser)
parser.add_argument("command", choices=["rebuild", "update", "compact"])
parser.add_argument(
"command",
choices=["rebuild", "update", "compact", "migrate"],
)
def handle(self, *args: Any, **options: Any) -> None:
if options["command"] == "compact":
llm_index_compact()
return
if options["command"] == "migrate":
llm_index_migrate()
return
llmindex_index(
rebuild=options["command"] == "rebuild",
iter_wrapper=lambda docs: self.track(
@@ -8,6 +8,7 @@ if TYPE_CHECKING:
from pytest_mock import MockerFixture
_COMPACT = "documents.management.commands.document_llmindex.llm_index_compact"
_MIGRATE = "documents.management.commands.document_llmindex.llm_index_migrate"
_INDEX = "documents.management.commands.document_llmindex.llmindex_index"
@@ -17,6 +18,11 @@ class TestDocumentLlmindexCommand:
call_command("document_llmindex", "compact")
mock_compact.assert_called_once_with()
def test_migrate_calls_llm_index_migrate(self, mocker: MockerFixture) -> None:
mock_migrate = mocker.patch(_MIGRATE)
call_command("document_llmindex", "migrate")
mock_migrate.assert_called_once_with()
def test_rebuild_calls_llmindex_index_with_rebuild_true(
self,
mocker: MockerFixture,
+25
View File
@@ -461,6 +461,31 @@ def llm_index_add_or_update_document(document: Document):
store.upsert_document(str(document.id), new_nodes)
def llm_index_migrate() -> None:
"""Apply any pending LLM index schema migrations, with no reindex.
Intended to run unconditionally on every startup (see the
init-llmindex-migrate container step and the bare-metal upgrade docs):
has_pending_migration() short-circuits to a metadata-only read once the
store is current, so a healthy install pays almost nothing here. Only
ever applies structural migrations -- a pending re-embed migration is
left for the explicit, deliberate rebuild path (``document_llmindex
update``/``rebuild``) to resolve, since re-embedding can be slow and,
for a metered embedding backend, cost money.
"""
if not AIConfig().llm_index_enabled:
return
with write_store() as store:
needs_reembed = _check_and_run_migrations(store)
if needs_reembed:
logger.warning(
"LLM index requires re-embedding, which this automatic migration "
"check will not do on its own -- it can be slow and, for a "
"metered embedding backend, cost money. Run "
"'document_llmindex rebuild' manually when ready.",
)
def llm_index_compact() -> None:
"""Compact the index immediately, rebuilding the table to reclaim space."""
with write_store() as store:
+104
View File
@@ -722,6 +722,110 @@ def test_llm_index_compact_uses_force(
mock_store.compact.assert_called_once_with(force=True)
@pytest.mark.django_db
class TestLlmIndexMigrate:
"""llm_index_migrate() is the cheap, startup-safe migration check -- see
the init-llmindex-migrate container step and the bare-metal upgrade docs.
"""
def test_skips_when_llm_index_disabled(
self,
temp_llm_index_dir: Path,
mocker: pytest_mock.MockerFixture,
) -> None:
"""
GIVEN:
- The LLM index is disabled
WHEN:
- llm_index_migrate() is called
THEN:
- The store is never opened (no stray db file for users who
never enabled AI features)
"""
mock_config = mocker.MagicMock()
mock_config.llm_index_enabled = False
mocker.patch("paperless_ai.indexing.AIConfig", return_value=mock_config)
mock_write_store = mocker.patch("paperless_ai.indexing.write_store")
indexing.llm_index_migrate()
mock_write_store.assert_not_called()
def test_runs_pending_structural_migration(
self,
temp_llm_index_dir: Path,
mocker: pytest_mock.MockerFixture,
caplog: pytest.LogCaptureFixture,
) -> None:
"""
GIVEN:
- The LLM index is enabled and a structural migration is pending
WHEN:
- llm_index_migrate() is called
THEN:
- check_and_run_migrations() runs, and no re-embed warning is
logged (the pending migration was structural, not re-embed)
"""
mock_config = mocker.MagicMock()
mock_config.llm_index_enabled = True
mocker.patch("paperless_ai.indexing.AIConfig", return_value=mock_config)
mock_store = mocker.MagicMock()
mock_store.has_pending_migration.return_value = True
mock_store.check_and_run_migrations.return_value = False
mocker.patch(
"paperless_ai.indexing.write_store",
return_value=mocker.MagicMock(
__enter__=mocker.MagicMock(return_value=mock_store),
__exit__=mocker.MagicMock(return_value=False),
),
)
with caplog.at_level("WARNING"):
indexing.llm_index_migrate()
mock_store.check_and_run_migrations.assert_called_once()
assert "re-embedding" not in caplog.text
def test_warns_without_rebuilding_when_reembed_pending(
self,
temp_llm_index_dir: Path,
mocker: pytest_mock.MockerFixture,
caplog: pytest.LogCaptureFixture,
) -> None:
"""
GIVEN:
- The LLM index is enabled and a pending migration requires
re-embedding
WHEN:
- llm_index_migrate() is called
THEN:
- A warning is logged telling the admin to rebuild manually, but
no rebuild is triggered automatically -- re-embedding can be
slow and, for a metered embedding backend, cost money, so it
must be a deliberate user action, never an automatic one
"""
mock_config = mocker.MagicMock()
mock_config.llm_index_enabled = True
mocker.patch("paperless_ai.indexing.AIConfig", return_value=mock_config)
mock_store = mocker.MagicMock()
mock_store.has_pending_migration.return_value = True
mock_store.check_and_run_migrations.return_value = True
mocker.patch(
"paperless_ai.indexing.write_store",
return_value=mocker.MagicMock(
__enter__=mocker.MagicMock(return_value=mock_store),
__exit__=mocker.MagicMock(return_value=False),
),
)
with caplog.at_level("WARNING"):
indexing.llm_index_migrate()
assert "re-embedding" in caplog.text
mock_store.drop_table.assert_not_called()
mock_store.add.assert_not_called()
@pytest.mark.django_db
class TestLlmIndexLocking:
"""Index mutation functions must go through write_store(), which holds the lock.
+1 -1
View File
@@ -671,7 +671,7 @@ class PaperlessSqliteVecVectorStore(BasePydanticVectorStore):
if migration.kind == "re-embed":
logger.warning(
"LLM index schema v%d -> v%d requires re-embedding (%s); "
"forcing full rebuild.",
"the caller must force a rebuild.",
migration.from_version,
migration.to_version,
migration.description,