Compare commits

..
Author SHA1 Message Date
shamoon 5a0270230d some chat component visual improvements 2026-10-11 08:58:42 -07:00
shamoon 39a2cd5d20 Ditch monospace 2026-10-11 08:54:46 -07:00
shamoon c60e5a6ce7 Update chat.component.html 2026-09-27 08:48:55 -07:00
shamoon 5a3fd7ef96 Use the markdown pipe 2026-09-27 08:43:33 -07:00
shamoon 50730ddc6c markdown pipe 2026-09-27 08:43:26 -07:00
shamoon 27ba1b3110 Add marked 2026-09-27 08:42:16 -07:00
GitHub Actions 937bf2fa7a Auto translate strings 2026-10-10 22:05:15 +00:00
Trenton HandClaude Sonnet 5.5 2f2381578c Fix: authorize document versions by their root and speed up permission id sets (#14396)
* Fix: authorize document versions by their root and speed up permission id sets

permitted_document_ids judged a version by its own owner and grants, so a
version whose owner had drifted from its root's was visible to the wrong
people and hidden from the right ones. Callers patched this individually by
mapping each document to its root first. The query itself was also slow on
MariaDB: the guardian grants were a UNION cast to integers and tested with
IN inside an OR with the owner checks, which MariaDB cannot materialize, so it
re-scans the user's grants for every document. At 20k documents that took
seconds for a user with a couple of hundred grants.

permitted_object_ids now casts the row key to a string, as guardian stores
object_pk, and tests it against a single uncorrelated UNION ALL of the user's
and groups' grants. No integer cast is needed and the user's groups are
matched with an IN subquery rather than a join through the membership table.
Postgres and SQLite build the grant set once, and MariaDB probes guardian's
unique indexes per row, which is cheap. A correlated EXISTS per grant also
fixed MariaDB but was up to 3x slower than before on Postgres and SQLite.

permitted_object_ids also takes an optional parent_field naming a
self-referencing foreign key whose target authorizes the row, and
permitted_document_ids passes root_document, so a version is visible exactly
when its root is. The helper that mapped documents to their roots at the call
sites is no longer needed, so the email, selection data, share link bundle,
trash, bulk download and bulk edit checks use the id set directly.

Co-authored-by: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-10-10 15:03:52 -07:00
shamoon e606dbffcd Chore: update the db options warning (#14398) 2026-10-10 07:17:54 -07:00
27 changed files with 263 additions and 497 deletions

No files matched your search

+1 -1
View File
@@ -26,7 +26,7 @@ module.exports = {
'abstract-paperless-service',
],
transformIgnorePatterns: [
'node_modules/(?!.*(\\.mjs$|tslib|lodash-es|normalize-diacritics|@angular/common/locales/.*\\.js$))',
'node_modules/(?!.*(\\.mjs$|tslib|lodash-es|normalize-diacritics|marked|@angular/common/locales/.*\\.js$))',
],
moduleNameMapper: {
...esmPreset.moduleNameMapper,
+1
View File
@@ -30,6 +30,7 @@
"bootstrap": "^5.3.8",
"file-saver": "^2.0.5",
"lodash-es": "^4.18.1",
"marked": "~18.0.14",
"mime-names": "^1.0.0",
"ngx-bootstrap-icons": "^1.9.3",
"ngx-color": "^10.1.0",
+10
View File
@@ -53,6 +53,9 @@ importers:
lodash-es:
specifier: ^4.18.1
version: 4.18.1
marked:
specifier: ~18.0.14
version: 18.0.14
mime-names:
specifier: ^1.0.0
version: 1.0.0
@@ -3226,6 +3229,11 @@ packages:
make-error@1.3.6:
resolution: {integrity: sha512-s8UhlNe7vPKomQhC1qFelMokr/Sc3AgNbso3n74mVPA5LTZwkB9NlXf4XPamLxJE8h0gh73rM94xvwRT2CVInw==}
marked@18.0.14:
resolution: {integrity: sha512-mBHK6FBHuBAlhgRe88w9F0O1AbwwXJUcQibUbC/QcdTbVGAD7aWza+xt3N6oT/jCZx3/OMeS+8rnuiHZcQ9s7A==}
engines: {node: '>= 20'}
hasBin: true
material-colors@1.2.6:
resolution: {integrity: sha512-6qE4B9deFBIa9YSpOc9O0Sgc43zTeVYbgDT5veRKSlB2+ZuHNoVVxA1L/ckMUayV9Ay9y7Z/SZCLcGteW9i7bg==}
@@ -7409,6 +7417,8 @@ snapshots:
make-error@1.3.6: {}
marked@18.0.14: {}
material-colors@1.2.6: {}
merge-stream@2.0.0: {}
@@ -5,14 +5,16 @@
</button>
<div ngbDropdownMenu class="dropdown-menu-end shadow p-3" aria-labelledby="chatDropdown">
<div class="chat-container bg-light p-2">
<div class="chat-messages font-monospace small">
<div class="chat-messages small">
@for (message of messages(); track message) {
<div class="message d-flex flex-row small" [class.justify-content-end]="message.role === 'user'">
<div class="p-2 m-2" [class.bg-body]="message.role === 'user'">
<span class="text-break">
{{ message.content }}
@if (message.role === 'assistant') {
<div class="chat-markdown text-break text-wrap" [innerHTML]="message.content | markdown"></div>
@if (message.isStreaming) { <span class="blinking-cursor">|</span> }
</span>
} @else {
<span class="text-break">{{ message.content }}</span>
}
@if (message.role === 'assistant' && message.references?.length) {
<div class="chat-references list-group mt-3">
@for (reference of message.references; track reference.id) {
@@ -8,10 +8,6 @@
white-space: pre-wrap;
}
.chat-references {
font-family: var(--bs-font-sans-serif);
}
.dropdown-toggle::after {
display: none;
}
@@ -40,3 +36,19 @@
opacity: 1;
}
}
.chat-markdown ::ng-deep {
h1, h2, h3, h4, h5, h6 {
font-size: 1em;
font-weight: bold;
}
th, td {
padding: 0.15rem 0.4rem;
border: 1px solid var(--bs-border-color);
}
> :last-child {
margin-bottom: 0;
}
}
@@ -207,4 +207,18 @@ describe('ChatComponent', () => {
component.searchInputKeyDown(event)
expect(component.sendMessage).not.toHaveBeenCalled()
})
it('should render markdown for assistant messages only', () => {
component.messages.set([
{ role: 'user', content: '**user**' },
{ role: 'assistant', content: '**assistant** <script>x</script>' },
])
fixture.detectChanges()
const el: HTMLElement = fixture.nativeElement
expect(el.querySelector('.chat-markdown strong')?.textContent).toBe(
'assistant'
)
expect(el.querySelector('.chat-markdown script')).toBeNull()
expect(el.textContent).toContain('**user**')
})
})
@@ -10,6 +10,7 @@ import { FormsModule, ReactiveFormsModule } from '@angular/forms'
import { NavigationEnd, Router, RouterModule } from '@angular/router'
import { NgbDropdownModule } from '@ng-bootstrap/ng-bootstrap'
import { NgxBootstrapIconsModule } from 'ngx-bootstrap-icons'
import { MarkdownPipe } from 'src/app/pipes/markdown.pipe'
import { filter, map } from 'rxjs'
import {
ChatMessage,
@@ -25,6 +26,7 @@ import {
RouterModule,
NgxBootstrapIconsModule,
NgbDropdownModule,
MarkdownPipe,
],
templateUrl: './chat.component.html',
styleUrl: './chat.component.scss',
@@ -0,0 +1,70 @@
import { MarkdownPipe } from './markdown.pipe'
describe('MarkdownPipe', () => {
const pipe = new MarkdownPipe()
it('should return empty string for empty input', () => {
expect(pipe.transform(null)).toEqual('')
expect(pipe.transform(undefined)).toEqual('')
expect(pipe.transform('')).toEqual('')
})
it('should render basic markdown', () => {
const html = pipe.transform(
'**bold** _em_ `code`\n\n- one\n- two\n\n| a | b |\n|---|---|\n| 1 | 2 |'
)
expect(html).toContain('<strong>bold</strong>')
expect(html).toContain('<em>em</em>')
expect(html).toContain('<code>code</code>')
expect(html).toContain('<li>one</li>')
expect(html).toContain('<table>')
})
it('should escape raw html', () => {
const html = pipe.transform(
'hi <img src=x onerror="alert(1)"> <b>x</b>\n\n<script>alert(1)</script>'
)
expect(html).not.toContain('<img')
expect(html).not.toContain('<script')
expect(html).not.toContain('<b>')
expect(html).toContain('&lt;script&gt;')
})
it('should not render images', () => {
const html = pipe.transform('![secret](https://evil.example/x.png?d=1)')
expect(html).not.toContain('<img')
expect(html).not.toContain('evil.example')
expect(html).toContain('secret')
})
it('should render safe links with target and rel', () => {
const html = pipe.transform('[docs](https://docs.paperless-ngx.com "Docs")')
expect(html).toContain(
'<a href="https://docs.paperless-ngx.com/" title="Docs" target="_blank" rel="noopener noreferrer nofollow">docs</a>'
)
expect(pipe.transform('[mail](mailto:a@b.c)')).toContain(
'href="mailto:a@b.c"'
)
})
it('should drop unsafe or relative links but keep their text', () => {
for (const href of [
'javascript:alert(1)',
'JaVaScRiPt:alert(1)',
'data:text/html,<script>alert(1)</script>',
'vbscript:msgbox',
'/api/documents/',
]) {
const html = pipe.transform(`[click](${href})`)
expect(html).not.toContain('<a')
expect(html).toContain('click')
}
})
it('should escape link titles', () => {
const html = pipe.transform(
'[x](https://a.example "a\\" onmouseover=\\"1")'
)
expect(html).not.toMatch(/"\s*onmouseover=/)
})
})
+57
View File
@@ -0,0 +1,57 @@
import { Pipe, PipeTransform } from '@angular/core'
import { Marked, Renderer, Tokens } from 'marked'
const ALLOWED_LINK_PROTOCOLS = ['http:', 'https:', 'mailto:']
function escapeHtml(text: string): string {
return text
.replace(/&/g, '&amp;')
.replace(/</g, '&lt;')
.replace(/>/g, '&gt;')
.replace(/"/g, '&quot;')
.replace(/'/g, '&#39;')
}
function safeHref(href: string): string | null {
try {
const url = new URL(href)
return ALLOWED_LINK_PROTOCOLS.includes(url.protocol) ? url.href : null
} catch {
return null
}
}
// Treat chat content as untrusted: no raw HTML, no images, and only
// absolute links. Angular sanitizer runs on top of this as well.
const renderer: Partial<Renderer> = {
html({ text }: Tokens.HTML | Tokens.Tag): string {
return escapeHtml(text)
},
image({ text }: Tokens.Image): string {
return escapeHtml(text)
},
link(this: Renderer, { href, title, tokens }: Tokens.Link): string {
const text = this.parser.parseInline(tokens)
const url = safeHref(href)
if (!url) return text
const titleAttr = title ? ` title="${escapeHtml(title)}"` : ''
return `<a href="${escapeHtml(url)}"${titleAttr} target="_blank" rel="noopener noreferrer nofollow">${text}</a>`
},
}
const markdown = new Marked({
async: false,
gfm: true,
breaks: true,
renderer,
})
@Pipe({
name: 'markdown',
})
export class MarkdownPipe implements PipeTransform {
transform(value: string | null | undefined): string {
if (!value) return ''
return markdown.parse(value) as string
}
}
@@ -67,22 +67,18 @@ class Command(PaperlessCommand):
if options.get("recreate"):
wipe_index(settings.INDEX_DIR)
documents = (
Document.objects.filter(root_document__isnull=True)
.select_related(
"correspondent",
"document_type",
"storage_path",
"owner",
)
.prefetch_related(
"tags",
"notes__user",
"custom_fields__field",
"versions",
"barcodes",
"versions__barcodes",
)
documents = Document.objects.select_related(
"correspondent",
"document_type",
"storage_path",
"owner",
).prefetch_related(
"tags",
"notes__user",
"custom_fields__field",
"versions",
"barcodes",
"versions__barcodes",
)
total = documents.count()
rebuild_kwargs = {}
-8
View File
@@ -28,7 +28,6 @@ from rest_framework.permissions import BasePermission
from rest_framework.permissions import DjangoObjectPermissions
from documents.models import Document
from documents.versioning import get_root_document
class PaperlessObjectPermissions(DjangoObjectPermissions):
@@ -680,14 +679,7 @@ def has_perms_owner_aware(user, perms, obj):
single-object check still has many production callers. Several callers
remain across ``documents/``, ``paperless_mail/``, and ``paperless_ai/``
-- grep for this function name before removing it.
A document version is authorized by its root document, like in
``permitted_document_ids``, so a version's own owner and grants never
matter. Fetch the root with ``select_related("root_document__owner")`` to
avoid extra queries.
"""
if isinstance(obj, Document):
obj = get_root_document(obj)
checker = ObjectPermissionChecker(user)
return obj.owner is None or obj.owner == user or checker.has_perm(perms, obj)
+3 -15
View File
@@ -284,14 +284,9 @@ class WriteBatch:
and adding the new version. This ensures stale document data (e.g., after
permission changes) doesn't persist in the index.
Only root documents are indexed, with their effective content, so a
version is indexed as its root document.
Args:
document: Django Document instance to index
"""
if document.root_document_id is not None:
document = document.root_document
self.remove(document.pk)
doc = self._backend._build_tantivy_doc(document)
self._writer.add_document(doc)
@@ -316,22 +311,20 @@ class WriteBatch:
An id with no matching document (e.g. deleted between the caller
collecting ids and the batch running) is silently skipped, matching
``add_or_update()``'s existing single-document deferred-task behavior
rather than erroring or leaving a stale index entry. The id of a
version stands for its root document.
rather than erroring or leaving a stale index entry.
Args:
ids: Primary keys of Document instances to index
"""
from documents.models import Document
from documents.versioning import annotate_effective_content
from documents.versioning import root_document_ids
ids = list(ids)
if not ids:
return
queryset = annotate_effective_content(
Document.objects.filter(pk__in=root_document_ids(ids))
Document.objects.filter(pk__in=ids)
.select_related("correspondent", "document_type", "storage_path", "owner")
.prefetch_related(
"tags",
@@ -1043,21 +1036,16 @@ class TantivyBackend:
excluded from results.
Args:
doc_id: Primary key of the reference document, or of one of its
versions
doc_id: Primary key of the reference document
user: User for permission filtering (None for no filtering)
limit: Maximum number of IDs to return (None = all matching docs)
Returns:
List of similar document IDs (excluding the original)
"""
from documents.versioning import root_document_ids
self._ensure_open()
searcher = self._index.searcher()
# Only root documents are indexed, so a version stands for its root
doc_id = next(iter(root_document_ids([doc_id])), doc_id)
id_query = tantivy.Query.term_query(self._schema, "id", doc_id)
results = searcher.search(id_query, limit=1)
+1 -3
View File
@@ -28,9 +28,7 @@ logger = logging.getLogger("paperless.search")
# columns dropped. tantivy compares schemas by ordered field list, so an
# index built by v1 rejects every write against the v2 schema.
# v3 - barcodes JSON field for stored barcode contents
# v4 - root documents only. Earlier indexes may hold document versions under
# their own id and metadata, so they are rebuilt without them.
SCHEMA_VERSION: Final[int] = 4
SCHEMA_VERSION: Final[int] = 3
# Present in the index directory from the moment a full rebuild starts until it
# finishes. If a rebuild is interrupted it is left behind, so the half-built
+2 -1
View File
@@ -90,6 +90,7 @@ from documents.templating.utils import convert_format_str_to_template_format
from documents.templating.workflows import validate_workflow_template
from documents.validators import uri_validator
from documents.validators import url_validator
from documents.versioning import get_root_document
from documents.versioning import has_prefetched_effective_content
from documents.versioning import sort_versions_newest_first
@@ -2894,7 +2895,7 @@ class ShareLinkSerializer(OwnedObjectSerializer):
and has_perms_owner_aware(
self.user,
"view_document",
document,
get_root_document(document),
)
):
return document
+8 -5
View File
@@ -490,20 +490,23 @@ def update_document_content_maybe_archive_file(
shutil.move(thumbnail, document.thumbnail_path)
document.refresh_from_db()
root_document = (
document.root_document if document.root_document_id else document
)
logger.info(
f"Updating index for document {document_id} ({document.archive_checksum})",
f"Updating index for document {root_document.pk} ({document.archive_checksum})",
)
from documents.search import get_backend
get_backend().add_or_update(document)
get_backend().add_or_update(root_document)
ai_config = AIConfig()
if ai_config.llm_index_enabled:
llm_index_add_or_update_document(document)
llm_index_add_or_update_document(root_document)
clear_document_caches(document.pk)
if document.root_document_id is not None:
clear_document_caches(document.root_document_id)
if root_document.pk != document.pk:
clear_document_caches(root_document.pk)
except Exception:
logger.exception(
-102
View File
@@ -293,80 +293,6 @@ class TestAddOrUpdateIds:
assert backend.search_ids("updated", user=None) == [doc.pk]
class TestVersionsAreIndexedAsTheirRoot:
"""Only root documents are indexed, with their effective content, so
every write path that is handed a version indexes its root instead."""
@staticmethod
def _root_with_version() -> tuple[Document, Document]:
root = DocumentFactory(title="Statement", content="stale text")
version = DocumentFactory(
title="Statement",
content="latest text",
root_document=root,
version_index=1,
)
return root, version
def test_add_or_update_indexes_the_root_of_a_version(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A root document with a version
WHEN:
- The version is passed to add_or_update
THEN:
- The root is indexed with the version's text, and the version is not
"""
root, version = self._root_with_version()
backend.add_or_update(version)
assert backend.search_ids("latest", user=None) == [root.pk]
assert backend.search_ids("stale", user=None) == []
def test_add_or_update_ids_indexes_each_root_once(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A root document with a version
WHEN:
- Both ids are passed to add_or_update_ids
THEN:
- The root is indexed once and the version is not indexed
"""
root, version = self._root_with_version()
with backend.batch_update() as batch:
batch.add_or_update_ids([version.pk, root.pk])
assert backend.search_ids("Statement", user=None) == [root.pk]
assert backend.search_ids("latest", user=None) == [root.pk]
def test_add_or_update_ids_resolves_a_lone_version(
self,
backend: TantivyBackend,
) -> None:
"""
GIVEN:
- A root document with a version
WHEN:
- Only the version's id is passed to add_or_update_ids
THEN:
- The root is indexed
"""
root, version = self._root_with_version()
with backend.batch_update() as batch:
batch.add_or_update_ids([version.pk])
assert backend.search_ids("latest", user=None) == [root.pk]
class TestSearch:
"""Test search query parsing and matching via search_ids."""
@@ -1136,34 +1062,6 @@ class TestMoreLikeThis:
assert 150 not in ids
assert 151 in ids
def test_more_like_this_ids_seeded_by_version_uses_root(
self,
backend: TantivyBackend,
) -> None:
"""A version is not indexed, so it must be looked up as its root."""
root = DocumentFactory.create(
title="Important document",
content="financial information report",
)
version = DocumentFactory.create(
title="Important document",
content="financial information report",
root_document=root,
version_index=1,
)
other = DocumentFactory.create(
title="Another document",
content="financial information report",
)
backend.add_or_update(root)
backend.add_or_update(other)
ids = backend.more_like_this_ids(doc_id=version.pk, user=None)
assert other.pk in ids
assert root.pk not in ids
assert version.pk not in ids
class TestSingleton:
"""Test get_backend() and reset_backend() singleton lifecycle."""
-53
View File
@@ -1196,59 +1196,6 @@ class TestDocumentSearchApi(DirectoriesMixin, APITestCase):
self.assertIn(d3.id, result_ids)
self.assertNotIn(d4.id, result_ids)
def test_search_more_like_version_uses_its_root(self) -> None:
"""
GIVEN:
- A document similar in content to a root document, and one that is not
- A version of the root document, which is never indexed
WHEN:
- API request for more like the version
THEN:
- The documents similar to the root are returned, not the version
"""
indexed = {}
for name, title, content, day in (
("root", "bank statement 1", "things i paid for in august", (2019, 3, 4)),
(
"similar",
"bank statement 3",
"things i paid for in september",
(2020, 7, 9),
),
(
"other",
"Quarterly Report",
"quarterly revenue profit margin",
(2021, 11, 30),
),
):
with time_machine.travel(
timezone.make_aware(datetime.datetime(*day)),
tick=False,
):
indexed[name] = DocumentFactory(
title=title,
content=content,
created=datetime.date(*day),
added=timezone.make_aware(datetime.datetime(*day)),
)
version = DocumentFactory(
root_document=indexed["root"],
version_index=1,
content="things i paid for in august",
)
backend = get_backend()
for document in indexed.values():
backend.add_or_update(document)
response = self.client.get(f"/api/documents/?more_like_id={version.id}")
self.assertEqual(response.status_code, status.HTTP_200_OK)
result_ids = [r["id"] for r in response.data["results"]]
self.assertIn(indexed["similar"].id, result_ids)
self.assertNotIn(indexed["other"].id, result_ids)
self.assertNotIn(version.id, result_ids)
def test_more_like_requires_id_of_existing_document(self) -> None:
"""
GIVEN:
-22
View File
@@ -22,7 +22,6 @@ from documents.models import Document
from documents.tasks import update_document_content_maybe_archive_file
from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.factories import DocumentFactory
sample_file: Path = Path(__file__).parent / "samples" / "simple.pdf"
@@ -117,27 +116,6 @@ class TestMakeIndex:
call_command("document_index", "reindex", skip_checks=True)
mock_get_backend.return_value.rebuild.assert_called_once()
def test_reindex_skips_versions(self, mocker: MockerFixture) -> None:
"""
GIVEN:
- A root document with a version
WHEN:
- The reindex command runs
THEN:
- Only the root document is handed to the rebuild, since a version
is indexed as its root
"""
root = DocumentFactory()
DocumentFactory(root_document=root, version_index=1)
mock_get_backend = mocker.patch(
"documents.management.commands.document_index.get_backend",
)
call_command("document_index", "reindex", skip_checks=True)
documents = mock_get_backend.return_value.rebuild.call_args.args[0]
assert list(documents.values_list("pk", flat=True)) == [root.pk]
def test_optimize(self) -> None:
"""Optimize command must execute without error (Tantivy handles optimization automatically)."""
call_command("document_index", "optimize", skip_checks=True)
@@ -18,7 +18,6 @@ from documents.models import Correspondent
from documents.models import DocumentType
from documents.models import StoragePath
from documents.models import Tag
from documents.permissions import has_perms_owner_aware
from documents.permissions import permitted_document_ids
from documents.permissions import permitted_object_ids
from documents.permissions import restrict_queryset_to_visible
@@ -471,92 +470,6 @@ class TestPermittedDocumentIdsVersions:
)
@pytest.mark.django_db
class TestHasPermsOwnerAwareVersions:
"""
The single-object check agrees with permitted_document_ids: a version is
authorized by its root document.
"""
@pytest.mark.parametrize(
("root_owner", "version_owner", "expected"),
[
pytest.param(
"other",
"nobody",
False,
id="unowned-version-of-private-root",
),
pytest.param("other", "user", False, id="own-version-of-private-root"),
pytest.param("user", "other", True, id="foreign-version-of-own-root"),
pytest.param("nobody", "other", True, id="private-version-of-unowned-root"),
],
)
def test_version_follows_root_owner(
self,
root_owner: str,
version_owner: str,
*,
expected: bool,
) -> None:
"""
GIVEN:
- A root document and a version with differing owners
WHEN:
- The single-object check runs for the version
THEN:
- The version is allowed exactly when its root is
"""
user = UserFactory()
owners = {"user": user, "other": UserFactory(), "nobody": None}
root = DocumentFactory(owner=owners[root_owner])
version = DocumentFactory(root_document=root, owner=owners[version_owner])
assert has_perms_owner_aware(user, "view_document", version) is expected
assert has_perms_owner_aware(user, "view_document", root) is expected
def test_grant_on_root_applies_and_grant_on_version_does_not(self) -> None:
"""
GIVEN:
- A private root with a version, and a second private root with a version
- The user may change only the first root, and was granted the second
root's version directly
WHEN:
- The single-object check runs for each version
THEN:
- Only the first root's version is allowed
"""
user = UserFactory()
shared_root = DocumentFactory(owner=UserFactory())
shared_version = DocumentFactory(root_document=shared_root, owner=UserFactory())
private_root = DocumentFactory(owner=UserFactory())
private_version = DocumentFactory(
root_document=private_root,
owner=UserFactory(),
)
grant_object(user, shared_root, "change_document")
grant_object(user, private_version, "change_document")
assert has_perms_owner_aware(user, "change_document", shared_version)
assert not has_perms_owner_aware(user, "change_document", private_version)
def test_other_models_use_their_own_owner(self) -> None:
"""
GIVEN:
- A tag owned by someone else, and one owned by the user
WHEN:
- The single-object check runs for each
THEN:
- Only the user's own tag is allowed without a grant
"""
user = UserFactory()
mine = TagFactory(owner=user)
theirs = TagFactory(owner=UserFactory())
assert has_perms_owner_aware(user, "view_tag", mine)
assert not has_perms_owner_aware(user, "view_tag", theirs)
@pytest.mark.django_db
class TestAiChatAllDocumentsPermissionBoundary:
"""
+9 -7
View File
@@ -310,7 +310,7 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
@mock.patch("documents.tasks.clear_document_caches")
@mock.patch("documents.search.get_backend")
def test_update_content_version_clears_caches_for_root(
def test_update_content_version_indexes_root(
self,
mock_get_backend: mock.Mock,
mock_clear_caches: mock.Mock,
@@ -321,8 +321,8 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
WHEN:
- Update content task is called for the version
THEN:
- The version's content is updated, not the root's
- The document is indexed
- The version's content is updated
- The root document is indexed rather than the version
- Caches are cleared for both
"""
root, version = self._create_root_with_version()
@@ -334,7 +334,8 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
"my document",
)
self.assertEqual(Document.objects.get(pk=root.pk).content, "root content")
mock_get_backend.return_value.add_or_update.assert_called_once()
indexed = mock_get_backend.return_value.add_or_update.call_args.args[0]
self.assertEqual(indexed.pk, root.pk)
mock_clear_caches.assert_has_calls(
[mock.call(version.pk), mock.call(root.pk)],
)
@@ -342,7 +343,7 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
@override_settings(AI_ENABLED=True, LLM_EMBEDDING_BACKEND="huggingface")
@mock.patch("documents.tasks.llm_index_add_or_update_document")
@mock.patch("documents.search.get_backend")
def test_update_content_version_updates_llm_index(
def test_update_content_version_updates_llm_index_for_root(
self,
mock_get_backend: mock.Mock,
mock_llm_index: mock.Mock,
@@ -354,13 +355,14 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
WHEN:
- Update content task is called for the version
THEN:
- The LLM index is updated
- The LLM index is updated for the root document, not the version
"""
_, version = self._create_root_with_version()
root, version = self._create_root_with_version()
tasks.update_document_content_maybe_archive_file(version.pk)
mock_llm_index.assert_called_once()
self.assertEqual(mock_llm_index.call_args.args[0].pk, root.pk)
class TestUpdateContentRemoteOCR(DirectoriesMixin, TestCase):
-17
View File
@@ -17,26 +17,9 @@ from django.db.models.functions import RowNumber
from documents.models import Document
if TYPE_CHECKING:
from collections.abc import Iterable
from rest_framework.request import Request
def root_document_ids(ids: Iterable[int]) -> QuerySet[int]:
"""
The ids of the root documents of the given documents: a root stands for
itself and a version for its root. Only the indexes' bookkeeping needs
this, since they hold root documents only.
"""
return (
Document.objects.filter(pk__in=ids)
.annotate(root_id=Coalesce("root_document_id", "id"))
.order_by()
.values_list("root_id", flat=True)
.distinct()
)
def versions_newest_first(documents: QuerySet[Document]) -> QuerySet[Document]:
"""
Sorts versions so the newest one comes first using version_index and not on id,
+11 -13
View File
@@ -329,10 +329,9 @@ def _get_tantivy_query_and_mode(params):
def _get_more_like_id(query_params: dict[str, Any], user: User | None) -> int:
try:
more_like_doc_id = int(query_params["more_like_id"])
more_like_doc = Document.objects.select_related(
"owner",
"root_document__owner",
).get(pk=more_like_doc_id)
more_like_doc = Document.objects.select_related("owner").get(
pk=more_like_doc_id,
)
except (TypeError, ValueError, Document.DoesNotExist):
raise PermissionDenied(_("Invalid more_like_id"))
@@ -343,8 +342,7 @@ def _get_more_like_id(query_params: dict[str, Any], user: User | None) -> int:
):
raise PermissionDenied(_("Insufficient permissions."))
# Only root documents are indexed, a version stands for its root
return more_like_doc.root_document_id or more_like_doc.pk
return more_like_doc_id
class SearchParams(NamedTuple):
@@ -1561,7 +1559,7 @@ class DocumentViewSet(
if request.user is not None and not has_perms_owner_aware(
request.user,
"change_document",
doc,
get_root_document(doc),
):
return HttpResponseForbidden("Insufficient permissions")
@@ -1624,7 +1622,7 @@ class DocumentViewSet(
if request.user is not None and not has_perms_owner_aware(
request.user,
"change_document",
doc,
get_root_document(doc),
):
return HttpResponseForbidden("Insufficient permissions")
@@ -1877,7 +1875,7 @@ class DocumentViewSet(
if currentUser is not None and not has_perms_owner_aware(
currentUser,
"view_document",
doc,
get_root_document(doc),
):
return HttpResponseForbidden("Insufficient permissions to view notes")
except Document.DoesNotExist:
@@ -1899,7 +1897,7 @@ class DocumentViewSet(
if currentUser is not None and not has_perms_owner_aware(
currentUser,
"change_document",
doc,
get_root_document(doc),
):
return HttpResponseForbidden(
"Insufficient permissions to create notes",
@@ -1942,7 +1940,7 @@ class DocumentViewSet(
if currentUser is not None and not has_perms_owner_aware(
currentUser,
"change_document",
doc,
get_root_document(doc),
):
return HttpResponseForbidden("Insufficient permissions to delete notes")
@@ -1992,7 +1990,7 @@ class DocumentViewSet(
if currentUser is not None and not has_perms_owner_aware(
currentUser,
"change_document",
doc,
get_root_document(doc),
):
return HttpResponseForbidden(
"Insufficient permissions to add share link",
@@ -2453,7 +2451,7 @@ class ChatStreamingView(GenericAPIView[Any]):
if not has_perms_owner_aware(
request.user,
"view_document",
document,
get_root_document(document),
):
return HttpResponseForbidden("Insufficient permissions")
+12 -12
View File
@@ -2,7 +2,7 @@ msgid ""
msgstr ""
"Project-Id-Version: paperless-ngx\n"
"Report-Msgid-Bugs-To: \n"
"POT-Creation-Date: 2026-10-08 18:39+0000\n"
"POT-Creation-Date: 2026-10-10 22:04+0000\n"
"PO-Revision-Date: 2022-02-17 04:17\n"
"Last-Translator: \n"
"Language-Team: English\n"
@@ -1653,7 +1653,7 @@ msgid "workflow runs"
msgstr ""
#: documents/serialisers.py:516 documents/serialisers.py:873
#: documents/serialisers.py:2903 documents/views.py:344 documents/views.py:2751
#: documents/serialisers.py:2903 documents/views.py:343 documents/views.py:2750
#: paperless_mail/serialisers.py:156
msgid "Insufficient permissions."
msgstr ""
@@ -1694,7 +1694,7 @@ msgstr ""
msgid "Duplicate document identifiers are not allowed."
msgstr ""
#: documents/serialisers.py:2989 documents/views.py:4807
#: documents/serialisers.py:2989 documents/views.py:4797
#, python-format
msgid "Documents not found: %(ids)s"
msgstr ""
@@ -1945,40 +1945,40 @@ msgstr ""
msgid ", "
msgstr ""
#: documents/views.py:337 documents/views.py:2748
#: documents/views.py:336 documents/views.py:2747
msgid "Invalid more_like_id"
msgstr ""
#: documents/views.py:1683
#: documents/views.py:1682
msgid "Invalid AI configuration."
msgstr ""
#: documents/views.py:1694
#: documents/views.py:1693
msgid "AI backend request timed out."
msgstr ""
#: documents/views.py:1706
#: documents/views.py:1705
msgid "AI backend rejected the request. Check logs for details."
msgstr ""
#: documents/views.py:2573 documents/views.py:2889
#: documents/views.py:2572 documents/views.py:2888
msgid "Specify only one of text, title_search, query, or more_like_id."
msgstr ""
#: documents/views.py:4823
#: documents/views.py:4813
#, python-format
msgid "Insufficient permissions to share document %(id)s."
msgstr ""
#: documents/views.py:4870
#: documents/views.py:4860
msgid "Bundle is already being processed."
msgstr ""
#: documents/views.py:4934
#: documents/views.py:4924
msgid "The share link bundle is still being prepared. Please try again later."
msgstr ""
#: documents/views.py:4948
#: documents/views.py:4938
msgid "The share link bundle is unavailable."
msgstr ""
+10 -11
View File
@@ -272,27 +272,26 @@ def check_deprecated_db_settings(
Detects legacy advanced options that should be migrated to
PAPERLESS_DB_OPTIONS. Returns one Warning per deprecated variable found.
"""
deprecated_vars: dict[str, str] = {
"PAPERLESS_DB_TIMEOUT": "timeout",
"PAPERLESS_DB_POOLSIZE": "pool.min_size / pool.max_size",
"PAPERLESS_DBSSLMODE": "sslmode",
"PAPERLESS_DBSSLROOTCERT": "sslrootcert",
"PAPERLESS_DBSSLCERT": "sslcert",
"PAPERLESS_DBSSLKEY": "sslkey",
}
deprecated_vars = (
"PAPERLESS_DB_TIMEOUT",
"PAPERLESS_DB_POOLSIZE",
"PAPERLESS_DBSSLMODE",
"PAPERLESS_DBSSLROOTCERT",
"PAPERLESS_DBSSLCERT",
"PAPERLESS_DBSSLKEY",
)
warnings: list[Warning] = []
for var_name, db_option_key in deprecated_vars.items():
for var_name in deprecated_vars:
if not os.getenv(var_name):
continue
warnings.append(
Warning(
f"Deprecated environment variable: {var_name}",
hint=(
f"{var_name} is no longer supported and will be removed in v3.2. "
f"{var_name} is deprecated. "
f"Set the equivalent option via PAPERLESS_DB_OPTIONS instead. "
f'Example: PAPERLESS_DB_OPTIONS=\'{{"{db_option_key}": "<value>"}}\'. '
"See https://docs.paperless-ngx.com/migration-v3/ for the full reference."
),
id="paperless.W001",
+11 -25
View File
@@ -211,14 +211,14 @@ class TestAuditLogChecks:
assert "auditlog table was found but audit log is disabled." in msgs[0].msg
DEPRECATED_VARS: dict[str, str] = {
"PAPERLESS_DB_TIMEOUT": "timeout",
"PAPERLESS_DB_POOLSIZE": "pool.min_size / pool.max_size",
"PAPERLESS_DBSSLMODE": "sslmode",
"PAPERLESS_DBSSLROOTCERT": "sslrootcert",
"PAPERLESS_DBSSLCERT": "sslcert",
"PAPERLESS_DBSSLKEY": "sslkey",
}
DEPRECATED_VARS = (
"PAPERLESS_DB_TIMEOUT",
"PAPERLESS_DB_POOLSIZE",
"PAPERLESS_DBSSLMODE",
"PAPERLESS_DBSSLROOTCERT",
"PAPERLESS_DBSSLCERT",
"PAPERLESS_DBSSLKEY",
)
class TestDeprecatedDbSettings:
@@ -234,26 +234,11 @@ class TestDeprecatedDbSettings:
result = check_deprecated_db_settings(None)
assert result == []
@pytest.mark.parametrize(
("env_var", "db_option_key"),
[
pytest.param("PAPERLESS_DB_TIMEOUT", "timeout", id="db-timeout"),
pytest.param(
"PAPERLESS_DB_POOLSIZE",
"pool.min_size / pool.max_size",
id="db-poolsize",
),
pytest.param("PAPERLESS_DBSSLMODE", "sslmode", id="ssl-mode"),
pytest.param("PAPERLESS_DBSSLROOTCERT", "sslrootcert", id="ssl-rootcert"),
pytest.param("PAPERLESS_DBSSLCERT", "sslcert", id="ssl-cert"),
pytest.param("PAPERLESS_DBSSLKEY", "sslkey", id="ssl-key"),
],
)
@pytest.mark.parametrize("env_var", DEPRECATED_VARS)
def test_single_deprecated_var_produces_one_warning(
self,
mocker: MockerFixture,
env_var: str,
db_option_key: str,
) -> None:
"""Each deprecated var in isolation produces exactly one warning."""
mocker.patch.dict(os.environ, {env_var: "some_value"}, clear=True)
@@ -264,7 +249,8 @@ class TestDeprecatedDbSettings:
assert isinstance(warning, Warning)
assert warning.id == "paperless.W001"
assert env_var in warning.hint
assert db_option_key in warning.hint
assert "PAPERLESS_DB_OPTIONS" in warning.hint
assert "https://docs.paperless-ngx.com/migration-v3/" in warning.hint
def test_multiple_deprecated_vars_produce_one_warning_each(
self,
+7 -13
View File
@@ -17,7 +17,6 @@ from documents.models import PaperlessTask
from documents.utils import IterWrapper
from documents.utils import QuerySetStream
from documents.utils import identity
from documents.versioning import root_document_ids
from paperless.config import AIConfig
from paperless_ai.db import db_connection_released
from paperless_ai.embedding import build_llm_index_text
@@ -444,11 +443,11 @@ def update_llm_index(
"Skipping LLM index update: migration check deferred; "
"will retry next run."
)
documents = (
Document.objects.filter(root_document__isnull=True)
.select_related("correspondent", "document_type", "storage_path")
.prefetch_related("tags", "notes", "custom_fields__field")
)
documents = Document.objects.select_related(
"correspondent",
"document_type",
"storage_path",
).prefetch_related("tags", "notes", "custom_fields__field")
no_documents = not documents.exists()
# Fast exit before touching config: nothing to index and no existing index.
@@ -484,7 +483,7 @@ def update_llm_index(
msg = "LLM index rebuilt successfully."
else:
scoped_documents = (
documents.filter(id__in=root_document_ids(document_ids))
documents.filter(id__in=document_ids)
if document_ids is not None
else documents
)
@@ -511,12 +510,7 @@ def update_llm_index(
def llm_index_add_or_update_document(document: Document):
"""
Add or atomically replace a document's chunks in the index. Only root
documents are indexed, so a version is indexed as its root document.
"""
if document.root_document_id is not None:
document = document.root_document
"""Add or atomically replace a document's chunks in the index."""
config = AIConfig()
new_nodes = build_document_node(
document,
@@ -388,83 +388,6 @@ def test_update_llm_index_partial_update(
assert after[str(doc2.pk)] == before[str(doc2.pk)]
@pytest.mark.django_db
class TestLlmIndexVersions:
"""The LLM index holds root documents only: a version is indexed as its root."""
def test_add_or_update_document_indexes_a_version_as_its_root(
self,
temp_llm_index_dir: Path,
mock_embed_model: FakeEmbedding,
) -> None:
"""
GIVEN:
- A root document with a version
WHEN:
- The version is passed to llm_index_add_or_update_document
THEN:
- Only the root document is in the index
"""
root = DocumentFactory(content="root content")
version = DocumentFactory(root_document=root, version_index=1)
indexing.llm_index_add_or_update_document(version)
with indexing.get_vector_store() as store:
indexed = store.get_modified_times()
assert set(indexed) == {str(root.pk)}
def test_rebuild_skips_versions(
self,
temp_llm_index_dir: Path,
mock_embed_model: FakeEmbedding,
) -> None:
"""
GIVEN:
- A root document with a version
WHEN:
- The LLM index is rebuilt
THEN:
- Only the root document is in the index
"""
root = DocumentFactory()
DocumentFactory(root_document=root, version_index=1)
indexing.update_llm_index(rebuild=True)
with indexing.get_vector_store() as store:
indexed = store.get_modified_times()
assert set(indexed) == {str(root.pk)}
def test_incremental_update_by_version_id_refreshes_the_root(
self,
temp_llm_index_dir: Path,
mock_embed_model: FakeEmbedding,
) -> None:
"""
GIVEN:
- An indexed root document with a version whose root was modified since
WHEN:
- An incremental update is scoped to the version's id
THEN:
- The root's entry is refreshed and no entry exists for the version
"""
root = DocumentFactory()
version = DocumentFactory(root_document=root, version_index=1)
indexing.update_llm_index(rebuild=True)
Document.objects.filter(pk=root.pk).update(modified=timezone.now())
root.refresh_from_db()
indexing.update_llm_index(document_ids=[version.pk])
with indexing.get_vector_store() as store:
indexed = store.get_modified_times()
assert indexed == {str(root.pk): root.modified.isoformat()}
@pytest.mark.django_db
def test_add_or_update_document_updates_existing_entry(
temp_llm_index_dir: Path,
@@ -714,7 +637,6 @@ class TestLlmIndexAddOrUpdateDocumentEmptyContent:
doc = MagicMock(spec=Document)
doc.id = 42
doc.root_document_id = None
# Must not raise
indexing.llm_index_add_or_update_document(doc)