mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-07-27 22:34:55 +00:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f00a6db939 |
@@ -932,8 +932,8 @@ for display in the web interface.
|
||||
| Document type | `never` | `auto` (default) | `always` |
|
||||
| -------------------------- | ------- | -------------------------- | -------- |
|
||||
| Scanned image (TIFF, JPEG) | No | **Yes** | Yes |
|
||||
| Image-based PDF | No | **Yes** (no embedded text) | Yes |
|
||||
| Born-digital PDF | No | No (has embedded text, optionally confirmed by tag) | Yes |
|
||||
| Image-based PDF | No | **Yes** (short/no text, untagged) | Yes |
|
||||
| Born-digital PDF | No | No (tagged or has embedded text) | Yes |
|
||||
| Plain text, email, HTML | No | No | No |
|
||||
| DOCX / ODT (via Tika) | Yes\* | Yes\* | Yes\* |
|
||||
|
||||
@@ -2135,7 +2135,7 @@ used with the OpenAI-compatible backend to target a custom provider or local gat
|
||||
|
||||
Defaults to None.
|
||||
|
||||
#### [`PAPERLESS_AI_LLM_OUTPUT_LANGUAGE=<str>`](#PAPERLESS_AI_LLM_OUTPUT_LANGUAGE) {#PAPERLESS_AI_LLM_OUTPUT_LANGUAGE}
|
||||
### [`PAPERLESS_AI_LLM_OUTPUT_LANGUAGE=<str>`](#PAPERLESS_AI_LLM_OUTPUT_LANGUAGE) {#PAPERLESS_AI_LLM_OUTPUT_LANGUAGE}
|
||||
|
||||
: The language to use for AI suggestions (results may vary by LLM model). If not supplied, defaults to the user's UI language setting or None.
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@
|
||||
</div>
|
||||
</pngx-page-header>
|
||||
|
||||
@if (!tasksService.completedFileTasks && tasksService.loading) {
|
||||
@if (!tasksService.completedTasks && tasksService.loading) {
|
||||
<div class="spinner-border spinner-border-sm fw-normal ms-2 me-auto" role="status"></div>
|
||||
<div class="visually-hidden" i18n>Loading...</div>
|
||||
}
|
||||
|
||||
@@ -50,13 +50,66 @@ describe('TasksService', () => {
|
||||
req.flush({ count: 0, results: [] })
|
||||
})
|
||||
|
||||
it('does not call tasks api endpoint on reload if already loading', () => {
|
||||
tasksService.loading = true
|
||||
it('cancels an in-progress reload when reloading again', () => {
|
||||
tasksService.reload()
|
||||
httpTestingController.expectNone(
|
||||
const staleReload = httpTestingController.expectOne(
|
||||
(req: HttpRequest<unknown>) =>
|
||||
req.url === `${environment.apiBaseUrl}tasks/`
|
||||
)
|
||||
tasksService.reload()
|
||||
|
||||
expect(staleReload.cancelled).toBe(true)
|
||||
httpTestingController
|
||||
.expectOne(
|
||||
(req: HttpRequest<unknown>) =>
|
||||
req.url === `${environment.apiBaseUrl}tasks/`
|
||||
)
|
||||
.flush({ count: 0, results: [] })
|
||||
})
|
||||
|
||||
it('continues reloading after a reload request fails', () => {
|
||||
tasksService.reload()
|
||||
httpTestingController
|
||||
.expectOne(
|
||||
(req: HttpRequest<unknown>) =>
|
||||
req.url === `${environment.apiBaseUrl}tasks/`
|
||||
)
|
||||
.flush('error', { status: 500, statusText: 'error' })
|
||||
|
||||
expect(tasksService.loading).toBe(false)
|
||||
|
||||
tasksService.reload()
|
||||
httpTestingController
|
||||
.expectOne(
|
||||
(req: HttpRequest<unknown>) =>
|
||||
req.url === `${environment.apiBaseUrl}tasks/`
|
||||
)
|
||||
.flush({ count: 0, results: [] })
|
||||
})
|
||||
|
||||
it('reloads after dismissing a task while a reload is already in progress', () => {
|
||||
tasksService.reload()
|
||||
const staleReload = httpTestingController.expectOne(
|
||||
(req: HttpRequest<unknown>) =>
|
||||
req.url === `${environment.apiBaseUrl}tasks/` &&
|
||||
req.params.get('acknowledged') === 'false'
|
||||
)
|
||||
|
||||
tasksService.dismissTasks(new Set([1])).subscribe()
|
||||
httpTestingController
|
||||
.expectOne(`${environment.apiBaseUrl}tasks/acknowledge/`)
|
||||
.flush([])
|
||||
|
||||
expect(staleReload.cancelled).toBe(true)
|
||||
httpTestingController
|
||||
.expectOne(
|
||||
(req: HttpRequest<unknown>) =>
|
||||
req.url === `${environment.apiBaseUrl}tasks/` &&
|
||||
req.params.get('acknowledged') === 'false'
|
||||
)
|
||||
.flush({ count: 0, results: [] })
|
||||
|
||||
expect(tasksService.needsAttentionTasks).toHaveLength(0)
|
||||
})
|
||||
|
||||
it('calls acknowledge_tasks api endpoint on dismiss and reloads', () => {
|
||||
@@ -166,12 +219,6 @@ describe('TasksService', () => {
|
||||
)
|
||||
|
||||
req.flush({ count: mockTasks.length, results: mockTasks })
|
||||
|
||||
expect(tasksService.allFileTasks).toHaveLength(5)
|
||||
expect(tasksService.completedFileTasks).toHaveLength(2)
|
||||
expect(tasksService.failedFileTasks).toHaveLength(1)
|
||||
expect(tasksService.queuedFileTasks).toHaveLength(1)
|
||||
expect(tasksService.startedFileTasks).toHaveLength(1)
|
||||
})
|
||||
|
||||
it('includes revoked tasks in needs attention', () => {
|
||||
|
||||
@@ -1,7 +1,15 @@
|
||||
import { HttpClient } from '@angular/common/http'
|
||||
import { Injectable, inject, signal } from '@angular/core'
|
||||
import { Observable, Subject } from 'rxjs'
|
||||
import { first, map, takeUntil, tap } from 'rxjs/operators'
|
||||
import { EMPTY, Observable, Subject } from 'rxjs'
|
||||
import {
|
||||
catchError,
|
||||
finalize,
|
||||
first,
|
||||
map,
|
||||
switchMap,
|
||||
takeUntil,
|
||||
tap,
|
||||
} from 'rxjs/operators'
|
||||
import {
|
||||
PaperlessTask,
|
||||
PaperlessTaskStatus,
|
||||
@@ -23,44 +31,48 @@ export class TasksService {
|
||||
|
||||
public loading: boolean = false
|
||||
|
||||
private readonly fileTasks = signal<PaperlessTask[]>([])
|
||||
private readonly tasks = signal<PaperlessTask[]>([])
|
||||
private readonly reloadNotifier = new Subject<void>()
|
||||
|
||||
private unsubscribeNotifer: Subject<any> = new Subject()
|
||||
|
||||
constructor() {
|
||||
this.reloadNotifier
|
||||
.pipe(
|
||||
switchMap(() => {
|
||||
this.loading = true
|
||||
return this.http
|
||||
.get<Results<PaperlessTask>>(`${this.baseUrl}${this.endpoint}/`, {
|
||||
params: {
|
||||
acknowledged: 'false',
|
||||
page_size: this.defaultReloadPageSize,
|
||||
},
|
||||
})
|
||||
.pipe(
|
||||
map((response) => response.results),
|
||||
takeUntil(this.unsubscribeNotifer),
|
||||
catchError(() => EMPTY),
|
||||
finalize(() => {
|
||||
this.loading = false
|
||||
})
|
||||
)
|
||||
})
|
||||
)
|
||||
.subscribe((tasks) => {
|
||||
this.tasks.set(tasks)
|
||||
})
|
||||
}
|
||||
|
||||
public get total(): number {
|
||||
return this.fileTasks().length
|
||||
return this.tasks().length
|
||||
}
|
||||
|
||||
public get allFileTasks(): PaperlessTask[] {
|
||||
return this.fileTasks().slice(0)
|
||||
}
|
||||
|
||||
public get queuedFileTasks(): PaperlessTask[] {
|
||||
return this.fileTasks().filter(
|
||||
(t) => t.status === PaperlessTaskStatus.Pending
|
||||
)
|
||||
}
|
||||
|
||||
public get startedFileTasks(): PaperlessTask[] {
|
||||
return this.fileTasks().filter(
|
||||
(t) => t.status === PaperlessTaskStatus.Started
|
||||
)
|
||||
}
|
||||
|
||||
public get completedFileTasks(): PaperlessTask[] {
|
||||
return this.fileTasks().filter(
|
||||
(t) => t.status === PaperlessTaskStatus.Success
|
||||
)
|
||||
}
|
||||
|
||||
public get failedFileTasks(): PaperlessTask[] {
|
||||
return this.fileTasks().filter(
|
||||
(t) => t.status === PaperlessTaskStatus.Failure
|
||||
)
|
||||
public get completedTasks(): PaperlessTask[] {
|
||||
return this.tasks().filter((t) => t.status === PaperlessTaskStatus.Success)
|
||||
}
|
||||
|
||||
public get needsAttentionTasks(): PaperlessTask[] {
|
||||
return this.fileTasks().filter((t) =>
|
||||
return this.tasks().filter((t) =>
|
||||
[PaperlessTaskStatus.Failure, PaperlessTaskStatus.Revoked].includes(
|
||||
t.status
|
||||
)
|
||||
@@ -68,22 +80,7 @@ export class TasksService {
|
||||
}
|
||||
|
||||
public reload() {
|
||||
if (this.loading) return
|
||||
this.loading = true
|
||||
|
||||
this.http
|
||||
.get<Results<PaperlessTask>>(`${this.baseUrl}${this.endpoint}/`, {
|
||||
params: {
|
||||
acknowledged: 'false',
|
||||
page_size: this.defaultReloadPageSize,
|
||||
},
|
||||
})
|
||||
.pipe(map((r) => r.results))
|
||||
.pipe(takeUntil(this.unsubscribeNotifer), first())
|
||||
.subscribe((r) => {
|
||||
this.fileTasks.set(r)
|
||||
this.loading = false
|
||||
})
|
||||
this.reloadNotifier.next()
|
||||
}
|
||||
|
||||
public list(
|
||||
|
||||
@@ -14,20 +14,4 @@ describe('text search utilities', () => {
|
||||
expect(matchesSearchText('Tax\u00e9s 2026', 'taxe 26')).toBeTruthy()
|
||||
expect(matchesSearchText('taxes 2026', 'tax receipt')).toBeFalsy()
|
||||
})
|
||||
|
||||
it('matches a large set of tag names without blocking input', () => {
|
||||
const tagNames = Array.from(
|
||||
{ length: 1280 },
|
||||
(_, index) =>
|
||||
`Customer Party ${index} Jos\u00e9 M\u00fcller \u00c5ngstr\u00f6m`
|
||||
)
|
||||
|
||||
const start = performance.now()
|
||||
const matches = tagNames.filter((name) => matchesSearchText(name, 'party'))
|
||||
const duration = performance.now() - start
|
||||
|
||||
expect(matches).toHaveLength(tagNames.length)
|
||||
// The previous implementation took roughly 500 ms for this workload.
|
||||
expect(duration).toBeLessThan(250)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -1,17 +1,10 @@
|
||||
import { diacritics } from 'normalize-diacritics/diacritics'
|
||||
import { normalizeSync } from 'normalize-diacritics'
|
||||
|
||||
export type SearchTextValue =
|
||||
string | number | boolean | bigint | null | undefined
|
||||
|
||||
export function normalizeSearchText(value: SearchTextValue): string {
|
||||
const normalized = diacritics.reduce(
|
||||
(text, replacement) => {
|
||||
return text.replace(replacement.diacritics, replacement.letter)
|
||||
},
|
||||
String(value ?? '')
|
||||
)
|
||||
|
||||
return normalized.toLocaleLowerCase()
|
||||
return normalizeSync(String(value ?? '')).toLocaleLowerCase()
|
||||
}
|
||||
|
||||
export function matchesSearchText(
|
||||
|
||||
@@ -160,14 +160,13 @@ def should_produce_archive(
|
||||
_log.debug("Archive: yes — image document, ARCHIVE_FILE_GENERATION=auto")
|
||||
return True
|
||||
if mime_type == "application/pdf":
|
||||
text = extract_pdf_text(document_path)
|
||||
has_text = text is not None and len(text) > 0
|
||||
if has_text and is_tagged_pdf(document_path):
|
||||
if is_tagged_pdf(document_path):
|
||||
_log.debug(
|
||||
"Archive: no — born-digital PDF (structure tags detected),"
|
||||
" ARCHIVE_FILE_GENERATION=auto",
|
||||
)
|
||||
return False
|
||||
text = extract_pdf_text(document_path)
|
||||
if text is None or len(text) <= PDF_TEXT_MIN_LENGTH:
|
||||
_log.debug(
|
||||
"Archive: yes — scanned PDF (text_length=%d ≤ %d),"
|
||||
|
||||
@@ -37,7 +37,6 @@ from drf_spectacular.utils import extend_schema_field
|
||||
from guardian.utils import get_group_obj_perms_model
|
||||
from guardian.utils import get_user_obj_perms_model
|
||||
from rest_framework import serializers
|
||||
from rest_framework.filters import BaseFilterBackend
|
||||
from rest_framework.filters import OrderingFilter
|
||||
from rest_framework_guardian.filters import ObjectPermissionsFilter
|
||||
|
||||
@@ -51,7 +50,6 @@ from documents.models import ShareLink
|
||||
from documents.models import ShareLinkBundle
|
||||
from documents.models import StoragePath
|
||||
from documents.models import Tag
|
||||
from documents.permissions import permitted_document_ids
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Callable
|
||||
@@ -1040,31 +1038,6 @@ class ObjectOwnedOrGrantedPermissionsFilter(ObjectPermissionsFilter):
|
||||
return objects_with_perms | objects_owned | objects_unowned
|
||||
|
||||
|
||||
class DocumentPermissionsFilter(BaseFilterBackend):
|
||||
"""
|
||||
A filter backend limiting Document results to those the requesting user
|
||||
owns, are unowned, or has explicit (user- or group-level) view
|
||||
permission on.
|
||||
|
||||
Unlike ``ObjectOwnedOrGrantedPermissionsFilter``, this does not build an
|
||||
``objects_with_perms | objects_owned | objects_unowned`` union of
|
||||
querysets derived from the same base queryset. When that base queryset
|
||||
already carries independent joins on a multi-valued relation (e.g. two
|
||||
separate joins from ``tags__id__all`` filtering on two tags), each
|
||||
OR-ed branch can end up pairing those joins' aliases differently,
|
||||
letting more than one row out of the join's cross product satisfy the
|
||||
combined WHERE -- returning the same document more than once. Filtering
|
||||
via a single ``id__in`` against ``permitted_document_ids`` (a plain
|
||||
subquery, not a join) sidesteps that entirely and is also cheaper than
|
||||
guardian's join-based permission check.
|
||||
"""
|
||||
|
||||
def filter_queryset(self, request, queryset, view):
|
||||
if request.user.is_superuser:
|
||||
return queryset
|
||||
return queryset.filter(id__in=permitted_document_ids(request.user))
|
||||
|
||||
|
||||
class ObjectOwnedPermissionsFilter(ObjectPermissionsFilter):
|
||||
"""
|
||||
A filter backend that limits results to those where the requesting user
|
||||
|
||||
@@ -163,7 +163,7 @@ def set_permissions_for_object(
|
||||
)
|
||||
|
||||
|
||||
def permitted_document_ids(user):
|
||||
def _permitted_document_ids(user):
|
||||
"""
|
||||
Return a queryset of document IDs the user may view, limited to non-deleted
|
||||
documents. This intentionally avoids ``get_objects_for_user`` to keep the
|
||||
@@ -220,7 +220,7 @@ def get_document_count_filter_for_user(user, related_name: str = "documents"):
|
||||
# Superuser: no permission filtering needed
|
||||
return Q(**{f"{related_name}__deleted_at__isnull": True})
|
||||
|
||||
permitted_ids = permitted_document_ids(user)
|
||||
permitted_ids = _permitted_document_ids(user)
|
||||
return Q(**{f"{related_name}__id__in": permitted_ids})
|
||||
|
||||
|
||||
@@ -311,7 +311,7 @@ def annotate_document_count_for_related_queryset(
|
||||
queryset,
|
||||
through_model=through_model,
|
||||
related_object_field=related_object_field,
|
||||
document_ids=permitted_document_ids(user),
|
||||
document_ids=_permitted_document_ids(user),
|
||||
target_field=target_field,
|
||||
)
|
||||
|
||||
|
||||
@@ -311,7 +311,6 @@ def sanity_check(*, raise_on_error: bool = True) -> str:
|
||||
def bulk_update_documents(document_ids) -> None:
|
||||
from documents.search import get_backend
|
||||
|
||||
document_ids = list(document_ids)
|
||||
documents = Document.objects.filter(id__in=document_ids)
|
||||
|
||||
for doc in documents:
|
||||
@@ -332,7 +331,6 @@ def bulk_update_documents(document_ids) -> None:
|
||||
if ai_config.llm_index_enabled:
|
||||
update_llm_index(
|
||||
rebuild=False,
|
||||
document_ids=document_ids,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -14,7 +14,6 @@ from unittest import mock
|
||||
import celery
|
||||
from dateutil import parser
|
||||
from django.conf import settings
|
||||
from django.contrib.auth.models import Group
|
||||
from django.contrib.auth.models import Permission
|
||||
from django.contrib.auth.models import User
|
||||
from django.core import mail
|
||||
@@ -48,8 +47,6 @@ from documents.models import Workflow
|
||||
from documents.models import WorkflowAction
|
||||
from documents.models import WorkflowTrigger
|
||||
from documents.signals.handlers import run_workflows
|
||||
from documents.tests.factories import DocumentFactory
|
||||
from documents.tests.factories import TagFactory
|
||||
from documents.tests.utils import ConsumeTaskMixin
|
||||
from documents.tests.utils import DirectoriesMixin
|
||||
from documents.tests.utils import read_streaming_response
|
||||
@@ -1215,91 +1212,6 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
|
||||
[u1_doc1.id],
|
||||
)
|
||||
|
||||
def test_document_owned_and_group_shared_not_duplicated_when_filtering_by_tags(
|
||||
self,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document owned by a user and also shared with a group the user belongs to
|
||||
WHEN:
|
||||
- The user filters documents by more than one tag (tags__id__all)
|
||||
THEN:
|
||||
- The document is returned exactly once, not once per permission path
|
||||
(regression test for https://github.com/paperless-ngx/paperless-ngx/issues/13331)
|
||||
"""
|
||||
user = User.objects.create_user("user1")
|
||||
user.user_permissions.add(*Permission.objects.filter(codename="view_document"))
|
||||
group = Group.objects.create(name="group1")
|
||||
user.groups.add(group)
|
||||
|
||||
tag1 = TagFactory()
|
||||
tag2 = TagFactory()
|
||||
doc = DocumentFactory(title="shared", owner=user)
|
||||
doc.tags.add(tag1, tag2)
|
||||
assign_perm("view_document", group, doc)
|
||||
|
||||
self.client.force_authenticate(user=user)
|
||||
response = self.client.get(
|
||||
f"/api/documents/?tags__id__all={tag1.id},{tag2.id}",
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
self.assertEqual(response.data["count"], 1)
|
||||
self.assertEqual(response.data["results"][0]["id"], doc.id)
|
||||
|
||||
def test_document_permission_filter_excludes_unrelated_documents(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document owned by one user, with no permission granted to another user
|
||||
WHEN:
|
||||
- The unrelated user requests the document list
|
||||
THEN:
|
||||
- The document does not appear in their results
|
||||
"""
|
||||
owner = User.objects.create_user("owner1")
|
||||
stranger = User.objects.create_user("stranger1")
|
||||
stranger.user_permissions.add(
|
||||
*Permission.objects.filter(codename="view_document"),
|
||||
)
|
||||
|
||||
DocumentFactory(title="private", owner=owner)
|
||||
|
||||
self.client.force_authenticate(user=stranger)
|
||||
response = self.client.get("/api/documents/")
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
self.assertEqual(response.data["count"], 0)
|
||||
|
||||
def test_document_permission_filter_only_visible_to_group_members(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A document shared with a group via object permissions
|
||||
WHEN:
|
||||
- A group member and a non-member both request the document list
|
||||
THEN:
|
||||
- Only the group member sees the document
|
||||
"""
|
||||
owner = User.objects.create_user("owner2")
|
||||
member = User.objects.create_user("member1")
|
||||
non_member = User.objects.create_user("nonmember1")
|
||||
for u in (member, non_member):
|
||||
u.user_permissions.add(*Permission.objects.filter(codename="view_document"))
|
||||
|
||||
group = Group.objects.create(name="group2")
|
||||
member.groups.add(group)
|
||||
|
||||
doc = DocumentFactory(title="shared2", owner=owner)
|
||||
assign_perm("view_document", group, doc)
|
||||
|
||||
self.client.force_authenticate(user=member)
|
||||
response = self.client.get("/api/documents/")
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
self.assertEqual(response.data["count"], 1)
|
||||
self.assertEqual(response.data["results"][0]["id"], doc.id)
|
||||
|
||||
self.client.force_authenticate(user=non_member)
|
||||
response = self.client.get("/api/documents/")
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
self.assertEqual(response.data["count"], 0)
|
||||
|
||||
def test_pagination_results(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
|
||||
@@ -166,28 +166,24 @@ class TestShouldProduceArchive:
|
||||
mocker: MockerFixture,
|
||||
settings,
|
||||
) -> None:
|
||||
"""Tagged PDFs (e.g. Word exports) with real text are treated as born-digital, even below PDF_TEXT_MIN_LENGTH."""
|
||||
"""Tagged PDFs (e.g. Word exports) are treated as born-digital regardless of text length."""
|
||||
settings.ARCHIVE_FILE_GENERATION = "auto"
|
||||
mocker.patch("documents.consumer.is_tagged_pdf", return_value=True)
|
||||
mocker.patch("documents.consumer.extract_pdf_text", return_value="tiny")
|
||||
parser = _parser_instance(can_produce=True, requires_rendition=False)
|
||||
assert (
|
||||
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
|
||||
is False
|
||||
)
|
||||
|
||||
def test_tagged_pdf_without_text_produces_archive(
|
||||
def test_tagged_pdf_does_not_call_pdftotext(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
settings,
|
||||
) -> None:
|
||||
"""A tagged PDF with no actual extractable text (e.g. some scanner firmware) is not
|
||||
trusted as born-digital — the tag alone must not bypass OCR."""
|
||||
"""When a PDF is tagged, pdftotext is not invoked (fast path)."""
|
||||
settings.ARCHIVE_FILE_GENERATION = "auto"
|
||||
mocker.patch("documents.consumer.is_tagged_pdf", return_value=True)
|
||||
mocker.patch("documents.consumer.extract_pdf_text", return_value=None)
|
||||
mock_extract = mocker.patch("documents.consumer.extract_pdf_text")
|
||||
parser = _parser_instance(can_produce=True, requires_rendition=False)
|
||||
assert (
|
||||
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
|
||||
is True
|
||||
)
|
||||
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
|
||||
mock_extract.assert_not_called()
|
||||
|
||||
@@ -401,10 +401,6 @@ class TestAIIndex(DirectoriesMixin, TestCase):
|
||||
"documents.tasks.update_llm_index",
|
||||
) as update_llm_index,
|
||||
):
|
||||
doc_ids = [doc.pk for doc in docs]
|
||||
tasks.bulk_update_documents(doc_ids)
|
||||
tasks.bulk_update_documents([doc.pk for doc in docs])
|
||||
self.assertEqual(update_document_in_llm_index.apply_async.call_count, 0)
|
||||
update_llm_index.assert_called_once_with(
|
||||
rebuild=False,
|
||||
document_ids=doc_ids,
|
||||
)
|
||||
update_llm_index.assert_called_once()
|
||||
|
||||
@@ -625,37 +625,6 @@ class TestAIChatStreamingView(DirectoriesMixin, TestCase):
|
||||
)
|
||||
self.assertEqual(response.status_code, 200)
|
||||
self.assertEqual(response["Content-Type"], "text/event-stream")
|
||||
mock_stream_chat.assert_called_once_with(
|
||||
query_str="question",
|
||||
documents=[self.document],
|
||||
output_language=None,
|
||||
)
|
||||
|
||||
@patch("documents.views.stream_chat_with_documents")
|
||||
@patch("documents.views.get_objects_for_user_owner_aware")
|
||||
@override_settings(AI_ENABLED=True)
|
||||
def test_post_uses_user_display_language(
|
||||
self,
|
||||
mock_get_objects,
|
||||
mock_stream_chat,
|
||||
) -> None:
|
||||
UiSettings.objects.create(user=self.user, settings={"language": "de-de"})
|
||||
self.grant_view_document_permission()
|
||||
mock_get_objects.return_value = [self.document]
|
||||
mock_stream_chat.return_value = iter([b"data"])
|
||||
|
||||
response = self.client.post(
|
||||
self.ENDPOINT,
|
||||
data='{"q": "question"}',
|
||||
content_type="application/json",
|
||||
)
|
||||
|
||||
self.assertEqual(response.status_code, 200)
|
||||
mock_stream_chat.assert_called_once_with(
|
||||
query_str="question",
|
||||
documents=[self.document],
|
||||
output_language="de-de",
|
||||
)
|
||||
|
||||
@patch("documents.views.stream_chat_with_documents")
|
||||
@override_settings(AI_ENABLED=True)
|
||||
|
||||
+12
-24
@@ -133,7 +133,6 @@ from documents.file_handling import format_filename
|
||||
from documents.filters import CorrespondentFilterSet
|
||||
from documents.filters import CustomFieldFilterSet
|
||||
from documents.filters import DocumentFilterSet
|
||||
from documents.filters import DocumentPermissionsFilter
|
||||
from documents.filters import DocumentsOrderingFilter
|
||||
from documents.filters import DocumentTypeFilterSet
|
||||
from documents.filters import ObjectOwnedOrGrantedPermissionsFilter
|
||||
@@ -654,20 +653,6 @@ class TagViewSet(PermissionsAwareDocumentCountMixin, ModelViewSet[Tag]):
|
||||
update_document_parent_tags(tag, new_parent)
|
||||
|
||||
|
||||
def _get_llm_output_language(ai_config: AIConfig, request) -> str | None:
|
||||
output_language = ai_config.llm_output_language
|
||||
if (
|
||||
not output_language
|
||||
and hasattr(request.user, "ui_settings")
|
||||
and isinstance(
|
||||
request.user.ui_settings.settings,
|
||||
dict,
|
||||
)
|
||||
):
|
||||
output_language = request.user.ui_settings.settings.get("language")
|
||||
return output_language
|
||||
|
||||
|
||||
@extend_schema_view(**generate_object_with_permissions_schema(DocumentTypeSerializer))
|
||||
class DocumentTypeViewSet(
|
||||
PermissionsAwareDocumentCountMixin,
|
||||
@@ -987,7 +972,7 @@ class DocumentViewSet(
|
||||
DjangoFilterBackend,
|
||||
SearchFilter,
|
||||
DocumentsOrderingFilter,
|
||||
DocumentPermissionsFilter,
|
||||
ObjectOwnedOrGrantedPermissionsFilter,
|
||||
)
|
||||
filterset_class = DocumentFilterSet
|
||||
search_fields = ("title", "correspondent__name", "effective_content")
|
||||
@@ -1529,7 +1514,16 @@ class DocumentViewSet(
|
||||
if not ai_config.ai_enabled:
|
||||
return HttpResponseBadRequest("AI is required for this feature")
|
||||
|
||||
output_language = _get_llm_output_language(ai_config=ai_config, request=request)
|
||||
output_language = ai_config.llm_output_language
|
||||
if (
|
||||
not output_language
|
||||
and hasattr(request.user, "ui_settings")
|
||||
and isinstance(
|
||||
request.user.ui_settings.settings,
|
||||
dict,
|
||||
)
|
||||
):
|
||||
output_language = request.user.ui_settings.settings.get("language") or None
|
||||
llm_cache_backend = (
|
||||
f"{ai_config.llm_backend}:{output_language}"
|
||||
if output_language
|
||||
@@ -2271,14 +2265,8 @@ class ChatStreamingView(GenericAPIView[Any]):
|
||||
Document,
|
||||
)
|
||||
|
||||
output_language = _get_llm_output_language(ai_config=ai_config, request=request)
|
||||
|
||||
response = StreamingHttpResponse(
|
||||
stream_chat_with_documents(
|
||||
query_str=question,
|
||||
documents=documents,
|
||||
output_language=output_language,
|
||||
),
|
||||
stream_chat_with_documents(query_str=question, documents=documents),
|
||||
content_type="text/event-stream",
|
||||
)
|
||||
return response
|
||||
|
||||
@@ -2,7 +2,7 @@ msgid ""
|
||||
msgstr ""
|
||||
"Project-Id-Version: paperless-ngx\n"
|
||||
"Report-Msgid-Bugs-To: \n"
|
||||
"POT-Creation-Date: 2026-07-27 19:34+0000\n"
|
||||
"POT-Creation-Date: 2026-07-25 07:12+0000\n"
|
||||
"PO-Revision-Date: 2022-02-17 04:17\n"
|
||||
"Last-Translator: \n"
|
||||
"Language-Team: English\n"
|
||||
@@ -21,39 +21,39 @@ msgstr ""
|
||||
msgid "Documents"
|
||||
msgstr ""
|
||||
|
||||
#: documents/filters.py:472
|
||||
#: documents/filters.py:470
|
||||
msgid "Value must be valid JSON."
|
||||
msgstr ""
|
||||
|
||||
#: documents/filters.py:491
|
||||
#: documents/filters.py:489
|
||||
msgid "Invalid custom field query expression"
|
||||
msgstr ""
|
||||
|
||||
#: documents/filters.py:501
|
||||
#: documents/filters.py:499
|
||||
msgid "Invalid expression list. Must be nonempty."
|
||||
msgstr ""
|
||||
|
||||
#: documents/filters.py:522
|
||||
#: documents/filters.py:520
|
||||
msgid "Invalid logical operator {op!r}"
|
||||
msgstr ""
|
||||
|
||||
#: documents/filters.py:536
|
||||
#: documents/filters.py:534
|
||||
msgid "Maximum number of query conditions exceeded."
|
||||
msgstr ""
|
||||
|
||||
#: documents/filters.py:600
|
||||
#: documents/filters.py:598
|
||||
msgid "{name!r} is not a valid custom field."
|
||||
msgstr ""
|
||||
|
||||
#: documents/filters.py:637
|
||||
#: documents/filters.py:635
|
||||
msgid "{data_type} does not support query expr {expr!r}."
|
||||
msgstr ""
|
||||
|
||||
#: documents/filters.py:752 documents/models.py:136
|
||||
#: documents/filters.py:750 documents/models.py:136
|
||||
msgid "Maximum nesting depth exceeded."
|
||||
msgstr ""
|
||||
|
||||
#: documents/filters.py:1094
|
||||
#: documents/filters.py:1067
|
||||
msgid "Custom field not found"
|
||||
msgstr ""
|
||||
|
||||
@@ -1352,7 +1352,7 @@ msgid "workflow runs"
|
||||
msgstr ""
|
||||
|
||||
#: documents/serialisers.py:522 documents/serialisers.py:874
|
||||
#: documents/serialisers.py:2763 documents/views.py:299 documents/views.py:2553
|
||||
#: documents/serialisers.py:2763 documents/views.py:298 documents/views.py:2541
|
||||
#: paperless_mail/serialisers.py:155
|
||||
msgid "Insufficient permissions."
|
||||
msgstr ""
|
||||
@@ -1393,7 +1393,7 @@ msgstr ""
|
||||
msgid "Duplicate document identifiers are not allowed."
|
||||
msgstr ""
|
||||
|
||||
#: documents/serialisers.py:2849 documents/views.py:4512
|
||||
#: documents/serialisers.py:2849 documents/views.py:4500
|
||||
#, python-format
|
||||
msgid "Documents not found: %(ids)s"
|
||||
msgstr ""
|
||||
@@ -1661,36 +1661,36 @@ msgstr ""
|
||||
msgid "Unable to parse URI {value}"
|
||||
msgstr ""
|
||||
|
||||
#: documents/views.py:292 documents/views.py:2550
|
||||
#: documents/views.py:291 documents/views.py:2538
|
||||
msgid "Invalid more_like_id"
|
||||
msgstr ""
|
||||
|
||||
#: documents/views.py:1562
|
||||
#: documents/views.py:1556
|
||||
msgid "Invalid AI configuration."
|
||||
msgstr ""
|
||||
|
||||
#: documents/views.py:1571
|
||||
#: documents/views.py:1565
|
||||
msgid "AI backend request timed out."
|
||||
msgstr ""
|
||||
|
||||
#: documents/views.py:2375 documents/views.py:2696
|
||||
#: documents/views.py:2363 documents/views.py:2684
|
||||
msgid "Specify only one of text, title_search, query, or more_like_id."
|
||||
msgstr ""
|
||||
|
||||
#: documents/views.py:4524
|
||||
#: documents/views.py:4512
|
||||
#, python-format
|
||||
msgid "Insufficient permissions to share document %(id)s."
|
||||
msgstr ""
|
||||
|
||||
#: documents/views.py:4570
|
||||
#: documents/views.py:4558
|
||||
msgid "Bundle is already being processed."
|
||||
msgstr ""
|
||||
|
||||
#: documents/views.py:4631
|
||||
#: documents/views.py:4619
|
||||
msgid "The share link bundle is still being prepared. Please try again later."
|
||||
msgstr ""
|
||||
|
||||
#: documents/views.py:4641
|
||||
#: documents/views.py:4629
|
||||
msgid "The share link bundle is unavailable."
|
||||
msgstr ""
|
||||
|
||||
|
||||
@@ -510,10 +510,8 @@ class RasterisedDocumentParser:
|
||||
|
||||
if mime_type == "application/pdf":
|
||||
text_original = self.extract_text(None, document_path)
|
||||
has_text = text_original is not None and len(text_original) > 0
|
||||
original_has_text = has_text and (
|
||||
is_tagged_pdf(document_path, log=self.log)
|
||||
or len(text_original) > PDF_TEXT_MIN_LENGTH
|
||||
original_has_text = is_tagged_pdf(document_path, log=self.log) or (
|
||||
text_original is not None and len(text_original) > PDF_TEXT_MIN_LENGTH
|
||||
)
|
||||
else:
|
||||
text_original = None
|
||||
|
||||
@@ -906,34 +906,6 @@ class TestSkipArchive:
|
||||
assert tesseract_parser.archive_path is None
|
||||
assert tesseract_parser.get_text()
|
||||
|
||||
def test_tagged_pdf_without_text_does_not_skip_ocr(
|
||||
self,
|
||||
mocker: MockerFixture,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
tesseract_samples_dir: Path,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- A PDF that reports itself as tagged (/MarkInfo /Marked true) but
|
||||
has no actual extractable text (some scanner firmware produces
|
||||
this — see GitHub issue #13349)
|
||||
- Mode: auto, produce_archive=False
|
||||
WHEN:
|
||||
- Document is parsed
|
||||
THEN:
|
||||
- The tag alone is not trusted as "has text"; OCRmyPDF still runs
|
||||
"""
|
||||
tesseract_parser.settings.mode = ModeChoices.AUTO
|
||||
mocker.patch("paperless.parsers.tesseract.is_tagged_pdf", return_value=True)
|
||||
mocker.patch.object(tesseract_parser, "extract_text", return_value=None)
|
||||
mock_ocr = mocker.patch("ocrmypdf.ocr")
|
||||
tesseract_parser.parse(
|
||||
tesseract_samples_dir / "multi-page-images.pdf",
|
||||
"application/pdf",
|
||||
produce_archive=False,
|
||||
)
|
||||
mock_ocr.assert_called()
|
||||
|
||||
def test_tagged_pdf_produces_pdfa_archive_without_ocr(
|
||||
self,
|
||||
tesseract_parser: RasterisedDocumentParser,
|
||||
|
||||
@@ -28,22 +28,11 @@ CHAT_PROMPT_TMPL = (
|
||||
"---------------------\n"
|
||||
"Using only the context above, answer the query. "
|
||||
"Do not use prior knowledge.\n"
|
||||
"{output_language_line}"
|
||||
"Query: {query_str}\n"
|
||||
"Answer:"
|
||||
)
|
||||
|
||||
|
||||
def _build_chat_prompt(output_language: str | None) -> str:
|
||||
output_language_line = (
|
||||
f"Respond in {output_language}.\n" if output_language is not None else ""
|
||||
)
|
||||
return CHAT_PROMPT_TMPL.replace(
|
||||
"{output_language_line}",
|
||||
output_language_line,
|
||||
)
|
||||
|
||||
|
||||
def _build_document_reference(
|
||||
document: Document,
|
||||
title: str | None = None,
|
||||
@@ -90,27 +79,15 @@ def _format_chat_metadata_trailer(references: list[dict[str, int | str]]) -> str
|
||||
)
|
||||
|
||||
|
||||
def stream_chat_with_documents(
|
||||
query_str: str,
|
||||
documents: list[Document],
|
||||
output_language: str | None = None,
|
||||
):
|
||||
def stream_chat_with_documents(query_str: str, documents: list[Document]):
|
||||
try:
|
||||
yield from _stream_chat_with_documents(
|
||||
query_str,
|
||||
documents,
|
||||
output_language=output_language,
|
||||
)
|
||||
yield from _stream_chat_with_documents(query_str, documents)
|
||||
except Exception as e:
|
||||
logger.exception("Failed to stream document chat response: %s", e)
|
||||
yield CHAT_ERROR_MESSAGE
|
||||
|
||||
|
||||
def _stream_chat_with_documents(
|
||||
query_str: str,
|
||||
documents: list[Document],
|
||||
output_language: str | None = None,
|
||||
):
|
||||
def _stream_chat_with_documents(query_str: str, documents: list[Document]):
|
||||
if not documents:
|
||||
yield CHAT_NO_CONTENT_MESSAGE
|
||||
return
|
||||
@@ -148,7 +125,7 @@ def _stream_chat_with_documents(
|
||||
|
||||
references = _get_document_references(documents, top_nodes)
|
||||
|
||||
prompt_template = PromptTemplate(template=_build_chat_prompt(output_language))
|
||||
prompt_template = PromptTemplate(template=CHAT_PROMPT_TMPL)
|
||||
response_synthesizer = get_response_synthesizer(
|
||||
llm=client.llm,
|
||||
prompt_helper=get_rag_prompt_helper(
|
||||
|
||||
@@ -7,6 +7,7 @@ if TYPE_CHECKING:
|
||||
from llama_index.core.base.embeddings.base import BaseEmbedding
|
||||
|
||||
from documents.models import Document
|
||||
from documents.models import Note
|
||||
from paperless.config import AIConfig
|
||||
from paperless.models import LLMEmbeddingBackend
|
||||
from paperless.network import PinnedHostAsyncHTTPTransport
|
||||
@@ -125,7 +126,7 @@ def build_llm_index_text(doc: Document) -> str:
|
||||
# prepend. Notes and Custom Fields stay in the body: Notes can be long free
|
||||
# text, Custom Fields are dynamic in count and best kept in the embedding.
|
||||
lines = [
|
||||
f"Notes: {','.join([str(c.note) for c in doc.notes.all()])}",
|
||||
f"Notes: {','.join([str(c.note) for c in Note.objects.filter(document=doc)])}",
|
||||
]
|
||||
|
||||
for instance in doc.custom_fields.all():
|
||||
|
||||
@@ -328,16 +328,8 @@ def update_llm_index(
|
||||
*,
|
||||
iter_wrapper: IterWrapper[Document] = identity,
|
||||
rebuild=False,
|
||||
document_ids: Iterable[int] | None = None,
|
||||
) -> str:
|
||||
"""Rebuild or incrementally update the LLM index.
|
||||
|
||||
``document_ids``, when given, scopes an incremental update to just those
|
||||
documents instead of scanning the whole library -- callers that already
|
||||
know which documents changed (e.g. a bulk edit) should pass this to avoid
|
||||
an O(library size) scan per call. Ignored whenever a rebuild actually
|
||||
happens, since a rebuild always covers the whole library regardless.
|
||||
"""
|
||||
"""Rebuild or incrementally update the LLM index."""
|
||||
with write_store() as store:
|
||||
try:
|
||||
with _exclude_readers():
|
||||
@@ -353,11 +345,7 @@ def update_llm_index(
|
||||
"LLM index migration requires re-embedding; forcing rebuild.",
|
||||
)
|
||||
rebuild = True
|
||||
documents = Document.objects.select_related(
|
||||
"correspondent",
|
||||
"document_type",
|
||||
"storage_path",
|
||||
).prefetch_related("tags", "notes", "custom_fields__field")
|
||||
documents = Document.objects.all()
|
||||
no_documents = not documents.exists()
|
||||
|
||||
# Fast exit before touching config: nothing to index and no existing index.
|
||||
@@ -391,14 +379,9 @@ def update_llm_index(
|
||||
store.add(nodes)
|
||||
msg = "LLM index rebuilt successfully."
|
||||
else:
|
||||
scoped_documents = (
|
||||
documents.filter(id__in=document_ids)
|
||||
if document_ids is not None
|
||||
else documents
|
||||
)
|
||||
existing = store.get_modified_times()
|
||||
changed = 0
|
||||
for document in iter_wrapper(scoped_documents):
|
||||
for document in iter_wrapper(documents):
|
||||
doc_id = str(document.id)
|
||||
if existing.get(doc_id) == document.modified.isoformat():
|
||||
continue
|
||||
|
||||
@@ -4,18 +4,13 @@ from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
import pytest_mock
|
||||
from django.db import connection
|
||||
from django.test import override_settings
|
||||
from django.test.utils import CaptureQueriesContext
|
||||
from django.utils import timezone
|
||||
from llama_index.core.schema import MetadataMode
|
||||
|
||||
from documents.models import Correspondent
|
||||
from documents.models import CustomField
|
||||
from documents.models import CustomFieldInstance
|
||||
from documents.models import Document
|
||||
from documents.models import DocumentType
|
||||
from documents.models import Note
|
||||
from documents.models import PaperlessTask
|
||||
from documents.signals import document_consumption_finished
|
||||
from documents.signals import document_updated
|
||||
@@ -202,8 +197,6 @@ def test_update_llm_index(
|
||||
mock_queryset = MagicMock()
|
||||
mock_queryset.exists.return_value = True
|
||||
mock_queryset.__iter__.return_value = iter([real_document])
|
||||
mock_queryset.select_related.return_value = mock_queryset
|
||||
mock_queryset.prefetch_related.return_value = mock_queryset
|
||||
mock_all.return_value = mock_queryset
|
||||
build_document_node.return_value = []
|
||||
indexing.update_llm_index(rebuild=True)
|
||||
@@ -223,8 +216,6 @@ def test_update_llm_index_rebuilds_on_model_name_change(
|
||||
mock_queryset = MagicMock()
|
||||
mock_queryset.exists.return_value = True
|
||||
mock_queryset.__iter__.return_value = iter([real_document])
|
||||
mock_queryset.select_related.return_value = mock_queryset
|
||||
mock_queryset.prefetch_related.return_value = mock_queryset
|
||||
mock_all.return_value = mock_queryset
|
||||
with patch(
|
||||
"paperless_ai.indexing.get_configured_model_name",
|
||||
@@ -237,8 +228,6 @@ def test_update_llm_index_rebuilds_on_model_name_change(
|
||||
mock_queryset = MagicMock()
|
||||
mock_queryset.exists.return_value = True
|
||||
mock_queryset.__iter__.return_value = iter([real_document])
|
||||
mock_queryset.select_related.return_value = mock_queryset
|
||||
mock_queryset.prefetch_related.return_value = mock_queryset
|
||||
mock_all.return_value = mock_queryset
|
||||
with patch(
|
||||
"paperless_ai.indexing.get_configured_model_name",
|
||||
@@ -269,8 +258,6 @@ def test_update_llm_index_partial_update(
|
||||
mock_queryset = MagicMock()
|
||||
mock_queryset.exists.return_value = True
|
||||
mock_queryset.__iter__.return_value = iter([real_document, doc2])
|
||||
mock_queryset.select_related.return_value = mock_queryset
|
||||
mock_queryset.prefetch_related.return_value = mock_queryset
|
||||
mock_all.return_value = mock_queryset
|
||||
|
||||
indexing.update_llm_index(rebuild=True)
|
||||
@@ -299,52 +286,6 @@ def test_update_llm_index_partial_update(
|
||||
assert store.table_exists(), (
|
||||
"Expected the vector store table to exist after incremental update"
|
||||
)
|
||||
before = store.get_modified_times()
|
||||
|
||||
# new doc, also touched by the scoped update below
|
||||
doc4 = DocumentFactory.create(title="Test Document 4", added=timezone.now())
|
||||
|
||||
# A further edit, scoped via document_ids to doc3 + doc4 -- doc2 must be
|
||||
# left exactly as it was, proving document_ids restricts the scan
|
||||
# instead of falling back to the whole library.
|
||||
doc3.modified = timezone.now()
|
||||
doc3.save()
|
||||
doc4.modified = timezone.now()
|
||||
doc4.save()
|
||||
|
||||
# Give both scoped documents a note and a custom field: build_llm_index_text
|
||||
# reads both per document, so without the notes/custom_fields__field
|
||||
# prefetch on scoped_documents, each additional document adds 3 more
|
||||
# queries (N+1 regression) instead of the query count staying flat.
|
||||
custom_field = CustomField.objects.create(
|
||||
name="Priority",
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
for doc in (doc3, doc4):
|
||||
Note.objects.create(document=doc, note=f"a note on {doc.title}")
|
||||
CustomFieldInstance.objects.create(
|
||||
document=doc,
|
||||
field=custom_field,
|
||||
value_text="high",
|
||||
)
|
||||
|
||||
with CaptureQueriesContext(connection) as ctx:
|
||||
result = indexing.update_llm_index(
|
||||
rebuild=False,
|
||||
document_ids=[doc3.pk, doc4.pk],
|
||||
)
|
||||
assert result == "LLM index updated successfully."
|
||||
# Notes/custom fields are prefetched in one batch query each (plus one
|
||||
# more for custom_fields__field), not re-queried per document -- an N+1
|
||||
# regression here would scale with document count instead of staying flat
|
||||
# (7 with the prefetch vs. 10 without it, for these 2 documents).
|
||||
assert len(ctx.captured_queries) <= 8
|
||||
|
||||
with indexing.get_vector_store() as store:
|
||||
after = store.get_modified_times()
|
||||
|
||||
assert after[str(doc3.pk)] == doc3.modified.isoformat()
|
||||
assert after[str(doc2.pk)] == before[str(doc2.pk)]
|
||||
|
||||
|
||||
@pytest.mark.django_db
|
||||
|
||||
@@ -12,7 +12,6 @@ from paperless_ai import chat
|
||||
from paperless_ai import indexing
|
||||
from paperless_ai.chat import CHAT_ERROR_MESSAGE
|
||||
from paperless_ai.chat import CHAT_METADATA_DELIMITER
|
||||
from paperless_ai.chat import _build_chat_prompt
|
||||
from paperless_ai.chat import stream_chat_with_documents
|
||||
|
||||
|
||||
@@ -60,26 +59,6 @@ def assert_chat_output(
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("output_language", "expected_language_line"),
|
||||
[
|
||||
(None, ""),
|
||||
("de-de", "Respond in de-de.\n"),
|
||||
],
|
||||
)
|
||||
def test_build_chat_prompt(
|
||||
output_language,
|
||||
expected_language_line,
|
||||
) -> None:
|
||||
prompt = _build_chat_prompt(output_language)
|
||||
|
||||
assert "{output_language_line}" not in prompt
|
||||
assert (
|
||||
prompt.split("Do not use prior knowledge.\n", maxsplit=1)[1]
|
||||
== f"{expected_language_line}Query: {{query_str}}\nAnswer:"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.django_db
|
||||
def test_stream_chat_with_one_document_retrieval(
|
||||
mock_document,
|
||||
|
||||
@@ -54,7 +54,6 @@ def mock_document():
|
||||
cf2.field.name = "Field2"
|
||||
cf2.value = "Value2"
|
||||
doc.custom_fields.all = MagicMock(return_value=[cf1, cf2])
|
||||
doc.notes.all = MagicMock(return_value=[])
|
||||
|
||||
return doc
|
||||
|
||||
@@ -220,26 +219,28 @@ def test_get_configured_model_name_explicit_overrides_default(mock_ai_config):
|
||||
|
||||
|
||||
def test_build_llm_index_text(mock_document):
|
||||
mock_document.notes.all = MagicMock(
|
||||
return_value=[MagicMock(note="Note1"), MagicMock(note="Note2")],
|
||||
)
|
||||
with patch("documents.models.Note.objects.filter") as mock_notes_filter:
|
||||
mock_notes_filter.return_value = [
|
||||
MagicMock(note="Note1"),
|
||||
MagicMock(note="Note2"),
|
||||
]
|
||||
|
||||
result = build_llm_index_text(mock_document)
|
||||
result = build_llm_index_text(mock_document)
|
||||
|
||||
# Structured fields live in node.metadata for LLM context -- not body text
|
||||
assert "Title: Test Title" not in result
|
||||
assert "Created: 2023-01-01" not in result
|
||||
assert "Tags: Tag1, Tag2" not in result
|
||||
assert "Document Type: Invoice" not in result
|
||||
assert "Correspondent: Test Correspondent" not in result
|
||||
assert "Filename:" not in result
|
||||
assert "Storage Path:" not in result
|
||||
assert "Archive Serial Number:" not in result
|
||||
# Structured fields live in node.metadata for LLM context -- not body text
|
||||
assert "Title: Test Title" not in result
|
||||
assert "Created: 2023-01-01" not in result
|
||||
assert "Tags: Tag1, Tag2" not in result
|
||||
assert "Document Type: Invoice" not in result
|
||||
assert "Correspondent: Test Correspondent" not in result
|
||||
assert "Filename:" not in result
|
||||
assert "Storage Path:" not in result
|
||||
assert "Archive Serial Number:" not in result
|
||||
|
||||
# Fields without a metadata equivalent stay in body text
|
||||
assert "Notes: Note1,Note2" in result
|
||||
assert "Content:\n\nThis is the document content." in result
|
||||
assert "Custom Field - Field1: Value1\nCustom Field - Field2: Value2" in result
|
||||
# Fields without a metadata equivalent stay in body text
|
||||
assert "Notes: Note1,Note2" in result
|
||||
assert "Content:\n\nThis is the document content." in result
|
||||
assert "Custom Field - Field1: Value1\nCustom Field - Field2: Value2" in result
|
||||
|
||||
|
||||
def test_build_llm_index_text_normalizes_ocr_punctuation_runs(mock_document):
|
||||
@@ -249,7 +250,8 @@ def test_build_llm_index_text_normalizes_ocr_punctuation_runs(mock_document):
|
||||
"Keep short punctuation like INV-100 and ellipses..."
|
||||
)
|
||||
|
||||
result = build_llm_index_text(mock_document)
|
||||
with patch("documents.models.Note.objects.filter", return_value=[]):
|
||||
result = build_llm_index_text(mock_document)
|
||||
|
||||
assert "Introduction 7" in result
|
||||
assert "Hardware Limitation 9" in result
|
||||
|
||||
Reference in New Issue
Block a user