Compare commits

..
Author SHA1 Message Date
stumpylog dfdc8aa3bb Refactor: extract QuerySetStream, shared by _StreamedDocuments and _DocumentViewerStream 2026-07-30 08:42:05 -07:00
stumpylog a9161e7d84 Test: consolidate TestStreamedDocuments into one test 2026-07-30 08:13:43 -07:00
stumpylogandClaude Sonnet 5 42208ec70c Perf: stream update_llm_index()'s document queryset instead of loading it whole
update_llm_index() built its documents queryset with
prefetch_related("tags", "notes", "custom_fields__field") and iterated
it directly -- no .iterator(), so a full rebuild pulled every Document
(including content, which can be megabytes each) plus every prefetch
cache into memory simultaneously.

Add _StreamedDocuments, a thin QuerySet wrapper matching the shape of
documents/search/_backend.py's _DocumentViewerStream: iterates via
.iterator(chunk_size=1000) so Django streams from a server-side cursor
(and, since 4.1, still runs prefetch_related per batch instead of all at
once), while __len__ still supports a real progress-bar total. Used for
both loop sites (the full rebuild and the modified-time-scoped update).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-07-28 15:28:52 -07:00
27 changed files with 249 additions and 402 deletions
+10 -10
View File
@@ -1849,7 +1849,7 @@
</context-group>
<context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">159</context>
<context context-type="linenumber">154</context>
</context-group>
<context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/filterable-dropdown/filterable-dropdown.component.html</context>
@@ -3972,11 +3972,11 @@
</context-group>
<context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">101</context>
<context context-type="linenumber">96</context>
</context-group>
<context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">107</context>
<context context-type="linenumber">102</context>
</context-group>
</trans-unit>
<trans-unit id="3800326155195149498" datatype="html">
@@ -3987,11 +3987,11 @@
</context-group>
<context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">102</context>
<context context-type="linenumber">97</context>
</context-group>
<context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">108</context>
<context context-type="linenumber">103</context>
</context-group>
</trans-unit>
<trans-unit id="7551700625201096185" datatype="html">
@@ -4002,14 +4002,14 @@
</context-group>
<context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">124</context>
<context context-type="linenumber">119</context>
</context-group>
</trans-unit>
<trans-unit id="3184700926171002527" datatype="html">
<source>Any</source>
<context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">157</context>
<context context-type="linenumber">152</context>
</context-group>
<context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/filterable-dropdown/filterable-dropdown.component.html</context>
@@ -4020,21 +4020,21 @@
<source>Not</source>
<context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">162</context>
<context context-type="linenumber">157</context>
</context-group>
</trans-unit>
<trans-unit id="6548676277933116532" datatype="html">
<source>Add query</source>
<context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">181</context>
<context context-type="linenumber">176</context>
</context-group>
</trans-unit>
<trans-unit id="5599577087865387184" datatype="html">
<source>Add expression</source>
<context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">184</context>
<context context-type="linenumber">179</context>
</context-group>
</trans-unit>
<trans-unit id="6312759212949884929" datatype="html">
@@ -182,7 +182,7 @@
container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim">
<i-bs class="me-2" name="stack"></i-bs><span><ng-container i18n>Attributes</ng-container></span>
</a>
@if (!slimSidebarEnabled && canSaveSettings) {
@if (!slimSidebarEnabled) {
<button
type="button"
class="btn btn-link btn-sm text-muted p-0 me-3 attributes-expand-btn"
@@ -68,11 +68,6 @@
></ng-select>
} @else if (getCustomFieldByID(atom.field)?.data_type === CustomFieldDataType.DocumentLink) {
<pngx-input-document-link [(ngModel)]="atom.value" class="w-25 form-select doc-link-select p-0" placeholder="Search docs..." i18n-placeholder [minimal]="true"></pngx-input-document-link>
} @else if (getCustomFieldByID(atom.field)?.data_type === CustomFieldDataType.Monetary) {
<input class="w-25 form-control rounded-end" type="text" inputmode="decimal"
[ngModel]="atom.value"
(ngModelChange)="setMonetaryValue(atom, $event)"
[disabled]="disabled">
} @else {
<input class="w-25 form-control rounded-end" type="text" [(ngModel)]="atom.value" [disabled]="disabled">
}
@@ -1,6 +1,5 @@
import { provideHttpClient, withInterceptorsFromDi } from '@angular/common/http'
import { provideHttpClientTesting } from '@angular/common/http/testing'
import { LOCALE_ID } from '@angular/core'
import { ComponentFixture, TestBed } from '@angular/core/testing'
import { FormsModule, ReactiveFormsModule } from '@angular/forms'
import { NgbDropdownModule } from '@ng-bootstrap/ng-bootstrap'
@@ -42,12 +41,6 @@ const customFields = [
],
},
},
{
id: 3,
name: 'Test Monetary Field',
data_type: CustomFieldDataType.Monetary,
extra_data: { default_currency: 'EUR' },
},
]
describe('CustomFieldsQueryDropdownComponent', () => {
@@ -68,7 +61,6 @@ describe('CustomFieldsQueryDropdownComponent', () => {
providers: [
provideHttpClient(withInterceptorsFromDi()),
provideHttpClientTesting(),
{ provide: LOCALE_ID, useValue: 'de' },
],
}).compileComponents()
@@ -158,22 +150,6 @@ describe('CustomFieldsQueryDropdownComponent', () => {
expect(options2).toEqual([])
})
it('should normalize localized monetary comparison values', () => {
const atom = new CustomFieldQueryAtom([3, 'exact', null])
component.setMonetaryValue(atom, '1.234,56')
expect(atom.value).toEqual('1234.56')
})
it('should preserve API-formatted monetary comparison values', () => {
const atom = new CustomFieldQueryAtom([3, 'exact', null])
component.setMonetaryValue(atom, '1234.56')
expect(atom.value).toEqual('1234.56')
})
it('should remove an element from the selection model', () => {
const expression = new CustomFieldQueryExpression()
const atom = new CustomFieldQueryAtom()
@@ -1,14 +1,9 @@
import {
getLocaleNumberSymbol,
NgTemplateOutlet,
NumberSymbol,
} from '@angular/common'
import { NgTemplateOutlet } from '@angular/common'
import {
Component,
EventEmitter,
inject,
Input,
LOCALE_ID,
Output,
QueryList,
signal,
@@ -217,7 +212,6 @@ export class CustomFieldQueriesModel {
})
export class CustomFieldsQueryDropdownComponent extends LoadingComponentWithPermissions {
protected customFieldsService = inject(CustomFieldsService)
private readonly locale = inject(LOCALE_ID)
public CustomFieldQueryComponentType = CustomFieldQueryElementType
public CustomFieldQueryOperator = CustomFieldQueryOperator
@@ -382,18 +376,4 @@ export class CustomFieldsQueryDropdownComponent extends LoadingComponentWithPerm
}
return []
}
setMonetaryValue(atom: CustomFieldQueryAtom, value: string) {
// Normalize the decimal symbol e.g. . vs , by locale
const decimalSymbol = getLocaleNumberSymbol(
this.locale,
NumberSymbol.Decimal
)
if (decimalSymbol !== '.' && value.includes(decimalSymbol)) {
const groupSymbol = getLocaleNumberSymbol(this.locale, NumberSymbol.Group)
value = value.split(groupSymbol).join('').split(decimalSymbol).join('.')
}
atom.value = value
}
}
@@ -129,25 +129,13 @@ describe('PngxPdfViewerComponent', () => {
;(component as any).applyScale()
expect(viewer.currentScaleValue).toBe(PdfZoomScale.PageFit)
expect(viewer.currentScale).toBe(2)
})
it('does not reapply scale for page-only changes', async () => {
await initComponent()
const pdf = (component as any).pdf as { numPages: number }
pdf.numPages = 3
const viewer = (component as any).pdfViewer as PDFViewer
viewer.setDocument(pdf)
const applyScaleSpy = jest.spyOn(component as any, 'applyScale')
component.page = 2
component.ngOnChanges({
page: new SimpleChange(1, 2, false),
})
expect(viewer.currentPageNumber).toBe(2)
;(component as any).lastViewerPage = 2
;(component as any).applyViewerState()
expect((component as any).lastViewerPage).toBeUndefined()
expect(applyScaleSpy).not.toHaveBeenCalled()
expect(applyScaleSpy).toHaveBeenCalled()
})
it('does not reset the viewer when it is already on the requested page', async () => {
@@ -116,10 +116,7 @@ export class PngxPdfViewerComponent
changes['zoomScale'] ||
changes['rotation']
) {
// Prevent loop with page / scale application see https://github.com/paperless-ngx/paperless-ngx/issues/13404
this.applyViewerState(
!!(changes['zoom'] || changes['zoomScale'] || changes['rotation'])
)
this.applyViewerState()
}
if (changes['searchQuery']) {
@@ -243,7 +240,7 @@ export class PngxPdfViewerComponent
}
}
private applyViewerState(applyScale = true): void {
private applyViewerState(): void {
if (!this.pdfViewer) {
return
}
@@ -267,7 +264,7 @@ export class PngxPdfViewerComponent
if (this.page === this.lastViewerPage) {
this.lastViewerPage = undefined
}
if (hasPages && applyScale) {
if (hasPages) {
this.applyScale()
}
this.dispatchFindIfReady()
@@ -51,8 +51,8 @@
*pngxIfPermissions="{ action: PermissionAction.Add, type: activeManagementList.permissionType }">
<i-bs name="plus-circle" class="me-1"></i-bs><ng-container i18n>Create</ng-container>
</button>
} @else if (customFieldsActive) {
<button type="button" class="btn btn-sm btn-outline-primary" (click)="addCustomField()"
} @else if (activeCustomFields) {
<button type="button" class="btn btn-sm btn-outline-primary" (click)="activeCustomFields.editField()"
*pngxIfPermissions="{ action: PermissionAction.Add, type: PermissionType.CustomField }">
<i-bs name="plus-circle" class="me-1"></i-bs><ng-container i18n>Add Field</ng-container>
</button>
@@ -18,7 +18,6 @@ import {
DocumentAttributesComponent,
DocumentAttributesSectionKind,
} from './document-attributes.component'
import { CustomFieldsComponent } from './custom-fields/custom-fields.component'
import { ManagementListComponent } from './management-list/management-list.component'
@Component({
@@ -208,29 +207,6 @@ describe('DocumentAttributesComponent', () => {
expect(component.activeSection.kind).toBe(
DocumentAttributesSectionKind.CustomFields
)
const customFields = Object.create(CustomFieldsComponent.prototype)
customFields.editField = jest.fn()
component.activeOutlet = {
componentInstance: customFields,
} as any
expect(component.activeCustomFields).toBe(customFields)
component.addCustomField()
expect(customFields.editField).toHaveBeenCalled()
})
it('should show the add field button before the custom fields instance is available', async () => {
jest.spyOn(permissionsService, 'currentUserCan').mockReturnValue(true)
fixture.detectChanges()
component.activeNavID.set(2)
await fixture.whenStable()
expect(component.activeCustomFields).toBeNull()
expect(
fixture.nativeElement.querySelector(
'pngx-page-header .btn-outline-primary'
)?.textContent
).toContain('Add Field')
expect(component.activeCustomFields).toBeDefined()
})
})
@@ -163,17 +163,12 @@ export class DocumentAttributesComponent implements OnInit, OnDestroy {
}
get activeCustomFields(): CustomFieldsComponent | null {
if (!this.customFieldsActive) return null
if (this.activeSection?.kind !== DocumentAttributesSectionKind.CustomFields)
return null
const instance = this.activeOutlet?.componentInstance
return instance instanceof CustomFieldsComponent ? instance : null
}
get customFieldsActive(): boolean {
return (
this.activeSection?.kind === DocumentAttributesSectionKind.CustomFields
)
}
get activeTabLabel(): string {
return this.activeSection?.label ?? ''
}
@@ -229,10 +224,6 @@ export class DocumentAttributesComponent implements OnInit, OnDestroy {
this.router.navigate(['attributes', nextSection])
}
addCustomField(): void {
this.activeCustomFields?.editField(null)
}
private getDefaultNavID(): DocumentAttributesNavIDs | null {
return this.visibleSections[0]?.id ?? null
}
+25 -15
View File
@@ -57,7 +57,9 @@ from paperless.models import ArchiveFileGenerationChoices
from paperless.parsers import ParserContext
from paperless.parsers import ParserProtocol
from paperless.parsers.registry import get_parser_registry
from paperless.parsers.utils import pdf_born_digital_text
from paperless.parsers.utils import PDF_TEXT_MIN_LENGTH
from paperless.parsers.utils import extract_pdf_text
from paperless.parsers.utils import is_tagged_pdf
LOGGING_NAME: Final[str] = "paperless.consumer"
@@ -136,45 +138,53 @@ def should_produce_archive(
# Must produce a PDF so the frontend can display the original format at all.
if parser.requires_pdf_rendition:
_log.debug("Archive: yes - parser requires PDF rendition for frontend display")
_log.debug("Archive: yes parser requires PDF rendition for frontend display")
return True
# Parser cannot produce an archive (e.g. TextDocumentParser).
if not parser.can_produce_archive:
_log.debug("Archive: no - parser cannot produce archives")
_log.debug("Archive: no parser cannot produce archives")
return False
generation = OcrConfig().archive_file_generation
if generation == ArchiveFileGenerationChoices.ALWAYS:
_log.debug("Archive: yes - ARCHIVE_FILE_GENERATION=always")
_log.debug("Archive: yes ARCHIVE_FILE_GENERATION=always")
return True
if generation == ArchiveFileGenerationChoices.NEVER:
_log.debug("Archive: no - ARCHIVE_FILE_GENERATION=never")
_log.debug("Archive: no ARCHIVE_FILE_GENERATION=never")
return False
# auto: produce archives for scanned/image documents; skip for born-digital PDFs.
if mime_type.startswith("image/"):
_log.debug("Archive: yes - image document, ARCHIVE_FILE_GENERATION=auto")
_log.debug("Archive: yes image document, ARCHIVE_FILE_GENERATION=auto")
return True
if mime_type == "application/pdf":
text, born_digital = pdf_born_digital_text(document_path, log=_log)
text_length = len(text) if text else 0
if born_digital:
text = extract_pdf_text(document_path)
has_text = text is not None and len(text) > 0
if has_text and is_tagged_pdf(document_path):
_log.debug(
"Archive: no - born-digital PDF (text_length=%d),"
"Archive: no born-digital PDF (structure tags detected),"
" ARCHIVE_FILE_GENERATION=auto",
text_length,
)
return False
if text is None or len(text) <= PDF_TEXT_MIN_LENGTH:
_log.debug(
"Archive: yes — scanned PDF (text_length=%d%d),"
" ARCHIVE_FILE_GENERATION=auto",
len(text) if text else 0,
PDF_TEXT_MIN_LENGTH,
)
return True
_log.debug(
"Archive: yes - scanned/textless PDF (text_length=%d),"
"Archive: no — born-digital PDF (text_length=%d > %d),"
" ARCHIVE_FILE_GENERATION=auto",
text_length,
len(text),
PDF_TEXT_MIN_LENGTH,
)
return True
return False
_log.debug(
"Archive: no - MIME type %r not eligible for auto archive generation",
"Archive: no MIME type %r not eligible for auto archive generation",
mime_type,
)
return False
-3
View File
@@ -36,9 +36,6 @@ def send_email(
TODO: re-evaluate this pending https://code.djangoproject.com/ticket/35581 / https://github.com/django/django/pull/18966
"""
if "\r" in subject or "\n" in subject:
subject = " ".join(line.strip(" \t") for line in subject.splitlines())
email = EmailMessage(
subject=subject,
body=body,
@@ -386,19 +386,10 @@ class Command(CryptMixin, PaperlessCommand):
raise DeserializationError(
f"{model.__name__} has no updatable fields; PK-only models are not supported by the importer",
)
# MySQL/MariaDB support upserts via ON DUPLICATE KEY UPDATE but,
# unlike PostgreSQL/SQLite, cannot target a specific unique field
# for the conflict -- passing unique_fields there raises
# NotSupportedError.
unique_fields = (
[model._meta.pk.attname]
if connection.features.supports_update_conflicts_with_target
else None
)
model.objects.bulk_create( # type: ignore[attr-defined]
instances,
update_conflicts=True,
unique_fields=unique_fields,
unique_fields=[model._meta.pk.attname],
update_fields=update_fields,
)
loaded_models.add(model)
@@ -61,7 +61,7 @@ class Command(PaperlessCommand):
)
table.add_column("Level", width=7, no_wrap=True)
table.add_column("Document", min_width=20)
table.add_column("Issue", ratio=1, overflow="fold")
table.add_column("Issue", ratio=1)
for doc_pk, doc_messages in messages.iter_messages():
if doc_pk is not None:
+6 -11
View File
@@ -39,6 +39,7 @@ from documents.search._tokenizer import ascii_fold
from documents.search._tokenizer import autocomplete_tokens
from documents.search._tokenizer import register_tokenizers
from documents.utils import IterWrapper
from documents.utils import QuerySetStream
from documents.utils import identity
if TYPE_CHECKING:
@@ -1035,14 +1036,15 @@ _EMPTY_VIEWER_GRANT: Final[ViewerGrant] = ViewerGrant(
)
class _DocumentViewerStream:
class _DocumentViewerStream(QuerySetStream["Document"]):
"""Yield document permission data while batch-loading grants.
Viewer permissions are fetched in batches (see
``_bulk_get_viewer_permissions``), but documents are yielded individually so a
progress bar wrapped around this stream advances per document rather than
jumping a whole chunk at a time. ``__len__`` lets the progress helper still
discover the total (it inspects ``QuerySet``/``Sized``).
jumping a whole chunk at a time. ``__len__`` (inherited from
``QuerySetStream``) lets the progress helper still discover the total (it
inspects ``QuerySet``/``Sized``).
The viewer and group ids travel with each document in the yielded pair
rather than through a separate mutable attribute, so the pairing survives
@@ -1051,18 +1053,11 @@ class _DocumentViewerStream:
generator in lock-step.
"""
def __init__(self, documents: QuerySet[Document], *, chunk_size: int) -> None:
self._documents = documents
self._chunk_size = chunk_size
def __len__(self) -> int:
return self._documents.count()
def __iter__(self) -> Iterator[tuple[Document, ViewerGrant]]:
# iterator(chunk_size=…) streams from a server-side cursor instead of
# materialising the whole queryset in memory; since Django 4.1 it still
# honours prefetch_related, running the prefetches one batch at a time.
documents = self._documents.iterator(chunk_size=self._chunk_size)
documents = self._queryset.iterator(chunk_size=self._chunk_size)
for chunk in chunked(documents, self._chunk_size):
grants_by_pk = _bulk_get_viewer_permissions([doc.pk for doc in chunk])
for doc in chunk:
+1 -1
View File
@@ -75,7 +75,7 @@ class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
{
"documents": [self.doc1.pk, self.doc2.pk],
"addresses": "hello@paperless-ngx.com,test@example.com",
"subject": "Bulk email\n test",
"subject": "Bulk email test",
"message": "Here are your documents",
},
),
+2 -2
View File
@@ -1329,7 +1329,7 @@ class PreConsumeTestCase(DirectoriesMixin, GetConsumerMixin, TestCase):
with self.get_consumer(self.test_file) as c:
c.run()
# Verify no pre-consume script subprocess was invoked
# (run_subprocess may still be called by pdf_born_digital_text via pdftotext)
# (run_subprocess may still be called by _extract_text_for_archive_check)
script_calls = [
call
for call in m.call_args_list
@@ -1354,7 +1354,7 @@ class PreConsumeTestCase(DirectoriesMixin, GetConsumerMixin, TestCase):
self.assertTrue(m.called)
# Find the call that invoked the pre-consume script
# (run_subprocess may also be called by pdf_born_digital_text via pdftotext)
# (run_subprocess may also be called by _extract_text_for_archive_check)
script_call = next(
call
for call in m.call_args_list
+42 -14
View File
@@ -134,32 +134,60 @@ class TestShouldProduceArchive:
assert should_produce_archive(parser, mime, Path("/tmp/doc")) is expected
@pytest.mark.parametrize(
("born_digital", "expected"),
("extracted_text", "expected"),
[
pytest.param(True, False, id="born-digital-skips-archive"),
pytest.param(False, True, id="not-born-digital-produces-archive"),
pytest.param(
"This is a born-digital PDF with lots of text content. " * 10,
False,
id="born-digital-long-text-skips-archive",
),
pytest.param(None, True, id="no-text-scanned-produces-archive"),
pytest.param("tiny", True, id="short-text-treated-as-scanned"),
],
)
def test_auto_pdf_archive_decision(
self,
mocker: MockerFixture,
settings,
born_digital: bool, # noqa: FBT001
extracted_text: str | None,
expected: bool, # noqa: FBT001
) -> None:
"""Archive decision tracks pdf_born_digital_text()'s verdict exactly.
should_produce_archive() defers entirely to pdf_born_digital_text()
for the has-real-text decision, so both callers of that predicate
(this function and RasterisedDocumentParser.parse()) always agree.
"""
settings.ARCHIVE_FILE_GENERATION = "auto"
mocker.patch(
"documents.consumer.pdf_born_digital_text",
return_value=("some text", born_digital),
)
mocker.patch("documents.consumer.is_tagged_pdf", return_value=False)
mocker.patch("documents.consumer.extract_pdf_text", return_value=extracted_text)
parser = _parser_instance(can_produce=True, requires_rendition=False)
assert (
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
is expected
)
def test_tagged_pdf_skips_archive_in_auto_mode(
self,
mocker: MockerFixture,
settings,
) -> None:
"""Tagged PDFs (e.g. Word exports) with real text are treated as born-digital, even below PDF_TEXT_MIN_LENGTH."""
settings.ARCHIVE_FILE_GENERATION = "auto"
mocker.patch("documents.consumer.is_tagged_pdf", return_value=True)
mocker.patch("documents.consumer.extract_pdf_text", return_value="tiny")
parser = _parser_instance(can_produce=True, requires_rendition=False)
assert (
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
is False
)
def test_tagged_pdf_without_text_produces_archive(
self,
mocker: MockerFixture,
settings,
) -> None:
"""A tagged PDF with no actual extractable text (e.g. some scanner firmware) is not
trusted as born-digital the tag alone must not bypass OCR."""
settings.ARCHIVE_FILE_GENERATION = "auto"
mocker.patch("documents.consumer.is_tagged_pdf", return_value=True)
mocker.patch("documents.consumer.extract_pdf_text", return_value=None)
parser = _parser_instance(can_produce=True, requires_rendition=False)
assert (
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
is True
)
+35
View File
@@ -0,0 +1,35 @@
import pytest_mock
from documents.utils import QuerySetStream
class TestQuerySetStream:
def test_len_and_iter_delegate_to_streaming_queryset_methods(
self,
mocker: pytest_mock.MockerFixture,
) -> None:
"""
GIVEN:
- A mock queryset
WHEN:
- A QuerySetStream wrapping it is measured and iterated
THEN:
- len() uses count() (not a materializing len()), and iteration
uses .iterator(chunk_size=...) (not plain iteration, which
would materialize the whole queryset, plus any prefetch
caches, into Django's own result cache at once)
"""
mock_queryset = mocker.MagicMock()
mock_queryset.count.return_value = 42
mock_queryset.iterator.return_value = iter(["row-1", "row-2"])
streamed = QuerySetStream(mock_queryset, chunk_size=1000)
assert len(streamed) == 42
assert list(streamed) == ["row-1", "row-2"]
# count.call_count isn't asserted exactly: list()'s own size-hint
# optimization calls len(streamed) again internally, on top of the
# explicit len() call above -- both legitimately delegate to
# count(), so only the delegation itself (not the call count) is
# the thing being verified here.
mock_queryset.count.assert_called_with()
mock_queryset.iterator.assert_called_once_with(chunk_size=1000)
+42
View File
@@ -3,16 +3,24 @@ import logging
import shutil
from collections.abc import Callable
from collections.abc import Iterable
from collections.abc import Iterator
from os import utime
from pathlib import Path
from subprocess import CompletedProcess
from subprocess import run
from typing import TYPE_CHECKING
from typing import Generic
from typing import TypeVar
from django.conf import settings
from PIL import Image
if TYPE_CHECKING:
from django.db.models import Model
from django.db.models import QuerySet
_T = TypeVar("_T")
_M = TypeVar("_M", bound="Model")
# A function that wraps an iterable — typically used to inject a progress bar.
IterWrapper = Callable[[Iterable[_T]], Iterable[_T]]
@@ -23,6 +31,40 @@ def identity(iterable: Iterable[_T]) -> Iterable[_T]:
return iterable
class QuerySetStream(Generic[_M]):
"""Stream a QuerySet via .iterator(chunk_size=...) instead of
materializing it (plus any prefetch caches) all at once, while still
supporting len() via count() so a progress bar wrapped around this
(e.g. via IterWrapper) shows a real total instead of falling back to
indeterminate.
Plain QuerySet iteration (``for row in queryset:``) is not lazy: Django
fetches every matching row in one query and caches the fully-hydrated
result in the queryset's own ``_result_cache`` before yielding the
first item -- wrapping that in a progress bar or any other iterable
adapter doesn't change this, since none of them alter how the
underlying queryset produces items. ``.iterator(chunk_size=...)`` is
the specific Django API that bypasses ``_result_cache`` and streams
from a server-side cursor instead, discarding each chunk once consumed
(and, since Django 4.1, still honours ``prefetch_related``, running the
prefetches one batch at a time rather than for the whole queryset).
Subclass to layer additional per-batch work on top (see
``documents.search._backend._DocumentViewerStream``) by overriding
``__iter__`` -- ``__len__`` and the constructor are inherited for free.
"""
def __init__(self, queryset: "QuerySet[_M]", *, chunk_size: int) -> None:
self._queryset = queryset
self._chunk_size = chunk_size
def __len__(self) -> int:
return self._queryset.count()
def __iter__(self) -> Iterator[_M]:
return iter(self._queryset.iterator(chunk_size=self._chunk_size))
def _coerce_to_path(
source: Path | str,
dest: Path | str,
+21 -6
View File
@@ -3,6 +3,7 @@ from __future__ import annotations
import importlib.resources
import logging
import os
import re
import shutil
import tempfile
from pathlib import Path
@@ -24,9 +25,9 @@ from paperless.config import OcrConfig
from paperless.models import CleanChoices
from paperless.models import ModeChoices
from paperless.models import OutputTypeChoices
from paperless.parsers.utils import PDF_TEXT_MIN_LENGTH
from paperless.parsers.utils import extract_pdf_text
from paperless.parsers.utils import is_born_digital_text
from paperless.parsers.utils import post_process_text
from paperless.parsers.utils import is_tagged_pdf
from paperless.parsers.utils import read_file_handle_unicode_errors
from paperless.version import __full_version_str__
@@ -509,10 +510,10 @@ class RasterisedDocumentParser:
if mime_type == "application/pdf":
text_original = self.extract_text(None, document_path)
original_has_text = is_born_digital_text(
text_original,
document_path,
log=self.log,
has_text = text_original is not None and len(text_original) > 0
original_has_text = has_text and (
is_tagged_pdf(document_path, log=self.log)
or len(text_original) > PDF_TEXT_MIN_LENGTH
)
else:
text_original = None
@@ -657,3 +658,17 @@ class RasterisedDocumentParser:
f"No text was found in {document_path}, the content will be empty.",
)
self.text = ""
def post_process_text(text: str | None) -> str | None:
if not text:
return None
collapsed_spaces = re.sub(r"([^\S\r\n]+)", " ", text)
no_leading_whitespace = re.sub(r"([\n\r]+)([^\S\n\r]+)", "\\1", collapsed_spaces)
no_trailing_whitespace = re.sub(r"([^\S\n\r]+)$", "", no_leading_whitespace)
# TODO: this needs a rework
# replace \0 prevents issues with saving to postgres.
# text may contain \0 when this character is present in PDF files.
return no_trailing_whitespace.strip().replace("\0", " ")
-82
View File
@@ -111,88 +111,6 @@ def extract_pdf_text(
return None
def post_process_text(text: str | None) -> str | None:
"""Normalize extracted PDF/OCR text: collapse whitespace, strip padding.
Returns ``None`` for ``None`` or whitespace-only input, so callers can
treat "no text" and "only layout padding" the same way.
"""
if not text:
return None
collapsed_spaces = re.sub(r"([^\S\r\n]+)", " ", text)
no_leading_whitespace = re.sub(r"([\n\r]+)([^\S\n\r]+)", "\\1", collapsed_spaces)
no_trailing_whitespace = re.sub(r"([^\S\n\r]+)$", "", no_leading_whitespace)
# replace \0 prevents issues with saving to postgres.
# text may contain \0 when this character is present in PDF files.
result = no_trailing_whitespace.strip().replace("\0", " ")
return result or None
def is_born_digital_text(
text: str | None,
path: Path,
log: logging.Logger | None = None,
) -> bool:
"""Decide whether already-extracted, normalized PDF text counts as born-digital.
This is the single source of truth for "does this PDF already have real
text", used both to decide whether to produce an archive file and to
decide whether OCR can be skipped. Both decisions must agree, or a
tagged-but-textless PDF can end up with no archive AND a forced OCR pass
(see GH #13387): raw ``pdftotext -layout`` output can be non-empty
(whitespace/form-feed padding) even when there is no real content, so
*text* must already be normalized via :func:`post_process_text`, not the
raw extraction.
Parameters
----------
text:
The normalized extracted text (or ``None``) to evaluate.
path:
Absolute path to the PDF file, used for the tagged-PDF check.
log:
Logger for warnings. Falls back to the module-level logger when omitted.
Returns
-------
bool
Whether the PDF counts as born-digital (has real text, and is either
tagged or exceeds ``PDF_TEXT_MIN_LENGTH``).
"""
if not text:
return False
return is_tagged_pdf(path, log=log) or len(text) > PDF_TEXT_MIN_LENGTH
def pdf_born_digital_text(
path: Path,
log: logging.Logger | None = None,
) -> tuple[str | None, bool]:
"""Extract a PDF's text and decide whether it should be treated as born-digital.
Convenience wrapper around :func:`is_born_digital_text` for callers that
don't already have the PDF's text extracted (e.g. the archive-generation
decision, which runs before any parser has touched the file).
Parameters
----------
path:
Absolute path to the PDF file.
log:
Logger for warnings. Falls back to the module-level logger when omitted.
Returns
-------
tuple[str | None, bool]
The normalized extracted text (or ``None``), and whether the PDF
counts as born-digital.
"""
text = post_process_text(extract_pdf_text(path, log=log))
return text, is_born_digital_text(text, path, log=log)
def read_file_handle_unicode_errors(
filepath: Path,
log: logging.Logger | None = None,
-17
View File
@@ -36,23 +36,6 @@ def samples_dir() -> Path:
return (Path(__file__).parent / "samples").resolve()
@pytest.fixture(scope="session")
def tagged_no_text_pdf_file(samples_dir: Path) -> Path:
"""Path to a tagged PDF whose only "text" is pdftotext layout padding.
Reproduces GH #13387: ``/MarkInfo /Marked true`` is set, but the only
extractable content is a form-feed byte, not real text. Lives here
rather than in parsers/conftest.py so both parser tests and
paperless/tests/test_parser_utils.py can use it.
Returns
-------
Path
Absolute path to ``tesseract/tagged-but-no-text.pdf``.
"""
return samples_dir / "tesseract" / "tagged-but-no-text.pdf"
@pytest.fixture(autouse=True)
def clean_registry() -> Generator[None, None, None]:
"""Reset the parser registry before and after every test.
@@ -21,7 +21,7 @@ from documents.parsers import run_convert
from paperless.models import ModeChoices
from paperless.parsers import ParserProtocol
from paperless.parsers.tesseract import RasterisedDocumentParser
from paperless.parsers.utils import is_tagged_pdf
from paperless.parsers.tesseract import post_process_text
if TYPE_CHECKING:
from pathlib import Path
@@ -151,6 +151,36 @@ class TestRasterisedDocumentParserLifecycle:
assert tempdir is not None and not tempdir.exists()
# ---------------------------------------------------------------------------
# post_process_text
# ---------------------------------------------------------------------------
class TestPostProcessText:
@pytest.mark.parametrize(
("source", "expected"),
[
pytest.param(
"simple string",
"simple string",
id="collapse-spaces",
),
pytest.param(
"simple newline\n testing string",
"simple newline\ntesting string",
id="preserve-newline",
),
pytest.param(
"utf-8 строка с пробелами в конце ", # noqa: RUF001
"utf-8 строка с пробелами в конце", # noqa: RUF001
id="utf8-trailing-spaces",
),
],
)
def test_post_process_text(self, source: str, expected: str) -> None:
assert post_process_text(source) == expected
# ---------------------------------------------------------------------------
# Page count
# ---------------------------------------------------------------------------
@@ -880,25 +910,25 @@ class TestSkipArchive:
self,
mocker: MockerFixture,
tesseract_parser: RasterisedDocumentParser,
tagged_no_text_pdf_file: Path,
tesseract_samples_dir: Path,
) -> None:
"""
GIVEN:
- A real PDF that reports itself as tagged (/MarkInfo /Marked
true) but whose only pdftotext output is layout padding (a
lone form-feed byte), not real text (see GitHub issue #13387,
originally reported against #13349's tagged-PDF handling)
- A PDF that reports itself as tagged (/MarkInfo /Marked true) but
has no actual extractable text (some scanner firmware produces
this see GitHub issue #13349)
- Mode: auto, produce_archive=False
WHEN:
- Document is parsed
THEN:
- The tag alone is not trusted as "has text"; OCRmyPDF still runs
"""
assert is_tagged_pdf(tagged_no_text_pdf_file) is True
tesseract_parser.settings.mode = ModeChoices.AUTO
mocker.patch("paperless.parsers.tesseract.is_tagged_pdf", return_value=True)
mocker.patch.object(tesseract_parser, "extract_text", return_value=None)
mock_ocr = mocker.patch("ocrmypdf.ocr")
tesseract_parser.parse(
tagged_no_text_pdf_file,
tesseract_samples_dir / "multi-page-images.pdf",
"application/pdf",
produce_archive=False,
)
-110
View File
@@ -4,18 +4,10 @@ from __future__ import annotations
import codecs
from pathlib import Path
from typing import TYPE_CHECKING
import pytest
from paperless.parsers.utils import is_tagged_pdf
from paperless.parsers.utils import pdf_born_digital_text
from paperless.parsers.utils import post_process_text
from paperless.parsers.utils import read_file_handle_unicode_errors
if TYPE_CHECKING:
from pytest_mock import MockerFixture
SAMPLES = Path(__file__).parent / "samples" / "tesseract"
@@ -68,105 +60,3 @@ class TestIsTaggedPdf:
bad = tmp_path / "bad.pdf"
bad.write_bytes(b"not a pdf")
assert is_tagged_pdf(bad) is False
class TestPostProcessText:
@pytest.mark.parametrize(
("source", "expected"),
[
pytest.param(
"simple string",
"simple string",
id="collapse-spaces",
),
pytest.param(
"simple newline\n testing string",
"simple newline\ntesting string",
id="preserve-newline",
),
pytest.param(
"utf-8 строка с пробелами в конце ", # noqa: RUF001
"utf-8 строка с пробелами в конце", # noqa: RUF001
id="utf8-trailing-spaces",
),
pytest.param(None, None, id="none-input"),
pytest.param("", None, id="empty-string"),
pytest.param(" \n\x0c \n ", None, id="whitespace-and-formfeed-only"),
],
)
def test_post_process_text(
self,
source: str | None,
expected: str | None,
) -> None:
assert post_process_text(source) == expected
class TestPdfBornDigitalText:
"""Regression coverage for GH #13387.
should_produce_archive() and RasterisedDocumentParser.parse() must agree
on whether a PDF has real text, so both go through this one function.
"""
@pytest.mark.parametrize(
("extracted", "tagged", "expected_text", "expected_born_digital"),
[
pytest.param("tiny", True, "tiny", True, id="tagged-with-real-text"),
pytest.param("tiny", False, "tiny", False, id="untagged-below-min-length"),
pytest.param(
"x" * 51,
False,
"x" * 51,
True,
id="untagged-above-min-length",
),
pytest.param(None, True, None, False, id="tagged-but-no-text"),
],
)
def test_born_digital_decision(
self,
mocker: MockerFixture,
tmp_path: Path,
extracted: str | None,
tagged: bool, # noqa: FBT001
expected_text: str | None,
expected_born_digital: bool, # noqa: FBT001
) -> None:
"""
GIVEN:
- A PDF whose pdftotext output and /MarkInfo tag status vary
WHEN:
- pdf_born_digital_text() is called
THEN:
- The normalized text and born-digital verdict match; the tag
alone never counts as "has text"
"""
mocker.patch(
"paperless.parsers.utils.extract_pdf_text",
return_value=extracted,
)
mocker.patch("paperless.parsers.utils.is_tagged_pdf", return_value=tagged)
text, born_digital = pdf_born_digital_text(tmp_path / "doc.pdf")
assert text == expected_text
assert born_digital is expected_born_digital
def test_tagged_but_textless_pdf_is_not_born_digital(
self,
tagged_no_text_pdf_file: Path,
) -> None:
"""
GIVEN:
- A real PDF that is tagged (/MarkInfo /Marked true) but whose
only "text" is layout padding (a stray form-feed byte)
WHEN:
- pdf_born_digital_text() is called with no mocking
THEN:
- The normalized text is None and the PDF is not treated as
born-digital. The raw, unnormalized pdftotext output is
non-empty for this file, which is exactly what caused the
archive decision to disagree with the OCR decision in #13387.
"""
text, born_digital = pdf_born_digital_text(tagged_no_text_pdf_file)
assert text is None
assert born_digital is False
+12 -2
View File
@@ -14,6 +14,7 @@ from filelock import Timeout
from documents.models import Document
from documents.models import PaperlessTask
from documents.utils import IterWrapper
from documents.utils import QuerySetStream
from documents.utils import identity
from paperless.config import AIConfig
from paperless_ai.db import db_connection_released
@@ -32,6 +33,11 @@ logger = logging.getLogger("paperless_ai.indexing")
RAG_NUM_OUTPUT = 512
RAG_CHUNK_OVERLAP = 200
# update_llm_index(): row count per .iterator() batch when streaming
# documents for a rebuild/update via QuerySetStream, matching
# _DocumentViewerStream's chunk size in documents/search/_backend.py.
_INDEX_STREAM_CHUNK_SIZE = 1000
def queue_llm_index_update_if_needed(*, rebuild: bool, reason: str) -> bool:
# NOTE: The check-then-enqueue sequence below is non-atomic (TOCTOU): two
@@ -385,7 +391,9 @@ def update_llm_index(
if rebuild or not store.table_exists():
logger.info("Rebuilding LLM index.")
store.drop_table()
for document in iter_wrapper(documents):
for document in iter_wrapper(
QuerySetStream(documents, chunk_size=_INDEX_STREAM_CHUNK_SIZE),
):
nodes = build_document_node(document, chunk_size=chunk_size)
_embed_nodes(nodes, embed_model)
store.add(nodes)
@@ -398,7 +406,9 @@ def update_llm_index(
)
existing = store.get_modified_times()
changed = 0
for document in iter_wrapper(scoped_documents):
for document in iter_wrapper(
QuerySetStream(scoped_documents, chunk_size=_INDEX_STREAM_CHUNK_SIZE),
):
doc_id = str(document.id)
if existing.get(doc_id) == document.modified.isoformat():
continue