Compare commits

..
27 changed files with 347 additions and 401 deletions
+5 -4
View File
@@ -948,10 +948,11 @@ for display in the web interface.
!!! note !!! note
The **remote OCR parser** (Azure AI) always produces a searchable The **remote OCR parser** (Azure AI) also honors this setting: when
PDF and stores it as the archive copy, regardless of this setting. no archive is requested (`never`, or `auto` with a born-digital PDF),
`ARCHIVE_FILE_GENERATION=never` has no effect when the remote the remote engine is skipped entirely and locally-extracted text is
parser handles a document. used instead, avoiding an unnecessary API call and a duplicate text
layer.
#### [`PAPERLESS_OCR_CLEAN=<mode>`](#PAPERLESS_OCR_CLEAN) {#PAPERLESS_OCR_CLEAN} #### [`PAPERLESS_OCR_CLEAN=<mode>`](#PAPERLESS_OCR_CLEAN) {#PAPERLESS_OCR_CLEAN}
+5 -4
View File
@@ -187,10 +187,11 @@ PAPERLESS_ARCHIVE_FILE_GENERATION=auto
### Remote OCR parser ### Remote OCR parser
If you use the **remote OCR parser** (Azure AI), note that it always produces a If you use the **remote OCR parser** (Azure AI), `ARCHIVE_FILE_GENERATION` is
searchable PDF and stores it as the archive copy. `ARCHIVE_FILE_GENERATION=never` honored the same way as for the local engine: when no archive is requested
has no effect for documents handled by the remote parser - the archive is produced (`never`, or `auto` with a born-digital PDF), the remote engine is skipped
unconditionally by the remote engine. entirely and locally-extracted text is used instead, avoiding an unnecessary
API call and a duplicate text layer.
## Search Index (Whoosh -> Tantivy) ## Search Index (Whoosh -> Tantivy)
+10 -10
View File
@@ -1849,7 +1849,7 @@
</context-group> </context-group>
<context-group purpose="location"> <context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context> <context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">159</context> <context context-type="linenumber">154</context>
</context-group> </context-group>
<context-group purpose="location"> <context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/filterable-dropdown/filterable-dropdown.component.html</context> <context context-type="sourcefile">src/app/components/common/filterable-dropdown/filterable-dropdown.component.html</context>
@@ -3972,11 +3972,11 @@
</context-group> </context-group>
<context-group purpose="location"> <context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context> <context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">101</context> <context context-type="linenumber">96</context>
</context-group> </context-group>
<context-group purpose="location"> <context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context> <context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">107</context> <context context-type="linenumber">102</context>
</context-group> </context-group>
</trans-unit> </trans-unit>
<trans-unit id="3800326155195149498" datatype="html"> <trans-unit id="3800326155195149498" datatype="html">
@@ -3987,11 +3987,11 @@
</context-group> </context-group>
<context-group purpose="location"> <context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context> <context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">102</context> <context context-type="linenumber">97</context>
</context-group> </context-group>
<context-group purpose="location"> <context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context> <context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">108</context> <context context-type="linenumber">103</context>
</context-group> </context-group>
</trans-unit> </trans-unit>
<trans-unit id="7551700625201096185" datatype="html"> <trans-unit id="7551700625201096185" datatype="html">
@@ -4002,14 +4002,14 @@
</context-group> </context-group>
<context-group purpose="location"> <context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context> <context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">124</context> <context context-type="linenumber">119</context>
</context-group> </context-group>
</trans-unit> </trans-unit>
<trans-unit id="3184700926171002527" datatype="html"> <trans-unit id="3184700926171002527" datatype="html">
<source>Any</source> <source>Any</source>
<context-group purpose="location"> <context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context> <context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">157</context> <context context-type="linenumber">152</context>
</context-group> </context-group>
<context-group purpose="location"> <context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/filterable-dropdown/filterable-dropdown.component.html</context> <context context-type="sourcefile">src/app/components/common/filterable-dropdown/filterable-dropdown.component.html</context>
@@ -4020,21 +4020,21 @@
<source>Not</source> <source>Not</source>
<context-group purpose="location"> <context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context> <context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">162</context> <context context-type="linenumber">157</context>
</context-group> </context-group>
</trans-unit> </trans-unit>
<trans-unit id="6548676277933116532" datatype="html"> <trans-unit id="6548676277933116532" datatype="html">
<source>Add query</source> <source>Add query</source>
<context-group purpose="location"> <context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context> <context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">181</context> <context context-type="linenumber">176</context>
</context-group> </context-group>
</trans-unit> </trans-unit>
<trans-unit id="5599577087865387184" datatype="html"> <trans-unit id="5599577087865387184" datatype="html">
<source>Add expression</source> <source>Add expression</source>
<context-group purpose="location"> <context-group purpose="location">
<context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context> <context context-type="sourcefile">src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component.html</context>
<context context-type="linenumber">184</context> <context context-type="linenumber">179</context>
</context-group> </context-group>
</trans-unit> </trans-unit>
<trans-unit id="6312759212949884929" datatype="html"> <trans-unit id="6312759212949884929" datatype="html">
@@ -182,7 +182,7 @@
container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim"> container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim">
<i-bs class="me-2" name="stack"></i-bs><span><ng-container i18n>Attributes</ng-container></span> <i-bs class="me-2" name="stack"></i-bs><span><ng-container i18n>Attributes</ng-container></span>
</a> </a>
@if (!slimSidebarEnabled && canSaveSettings) { @if (!slimSidebarEnabled) {
<button <button
type="button" type="button"
class="btn btn-link btn-sm text-muted p-0 me-3 attributes-expand-btn" class="btn btn-link btn-sm text-muted p-0 me-3 attributes-expand-btn"
@@ -68,11 +68,6 @@
></ng-select> ></ng-select>
} @else if (getCustomFieldByID(atom.field)?.data_type === CustomFieldDataType.DocumentLink) { } @else if (getCustomFieldByID(atom.field)?.data_type === CustomFieldDataType.DocumentLink) {
<pngx-input-document-link [(ngModel)]="atom.value" class="w-25 form-select doc-link-select p-0" placeholder="Search docs..." i18n-placeholder [minimal]="true"></pngx-input-document-link> <pngx-input-document-link [(ngModel)]="atom.value" class="w-25 form-select doc-link-select p-0" placeholder="Search docs..." i18n-placeholder [minimal]="true"></pngx-input-document-link>
} @else if (getCustomFieldByID(atom.field)?.data_type === CustomFieldDataType.Monetary) {
<input class="w-25 form-control rounded-end" type="text" inputmode="decimal"
[ngModel]="atom.value"
(ngModelChange)="setMonetaryValue(atom, $event)"
[disabled]="disabled">
} @else { } @else {
<input class="w-25 form-control rounded-end" type="text" [(ngModel)]="atom.value" [disabled]="disabled"> <input class="w-25 form-control rounded-end" type="text" [(ngModel)]="atom.value" [disabled]="disabled">
} }
@@ -1,6 +1,5 @@
import { provideHttpClient, withInterceptorsFromDi } from '@angular/common/http' import { provideHttpClient, withInterceptorsFromDi } from '@angular/common/http'
import { provideHttpClientTesting } from '@angular/common/http/testing' import { provideHttpClientTesting } from '@angular/common/http/testing'
import { LOCALE_ID } from '@angular/core'
import { ComponentFixture, TestBed } from '@angular/core/testing' import { ComponentFixture, TestBed } from '@angular/core/testing'
import { FormsModule, ReactiveFormsModule } from '@angular/forms' import { FormsModule, ReactiveFormsModule } from '@angular/forms'
import { NgbDropdownModule } from '@ng-bootstrap/ng-bootstrap' import { NgbDropdownModule } from '@ng-bootstrap/ng-bootstrap'
@@ -42,12 +41,6 @@ const customFields = [
], ],
}, },
}, },
{
id: 3,
name: 'Test Monetary Field',
data_type: CustomFieldDataType.Monetary,
extra_data: { default_currency: 'EUR' },
},
] ]
describe('CustomFieldsQueryDropdownComponent', () => { describe('CustomFieldsQueryDropdownComponent', () => {
@@ -68,7 +61,6 @@ describe('CustomFieldsQueryDropdownComponent', () => {
providers: [ providers: [
provideHttpClient(withInterceptorsFromDi()), provideHttpClient(withInterceptorsFromDi()),
provideHttpClientTesting(), provideHttpClientTesting(),
{ provide: LOCALE_ID, useValue: 'de' },
], ],
}).compileComponents() }).compileComponents()
@@ -158,22 +150,6 @@ describe('CustomFieldsQueryDropdownComponent', () => {
expect(options2).toEqual([]) expect(options2).toEqual([])
}) })
it('should normalize localized monetary comparison values', () => {
const atom = new CustomFieldQueryAtom([3, 'exact', null])
component.setMonetaryValue(atom, '1.234,56')
expect(atom.value).toEqual('1234.56')
})
it('should preserve API-formatted monetary comparison values', () => {
const atom = new CustomFieldQueryAtom([3, 'exact', null])
component.setMonetaryValue(atom, '1234.56')
expect(atom.value).toEqual('1234.56')
})
it('should remove an element from the selection model', () => { it('should remove an element from the selection model', () => {
const expression = new CustomFieldQueryExpression() const expression = new CustomFieldQueryExpression()
const atom = new CustomFieldQueryAtom() const atom = new CustomFieldQueryAtom()
@@ -1,14 +1,9 @@
import { import { NgTemplateOutlet } from '@angular/common'
getLocaleNumberSymbol,
NgTemplateOutlet,
NumberSymbol,
} from '@angular/common'
import { import {
Component, Component,
EventEmitter, EventEmitter,
inject, inject,
Input, Input,
LOCALE_ID,
Output, Output,
QueryList, QueryList,
signal, signal,
@@ -217,7 +212,6 @@ export class CustomFieldQueriesModel {
}) })
export class CustomFieldsQueryDropdownComponent extends LoadingComponentWithPermissions { export class CustomFieldsQueryDropdownComponent extends LoadingComponentWithPermissions {
protected customFieldsService = inject(CustomFieldsService) protected customFieldsService = inject(CustomFieldsService)
private readonly locale = inject(LOCALE_ID)
public CustomFieldQueryComponentType = CustomFieldQueryElementType public CustomFieldQueryComponentType = CustomFieldQueryElementType
public CustomFieldQueryOperator = CustomFieldQueryOperator public CustomFieldQueryOperator = CustomFieldQueryOperator
@@ -382,18 +376,4 @@ export class CustomFieldsQueryDropdownComponent extends LoadingComponentWithPerm
} }
return [] return []
} }
setMonetaryValue(atom: CustomFieldQueryAtom, value: string) {
// Normalize the decimal symbol e.g. . vs , by locale
const decimalSymbol = getLocaleNumberSymbol(
this.locale,
NumberSymbol.Decimal
)
if (decimalSymbol !== '.' && value.includes(decimalSymbol)) {
const groupSymbol = getLocaleNumberSymbol(this.locale, NumberSymbol.Group)
value = value.split(groupSymbol).join('').split(decimalSymbol).join('.')
}
atom.value = value
}
} }
@@ -129,25 +129,13 @@ describe('PngxPdfViewerComponent', () => {
;(component as any).applyScale() ;(component as any).applyScale()
expect(viewer.currentScaleValue).toBe(PdfZoomScale.PageFit) expect(viewer.currentScaleValue).toBe(PdfZoomScale.PageFit)
expect(viewer.currentScale).toBe(2) expect(viewer.currentScale).toBe(2)
})
it('does not reapply scale for page-only changes', async () => {
await initComponent()
const pdf = (component as any).pdf as { numPages: number }
pdf.numPages = 3
const viewer = (component as any).pdfViewer as PDFViewer
viewer.setDocument(pdf)
const applyScaleSpy = jest.spyOn(component as any, 'applyScale') const applyScaleSpy = jest.spyOn(component as any, 'applyScale')
component.page = 2 component.page = 2
;(component as any).lastViewerPage = 2
component.ngOnChanges({ ;(component as any).applyViewerState()
page: new SimpleChange(1, 2, false),
})
expect(viewer.currentPageNumber).toBe(2)
expect((component as any).lastViewerPage).toBeUndefined() expect((component as any).lastViewerPage).toBeUndefined()
expect(applyScaleSpy).not.toHaveBeenCalled() expect(applyScaleSpy).toHaveBeenCalled()
}) })
it('does not reset the viewer when it is already on the requested page', async () => { it('does not reset the viewer when it is already on the requested page', async () => {
@@ -116,10 +116,7 @@ export class PngxPdfViewerComponent
changes['zoomScale'] || changes['zoomScale'] ||
changes['rotation'] changes['rotation']
) { ) {
// Prevent loop with page / scale application see https://github.com/paperless-ngx/paperless-ngx/issues/13404 this.applyViewerState()
this.applyViewerState(
!!(changes['zoom'] || changes['zoomScale'] || changes['rotation'])
)
} }
if (changes['searchQuery']) { if (changes['searchQuery']) {
@@ -243,7 +240,7 @@ export class PngxPdfViewerComponent
} }
} }
private applyViewerState(applyScale = true): void { private applyViewerState(): void {
if (!this.pdfViewer) { if (!this.pdfViewer) {
return return
} }
@@ -267,7 +264,7 @@ export class PngxPdfViewerComponent
if (this.page === this.lastViewerPage) { if (this.page === this.lastViewerPage) {
this.lastViewerPage = undefined this.lastViewerPage = undefined
} }
if (hasPages && applyScale) { if (hasPages) {
this.applyScale() this.applyScale()
} }
this.dispatchFindIfReady() this.dispatchFindIfReady()
@@ -51,8 +51,8 @@
*pngxIfPermissions="{ action: PermissionAction.Add, type: activeManagementList.permissionType }"> *pngxIfPermissions="{ action: PermissionAction.Add, type: activeManagementList.permissionType }">
<i-bs name="plus-circle" class="me-1"></i-bs><ng-container i18n>Create</ng-container> <i-bs name="plus-circle" class="me-1"></i-bs><ng-container i18n>Create</ng-container>
</button> </button>
} @else if (customFieldsActive) { } @else if (activeCustomFields) {
<button type="button" class="btn btn-sm btn-outline-primary" (click)="addCustomField()" <button type="button" class="btn btn-sm btn-outline-primary" (click)="activeCustomFields.editField()"
*pngxIfPermissions="{ action: PermissionAction.Add, type: PermissionType.CustomField }"> *pngxIfPermissions="{ action: PermissionAction.Add, type: PermissionType.CustomField }">
<i-bs name="plus-circle" class="me-1"></i-bs><ng-container i18n>Add Field</ng-container> <i-bs name="plus-circle" class="me-1"></i-bs><ng-container i18n>Add Field</ng-container>
</button> </button>
@@ -18,7 +18,6 @@ import {
DocumentAttributesComponent, DocumentAttributesComponent,
DocumentAttributesSectionKind, DocumentAttributesSectionKind,
} from './document-attributes.component' } from './document-attributes.component'
import { CustomFieldsComponent } from './custom-fields/custom-fields.component'
import { ManagementListComponent } from './management-list/management-list.component' import { ManagementListComponent } from './management-list/management-list.component'
@Component({ @Component({
@@ -208,29 +207,6 @@ describe('DocumentAttributesComponent', () => {
expect(component.activeSection.kind).toBe( expect(component.activeSection.kind).toBe(
DocumentAttributesSectionKind.CustomFields DocumentAttributesSectionKind.CustomFields
) )
const customFields = Object.create(CustomFieldsComponent.prototype) expect(component.activeCustomFields).toBeDefined()
customFields.editField = jest.fn()
component.activeOutlet = {
componentInstance: customFields,
} as any
expect(component.activeCustomFields).toBe(customFields)
component.addCustomField()
expect(customFields.editField).toHaveBeenCalled()
})
it('should show the add field button before the custom fields instance is available', async () => {
jest.spyOn(permissionsService, 'currentUserCan').mockReturnValue(true)
fixture.detectChanges()
component.activeNavID.set(2)
await fixture.whenStable()
expect(component.activeCustomFields).toBeNull()
expect(
fixture.nativeElement.querySelector(
'pngx-page-header .btn-outline-primary'
)?.textContent
).toContain('Add Field')
}) })
}) })
@@ -163,17 +163,12 @@ export class DocumentAttributesComponent implements OnInit, OnDestroy {
} }
get activeCustomFields(): CustomFieldsComponent | null { get activeCustomFields(): CustomFieldsComponent | null {
if (!this.customFieldsActive) return null if (this.activeSection?.kind !== DocumentAttributesSectionKind.CustomFields)
return null
const instance = this.activeOutlet?.componentInstance const instance = this.activeOutlet?.componentInstance
return instance instanceof CustomFieldsComponent ? instance : null return instance instanceof CustomFieldsComponent ? instance : null
} }
get customFieldsActive(): boolean {
return (
this.activeSection?.kind === DocumentAttributesSectionKind.CustomFields
)
}
get activeTabLabel(): string { get activeTabLabel(): string {
return this.activeSection?.label ?? '' return this.activeSection?.label ?? ''
} }
@@ -229,10 +224,6 @@ export class DocumentAttributesComponent implements OnInit, OnDestroy {
this.router.navigate(['attributes', nextSection]) this.router.navigate(['attributes', nextSection])
} }
addCustomField(): void {
this.activeCustomFields?.editField(null)
}
private getDefaultNavID(): DocumentAttributesNavIDs | null { private getDefaultNavID(): DocumentAttributesNavIDs | null {
return this.visibleSections[0]?.id ?? null return this.visibleSections[0]?.id ?? null
} }
+25 -15
View File
@@ -57,7 +57,9 @@ from paperless.models import ArchiveFileGenerationChoices
from paperless.parsers import ParserContext from paperless.parsers import ParserContext
from paperless.parsers import ParserProtocol from paperless.parsers import ParserProtocol
from paperless.parsers.registry import get_parser_registry from paperless.parsers.registry import get_parser_registry
from paperless.parsers.utils import pdf_born_digital_text from paperless.parsers.utils import PDF_TEXT_MIN_LENGTH
from paperless.parsers.utils import extract_pdf_text
from paperless.parsers.utils import is_tagged_pdf
LOGGING_NAME: Final[str] = "paperless.consumer" LOGGING_NAME: Final[str] = "paperless.consumer"
@@ -136,45 +138,53 @@ def should_produce_archive(
# Must produce a PDF so the frontend can display the original format at all. # Must produce a PDF so the frontend can display the original format at all.
if parser.requires_pdf_rendition: if parser.requires_pdf_rendition:
_log.debug("Archive: yes - parser requires PDF rendition for frontend display") _log.debug("Archive: yes parser requires PDF rendition for frontend display")
return True return True
# Parser cannot produce an archive (e.g. TextDocumentParser). # Parser cannot produce an archive (e.g. TextDocumentParser).
if not parser.can_produce_archive: if not parser.can_produce_archive:
_log.debug("Archive: no - parser cannot produce archives") _log.debug("Archive: no parser cannot produce archives")
return False return False
generation = OcrConfig().archive_file_generation generation = OcrConfig().archive_file_generation
if generation == ArchiveFileGenerationChoices.ALWAYS: if generation == ArchiveFileGenerationChoices.ALWAYS:
_log.debug("Archive: yes - ARCHIVE_FILE_GENERATION=always") _log.debug("Archive: yes ARCHIVE_FILE_GENERATION=always")
return True return True
if generation == ArchiveFileGenerationChoices.NEVER: if generation == ArchiveFileGenerationChoices.NEVER:
_log.debug("Archive: no - ARCHIVE_FILE_GENERATION=never") _log.debug("Archive: no ARCHIVE_FILE_GENERATION=never")
return False return False
# auto: produce archives for scanned/image documents; skip for born-digital PDFs. # auto: produce archives for scanned/image documents; skip for born-digital PDFs.
if mime_type.startswith("image/"): if mime_type.startswith("image/"):
_log.debug("Archive: yes - image document, ARCHIVE_FILE_GENERATION=auto") _log.debug("Archive: yes image document, ARCHIVE_FILE_GENERATION=auto")
return True return True
if mime_type == "application/pdf": if mime_type == "application/pdf":
text, born_digital = pdf_born_digital_text(document_path, log=_log) text = extract_pdf_text(document_path)
text_length = len(text) if text else 0 has_text = text is not None and len(text) > 0
if born_digital: if has_text and is_tagged_pdf(document_path):
_log.debug( _log.debug(
"Archive: no - born-digital PDF (text_length=%d)," "Archive: no born-digital PDF (structure tags detected),"
" ARCHIVE_FILE_GENERATION=auto", " ARCHIVE_FILE_GENERATION=auto",
text_length,
) )
return False return False
if text is None or len(text) <= PDF_TEXT_MIN_LENGTH:
_log.debug(
"Archive: yes — scanned PDF (text_length=%d%d),"
" ARCHIVE_FILE_GENERATION=auto",
len(text) if text else 0,
PDF_TEXT_MIN_LENGTH,
)
return True
_log.debug( _log.debug(
"Archive: yes - scanned/textless PDF (text_length=%d)," "Archive: no — born-digital PDF (text_length=%d > %d),"
" ARCHIVE_FILE_GENERATION=auto", " ARCHIVE_FILE_GENERATION=auto",
text_length, len(text),
PDF_TEXT_MIN_LENGTH,
) )
return True return False
_log.debug( _log.debug(
"Archive: no - MIME type %r not eligible for auto archive generation", "Archive: no MIME type %r not eligible for auto archive generation",
mime_type, mime_type,
) )
return False return False
-3
View File
@@ -36,9 +36,6 @@ def send_email(
TODO: re-evaluate this pending https://code.djangoproject.com/ticket/35581 / https://github.com/django/django/pull/18966 TODO: re-evaluate this pending https://code.djangoproject.com/ticket/35581 / https://github.com/django/django/pull/18966
""" """
if "\r" in subject or "\n" in subject:
subject = " ".join(line.strip(" \t") for line in subject.splitlines())
email = EmailMessage( email = EmailMessage(
subject=subject, subject=subject,
body=body, body=body,
@@ -386,19 +386,10 @@ class Command(CryptMixin, PaperlessCommand):
raise DeserializationError( raise DeserializationError(
f"{model.__name__} has no updatable fields; PK-only models are not supported by the importer", f"{model.__name__} has no updatable fields; PK-only models are not supported by the importer",
) )
# MySQL/MariaDB support upserts via ON DUPLICATE KEY UPDATE but,
# unlike PostgreSQL/SQLite, cannot target a specific unique field
# for the conflict -- passing unique_fields there raises
# NotSupportedError.
unique_fields = (
[model._meta.pk.attname]
if connection.features.supports_update_conflicts_with_target
else None
)
model.objects.bulk_create( # type: ignore[attr-defined] model.objects.bulk_create( # type: ignore[attr-defined]
instances, instances,
update_conflicts=True, update_conflicts=True,
unique_fields=unique_fields, unique_fields=[model._meta.pk.attname],
update_fields=update_fields, update_fields=update_fields,
) )
loaded_models.add(model) loaded_models.add(model)
@@ -61,7 +61,7 @@ class Command(PaperlessCommand):
) )
table.add_column("Level", width=7, no_wrap=True) table.add_column("Level", width=7, no_wrap=True)
table.add_column("Document", min_width=20) table.add_column("Document", min_width=20)
table.add_column("Issue", ratio=1, overflow="fold") table.add_column("Issue", ratio=1)
for doc_pk, doc_messages in messages.iter_messages(): for doc_pk, doc_messages in messages.iter_messages():
if doc_pk is not None: if doc_pk is not None:
+1 -1
View File
@@ -75,7 +75,7 @@ class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
{ {
"documents": [self.doc1.pk, self.doc2.pk], "documents": [self.doc1.pk, self.doc2.pk],
"addresses": "hello@paperless-ngx.com,test@example.com", "addresses": "hello@paperless-ngx.com,test@example.com",
"subject": "Bulk email\n test", "subject": "Bulk email test",
"message": "Here are your documents", "message": "Here are your documents",
}, },
), ),
+2 -2
View File
@@ -1329,7 +1329,7 @@ class PreConsumeTestCase(DirectoriesMixin, GetConsumerMixin, TestCase):
with self.get_consumer(self.test_file) as c: with self.get_consumer(self.test_file) as c:
c.run() c.run()
# Verify no pre-consume script subprocess was invoked # Verify no pre-consume script subprocess was invoked
# (run_subprocess may still be called by pdf_born_digital_text via pdftotext) # (run_subprocess may still be called by _extract_text_for_archive_check)
script_calls = [ script_calls = [
call call
for call in m.call_args_list for call in m.call_args_list
@@ -1354,7 +1354,7 @@ class PreConsumeTestCase(DirectoriesMixin, GetConsumerMixin, TestCase):
self.assertTrue(m.called) self.assertTrue(m.called)
# Find the call that invoked the pre-consume script # Find the call that invoked the pre-consume script
# (run_subprocess may also be called by pdf_born_digital_text via pdftotext) # (run_subprocess may also be called by _extract_text_for_archive_check)
script_call = next( script_call = next(
call call
for call in m.call_args_list for call in m.call_args_list
+42 -14
View File
@@ -134,32 +134,60 @@ class TestShouldProduceArchive:
assert should_produce_archive(parser, mime, Path("/tmp/doc")) is expected assert should_produce_archive(parser, mime, Path("/tmp/doc")) is expected
@pytest.mark.parametrize( @pytest.mark.parametrize(
("born_digital", "expected"), ("extracted_text", "expected"),
[ [
pytest.param(True, False, id="born-digital-skips-archive"), pytest.param(
pytest.param(False, True, id="not-born-digital-produces-archive"), "This is a born-digital PDF with lots of text content. " * 10,
False,
id="born-digital-long-text-skips-archive",
),
pytest.param(None, True, id="no-text-scanned-produces-archive"),
pytest.param("tiny", True, id="short-text-treated-as-scanned"),
], ],
) )
def test_auto_pdf_archive_decision( def test_auto_pdf_archive_decision(
self, self,
mocker: MockerFixture, mocker: MockerFixture,
settings, settings,
born_digital: bool, # noqa: FBT001 extracted_text: str | None,
expected: bool, # noqa: FBT001 expected: bool, # noqa: FBT001
) -> None: ) -> None:
"""Archive decision tracks pdf_born_digital_text()'s verdict exactly.
should_produce_archive() defers entirely to pdf_born_digital_text()
for the has-real-text decision, so both callers of that predicate
(this function and RasterisedDocumentParser.parse()) always agree.
"""
settings.ARCHIVE_FILE_GENERATION = "auto" settings.ARCHIVE_FILE_GENERATION = "auto"
mocker.patch( mocker.patch("documents.consumer.is_tagged_pdf", return_value=False)
"documents.consumer.pdf_born_digital_text", mocker.patch("documents.consumer.extract_pdf_text", return_value=extracted_text)
return_value=("some text", born_digital),
)
parser = _parser_instance(can_produce=True, requires_rendition=False) parser = _parser_instance(can_produce=True, requires_rendition=False)
assert ( assert (
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf")) should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
is expected is expected
) )
def test_tagged_pdf_skips_archive_in_auto_mode(
self,
mocker: MockerFixture,
settings,
) -> None:
"""Tagged PDFs (e.g. Word exports) with real text are treated as born-digital, even below PDF_TEXT_MIN_LENGTH."""
settings.ARCHIVE_FILE_GENERATION = "auto"
mocker.patch("documents.consumer.is_tagged_pdf", return_value=True)
mocker.patch("documents.consumer.extract_pdf_text", return_value="tiny")
parser = _parser_instance(can_produce=True, requires_rendition=False)
assert (
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
is False
)
def test_tagged_pdf_without_text_produces_archive(
self,
mocker: MockerFixture,
settings,
) -> None:
"""A tagged PDF with no actual extractable text (e.g. some scanner firmware) is not
trusted as born-digital — the tag alone must not bypass OCR."""
settings.ARCHIVE_FILE_GENERATION = "auto"
mocker.patch("documents.consumer.is_tagged_pdf", return_value=True)
mocker.patch("documents.consumer.extract_pdf_text", return_value=None)
parser = _parser_instance(can_produce=True, requires_rendition=False)
assert (
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
is True
)
+33 -7
View File
@@ -3,7 +3,9 @@ Built-in remote-OCR document parser.
Handles documents by sending them to a configured remote OCR engine Handles documents by sending them to a configured remote OCR engine
(currently Azure AI Vision / Document Intelligence) and retrieving both (currently Azure AI Vision / Document Intelligence) and retrieving both
the extracted text and a searchable PDF with an embedded text layer. the extracted text and a searchable PDF with an embedded text layer. For
born-digital PDFs that need no archive copy, the remote call is skipped
entirely in favor of locally-extracted text (see ``RemoteDocumentParser.parse``).
When no engine is configured, ``score()`` returns ``None`` so the parser When no engine is configured, ``score()`` returns ``None`` so the parser
is effectively invisible to the registry — the tesseract parser handles is effectively invisible to the registry — the tesseract parser handles
@@ -21,6 +23,8 @@ from typing import Self
from django.conf import settings from django.conf import settings
from paperless.parsers.utils import extract_pdf_text
from paperless.parsers.utils import post_process_text
from paperless.version import __full_version_str__ from paperless.version import __full_version_str__
if TYPE_CHECKING: if TYPE_CHECKING:
@@ -69,8 +73,11 @@ class RemoteDocumentParser:
"""Parse documents via a remote OCR API (currently Azure AI Vision). """Parse documents via a remote OCR API (currently Azure AI Vision).
This parser sends documents to a remote engine that returns both This parser sends documents to a remote engine that returns both
extracted text and a searchable PDF with an embedded text layer. extracted text and a searchable PDF with an embedded text layer,
It does not depend on Tesseract or ocrmypdf. except when ``parse()`` is called with ``produce_archive=False`` for
a PDF, in which case the remote call is skipped and only locally
extracted text is returned (no archive). It does not depend on
Tesseract or ocrmypdf.
Class attributes Class attributes
---------------- ----------------
@@ -159,8 +166,11 @@ class RemoteDocumentParser:
Returns Returns
------- -------
bool bool
Always True — the remote engine always returns a PDF with an Always True — the remote engine is capable of returning a PDF
embedded text layer that serves as the archive copy. with an embedded text layer to serve as the archive copy.
Whether it actually does so for a given document depends on
``produce_archive`` passed to :meth:`parse` (see there for when
the remote engine call, and thus archive generation, is skipped).
""" """
return True return True
@@ -217,6 +227,12 @@ class RemoteDocumentParser:
) -> None: ) -> None:
"""Send the document to the remote engine and store results. """Send the document to the remote engine and store results.
When *produce_archive* is False for a PDF, the caller (via
``documents.consumer.should_produce_archive``) has already determined
that the document is born-digital and needs no archive — skip the
remote engine entirely rather than re-OCRing it and creating a
duplicate text layer.
Parameters Parameters
---------- ----------
document_path: document_path:
@@ -224,8 +240,8 @@ class RemoteDocumentParser:
mime_type: mime_type:
Detected MIME type of the document. Detected MIME type of the document.
produce_archive: produce_archive:
Ignored — the remote engine always returns a searchable PDF, Whether an archive copy is wanted. For PDFs, False skips the
which is stored as the archive copy regardless of this flag. remote engine and uses locally-extracted text instead.
""" """
config = RemoteEngineConfig( config = RemoteEngineConfig(
engine=settings.REMOTE_OCR_ENGINE, engine=settings.REMOTE_OCR_ENGINE,
@@ -240,6 +256,16 @@ class RemoteDocumentParser:
self._text = "" self._text = ""
return return
if not produce_archive and mime_type == "application/pdf":
logger.debug(
"Remote OCR: skipped — no archive requested, "
"using locally-extracted text",
)
self._text = (
post_process_text(extract_pdf_text(document_path, log=logger)) or ""
)
return
if config.engine == "azureai": if config.engine == "azureai":
self._text = self._azure_ai_vision_parse(document_path, config) self._text = self._azure_ai_vision_parse(document_path, config)
+3 -3
View File
@@ -25,7 +25,7 @@ from paperless.models import CleanChoices
from paperless.models import ModeChoices from paperless.models import ModeChoices
from paperless.models import OutputTypeChoices from paperless.models import OutputTypeChoices
from paperless.parsers.utils import extract_pdf_text from paperless.parsers.utils import extract_pdf_text
from paperless.parsers.utils import is_born_digital_text from paperless.parsers.utils import pdf_has_digital_text
from paperless.parsers.utils import post_process_text from paperless.parsers.utils import post_process_text
from paperless.parsers.utils import read_file_handle_unicode_errors from paperless.parsers.utils import read_file_handle_unicode_errors
from paperless.version import __full_version_str__ from paperless.version import __full_version_str__
@@ -509,9 +509,9 @@ class RasterisedDocumentParser:
if mime_type == "application/pdf": if mime_type == "application/pdf":
text_original = self.extract_text(None, document_path) text_original = self.extract_text(None, document_path)
original_has_text = is_born_digital_text( original_has_text = pdf_has_digital_text(
text_original,
document_path, document_path,
text_original,
log=self.log, log=self.log,
) )
else: else:
+57 -82
View File
@@ -65,6 +65,63 @@ def is_tagged_pdf(
return False return False
def pdf_has_digital_text(
path: Path,
text: str | None,
log: logging.Logger | None = None,
) -> bool:
"""Return True if a PDF already has a usable, born-digital text layer.
Combines the tagged-PDF check with an extracted-text length check.
Shared by the tesseract and remote OCR parsers to decide whether
OCR_MODE=auto/off should skip (re-)OCRing a document.
Parameters
----------
path:
Absolute path to the PDF file.
text:
Text already extracted from the PDF (e.g. via ``extract_pdf_text``),
or ``None``.
log:
Logger for warnings. Falls back to the module-level logger when omitted.
Returns
-------
bool
``True`` when the document already contains a text layer.
"""
return is_tagged_pdf(path, log=log) or (
text is not None and len(text) > PDF_TEXT_MIN_LENGTH
)
def post_process_text(text: str | None) -> str | None:
"""Normalise whitespace in extracted OCR/PDF text and strip NUL bytes.
Parameters
----------
text:
Raw extracted text, or ``None``.
Returns
-------
str | None
Cleaned text, or ``None`` when *text* is falsy.
"""
if not text:
return None
collapsed_spaces = re.sub(r"([^\S\r\n]+)", " ", text)
no_leading_whitespace = re.sub(r"([\n\r]+)([^\S\n\r]+)", "\\1", collapsed_spaces)
no_trailing_whitespace = re.sub(r"([^\S\n\r]+)$", "", no_leading_whitespace)
# TODO: this needs a rework
# replace \0 prevents issues with saving to postgres.
# text may contain \0 when this character is present in PDF files.
return no_trailing_whitespace.strip().replace("\0", " ")
def extract_pdf_text( def extract_pdf_text(
path: Path, path: Path,
log: logging.Logger | None = None, log: logging.Logger | None = None,
@@ -111,88 +168,6 @@ def extract_pdf_text(
return None return None
def post_process_text(text: str | None) -> str | None:
"""Normalize extracted PDF/OCR text: collapse whitespace, strip padding.
Returns ``None`` for ``None`` or whitespace-only input, so callers can
treat "no text" and "only layout padding" the same way.
"""
if not text:
return None
collapsed_spaces = re.sub(r"([^\S\r\n]+)", " ", text)
no_leading_whitespace = re.sub(r"([\n\r]+)([^\S\n\r]+)", "\\1", collapsed_spaces)
no_trailing_whitespace = re.sub(r"([^\S\n\r]+)$", "", no_leading_whitespace)
# replace \0 prevents issues with saving to postgres.
# text may contain \0 when this character is present in PDF files.
result = no_trailing_whitespace.strip().replace("\0", " ")
return result or None
def is_born_digital_text(
text: str | None,
path: Path,
log: logging.Logger | None = None,
) -> bool:
"""Decide whether already-extracted, normalized PDF text counts as born-digital.
This is the single source of truth for "does this PDF already have real
text", used both to decide whether to produce an archive file and to
decide whether OCR can be skipped. Both decisions must agree, or a
tagged-but-textless PDF can end up with no archive AND a forced OCR pass
(see GH #13387): raw ``pdftotext -layout`` output can be non-empty
(whitespace/form-feed padding) even when there is no real content, so
*text* must already be normalized via :func:`post_process_text`, not the
raw extraction.
Parameters
----------
text:
The normalized extracted text (or ``None``) to evaluate.
path:
Absolute path to the PDF file, used for the tagged-PDF check.
log:
Logger for warnings. Falls back to the module-level logger when omitted.
Returns
-------
bool
Whether the PDF counts as born-digital (has real text, and is either
tagged or exceeds ``PDF_TEXT_MIN_LENGTH``).
"""
if not text:
return False
return is_tagged_pdf(path, log=log) or len(text) > PDF_TEXT_MIN_LENGTH
def pdf_born_digital_text(
path: Path,
log: logging.Logger | None = None,
) -> tuple[str | None, bool]:
"""Extract a PDF's text and decide whether it should be treated as born-digital.
Convenience wrapper around :func:`is_born_digital_text` for callers that
don't already have the PDF's text extracted (e.g. the archive-generation
decision, which runs before any parser has touched the file).
Parameters
----------
path:
Absolute path to the PDF file.
log:
Logger for warnings. Falls back to the module-level logger when omitted.
Returns
-------
tuple[str | None, bool]
The normalized extracted text (or ``None``), and whether the PDF
counts as born-digital.
"""
text = post_process_text(extract_pdf_text(path, log=log))
return text, is_born_digital_text(text, path, log=log)
def read_file_handle_unicode_errors( def read_file_handle_unicode_errors(
filepath: Path, filepath: Path,
log: logging.Logger | None = None, log: logging.Logger | None = None,
-17
View File
@@ -36,23 +36,6 @@ def samples_dir() -> Path:
return (Path(__file__).parent / "samples").resolve() return (Path(__file__).parent / "samples").resolve()
@pytest.fixture(scope="session")
def tagged_no_text_pdf_file(samples_dir: Path) -> Path:
"""Path to a tagged PDF whose only "text" is pdftotext layout padding.
Reproduces GH #13387: ``/MarkInfo /Marked true`` is set, but the only
extractable content is a form-feed byte, not real text. Lives here
rather than in parsers/conftest.py so both parser tests and
paperless/tests/test_parser_utils.py can use it.
Returns
-------
Path
Absolute path to ``tesseract/tagged-but-no-text.pdf``.
"""
return samples_dir / "tesseract" / "tagged-but-no-text.pdf"
@pytest.fixture(autouse=True) @pytest.fixture(autouse=True)
def clean_registry() -> Generator[None, None, None]: def clean_registry() -> Generator[None, None, None]:
"""Reset the parser registry before and after every test. """Reset the parser registry before and after every test.
@@ -336,6 +336,117 @@ class TestRemoteParserParse:
assert remote_parser.get_date() is None assert remote_parser.get_date() is None
# ---------------------------------------------------------------------------
# parse() — produce_archive=False skips the remote engine (PDFs only)
# ---------------------------------------------------------------------------
class TestRemoteParserSkipsWhenNoArchiveWanted:
"""When the caller has already decided no archive is needed for a PDF
(documents.consumer.should_produce_archive), the remote engine call is
skipped entirely in favor of locally-extracted text.
"""
def test_pdf_skips_azure_when_no_archive_requested(
self,
remote_parser: RemoteDocumentParser,
simple_digital_pdf_file: Path,
azure_client: Mock,
) -> None:
"""
GIVEN: produce_archive=False for a PDF
WHEN: parse() is called
THEN: Azure is never invoked, no archive is produced, and text
comes from local pdftotext extraction
"""
remote_parser.parse(
simple_digital_pdf_file,
"application/pdf",
produce_archive=False,
)
azure_client.begin_analyze_document.assert_not_called()
assert remote_parser.get_archive_path() is None
assert remote_parser.get_text() != ""
def test_pdf_no_archive_requested_text_matches_local_extraction(
self,
remote_parser: RemoteDocumentParser,
simple_digital_pdf_file: Path,
azure_client: Mock,
mocker: MockerFixture,
) -> None:
"""
GIVEN: produce_archive=False for a PDF
WHEN: parse() is called
THEN: the returned text is exactly the locally-extracted text,
not anything from the (unused) Azure mock
"""
mocker.patch(
"paperless.parsers.remote.extract_pdf_text",
return_value="Local digital text.",
)
remote_parser.parse(
simple_digital_pdf_file,
"application/pdf",
produce_archive=False,
)
assert remote_parser.get_text() == "Local digital text."
def test_pdf_no_archive_requested_closes_no_client(
self,
remote_parser: RemoteDocumentParser,
simple_digital_pdf_file: Path,
azure_client: Mock,
) -> None:
remote_parser.parse(
simple_digital_pdf_file,
"application/pdf",
produce_archive=False,
)
azure_client.close.assert_not_called()
def test_non_pdf_still_calls_azure_when_no_archive_requested(
self,
remote_parser: RemoteDocumentParser,
simple_digital_pdf_file: Path,
azure_client: Mock,
) -> None:
"""
Images have no local-text fallback, so produce_archive=False does
not skip the remote engine for non-PDF MIME types.
"""
remote_parser.parse(
simple_digital_pdf_file,
"image/png",
produce_archive=False,
)
azure_client.begin_analyze_document.assert_called_once()
assert remote_parser.get_text() == _DEFAULT_TEXT
@pytest.mark.usefixtures("no_engine_settings")
def test_unconfigured_engine_takes_precedence_over_skip(
self,
remote_parser: RemoteDocumentParser,
simple_digital_pdf_file: Path,
) -> None:
"""An unconfigured engine still short-circuits before the
produce_archive check, returning empty text as before.
"""
remote_parser.parse(
simple_digital_pdf_file,
"application/pdf",
produce_archive=False,
)
assert remote_parser.get_text() == ""
assert remote_parser.get_archive_path() is None
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# parse() — Azure failure path # parse() — Azure failure path
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@@ -21,7 +21,7 @@ from documents.parsers import run_convert
from paperless.models import ModeChoices from paperless.models import ModeChoices
from paperless.parsers import ParserProtocol from paperless.parsers import ParserProtocol
from paperless.parsers.tesseract import RasterisedDocumentParser from paperless.parsers.tesseract import RasterisedDocumentParser
from paperless.parsers.utils import is_tagged_pdf from paperless.parsers.utils import post_process_text
if TYPE_CHECKING: if TYPE_CHECKING:
from pathlib import Path from pathlib import Path
@@ -151,6 +151,36 @@ class TestRasterisedDocumentParserLifecycle:
assert tempdir is not None and not tempdir.exists() assert tempdir is not None and not tempdir.exists()
# ---------------------------------------------------------------------------
# post_process_text
# ---------------------------------------------------------------------------
class TestPostProcessText:
@pytest.mark.parametrize(
("source", "expected"),
[
pytest.param(
"simple string",
"simple string",
id="collapse-spaces",
),
pytest.param(
"simple newline\n testing string",
"simple newline\ntesting string",
id="preserve-newline",
),
pytest.param(
"utf-8 строка с пробелами в конце ", # noqa: RUF001
"utf-8 строка с пробелами в конце", # noqa: RUF001
id="utf8-trailing-spaces",
),
],
)
def test_post_process_text(self, source: str, expected: str) -> None:
assert post_process_text(source) == expected
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# Page count # Page count
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@@ -880,25 +910,25 @@ class TestSkipArchive:
self, self,
mocker: MockerFixture, mocker: MockerFixture,
tesseract_parser: RasterisedDocumentParser, tesseract_parser: RasterisedDocumentParser,
tagged_no_text_pdf_file: Path, tesseract_samples_dir: Path,
) -> None: ) -> None:
""" """
GIVEN: GIVEN:
- A real PDF that reports itself as tagged (/MarkInfo /Marked - A PDF that reports itself as tagged (/MarkInfo /Marked true) but
true) but whose only pdftotext output is layout padding (a has no actual extractable text (some scanner firmware produces
lone form-feed byte), not real text (see GitHub issue #13387, this — see GitHub issue #13349)
originally reported against #13349's tagged-PDF handling)
- Mode: auto, produce_archive=False - Mode: auto, produce_archive=False
WHEN: WHEN:
- Document is parsed - Document is parsed
THEN: THEN:
- The tag alone is not trusted as "has text"; OCRmyPDF still runs - The tag alone is not trusted as "has text"; OCRmyPDF still runs
""" """
assert is_tagged_pdf(tagged_no_text_pdf_file) is True
tesseract_parser.settings.mode = ModeChoices.AUTO tesseract_parser.settings.mode = ModeChoices.AUTO
mocker.patch("paperless.parsers.tesseract.is_tagged_pdf", return_value=True)
mocker.patch.object(tesseract_parser, "extract_text", return_value=None)
mock_ocr = mocker.patch("ocrmypdf.ocr") mock_ocr = mocker.patch("ocrmypdf.ocr")
tesseract_parser.parse( tesseract_parser.parse(
tagged_no_text_pdf_file, tesseract_samples_dir / "multi-page-images.pdf",
"application/pdf", "application/pdf",
produce_archive=False, produce_archive=False,
) )
-110
View File
@@ -4,18 +4,10 @@ from __future__ import annotations
import codecs import codecs
from pathlib import Path from pathlib import Path
from typing import TYPE_CHECKING
import pytest
from paperless.parsers.utils import is_tagged_pdf from paperless.parsers.utils import is_tagged_pdf
from paperless.parsers.utils import pdf_born_digital_text
from paperless.parsers.utils import post_process_text
from paperless.parsers.utils import read_file_handle_unicode_errors from paperless.parsers.utils import read_file_handle_unicode_errors
if TYPE_CHECKING:
from pytest_mock import MockerFixture
SAMPLES = Path(__file__).parent / "samples" / "tesseract" SAMPLES = Path(__file__).parent / "samples" / "tesseract"
@@ -68,105 +60,3 @@ class TestIsTaggedPdf:
bad = tmp_path / "bad.pdf" bad = tmp_path / "bad.pdf"
bad.write_bytes(b"not a pdf") bad.write_bytes(b"not a pdf")
assert is_tagged_pdf(bad) is False assert is_tagged_pdf(bad) is False
class TestPostProcessText:
@pytest.mark.parametrize(
("source", "expected"),
[
pytest.param(
"simple string",
"simple string",
id="collapse-spaces",
),
pytest.param(
"simple newline\n testing string",
"simple newline\ntesting string",
id="preserve-newline",
),
pytest.param(
"utf-8 строка с пробелами в конце ", # noqa: RUF001
"utf-8 строка с пробелами в конце", # noqa: RUF001
id="utf8-trailing-spaces",
),
pytest.param(None, None, id="none-input"),
pytest.param("", None, id="empty-string"),
pytest.param(" \n\x0c \n ", None, id="whitespace-and-formfeed-only"),
],
)
def test_post_process_text(
self,
source: str | None,
expected: str | None,
) -> None:
assert post_process_text(source) == expected
class TestPdfBornDigitalText:
"""Regression coverage for GH #13387.
should_produce_archive() and RasterisedDocumentParser.parse() must agree
on whether a PDF has real text, so both go through this one function.
"""
@pytest.mark.parametrize(
("extracted", "tagged", "expected_text", "expected_born_digital"),
[
pytest.param("tiny", True, "tiny", True, id="tagged-with-real-text"),
pytest.param("tiny", False, "tiny", False, id="untagged-below-min-length"),
pytest.param(
"x" * 51,
False,
"x" * 51,
True,
id="untagged-above-min-length",
),
pytest.param(None, True, None, False, id="tagged-but-no-text"),
],
)
def test_born_digital_decision(
self,
mocker: MockerFixture,
tmp_path: Path,
extracted: str | None,
tagged: bool, # noqa: FBT001
expected_text: str | None,
expected_born_digital: bool, # noqa: FBT001
) -> None:
"""
GIVEN:
- A PDF whose pdftotext output and /MarkInfo tag status vary
WHEN:
- pdf_born_digital_text() is called
THEN:
- The normalized text and born-digital verdict match; the tag
alone never counts as "has text"
"""
mocker.patch(
"paperless.parsers.utils.extract_pdf_text",
return_value=extracted,
)
mocker.patch("paperless.parsers.utils.is_tagged_pdf", return_value=tagged)
text, born_digital = pdf_born_digital_text(tmp_path / "doc.pdf")
assert text == expected_text
assert born_digital is expected_born_digital
def test_tagged_but_textless_pdf_is_not_born_digital(
self,
tagged_no_text_pdf_file: Path,
) -> None:
"""
GIVEN:
- A real PDF that is tagged (/MarkInfo /Marked true) but whose
only "text" is layout padding (a stray form-feed byte)
WHEN:
- pdf_born_digital_text() is called with no mocking
THEN:
- The normalized text is None and the PDF is not treated as
born-digital. The raw, unnormalized pdftotext output is
non-empty for this file, which is exactly what caused the
archive decision to disagree with the OCR decision in #13387.
"""
text, born_digital = pdf_born_digital_text(tagged_no_text_pdf_file)
assert text is None
assert born_digital is False