Compare commits

..
Author SHA1 Message Date
350684cd6b Fix: consolidate born-digital PDF detection between archive decision and OCR (#13409)
* Fix: unify born-digital PDF detection between archive decision and OCR

should_produce_archive() and RasterisedDocumentParser.parse() each
reimplemented the "does this PDF have real text" check independently,
using different normalization of pdftotext output. Raw pdftotext output
can be non-empty (whitespace/form-feed layout padding) even when there
is no real content, so the two checks could disagree: consumer.py
treated a tagged-but-textless PDF as born-digital and skipped the
archive, while the parser's own (stricter, normalized) check found no
text and ran OCR anyway, leaving the document with no archive despite
real OCR text (GH #13387).

Both call sites now share one predicate, pdf_born_digital_text() in
paperless/parsers/utils.py, so they can no longer drift apart.

* Fix: restore extract_text seam for born-digital detection in parse()

parse() had switched to calling pdf_born_digital_text() directly for
its initial text/born-digital check, bypassing the parser's own
extract_text instance method. That broke test mockability (tests patch
tesseract_parser.extract_text to control the born-digital decision)
and caused CI failures with mismatched OCR call counts and text.

Split pdf_born_digital_text() into is_born_digital_text(text, path,
log) - a pure decision function - and a thin pdf_born_digital_text()
wrapper for callers without text in hand (consumer.should_produce_archive).
parse() now extracts via self.extract_text(None, document_path) and
passes the result to is_born_digital_text(), restoring the seam with
no change to production behavior.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>

* Cleanup: simplify born-digital detection, close #13387 test gap

Simplification pass over the born-digital detection consolidation:
- is_born_digital_text(): drop the has_text temp for an early return.
- consumer.py: standardize the archive-decision log lines on plain
  hyphens (was a mix of em-dash and hyphen) and hoist the duplicated
  text_length computation.
- Parametrize TestPdfBornDigitalText instead of four near-identical
  tests.

Code review follow-up: the existing tests only ever exercised
pdf_born_digital_text() through mocks, so the actual #13387 scenario
(a tagged PDF whose only "text" is layout padding) was never checked
against real pdftotext/pikepdf output - a regression in the
normalize-before-decide logic itself would have gone undetected.
Moved tagged_no_text_pdf_file from parsers/conftest.py up to the
shared paperless/tests/conftest.py (it was previously only visible to
tests under parsers/) and added a non-mocked regression test against
the real sample file.

Also fixed two stale comments in test_consumer.py referencing a
_extract_text_for_archive_check helper that no longer exists.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>

---------

Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>
2026-07-29 12:41:09 -07:00
shamoonandGitHub 668fa77428 Fix: prevent scale vs page loop in pngx PDF viewer (#13406) 2026-07-29 09:25:37 -07:00
shamoonandGitHub 5bd72014a6 Fix: handle hidden line breaks in email subjects when sending email (#13402) 2026-07-29 14:35:30 +00:00
Trenton HandGitHub bbb9c86ba4 Fix: avoid NotSupportedError from document_importer on MariaDB (#13400) 2026-07-29 07:14:13 -07:00
35 changed files with 298 additions and 534 deletions
@@ -111,7 +111,7 @@
routerLinkActive="active" (click)="closeMenu()" [ngbPopover]="view.name" routerLinkActive="active" (click)="closeMenu()" [ngbPopover]="view.name"
[disablePopover]="!slimSidebarEnabled" placement="end" container="body" triggers="mouseenter:mouseleave" [disablePopover]="!slimSidebarEnabled" placement="end" container="body" triggers="mouseenter:mouseleave"
popoverClass="popover-slim"> popoverClass="popover-slim">
<i-bs class="me-2" [name]="view.icon || 'funnel'"></i-bs><span><div class="d-inline-flex view-name"><span class="overflow-hidden" [class.text-wrap]="!slimSidebarEnabled">{{view.name}}</span></div> <i-bs class="me-2" name="funnel"></i-bs><span><div class="d-inline-flex view-name"><span class="overflow-hidden" [class.text-wrap]="!slimSidebarEnabled">{{view.name}}</span></div>
@if (showSidebarCounts && !slimSidebarEnabled) { @if (showSidebarCounts && !slimSidebarEnabled) {
<span class="badge bg-info text-dark ms-2 d-inline">{{ savedViewService.getDocumentCount(view) }}</span> <span class="badge bg-info text-dark ms-2 d-inline">{{ savedViewService.getDocumentCount(view) }}</span>
} }
@@ -36,16 +36,7 @@
(focus)="clearLastSearchTerm()" (focus)="clearLastSearchTerm()"
(clear)="clearLastSearchTerm()" (clear)="clearLastSearchTerm()"
(blur)="onBlur()"> (blur)="onBlur()">
<ng-template ng-label-tmp let-item="item">
@if (iconField && item[iconField]) {
<i-bs class="me-2" [name]="item[iconField]"></i-bs>
}
<span [title]="item[bindLabel]">{{item[bindLabel]}}</span>
</ng-template>
<ng-template ng-option-tmp let-item="item"> <ng-template ng-option-tmp let-item="item">
@if (iconField && item[iconField]) {
<i-bs class="me-2" [name]="item[iconField]"></i-bs>
}
<span [title]="item[bindLabel]">{{item[bindLabel]}}</span> <span [title]="item[bindLabel]">{{item[bindLabel]}}</span>
</ng-template> </ng-template>
</ng-select> </ng-select>
@@ -35,7 +35,7 @@ import { AbstractInputComponent } from '../abstract-input'
NgxBootstrapIconsModule, NgxBootstrapIconsModule,
], ],
}) })
export class SelectComponent extends AbstractInputComponent<number | string> { export class SelectComponent extends AbstractInputComponent<number> {
constructor() { constructor() {
super() super()
this.addItemRef = this.addItem.bind(this) this.addItemRef = this.addItem.bind(this)
@@ -100,9 +100,6 @@ export class SelectComponent extends AbstractInputComponent<number | string> {
@Input() @Input()
bindLabel: string = 'name' bindLabel: string = 'name'
@Input()
iconField: string
public searchFn = (term: string, item: any): boolean => public searchFn = (term: string, item: any): boolean =>
matchesSearchText(item?.[this.bindLabel], term) matchesSearchText(item?.[this.bindLabel], term)
@@ -129,13 +129,25 @@ describe('PngxPdfViewerComponent', () => {
;(component as any).applyScale() ;(component as any).applyScale()
expect(viewer.currentScaleValue).toBe(PdfZoomScale.PageFit) expect(viewer.currentScaleValue).toBe(PdfZoomScale.PageFit)
expect(viewer.currentScale).toBe(2) expect(viewer.currentScale).toBe(2)
})
it('does not reapply scale for page-only changes', async () => {
await initComponent()
const pdf = (component as any).pdf as { numPages: number }
pdf.numPages = 3
const viewer = (component as any).pdfViewer as PDFViewer
viewer.setDocument(pdf)
const applyScaleSpy = jest.spyOn(component as any, 'applyScale') const applyScaleSpy = jest.spyOn(component as any, 'applyScale')
component.page = 2 component.page = 2
;(component as any).lastViewerPage = 2
;(component as any).applyViewerState() component.ngOnChanges({
page: new SimpleChange(1, 2, false),
})
expect(viewer.currentPageNumber).toBe(2)
expect((component as any).lastViewerPage).toBeUndefined() expect((component as any).lastViewerPage).toBeUndefined()
expect(applyScaleSpy).toHaveBeenCalled() expect(applyScaleSpy).not.toHaveBeenCalled()
}) })
it('does not reset the viewer when it is already on the requested page', async () => { it('does not reset the viewer when it is already on the requested page', async () => {
@@ -116,7 +116,10 @@ export class PngxPdfViewerComponent
changes['zoomScale'] || changes['zoomScale'] ||
changes['rotation'] changes['rotation']
) { ) {
this.applyViewerState() // Prevent loop with page / scale application see https://github.com/paperless-ngx/paperless-ngx/issues/13404
this.applyViewerState(
!!(changes['zoom'] || changes['zoomScale'] || changes['rotation'])
)
} }
if (changes['searchQuery']) { if (changes['searchQuery']) {
@@ -240,7 +243,7 @@ export class PngxPdfViewerComponent
} }
} }
private applyViewerState(): void { private applyViewerState(applyScale = true): void {
if (!this.pdfViewer) { if (!this.pdfViewer) {
return return
} }
@@ -264,7 +267,7 @@ export class PngxPdfViewerComponent
if (this.page === this.lastViewerPage) { if (this.page === this.lastViewerPage) {
this.lastViewerPage = undefined this.lastViewerPage = undefined
} }
if (hasPages) { if (hasPages && applyScale) {
this.applyScale() this.applyScale()
} }
this.dispatchFindIfReady() this.dispatchFindIfReady()
@@ -1,7 +1,6 @@
<pngx-widget-frame <pngx-widget-frame
*pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.Document }" *pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.Document }"
[title]="savedView.name" [title]="savedView.name"
[titleIcon]="savedView.icon || 'funnel'"
[loading]="false" [loading]="false"
[draggable]="savedView" [draggable]="savedView"
> >
@@ -8,12 +8,7 @@
<i-bs name="grip-vertical"></i-bs> <i-bs name="grip-vertical"></i-bs>
</div> </div>
} }
<h6 class="card-title mb-0"> <h6 class="card-title mb-0">{{title()}}</h6>
@if (titleIcon()) {
<i-bs class="me-2" [name]="titleIcon()"></i-bs>
}
{{title()}}
</h6>
<ng-content select="[title-badge]"></ng-content> <ng-content select="[title-badge]"></ng-content>
@if (badge() !== null && badge() !== undefined) { @if (badge() !== null && badge() !== undefined) {
<span class="badge bg-info text-dark ms-2">{{badge()}}</span> <span class="badge bg-info text-dark ms-2">{{badge()}}</span>
@@ -16,8 +16,6 @@ export class WidgetFrameComponent implements AfterViewInit {
title = input<string>() title = input<string>()
titleIcon = input<string>()
draggable = input<any>() draggable = input<any>()
cardless = input(false) cardless = input(false)
@@ -97,9 +97,7 @@
<div class="dropdown-menu shadow dropdown-menu-right" ngbDropdownMenu> <div class="dropdown-menu shadow dropdown-menu-right" ngbDropdownMenu>
@if (!list.activeSavedViewId) { @if (!list.activeSavedViewId) {
@for (view of savedViewService.allViews; track view) { @for (view of savedViewService.allViews; track view) {
<button ngbDropdownItem (click)="loadViewConfig(view.id)"> <button ngbDropdownItem (click)="loadViewConfig(view.id)">{{view.name}}</button>
<i-bs class="me-2" [name]="view.icon || 'funnel'"></i-bs>{{view.name}}
</button>
} }
@if (savedViewService.allViews.length > 0) { @if (savedViewService.allViews.length > 0) {
<div class="dropdown-divider"></div> <div class="dropdown-divider"></div>
@@ -457,7 +457,6 @@ export class DocumentListComponent
modal.componentInstance.buttonsEnabled.set(false) modal.componentInstance.buttonsEnabled.set(false)
let savedView: SavedView = { let savedView: SavedView = {
name: formValue.name, name: formValue.name,
icon: formValue.icon,
filter_rules: this.list.filterRules, filter_rules: this.list.filterRules,
sort_reverse: this.list.sortReverse, sort_reverse: this.list.sortReverse,
sort_field: this.list.sortField, sort_field: this.list.sortField,
@@ -6,14 +6,6 @@
</div> </div>
<div class="modal-body"> <div class="modal-body">
<pngx-input-text i18n-title title="Name" formControlName="name" [error]="error()?.name" autocomplete="off"></pngx-input-text> <pngx-input-text i18n-title title="Name" formControlName="name" [error]="error()?.name" autocomplete="off"></pngx-input-text>
<pngx-input-select
i18n-title
title="Icon"
formControlName="icon"
[items]="savedViewIcons"
iconField="icon"
[error]="error()?.icon">
</pngx-input-select>
<pngx-input-check i18n-title title="Show in sidebar" formControlName="showInSideBar"></pngx-input-check> <pngx-input-check i18n-title title="Show in sidebar" formControlName="showInSideBar"></pngx-input-check>
<pngx-input-check i18n-title title="Show on dashboard" formControlName="showOnDashboard"></pngx-input-check> <pngx-input-check i18n-title title="Show on dashboard" formControlName="showOnDashboard"></pngx-input-check>
<pngx-permissions-form accordion="true" formControlName="permissions_form"></pngx-permissions-form> <pngx-permissions-form accordion="true" formControlName="permissions_form"></pngx-permissions-form>
@@ -9,7 +9,6 @@ import { CheckComponent } from '../../common/input/check/check.component'
import { PermissionsFormComponent } from '../../common/input/permissions/permissions-form/permissions-form.component' import { PermissionsFormComponent } from '../../common/input/permissions/permissions-form/permissions-form.component'
import { PermissionsGroupComponent } from '../../common/input/permissions/permissions-group/permissions-group.component' import { PermissionsGroupComponent } from '../../common/input/permissions/permissions-group/permissions-group.component'
import { PermissionsUserComponent } from '../../common/input/permissions/permissions-user/permissions-user.component' import { PermissionsUserComponent } from '../../common/input/permissions/permissions-user/permissions-user.component'
import { SelectComponent } from '../../common/input/select/select.component'
import { TextComponent } from '../../common/input/text/text.component' import { TextComponent } from '../../common/input/text/text.component'
import { SaveViewConfigDialogComponent } from './save-view-config-dialog.component' import { SaveViewConfigDialogComponent } from './save-view-config-dialog.component'
@@ -41,7 +40,6 @@ describe('SaveViewConfigDialogComponent', () => {
ReactiveFormsModule, ReactiveFormsModule,
SaveViewConfigDialogComponent, SaveViewConfigDialogComponent,
TextComponent, TextComponent,
SelectComponent,
CheckComponent, CheckComponent,
PermissionsFormComponent, PermissionsFormComponent,
PermissionsUserComponent, PermissionsUserComponent,
@@ -65,7 +63,6 @@ describe('SaveViewConfigDialogComponent', () => {
expect(component.defaultName()).toEqual(name) expect(component.defaultName()).toEqual(name)
expect(result).toEqual({ expect(result).toEqual({
name, name,
icon: 'funnel',
showInSideBar: false, showInSideBar: false,
showOnDashboard: false, showOnDashboard: false,
}) })
@@ -97,7 +94,6 @@ describe('SaveViewConfigDialogComponent', () => {
component.save() component.save()
expect(result).toEqual({ expect(result).toEqual({
name, name,
icon: 'funnel',
showInSideBar: true, showInSideBar: true,
showOnDashboard: true, showOnDashboard: true,
}) })
@@ -117,7 +113,6 @@ describe('SaveViewConfigDialogComponent', () => {
component.save() component.save()
expect(result).toEqual({ expect(result).toEqual({
name: '', name: '',
icon: 'funnel',
showInSideBar: false, showInSideBar: false,
showOnDashboard: false, showOnDashboard: false,
permissions_form: permissions, permissions_form: permissions,
@@ -13,14 +13,9 @@ import {
ReactiveFormsModule, ReactiveFormsModule,
} from '@angular/forms' } from '@angular/forms'
import { NgbActiveModal } from '@ng-bootstrap/ng-bootstrap' import { NgbActiveModal } from '@ng-bootstrap/ng-bootstrap'
import {
DEFAULT_SAVED_VIEW_ICON,
SAVED_VIEW_ICONS,
} from 'src/app/data/saved-view-icons'
import { User } from 'src/app/data/user' import { User } from 'src/app/data/user'
import { CheckComponent } from '../../common/input/check/check.component' import { CheckComponent } from '../../common/input/check/check.component'
import { PermissionsFormComponent } from '../../common/input/permissions/permissions-form/permissions-form.component' import { PermissionsFormComponent } from '../../common/input/permissions/permissions-form/permissions-form.component'
import { SelectComponent } from '../../common/input/select/select.component'
import { TextComponent } from '../../common/input/text/text.component' import { TextComponent } from '../../common/input/text/text.component'
@Component({ @Component({
@@ -29,7 +24,6 @@ import { TextComponent } from '../../common/input/text/text.component'
styleUrls: ['./save-view-config-dialog.component.scss'], styleUrls: ['./save-view-config-dialog.component.scss'],
imports: [ imports: [
CheckComponent, CheckComponent,
SelectComponent,
TextComponent, TextComponent,
PermissionsFormComponent, PermissionsFormComponent,
FormsModule, FormsModule,
@@ -47,7 +41,6 @@ export class SaveViewConfigDialogComponent implements OnInit {
public saveClicked = new EventEmitter() public saveClicked = new EventEmitter()
users: User[] users: User[]
readonly savedViewIcons = SAVED_VIEW_ICONS
setDefaultName(value: string) { setDefaultName(value: string) {
this.defaultName.set(value) this.defaultName.set(value)
@@ -56,7 +49,6 @@ export class SaveViewConfigDialogComponent implements OnInit {
saveViewConfigForm = new FormGroup({ saveViewConfigForm = new FormGroup({
name: new FormControl(''), name: new FormControl(''),
icon: new FormControl(DEFAULT_SAVED_VIEW_ICON),
showInSideBar: new FormControl(false), showInSideBar: new FormControl(false),
showOnDashboard: new FormControl(false), showOnDashboard: new FormControl(false),
permissions_form: new FormControl(null), permissions_form: new FormControl(null),
@@ -73,7 +65,6 @@ export class SaveViewConfigDialogComponent implements OnInit {
const formValue = this.saveViewConfigForm.value const formValue = this.saveViewConfigForm.value
const saveViewConfig = { const saveViewConfig = {
name: formValue.name, name: formValue.name,
icon: formValue.icon,
showInSideBar: formValue.showInSideBar, showInSideBar: formValue.showInSideBar,
showOnDashboard: formValue.showOnDashboard, showOnDashboard: formValue.showOnDashboard,
} }
@@ -11,19 +11,10 @@
<li class="list-group-item py-3"> <li class="list-group-item py-3">
<div [formGroupName]="view.id"> <div [formGroupName]="view.id">
<div class="row"> <div class="row">
<div class="col-md"> <div class="col">
<pngx-input-text title="Name" formControlName="name"></pngx-input-text> <pngx-input-text title="Name" formControlName="name"></pngx-input-text>
</div> </div>
<div class="col-md"> <div class="col">
<pngx-input-select
i18n-title
title="Icon"
formControlName="icon"
[items]="savedViewIcons"
iconField="icon">
</pngx-input-select>
</div>
<div class="col-md">
<div class="form-check form-switch mt-3"> <div class="form-check form-switch mt-3">
<input type="checkbox" class="form-check-input" id="show_on_dashboard_{{view.id}}" formControlName="show_on_dashboard"> <input type="checkbox" class="form-check-input" id="show_on_dashboard_{{view.id}}" formControlName="show_on_dashboard">
<label class="form-check-label" for="show_on_dashboard_{{view.id}}" i18n>Show on dashboard</label> <label class="form-check-label" for="show_on_dashboard_{{view.id}}" i18n>Show on dashboard</label>
@@ -25,20 +25,8 @@ import { PageHeaderComponent } from '../../common/page-header/page-header.compon
import { SavedViewsComponent } from './saved-views.component' import { SavedViewsComponent } from './saved-views.component'
const savedViews = [ const savedViews = [
{ { id: 1, name: 'view1', show_in_sidebar: true, show_on_dashboard: true },
id: 1, { id: 2, name: 'view2', show_in_sidebar: false, show_on_dashboard: false },
name: 'view1',
icon: 'archive',
show_in_sidebar: true,
show_on_dashboard: true,
},
{
id: 2,
name: 'view2',
icon: 'funnel',
show_in_sidebar: false,
show_on_dashboard: false,
},
] ]
describe('SavedViewsComponent', () => { describe('SavedViewsComponent', () => {
@@ -169,24 +157,6 @@ describe('SavedViewsComponent', () => {
expect(patchBody.show_in_sidebar).toBeUndefined() expect(patchBody.show_in_sidebar).toBeUndefined()
}) })
it('should persist a changed icon', () => {
const patchSpy = jest.spyOn(savedViewService, 'patchMany')
const view = savedViews[0]
const iconControl = component.savedViewsForm
.get('savedViews')
.get(view.id.toString())
.get('icon')
iconControl.setValue('bell')
iconControl.markAsDirty()
component.save()
expect(patchSpy.mock.calls[0][0][0]).toMatchObject({
id: view.id,
icon: 'bell',
})
})
it('should persist visibility changes to user settings', () => { it('should persist visibility changes to user settings', () => {
const patchSpy = jest.spyOn(savedViewService, 'patchMany') const patchSpy = jest.spyOn(savedViewService, 'patchMany')
const updateVisibilitySpy = jest const updateVisibilitySpy = jest
@@ -13,10 +13,6 @@ import { BehaviorSubject, Observable, of, switchMap, takeUntil } from 'rxjs'
import { PermissionsDialogComponent } from 'src/app/components/common/permissions-dialog/permissions-dialog.component' import { PermissionsDialogComponent } from 'src/app/components/common/permissions-dialog/permissions-dialog.component'
import { DisplayMode } from 'src/app/data/document' import { DisplayMode } from 'src/app/data/document'
import { SavedView } from 'src/app/data/saved-view' import { SavedView } from 'src/app/data/saved-view'
import {
DEFAULT_SAVED_VIEW_ICON,
SAVED_VIEW_ICONS,
} from 'src/app/data/saved-view-icons'
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive' import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
import { import {
PermissionAction, PermissionAction,
@@ -28,7 +24,6 @@ import { ToastService } from 'src/app/services/toast.service'
import { ConfirmButtonComponent } from '../../common/confirm-button/confirm-button.component' import { ConfirmButtonComponent } from '../../common/confirm-button/confirm-button.component'
import { DragDropSelectComponent } from '../../common/input/drag-drop-select/drag-drop-select.component' import { DragDropSelectComponent } from '../../common/input/drag-drop-select/drag-drop-select.component'
import { NumberComponent } from '../../common/input/number/number.component' import { NumberComponent } from '../../common/input/number/number.component'
import { SelectComponent } from '../../common/input/select/select.component'
import { TextComponent } from '../../common/input/text/text.component' import { TextComponent } from '../../common/input/text/text.component'
import { PageHeaderComponent } from '../../common/page-header/page-header.component' import { PageHeaderComponent } from '../../common/page-header/page-header.component'
import { LoadingComponentWithPermissions } from '../../loading-component/loading.component' import { LoadingComponentWithPermissions } from '../../loading-component/loading.component'
@@ -40,7 +35,6 @@ import { LoadingComponentWithPermissions } from '../../loading-component/loading
PageHeaderComponent, PageHeaderComponent,
ConfirmButtonComponent, ConfirmButtonComponent,
NumberComponent, NumberComponent,
SelectComponent,
TextComponent, TextComponent,
IfPermissionsDirective, IfPermissionsDirective,
DragDropSelectComponent, DragDropSelectComponent,
@@ -61,7 +55,6 @@ export class SavedViewsComponent
private readonly modalService = inject(NgbModal) private readonly modalService = inject(NgbModal)
DisplayMode = DisplayMode DisplayMode = DisplayMode
readonly savedViewIcons = SAVED_VIEW_ICONS
readonly savedViews = signal<SavedView[]>(undefined) readonly savedViews = signal<SavedView[]>(undefined)
private savedViewsGroup = new FormGroup({}) private savedViewsGroup = new FormGroup({})
@@ -112,7 +105,6 @@ export class SavedViewsComponent
storeData.savedViews[view.id.toString()] = { storeData.savedViews[view.id.toString()] = {
id: view.id, id: view.id,
name: view.name, name: view.name,
icon: view.icon ?? DEFAULT_SAVED_VIEW_ICON,
show_on_dashboard: view.show_on_dashboard, show_on_dashboard: view.show_on_dashboard,
show_in_sidebar: view.show_in_sidebar, show_in_sidebar: view.show_in_sidebar,
page_size: view.page_size, page_size: view.page_size,
@@ -125,7 +117,6 @@ export class SavedViewsComponent
new FormGroup({ new FormGroup({
id: new FormControl({ value: null, disabled: !canEdit }), id: new FormControl({ value: null, disabled: !canEdit }),
name: new FormControl({ value: null, disabled: !canEdit }), name: new FormControl({ value: null, disabled: !canEdit }),
icon: new FormControl({ value: null, disabled: !canEdit }),
show_on_dashboard: new FormControl({ show_on_dashboard: new FormControl({
value: null, value: null,
disabled: false, disabled: false,
@@ -204,7 +195,6 @@ export class SavedViewsComponent
const modelFieldsChanged = const modelFieldsChanged =
group.get('name')?.dirty || group.get('name')?.dirty ||
group.get('icon')?.dirty ||
group.get('page_size')?.dirty || group.get('page_size')?.dirty ||
group.get('display_mode')?.dirty || group.get('display_mode')?.dirty ||
group.get('display_fields')?.dirty group.get('display_fields')?.dirty
-89
View File
@@ -1,89 +0,0 @@
export const DEFAULT_SAVED_VIEW_ICON = 'funnel'
export const SAVED_VIEW_ICONS = [
{ id: 'archive', name: $localize`Archive`, icon: 'archive' },
{ id: 'bank', name: $localize`Bank`, icon: 'bank' },
{ id: 'basket', name: $localize`Basket`, icon: 'basket' },
{ id: 'bell', name: $localize`Bell`, icon: 'bell' },
{ id: 'bookmark', name: $localize`Bookmark`, icon: 'bookmark' },
{ id: 'boxes', name: $localize`Boxes`, icon: 'boxes' },
{ id: 'briefcase', name: $localize`Briefcase`, icon: 'briefcase' },
{ id: 'building', name: $localize`Building`, icon: 'building' },
{ id: 'calculator', name: $localize`Calculator`, icon: 'calculator' },
{ id: 'calendar', name: $localize`Calendar`, icon: 'calendar' },
{ id: 'camera', name: $localize`Camera`, icon: 'camera' },
{
id: 'card-checklist',
name: $localize`Checklist`,
icon: 'card-checklist',
},
{ id: 'cash', name: $localize`Cash`, icon: 'cash' },
{ id: 'chat-left-text', name: $localize`Chat`, icon: 'chat-left-text' },
{ id: 'check-circle', name: $localize`Check`, icon: 'check-circle' },
{ id: 'clipboard', name: $localize`Clipboard`, icon: 'clipboard' },
{ id: 'clock-history', name: $localize`Clock`, icon: 'clock-history' },
{ id: 'credit-card', name: $localize`Credit card`, icon: 'credit-card' },
{ id: 'download', name: $localize`Download`, icon: 'download' },
{ id: 'envelope', name: $localize`Envelope`, icon: 'envelope' },
{
id: 'exclamation-triangle',
name: $localize`Warning`,
icon: 'exclamation-triangle',
},
{ id: 'file-earmark', name: $localize`File`, icon: 'file-earmark' },
{
id: 'file-earmark-check',
name: $localize`Checked file`,
icon: 'file-earmark-check',
},
{
id: 'file-earmark-lock',
name: $localize`Locked file`,
icon: 'file-earmark-lock',
},
{
id: 'file-earmark-medical',
name: $localize`Medical file`,
icon: 'file-earmark-medical',
},
{
id: 'file-earmark-person',
name: $localize`Person file`,
icon: 'file-earmark-person',
},
{
id: 'file-earmark-spreadsheet',
name: $localize`Spreadsheet`,
icon: 'file-earmark-spreadsheet',
},
{ id: 'file-text', name: $localize`Text file`, icon: 'file-text' },
{ id: 'files', name: $localize`Files`, icon: 'files' },
{ id: 'folder', name: $localize`Folder`, icon: 'folder' },
{ id: 'funnel', name: $localize`Filter`, icon: 'funnel' },
{ id: 'gear', name: $localize`Gear`, icon: 'gear' },
{ id: 'globe2', name: $localize`Globe`, icon: 'globe2' },
{ id: 'hash', name: $localize`Hash`, icon: 'hash' },
{ id: 'heart', name: $localize`Heart`, icon: 'heart' },
{ id: 'house', name: $localize`House`, icon: 'house' },
{ id: 'inbox', name: $localize`Inbox`, icon: 'inbox' },
{ id: 'journals', name: $localize`Journals`, icon: 'journals' },
{ id: 'list-task', name: $localize`Task list`, icon: 'list-task' },
{ id: 'newspaper', name: $localize`Newspaper`, icon: 'newspaper' },
{ id: 'paperclip', name: $localize`Attachment`, icon: 'paperclip' },
{ id: 'people', name: $localize`People`, icon: 'people' },
{ id: 'person', name: $localize`Person`, icon: 'person' },
{ id: 'printer', name: $localize`Printer`, icon: 'printer' },
{ id: 'receipt', name: $localize`Receipt`, icon: 'receipt' },
{ id: 'safe', name: $localize`Safe`, icon: 'safe' },
{ id: 'search', name: $localize`Search`, icon: 'search' },
{ id: 'send', name: $localize`Send`, icon: 'send' },
{ id: 'shop', name: $localize`Shop`, icon: 'shop' },
{ id: 'stack', name: $localize`Stack`, icon: 'stack' },
{ id: 'stars', name: $localize`Stars`, icon: 'stars' },
{ id: 'tag', name: $localize`Tag`, icon: 'tag' },
{ id: 'tags', name: $localize`Tags`, icon: 'tags' },
{ id: 'telephone', name: $localize`Telephone`, icon: 'telephone' },
{ id: 'truck', name: $localize`Truck`, icon: 'truck' },
{ id: 'upc-scan', name: $localize`Barcode`, icon: 'upc-scan' },
{ id: 'wallet2', name: $localize`Wallet`, icon: 'wallet2' },
]
-2
View File
@@ -5,8 +5,6 @@ import { ObjectWithPermissions } from './object-with-permissions'
export interface SavedView extends ObjectWithPermissions { export interface SavedView extends ObjectWithPermissions {
name?: string name?: string
icon?: string
show_on_dashboard?: boolean show_on_dashboard?: boolean
show_in_sidebar?: boolean show_in_sidebar?: boolean
-46
View File
@@ -35,27 +35,19 @@ import {
arrowRightShort, arrowRightShort,
arrowUpRight, arrowUpRight,
asterisk, asterisk,
bank,
basket,
bell, bell,
bodyText, bodyText,
bookmark,
boxArrowUp, boxArrowUp,
boxArrowUpRight, boxArrowUpRight,
boxes, boxes,
braces, braces,
briefcase,
building,
calculator,
calendar, calendar,
calendarEvent, calendarEvent,
calendarEventFill, calendarEventFill,
camera,
cardChecklist, cardChecklist,
cardHeading, cardHeading,
caretDown, caretDown,
caretUp, caretUp,
cash,
chatLeftText, chatLeftText,
chatSquareDots, chatSquareDots,
check, check,
@@ -73,7 +65,6 @@ import {
clipboardCheckFill, clipboardCheckFill,
clipboardFill, clipboardFill,
clockHistory, clockHistory,
creditCard,
dash, dash,
dashCircle, dashCircle,
diagram3, diagram3,
@@ -92,12 +83,9 @@ import {
fileEarmarkDiff, fileEarmarkDiff,
fileEarmarkFill, fileEarmarkFill,
fileEarmarkLock, fileEarmarkLock,
fileEarmarkMedical,
fileEarmarkMinus, fileEarmarkMinus,
fileEarmarkPerson,
fileEarmarkPlus, fileEarmarkPlus,
fileEarmarkRichtext, fileEarmarkRichtext,
fileEarmarkSpreadsheet,
fileText, fileText,
files, files,
filter, filter,
@@ -105,15 +93,12 @@ import {
folderFill, folderFill,
funnel, funnel,
gear, gear,
globe2,
google, google,
grid, grid,
gripVertical, gripVertical,
hash, hash,
hddStack, hddStack,
heart,
house, house,
inbox,
infoCircle, infoCircle,
journals, journals,
link, link,
@@ -121,9 +106,7 @@ import {
listTask, listTask,
listUl, listUl,
microsoft, microsoft,
newspaper,
nodePlus, nodePlus,
paperclip,
pencil, pencil,
people, people,
peopleFill, peopleFill,
@@ -138,12 +121,9 @@ import {
plusCircle, plusCircle,
printer, printer,
questionCircle, questionCircle,
receipt,
safe,
scissors, scissors,
search, search,
send, send,
shop,
slashCircle, slashCircle,
sliders2Vertical, sliders2Vertical,
sortAlphaDown, sortAlphaDown,
@@ -153,17 +133,14 @@ import {
tag, tag,
tagFill, tagFill,
tags, tags,
telephone,
textIndentLeft, textIndentLeft,
textLeft, textLeft,
threeDots, threeDots,
threeDotsVertical, threeDotsVertical,
trash, trash,
truck,
uiRadios, uiRadios,
unlock, unlock,
upcScan, upcScan,
wallet2,
windowStack, windowStack,
x, x,
xCircle, xCircle,
@@ -281,22 +258,15 @@ const icons = {
arrowRightShort, arrowRightShort,
arrowUpRight, arrowUpRight,
asterisk, asterisk,
bank,
basket,
bell, bell,
braces, braces,
bodyText, bodyText,
bookmark,
boxArrowUp, boxArrowUp,
boxArrowUpRight, boxArrowUpRight,
boxes, boxes,
briefcase,
building,
calculator,
calendar, calendar,
calendarEvent, calendarEvent,
calendarEventFill, calendarEventFill,
camera,
cardChecklist, cardChecklist,
cardHeading, cardHeading,
caretDown, caretDown,
@@ -318,8 +288,6 @@ const icons = {
clipboardCheckFill, clipboardCheckFill,
clipboardFill, clipboardFill,
clockHistory, clockHistory,
cash,
creditCard,
dash, dash,
dashCircle, dashCircle,
diagram3, diagram3,
@@ -338,12 +306,9 @@ const icons = {
fileEarmarkDiff, fileEarmarkDiff,
fileEarmarkFill, fileEarmarkFill,
fileEarmarkLock, fileEarmarkLock,
fileEarmarkMedical,
fileEarmarkMinus, fileEarmarkMinus,
fileEarmarkPerson,
fileEarmarkPlus, fileEarmarkPlus,
fileEarmarkRichtext, fileEarmarkRichtext,
fileEarmarkSpreadsheet,
files, files,
fileText, fileText,
filter, filter,
@@ -351,15 +316,12 @@ const icons = {
folderFill, folderFill,
funnel, funnel,
gear, gear,
globe2,
google, google,
grid, grid,
gripVertical, gripVertical,
hash, hash,
hddStack, hddStack,
heart,
house, house,
inbox,
infoCircle, infoCircle,
journals, journals,
link, link,
@@ -367,10 +329,8 @@ const icons = {
listTask, listTask,
listUl, listUl,
microsoft, microsoft,
newspaper,
nodePlus, nodePlus,
pencil, pencil,
paperclip,
people, people,
peopleFill, peopleFill,
person, person,
@@ -384,13 +344,10 @@ const icons = {
plusCircle, plusCircle,
printer, printer,
questionCircle, questionCircle,
receipt,
safe,
scissors, scissors,
search, search,
send, send,
slashCircle, slashCircle,
shop,
sliders2Vertical, sliders2Vertical,
sortAlphaDown, sortAlphaDown,
sortAlphaUpAlt, sortAlphaUpAlt,
@@ -401,15 +358,12 @@ const icons = {
tags, tags,
textIndentLeft, textIndentLeft,
textLeft, textLeft,
telephone,
threeDots, threeDots,
threeDotsVertical, threeDotsVertical,
trash, trash,
truck,
uiRadios, uiRadios,
unlock, unlock,
upcScan, upcScan,
wallet2,
windowStack, windowStack,
x, x,
xCircle, xCircle,
+15 -25
View File
@@ -57,9 +57,7 @@ from paperless.models import ArchiveFileGenerationChoices
from paperless.parsers import ParserContext from paperless.parsers import ParserContext
from paperless.parsers import ParserProtocol from paperless.parsers import ParserProtocol
from paperless.parsers.registry import get_parser_registry from paperless.parsers.registry import get_parser_registry
from paperless.parsers.utils import PDF_TEXT_MIN_LENGTH from paperless.parsers.utils import pdf_born_digital_text
from paperless.parsers.utils import extract_pdf_text
from paperless.parsers.utils import is_tagged_pdf
LOGGING_NAME: Final[str] = "paperless.consumer" LOGGING_NAME: Final[str] = "paperless.consumer"
@@ -138,53 +136,45 @@ def should_produce_archive(
# Must produce a PDF so the frontend can display the original format at all. # Must produce a PDF so the frontend can display the original format at all.
if parser.requires_pdf_rendition: if parser.requires_pdf_rendition:
_log.debug("Archive: yes parser requires PDF rendition for frontend display") _log.debug("Archive: yes - parser requires PDF rendition for frontend display")
return True return True
# Parser cannot produce an archive (e.g. TextDocumentParser). # Parser cannot produce an archive (e.g. TextDocumentParser).
if not parser.can_produce_archive: if not parser.can_produce_archive:
_log.debug("Archive: no parser cannot produce archives") _log.debug("Archive: no - parser cannot produce archives")
return False return False
generation = OcrConfig().archive_file_generation generation = OcrConfig().archive_file_generation
if generation == ArchiveFileGenerationChoices.ALWAYS: if generation == ArchiveFileGenerationChoices.ALWAYS:
_log.debug("Archive: yes ARCHIVE_FILE_GENERATION=always") _log.debug("Archive: yes - ARCHIVE_FILE_GENERATION=always")
return True return True
if generation == ArchiveFileGenerationChoices.NEVER: if generation == ArchiveFileGenerationChoices.NEVER:
_log.debug("Archive: no ARCHIVE_FILE_GENERATION=never") _log.debug("Archive: no - ARCHIVE_FILE_GENERATION=never")
return False return False
# auto: produce archives for scanned/image documents; skip for born-digital PDFs. # auto: produce archives for scanned/image documents; skip for born-digital PDFs.
if mime_type.startswith("image/"): if mime_type.startswith("image/"):
_log.debug("Archive: yes image document, ARCHIVE_FILE_GENERATION=auto") _log.debug("Archive: yes - image document, ARCHIVE_FILE_GENERATION=auto")
return True return True
if mime_type == "application/pdf": if mime_type == "application/pdf":
text = extract_pdf_text(document_path) text, born_digital = pdf_born_digital_text(document_path, log=_log)
has_text = text is not None and len(text) > 0 text_length = len(text) if text else 0
if has_text and is_tagged_pdf(document_path): if born_digital:
_log.debug( _log.debug(
"Archive: no born-digital PDF (structure tags detected)," "Archive: no - born-digital PDF (text_length=%d),"
" ARCHIVE_FILE_GENERATION=auto", " ARCHIVE_FILE_GENERATION=auto",
text_length,
) )
return False return False
if text is None or len(text) <= PDF_TEXT_MIN_LENGTH:
_log.debug(
"Archive: yes — scanned PDF (text_length=%d%d),"
" ARCHIVE_FILE_GENERATION=auto",
len(text) if text else 0,
PDF_TEXT_MIN_LENGTH,
)
return True
_log.debug( _log.debug(
"Archive: no — born-digital PDF (text_length=%d > %d)," "Archive: yes - scanned/textless PDF (text_length=%d),"
" ARCHIVE_FILE_GENERATION=auto", " ARCHIVE_FILE_GENERATION=auto",
len(text), text_length,
PDF_TEXT_MIN_LENGTH,
) )
return False return True
_log.debug( _log.debug(
"Archive: no MIME type %r not eligible for auto archive generation", "Archive: no - MIME type %r not eligible for auto archive generation",
mime_type, mime_type,
) )
return False return False
+3
View File
@@ -36,6 +36,9 @@ def send_email(
TODO: re-evaluate this pending https://code.djangoproject.com/ticket/35581 / https://github.com/django/django/pull/18966 TODO: re-evaluate this pending https://code.djangoproject.com/ticket/35581 / https://github.com/django/django/pull/18966
""" """
if "\r" in subject or "\n" in subject:
subject = " ".join(line.strip(" \t") for line in subject.splitlines())
email = EmailMessage( email = EmailMessage(
subject=subject, subject=subject,
body=body, body=body,
@@ -386,10 +386,19 @@ class Command(CryptMixin, PaperlessCommand):
raise DeserializationError( raise DeserializationError(
f"{model.__name__} has no updatable fields; PK-only models are not supported by the importer", f"{model.__name__} has no updatable fields; PK-only models are not supported by the importer",
) )
# MySQL/MariaDB support upserts via ON DUPLICATE KEY UPDATE but,
# unlike PostgreSQL/SQLite, cannot target a specific unique field
# for the conflict -- passing unique_fields there raises
# NotSupportedError.
unique_fields = (
[model._meta.pk.attname]
if connection.features.supports_update_conflicts_with_target
else None
)
model.objects.bulk_create( # type: ignore[attr-defined] model.objects.bulk_create( # type: ignore[attr-defined]
instances, instances,
update_conflicts=True, update_conflicts=True,
unique_fields=[model._meta.pk.attname], unique_fields=unique_fields,
update_fields=update_fields, update_fields=update_fields,
) )
loaded_models.add(model) loaded_models.add(model)
@@ -1,79 +0,0 @@
from django.db import migrations
from django.db import models
class Migration(migrations.Migration):
dependencies = [
("documents", "0022_add_perf_indexes"),
]
operations = [
migrations.AddField(
model_name="savedview",
name="icon",
field=models.CharField(
choices=[
("archive", "Archive"),
("bank", "Bank"),
("basket", "Basket"),
("bell", "Bell"),
("bookmark", "Bookmark"),
("boxes", "Boxes"),
("briefcase", "Briefcase"),
("building", "Building"),
("calculator", "Calculator"),
("calendar", "Calendar"),
("camera", "Camera"),
("card-checklist", "Checklist"),
("cash", "Cash"),
("chat-left-text", "Chat"),
("check-circle", "Check"),
("clipboard", "Clipboard"),
("clock-history", "Clock"),
("credit-card", "Credit card"),
("download", "Download"),
("envelope", "Envelope"),
("exclamation-triangle", "Warning"),
("file-earmark", "File"),
("file-earmark-check", "Checked file"),
("file-earmark-lock", "Locked file"),
("file-earmark-medical", "Medical file"),
("file-earmark-person", "Person file"),
("file-earmark-spreadsheet", "Spreadsheet"),
("file-text", "Text file"),
("files", "Files"),
("folder", "Folder"),
("funnel", "Filter"),
("gear", "Gear"),
("globe2", "Globe"),
("hash", "Hash"),
("heart", "Heart"),
("house", "House"),
("inbox", "Inbox"),
("journals", "Journals"),
("list-task", "Task list"),
("newspaper", "Newspaper"),
("paperclip", "Attachment"),
("people", "People"),
("person", "Person"),
("printer", "Printer"),
("receipt", "Receipt"),
("safe", "Safe"),
("search", "Search"),
("send", "Send"),
("shop", "Shop"),
("stack", "Stack"),
("stars", "Stars"),
("tag", "Tag"),
("tags", "Tags"),
("telephone", "Telephone"),
("truck", "Truck"),
("upc-scan", "Barcode"),
("wallet2", "Wallet"),
],
default="funnel",
max_length=64,
verbose_name="icon",
),
),
]
-69
View File
@@ -519,68 +519,6 @@ class Document(SoftDeleteModel, ModelWithOwner): # type: ignore[django-manager-
class SavedView(ModelWithOwner): class SavedView(ModelWithOwner):
class Icon(models.TextChoices):
ARCHIVE = ("archive", _("Archive"))
BANK = ("bank", _("Bank"))
BASKET = ("basket", _("Basket"))
BELL = ("bell", _("Bell"))
BOOKMARK = ("bookmark", _("Bookmark"))
BOXES = ("boxes", _("Boxes"))
BRIEFCASE = ("briefcase", _("Briefcase"))
BUILDING = ("building", _("Building"))
CALCULATOR = ("calculator", _("Calculator"))
CALENDAR = ("calendar", _("Calendar"))
CAMERA = ("camera", _("Camera"))
CARD_CHECKLIST = ("card-checklist", _("Checklist"))
CASH = ("cash", _("Cash"))
CHAT_LEFT_TEXT = ("chat-left-text", _("Chat"))
CHECK_CIRCLE = ("check-circle", _("Check"))
CLIPBOARD = ("clipboard", _("Clipboard"))
CLOCK_HISTORY = ("clock-history", _("Clock"))
CREDIT_CARD = ("credit-card", _("Credit card"))
DOWNLOAD = ("download", _("Download"))
ENVELOPE = ("envelope", _("Envelope"))
EXCLAMATION_TRIANGLE = ("exclamation-triangle", _("Warning"))
FILE_EARMARK = ("file-earmark", _("File"))
FILE_EARMARK_CHECK = ("file-earmark-check", _("Checked file"))
FILE_EARMARK_LOCK = ("file-earmark-lock", _("Locked file"))
FILE_EARMARK_MEDICAL = ("file-earmark-medical", _("Medical file"))
FILE_EARMARK_PERSON = ("file-earmark-person", _("Person file"))
FILE_EARMARK_SPREADSHEET = (
"file-earmark-spreadsheet",
_("Spreadsheet"),
)
FILE_TEXT = ("file-text", _("Text file"))
FILES = ("files", _("Files"))
FOLDER = ("folder", _("Folder"))
FUNNEL = ("funnel", _("Filter"))
GEAR = ("gear", _("Gear"))
GLOBE = ("globe2", _("Globe"))
HASH = ("hash", _("Hash"))
HEART = ("heart", _("Heart"))
HOUSE = ("house", _("House"))
INBOX = ("inbox", _("Inbox"))
JOURNALS = ("journals", _("Journals"))
LIST_TASK = ("list-task", _("Task list"))
NEWSPAPER = ("newspaper", _("Newspaper"))
PAPERCLIP = ("paperclip", _("Attachment"))
PEOPLE = ("people", _("People"))
PERSON = ("person", _("Person"))
PRINTER = ("printer", _("Printer"))
RECEIPT = ("receipt", _("Receipt"))
SAFE = ("safe", _("Safe"))
SEARCH = ("search", _("Search"))
SEND = ("send", _("Send"))
SHOP = ("shop", _("Shop"))
STACK = ("stack", _("Stack"))
STARS = ("stars", _("Stars"))
TAG = ("tag", _("Tag"))
TAGS = ("tags", _("Tags"))
TELEPHONE = ("telephone", _("Telephone"))
TRUCK = ("truck", _("Truck"))
UPC_SCAN = ("upc-scan", _("Barcode"))
WALLET = ("wallet2", _("Wallet"))
class DisplayMode(models.TextChoices): class DisplayMode(models.TextChoices):
TABLE = ("table", _("Table")) TABLE = ("table", _("Table"))
SMALL_CARDS = ("smallCards", _("Small Cards")) SMALL_CARDS = ("smallCards", _("Small Cards"))
@@ -603,13 +541,6 @@ class SavedView(ModelWithOwner):
name = models.CharField(_("name"), max_length=128) name = models.CharField(_("name"), max_length=128)
icon = models.CharField(
_("icon"),
max_length=64,
choices=Icon.choices,
default=Icon.FUNNEL,
)
sort_field = models.CharField( sort_field = models.CharField(
_("sort field"), _("sort field"),
max_length=128, max_length=128,
-1
View File
@@ -1389,7 +1389,6 @@ class SavedViewSerializer(OwnedObjectSerializer):
fields = [ fields = [
"id", "id",
"name", "name",
"icon",
"sort_field", "sort_field",
"sort_reverse", "sort_reverse",
"filter_rules", "filter_rules",
+1 -10
View File
@@ -2880,20 +2880,18 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
v1 = SavedView.objects.get(name="test") v1 = SavedView.objects.get(name="test")
self.assertEqual(v1.sort_field, "created2") self.assertEqual(v1.sort_field, "created2")
self.assertEqual(v1.icon, SavedView.Icon.FUNNEL)
self.assertEqual(v1.filter_rules.count(), 1) self.assertEqual(v1.filter_rules.count(), 1)
self.assertEqual(v1.owner, self.user) self.assertEqual(v1.owner, self.user)
response = self.client.patch( response = self.client.patch(
f"/api/saved_views/{v1.id}/", f"/api/saved_views/{v1.id}/",
{"sort_reverse": True, "icon": SavedView.Icon.RECEIPT}, {"sort_reverse": True},
format="json", format="json",
) )
v1 = SavedView.objects.get(id=v1.id) v1 = SavedView.objects.get(id=v1.id)
self.assertEqual(response.status_code, status.HTTP_200_OK) self.assertEqual(response.status_code, status.HTTP_200_OK)
self.assertTrue(v1.sort_reverse) self.assertTrue(v1.sort_reverse)
self.assertEqual(v1.icon, SavedView.Icon.RECEIPT)
self.assertEqual(v1.filter_rules.count(), 1) self.assertEqual(v1.filter_rules.count(), 1)
view["filter_rules"] = [{"rule_type": 12, "value": "secret"}] view["filter_rules"] = [{"rule_type": 12, "value": "secret"}]
@@ -2913,13 +2911,6 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
v1 = SavedView.objects.get(id=v1.id) v1 = SavedView.objects.get(id=v1.id)
self.assertEqual(v1.filter_rules.count(), 0) self.assertEqual(v1.filter_rules.count(), 0)
response = self.client.patch(
f"/api/saved_views/{v1.id}/",
{"icon": "not-an-icon"},
format="json",
)
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
def test_saved_view_display_options(self) -> None: def test_saved_view_display_options(self) -> None:
""" """
GIVEN: GIVEN:
+1 -1
View File
@@ -75,7 +75,7 @@ class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
{ {
"documents": [self.doc1.pk, self.doc2.pk], "documents": [self.doc1.pk, self.doc2.pk],
"addresses": "hello@paperless-ngx.com,test@example.com", "addresses": "hello@paperless-ngx.com,test@example.com",
"subject": "Bulk email test", "subject": "Bulk email\n test",
"message": "Here are your documents", "message": "Here are your documents",
}, },
), ),
+2 -2
View File
@@ -1329,7 +1329,7 @@ class PreConsumeTestCase(DirectoriesMixin, GetConsumerMixin, TestCase):
with self.get_consumer(self.test_file) as c: with self.get_consumer(self.test_file) as c:
c.run() c.run()
# Verify no pre-consume script subprocess was invoked # Verify no pre-consume script subprocess was invoked
# (run_subprocess may still be called by _extract_text_for_archive_check) # (run_subprocess may still be called by pdf_born_digital_text via pdftotext)
script_calls = [ script_calls = [
call call
for call in m.call_args_list for call in m.call_args_list
@@ -1354,7 +1354,7 @@ class PreConsumeTestCase(DirectoriesMixin, GetConsumerMixin, TestCase):
self.assertTrue(m.called) self.assertTrue(m.called)
# Find the call that invoked the pre-consume script # Find the call that invoked the pre-consume script
# (run_subprocess may also be called by _extract_text_for_archive_check) # (run_subprocess may also be called by pdf_born_digital_text via pdftotext)
script_call = next( script_call = next(
call call
for call in m.call_args_list for call in m.call_args_list
+14 -42
View File
@@ -134,60 +134,32 @@ class TestShouldProduceArchive:
assert should_produce_archive(parser, mime, Path("/tmp/doc")) is expected assert should_produce_archive(parser, mime, Path("/tmp/doc")) is expected
@pytest.mark.parametrize( @pytest.mark.parametrize(
("extracted_text", "expected"), ("born_digital", "expected"),
[ [
pytest.param( pytest.param(True, False, id="born-digital-skips-archive"),
"This is a born-digital PDF with lots of text content. " * 10, pytest.param(False, True, id="not-born-digital-produces-archive"),
False,
id="born-digital-long-text-skips-archive",
),
pytest.param(None, True, id="no-text-scanned-produces-archive"),
pytest.param("tiny", True, id="short-text-treated-as-scanned"),
], ],
) )
def test_auto_pdf_archive_decision( def test_auto_pdf_archive_decision(
self, self,
mocker: MockerFixture, mocker: MockerFixture,
settings, settings,
extracted_text: str | None, born_digital: bool, # noqa: FBT001
expected: bool, # noqa: FBT001 expected: bool, # noqa: FBT001
) -> None: ) -> None:
"""Archive decision tracks pdf_born_digital_text()'s verdict exactly.
should_produce_archive() defers entirely to pdf_born_digital_text()
for the has-real-text decision, so both callers of that predicate
(this function and RasterisedDocumentParser.parse()) always agree.
"""
settings.ARCHIVE_FILE_GENERATION = "auto" settings.ARCHIVE_FILE_GENERATION = "auto"
mocker.patch("documents.consumer.is_tagged_pdf", return_value=False) mocker.patch(
mocker.patch("documents.consumer.extract_pdf_text", return_value=extracted_text) "documents.consumer.pdf_born_digital_text",
return_value=("some text", born_digital),
)
parser = _parser_instance(can_produce=True, requires_rendition=False) parser = _parser_instance(can_produce=True, requires_rendition=False)
assert ( assert (
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf")) should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
is expected is expected
) )
def test_tagged_pdf_skips_archive_in_auto_mode(
self,
mocker: MockerFixture,
settings,
) -> None:
"""Tagged PDFs (e.g. Word exports) with real text are treated as born-digital, even below PDF_TEXT_MIN_LENGTH."""
settings.ARCHIVE_FILE_GENERATION = "auto"
mocker.patch("documents.consumer.is_tagged_pdf", return_value=True)
mocker.patch("documents.consumer.extract_pdf_text", return_value="tiny")
parser = _parser_instance(can_produce=True, requires_rendition=False)
assert (
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
is False
)
def test_tagged_pdf_without_text_produces_archive(
self,
mocker: MockerFixture,
settings,
) -> None:
"""A tagged PDF with no actual extractable text (e.g. some scanner firmware) is not
trusted as born-digital the tag alone must not bypass OCR."""
settings.ARCHIVE_FILE_GENERATION = "auto"
mocker.patch("documents.consumer.is_tagged_pdf", return_value=True)
mocker.patch("documents.consumer.extract_pdf_text", return_value=None)
parser = _parser_instance(can_produce=True, requires_rendition=False)
assert (
should_produce_archive(parser, "application/pdf", Path("/tmp/doc.pdf"))
is True
)
+6 -21
View File
@@ -3,7 +3,6 @@ from __future__ import annotations
import importlib.resources import importlib.resources
import logging import logging
import os import os
import re
import shutil import shutil
import tempfile import tempfile
from pathlib import Path from pathlib import Path
@@ -25,9 +24,9 @@ from paperless.config import OcrConfig
from paperless.models import CleanChoices from paperless.models import CleanChoices
from paperless.models import ModeChoices from paperless.models import ModeChoices
from paperless.models import OutputTypeChoices from paperless.models import OutputTypeChoices
from paperless.parsers.utils import PDF_TEXT_MIN_LENGTH
from paperless.parsers.utils import extract_pdf_text from paperless.parsers.utils import extract_pdf_text
from paperless.parsers.utils import is_tagged_pdf from paperless.parsers.utils import is_born_digital_text
from paperless.parsers.utils import post_process_text
from paperless.parsers.utils import read_file_handle_unicode_errors from paperless.parsers.utils import read_file_handle_unicode_errors
from paperless.version import __full_version_str__ from paperless.version import __full_version_str__
@@ -510,10 +509,10 @@ class RasterisedDocumentParser:
if mime_type == "application/pdf": if mime_type == "application/pdf":
text_original = self.extract_text(None, document_path) text_original = self.extract_text(None, document_path)
has_text = text_original is not None and len(text_original) > 0 original_has_text = is_born_digital_text(
original_has_text = has_text and ( text_original,
is_tagged_pdf(document_path, log=self.log) document_path,
or len(text_original) > PDF_TEXT_MIN_LENGTH log=self.log,
) )
else: else:
text_original = None text_original = None
@@ -658,17 +657,3 @@ class RasterisedDocumentParser:
f"No text was found in {document_path}, the content will be empty.", f"No text was found in {document_path}, the content will be empty.",
) )
self.text = "" self.text = ""
def post_process_text(text: str | None) -> str | None:
if not text:
return None
collapsed_spaces = re.sub(r"([^\S\r\n]+)", " ", text)
no_leading_whitespace = re.sub(r"([\n\r]+)([^\S\n\r]+)", "\\1", collapsed_spaces)
no_trailing_whitespace = re.sub(r"([^\S\n\r]+)$", "", no_leading_whitespace)
# TODO: this needs a rework
# replace \0 prevents issues with saving to postgres.
# text may contain \0 when this character is present in PDF files.
return no_trailing_whitespace.strip().replace("\0", " ")
+82
View File
@@ -111,6 +111,88 @@ def extract_pdf_text(
return None return None
def post_process_text(text: str | None) -> str | None:
"""Normalize extracted PDF/OCR text: collapse whitespace, strip padding.
Returns ``None`` for ``None`` or whitespace-only input, so callers can
treat "no text" and "only layout padding" the same way.
"""
if not text:
return None
collapsed_spaces = re.sub(r"([^\S\r\n]+)", " ", text)
no_leading_whitespace = re.sub(r"([\n\r]+)([^\S\n\r]+)", "\\1", collapsed_spaces)
no_trailing_whitespace = re.sub(r"([^\S\n\r]+)$", "", no_leading_whitespace)
# replace \0 prevents issues with saving to postgres.
# text may contain \0 when this character is present in PDF files.
result = no_trailing_whitespace.strip().replace("\0", " ")
return result or None
def is_born_digital_text(
text: str | None,
path: Path,
log: logging.Logger | None = None,
) -> bool:
"""Decide whether already-extracted, normalized PDF text counts as born-digital.
This is the single source of truth for "does this PDF already have real
text", used both to decide whether to produce an archive file and to
decide whether OCR can be skipped. Both decisions must agree, or a
tagged-but-textless PDF can end up with no archive AND a forced OCR pass
(see GH #13387): raw ``pdftotext -layout`` output can be non-empty
(whitespace/form-feed padding) even when there is no real content, so
*text* must already be normalized via :func:`post_process_text`, not the
raw extraction.
Parameters
----------
text:
The normalized extracted text (or ``None``) to evaluate.
path:
Absolute path to the PDF file, used for the tagged-PDF check.
log:
Logger for warnings. Falls back to the module-level logger when omitted.
Returns
-------
bool
Whether the PDF counts as born-digital (has real text, and is either
tagged or exceeds ``PDF_TEXT_MIN_LENGTH``).
"""
if not text:
return False
return is_tagged_pdf(path, log=log) or len(text) > PDF_TEXT_MIN_LENGTH
def pdf_born_digital_text(
path: Path,
log: logging.Logger | None = None,
) -> tuple[str | None, bool]:
"""Extract a PDF's text and decide whether it should be treated as born-digital.
Convenience wrapper around :func:`is_born_digital_text` for callers that
don't already have the PDF's text extracted (e.g. the archive-generation
decision, which runs before any parser has touched the file).
Parameters
----------
path:
Absolute path to the PDF file.
log:
Logger for warnings. Falls back to the module-level logger when omitted.
Returns
-------
tuple[str | None, bool]
The normalized extracted text (or ``None``), and whether the PDF
counts as born-digital.
"""
text = post_process_text(extract_pdf_text(path, log=log))
return text, is_born_digital_text(text, path, log=log)
def read_file_handle_unicode_errors( def read_file_handle_unicode_errors(
filepath: Path, filepath: Path,
log: logging.Logger | None = None, log: logging.Logger | None = None,
+17
View File
@@ -36,6 +36,23 @@ def samples_dir() -> Path:
return (Path(__file__).parent / "samples").resolve() return (Path(__file__).parent / "samples").resolve()
@pytest.fixture(scope="session")
def tagged_no_text_pdf_file(samples_dir: Path) -> Path:
"""Path to a tagged PDF whose only "text" is pdftotext layout padding.
Reproduces GH #13387: ``/MarkInfo /Marked true`` is set, but the only
extractable content is a form-feed byte, not real text. Lives here
rather than in parsers/conftest.py so both parser tests and
paperless/tests/test_parser_utils.py can use it.
Returns
-------
Path
Absolute path to ``tesseract/tagged-but-no-text.pdf``.
"""
return samples_dir / "tesseract" / "tagged-but-no-text.pdf"
@pytest.fixture(autouse=True) @pytest.fixture(autouse=True)
def clean_registry() -> Generator[None, None, None]: def clean_registry() -> Generator[None, None, None]:
"""Reset the parser registry before and after every test. """Reset the parser registry before and after every test.
@@ -21,7 +21,7 @@ from documents.parsers import run_convert
from paperless.models import ModeChoices from paperless.models import ModeChoices
from paperless.parsers import ParserProtocol from paperless.parsers import ParserProtocol
from paperless.parsers.tesseract import RasterisedDocumentParser from paperless.parsers.tesseract import RasterisedDocumentParser
from paperless.parsers.tesseract import post_process_text from paperless.parsers.utils import is_tagged_pdf
if TYPE_CHECKING: if TYPE_CHECKING:
from pathlib import Path from pathlib import Path
@@ -151,36 +151,6 @@ class TestRasterisedDocumentParserLifecycle:
assert tempdir is not None and not tempdir.exists() assert tempdir is not None and not tempdir.exists()
# ---------------------------------------------------------------------------
# post_process_text
# ---------------------------------------------------------------------------
class TestPostProcessText:
@pytest.mark.parametrize(
("source", "expected"),
[
pytest.param(
"simple string",
"simple string",
id="collapse-spaces",
),
pytest.param(
"simple newline\n testing string",
"simple newline\ntesting string",
id="preserve-newline",
),
pytest.param(
"utf-8 строка с пробелами в конце ", # noqa: RUF001
"utf-8 строка с пробелами в конце", # noqa: RUF001
id="utf8-trailing-spaces",
),
],
)
def test_post_process_text(self, source: str, expected: str) -> None:
assert post_process_text(source) == expected
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# Page count # Page count
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@@ -910,25 +880,25 @@ class TestSkipArchive:
self, self,
mocker: MockerFixture, mocker: MockerFixture,
tesseract_parser: RasterisedDocumentParser, tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path, tagged_no_text_pdf_file: Path,
) -> None: ) -> None:
""" """
GIVEN: GIVEN:
- A PDF that reports itself as tagged (/MarkInfo /Marked true) but - A real PDF that reports itself as tagged (/MarkInfo /Marked
has no actual extractable text (some scanner firmware produces true) but whose only pdftotext output is layout padding (a
this see GitHub issue #13349) lone form-feed byte), not real text (see GitHub issue #13387,
originally reported against #13349's tagged-PDF handling)
- Mode: auto, produce_archive=False - Mode: auto, produce_archive=False
WHEN: WHEN:
- Document is parsed - Document is parsed
THEN: THEN:
- The tag alone is not trusted as "has text"; OCRmyPDF still runs - The tag alone is not trusted as "has text"; OCRmyPDF still runs
""" """
assert is_tagged_pdf(tagged_no_text_pdf_file) is True
tesseract_parser.settings.mode = ModeChoices.AUTO tesseract_parser.settings.mode = ModeChoices.AUTO
mocker.patch("paperless.parsers.tesseract.is_tagged_pdf", return_value=True)
mocker.patch.object(tesseract_parser, "extract_text", return_value=None)
mock_ocr = mocker.patch("ocrmypdf.ocr") mock_ocr = mocker.patch("ocrmypdf.ocr")
tesseract_parser.parse( tesseract_parser.parse(
tesseract_samples_dir / "multi-page-images.pdf", tagged_no_text_pdf_file,
"application/pdf", "application/pdf",
produce_archive=False, produce_archive=False,
) )
+110
View File
@@ -4,10 +4,18 @@ from __future__ import annotations
import codecs import codecs
from pathlib import Path from pathlib import Path
from typing import TYPE_CHECKING
import pytest
from paperless.parsers.utils import is_tagged_pdf from paperless.parsers.utils import is_tagged_pdf
from paperless.parsers.utils import pdf_born_digital_text
from paperless.parsers.utils import post_process_text
from paperless.parsers.utils import read_file_handle_unicode_errors from paperless.parsers.utils import read_file_handle_unicode_errors
if TYPE_CHECKING:
from pytest_mock import MockerFixture
SAMPLES = Path(__file__).parent / "samples" / "tesseract" SAMPLES = Path(__file__).parent / "samples" / "tesseract"
@@ -60,3 +68,105 @@ class TestIsTaggedPdf:
bad = tmp_path / "bad.pdf" bad = tmp_path / "bad.pdf"
bad.write_bytes(b"not a pdf") bad.write_bytes(b"not a pdf")
assert is_tagged_pdf(bad) is False assert is_tagged_pdf(bad) is False
class TestPostProcessText:
@pytest.mark.parametrize(
("source", "expected"),
[
pytest.param(
"simple string",
"simple string",
id="collapse-spaces",
),
pytest.param(
"simple newline\n testing string",
"simple newline\ntesting string",
id="preserve-newline",
),
pytest.param(
"utf-8 строка с пробелами в конце ", # noqa: RUF001
"utf-8 строка с пробелами в конце", # noqa: RUF001
id="utf8-trailing-spaces",
),
pytest.param(None, None, id="none-input"),
pytest.param("", None, id="empty-string"),
pytest.param(" \n\x0c \n ", None, id="whitespace-and-formfeed-only"),
],
)
def test_post_process_text(
self,
source: str | None,
expected: str | None,
) -> None:
assert post_process_text(source) == expected
class TestPdfBornDigitalText:
"""Regression coverage for GH #13387.
should_produce_archive() and RasterisedDocumentParser.parse() must agree
on whether a PDF has real text, so both go through this one function.
"""
@pytest.mark.parametrize(
("extracted", "tagged", "expected_text", "expected_born_digital"),
[
pytest.param("tiny", True, "tiny", True, id="tagged-with-real-text"),
pytest.param("tiny", False, "tiny", False, id="untagged-below-min-length"),
pytest.param(
"x" * 51,
False,
"x" * 51,
True,
id="untagged-above-min-length",
),
pytest.param(None, True, None, False, id="tagged-but-no-text"),
],
)
def test_born_digital_decision(
self,
mocker: MockerFixture,
tmp_path: Path,
extracted: str | None,
tagged: bool, # noqa: FBT001
expected_text: str | None,
expected_born_digital: bool, # noqa: FBT001
) -> None:
"""
GIVEN:
- A PDF whose pdftotext output and /MarkInfo tag status vary
WHEN:
- pdf_born_digital_text() is called
THEN:
- The normalized text and born-digital verdict match; the tag
alone never counts as "has text"
"""
mocker.patch(
"paperless.parsers.utils.extract_pdf_text",
return_value=extracted,
)
mocker.patch("paperless.parsers.utils.is_tagged_pdf", return_value=tagged)
text, born_digital = pdf_born_digital_text(tmp_path / "doc.pdf")
assert text == expected_text
assert born_digital is expected_born_digital
def test_tagged_but_textless_pdf_is_not_born_digital(
self,
tagged_no_text_pdf_file: Path,
) -> None:
"""
GIVEN:
- A real PDF that is tagged (/MarkInfo /Marked true) but whose
only "text" is layout padding (a stray form-feed byte)
WHEN:
- pdf_born_digital_text() is called with no mocking
THEN:
- The normalized text is None and the PDF is not treated as
born-digital. The raw, unnormalized pdftotext output is
non-empty for this file, which is exactly what caused the
archive decision to disagree with the OCR decision in #13387.
"""
text, born_digital = pdf_born_digital_text(tagged_no_text_pdf_file)
assert text is None
assert born_digital is False