Compare commits

...
Author SHA1 Message Date
Trenton HandGitHub d72ad0087d Merge branch 'dev' into feature-export-zip-compression 2026-08-12 12:39:27 -07:00
GitHub Actions 4789fe9a52 Auto translate strings 2026-08-12 19:04:59 +00:00
shamoonandGitHub e150c8c7c0 Enhancement: customizable icons for saved views (#13388) 2026-08-12 19:03:24 +00:00
stumpylog a19ef32a7f Cover some of the weird cases 2026-08-12 11:24:38 -07:00
stumpylog db75ef5d0d Cleanup 2026-08-12 10:57:02 -07:00
stumpylog f960874e58 Claude suggested test combinations with subtests 2026-08-12 10:47:29 -07:00
stumpylog b5ba46adb5 Feature: Allow configuring the compression type and compression levels during export
Building on the zip export improvements, this now allows users to further configure the
zip to fit their needs.  A simple stored zip for speed, or a high compression zstd for
the smallest archive.  Full validation of the method and levels at the command line
2026-08-12 10:31:35 -07:00
JaydenandGitHub 879cd4a30a Enhancement: Add --url argument to document_fuzzy_match to improve output (#13123)
Added a new --url argument to specify the base URL of the Paperless instance, allowing matched documents to be displayed as clickable links. Updated the logic to fetch document titles based on the presence of the base URL.
2026-08-12 08:23:05 -07:00
Trenton HandGitHub 6a02b87dde Feature: Updates remote OCR parser to respect the OCR mode setting (#13408)
* Have the remote parser respect the provided produce_archive_file setting, as already determined via the consumer checks

* Updates the documentation to be correct about the respecting now

* merge conflict fixing
2026-08-11 19:23:46 +00:00
Trenton HandGitHub 59a2651804 Fix: pass document chat queries as a QuerySet instead of a materialized list (#13638)
In tracemalloc based profiling, not materializing the whole Document list
reduced memory to approximately 20% of the baseline, with a peak memory
that scaled with the library size.  Now, the lazt queryset is used and only
the needed pk value is actually contributing to memory
2026-08-11 15:25:08 +00:00
GitHub Actions a99f63e059 Auto translate strings 2026-08-11 14:11:20 +00:00
shamoonandGitHub 939cb52f6e Fix: add pagination to saved views management page (#13646) 2026-08-11 07:08:30 -07:00
shamoonandGitHub 855669ddf9 Fix: fixes for workflow assign custom field values (#13630) 2026-08-10 07:38:07 -07:00
47 changed files with 2986 additions and 872 deletions
+16
View File
@@ -299,6 +299,8 @@ optional arguments:
-sm, --split-manifest -sm, --split-manifest
-z, --zip -z, --zip
-zn, --zip-name -zn, --zip-name
--zip-compression
--zip-compression-level
--data-only --data-only
--no-progress-bar --no-progress-bar
--passphrase --passphrase
@@ -361,6 +363,19 @@ If `-z` or `--zip` is provided, the export will be a zip file
in the target directory, named according to the current local date or the in the target directory, named according to the current local date or the
value set in `-zn` or `--zip-name`. value set in `-zn` or `--zip-name`.
The compression method for the zip can be set with `--zip-compression`
(`stored`, `deflated` (default), `bzip2`, `lzma`, or `zstd`) and tuned with
`--zip-compression-level` (deflated: 09, bzip2: 19, zstd: -2222; ignored
for `stored` and `lzma`). Both options require `--zip`.
!!! warning
`zstd` compression requires Python 3.14 or newer on **both** the machine
creating the export and any machine importing it. An archive compressed with
`zstd` (or `lzma`/`bzip2` where those modules are unavailable) cannot be
imported on a runtime that lacks the codec; the importer will refuse it with
a clear error. The default `deflated` is universally readable.
If `--data-only` is provided, only the database will be exported. This option is intended If `--data-only` is provided, only the database will be exported. This option is intended
to facilitate database upgrades without needing to clean documents and thumbnails from the media directory. to facilitate database upgrades without needing to clean documents and thumbnails from the media directory.
@@ -699,6 +714,7 @@ document_fuzzy_match [--ratio] [--processes N]
| --ratio | No | 85.0 | a number between 0 and 100, setting how similar a document must be for it to be reported. Higher numbers mean more similarity. | | --ratio | No | 85.0 | a number between 0 and 100, setting how similar a document must be for it to be reported. Higher numbers mean more similarity. |
| --processes | No | 1/4 of system cores | Number of processes to use for matching. Setting 1 disables multiple processes | | --processes | No | 1/4 of system cores | Number of processes to use for matching. Setting 1 disables multiple processes |
| --delete | No | False | If provided, one document of a matched pair above the ratio will be deleted. | | --delete | No | False | If provided, one document of a matched pair above the ratio will be deleted. |
| --url | No | blank | If an instance URL is provided, the output table will show URLs to each documents instead of the document ID and name. |
!!! warning !!! warning
+5 -4
View File
@@ -948,10 +948,11 @@ for display in the web interface.
!!! note !!! note
The **remote OCR parser** (Azure AI) always produces a searchable The **remote OCR parser** (Azure AI) also honors this setting: when
PDF and stores it as the archive copy, regardless of this setting. no archive is requested (`never`, or `auto` with a born-digital PDF),
`ARCHIVE_FILE_GENERATION=never` has no effect when the remote the remote engine is skipped entirely and locally-extracted text is
parser handles a document. used instead, avoiding an unnecessary API call and a duplicate text
layer.
#### [`PAPERLESS_OCR_CLEAN=<mode>`](#PAPERLESS_OCR_CLEAN) {#PAPERLESS_OCR_CLEAN} #### [`PAPERLESS_OCR_CLEAN=<mode>`](#PAPERLESS_OCR_CLEAN) {#PAPERLESS_OCR_CLEAN}
+5 -4
View File
@@ -187,10 +187,11 @@ PAPERLESS_ARCHIVE_FILE_GENERATION=auto
### Remote OCR parser ### Remote OCR parser
If you use the **remote OCR parser** (Azure AI), note that it always produces a If you use the **remote OCR parser** (Azure AI), `ARCHIVE_FILE_GENERATION` is
searchable PDF and stores it as the archive copy. `ARCHIVE_FILE_GENERATION=never` honored the same way as for the local engine: when no archive is requested
has no effect for documents handled by the remote parser - the archive is produced (`never`, or `auto` with a born-digital PDF), the remote engine is skipped
unconditionally by the remote engine. entirely and locally-extracted text is used instead, avoiding an unnecessary
API call and a duplicate text layer.
## Search Index (Whoosh -> Tantivy) ## Search Index (Whoosh -> Tantivy)
+3 -1
View File
@@ -576,7 +576,9 @@ The following workflow action types are available:
- Tags, correspondent, document type and storage path - Tags, correspondent, document type and storage path
- Document owner - Document owner
- View and / or edit permissions to users or groups - View and / or edit permissions to users or groups
- Custom fields. Note that no value for the field will be set - Custom fields, optionally with a value. If no value is set, the field is only added to the
document and any value it may already have is left untouched. If a value is set, it will
overwrite an existing value of that field on the document.
##### Removal {#workflow-action-removal} ##### Removal {#workflow-action-removal}
+491 -99
View File
File diff suppressed because it is too large Load Diff
@@ -111,7 +111,7 @@
routerLinkActive="active" (click)="closeMenu()" [ngbPopover]="view.name" routerLinkActive="active" (click)="closeMenu()" [ngbPopover]="view.name"
[disablePopover]="!slimSidebarEnabled" placement="end" container="body" triggers="mouseenter:mouseleave" [disablePopover]="!slimSidebarEnabled" placement="end" container="body" triggers="mouseenter:mouseleave"
popoverClass="popover-slim"> popoverClass="popover-slim">
<i-bs class="me-2" name="funnel"></i-bs><span><div class="d-inline-flex view-name"><span class="overflow-hidden" [class.text-wrap]="!slimSidebarEnabled">{{view.name}}</span></div> <i-bs class="me-2" [name]="view.icon || 'funnel'"></i-bs><span><div class="d-inline-flex view-name"><span class="overflow-hidden" [class.text-wrap]="!slimSidebarEnabled">{{view.name}}</span></div>
@if (showSidebarCounts && !slimSidebarEnabled) { @if (showSidebarCounts && !slimSidebarEnabled) {
<span class="badge bg-info text-dark ms-2 d-inline">{{ savedViewService.getDocumentCount(view) }}</span> <span class="badge bg-info text-dark ms-2 d-inline">{{ savedViewService.getDocumentCount(view) }}</span>
} }
@@ -52,10 +52,10 @@ describe('CustomFieldsValuesComponent', () => {
}) })
it('should set selectedFields and map values correctly', () => { it('should set selectedFields and map values correctly', () => {
component.value = { 1: 'value1' } component.value = { 1: 'value1', 3: 0, 4: false }
component.selectedFields = [1, 2] component.selectedFields = [1, 2, 3, 4]
expect(component.selectedFields).toEqual([1, 2]) expect(component.selectedFields).toEqual([1, 2, 3, 4])
expect(component.value).toEqual({ 1: 'value1', 2: null }) expect(component.value).toEqual({ 1: 'value1', 2: null, 3: 0, 4: false })
}) })
it('should return the correct custom field by id', () => { it('should return the correct custom field by id', () => {
@@ -77,7 +77,7 @@ export class CustomFieldsValuesComponent extends AbstractInputComponent<Object>
this._selectedFields = newFields this._selectedFields = newFields
// map the selected fields to an object with field_id as key and value as value // map the selected fields to an object with field_id as key and value as value
this.value = newFields.reduce((acc, fieldId) => { this.value = newFields.reduce((acc, fieldId) => {
acc[fieldId] = this.value?.[fieldId] || null acc[fieldId] = this.value?.[fieldId] ?? null
return acc return acc
}, {}) }, {})
this.onChange(this.value) this.onChange(this.value)
@@ -36,7 +36,16 @@
(focus)="clearLastSearchTerm()" (focus)="clearLastSearchTerm()"
(clear)="clearLastSearchTerm()" (clear)="clearLastSearchTerm()"
(blur)="onBlur()"> (blur)="onBlur()">
<ng-template ng-label-tmp let-item="item">
@if (iconField && item[iconField]) {
<i-bs class="me-2" [name]="item[iconField]"></i-bs>
}
<span [title]="item[bindLabel]">{{item[bindLabel]}}</span>
</ng-template>
<ng-template ng-option-tmp let-item="item"> <ng-template ng-option-tmp let-item="item">
@if (iconField && item[iconField]) {
<i-bs class="me-2" [name]="item[iconField]"></i-bs>
}
<span [title]="item[bindLabel]">{{item[bindLabel]}}</span> <span [title]="item[bindLabel]">{{item[bindLabel]}}</span>
</ng-template> </ng-template>
</ng-select> </ng-select>
@@ -35,7 +35,7 @@ import { AbstractInputComponent } from '../abstract-input'
NgxBootstrapIconsModule, NgxBootstrapIconsModule,
], ],
}) })
export class SelectComponent extends AbstractInputComponent<number> { export class SelectComponent extends AbstractInputComponent<number | string> {
constructor() { constructor() {
super() super()
this.addItemRef = this.addItem.bind(this) this.addItemRef = this.addItem.bind(this)
@@ -100,6 +100,9 @@ export class SelectComponent extends AbstractInputComponent<number> {
@Input() @Input()
bindLabel: string = 'name' bindLabel: string = 'name'
@Input()
iconField: string
public searchFn = (term: string, item: any): boolean => public searchFn = (term: string, item: any): boolean =>
matchesSearchText(item?.[this.bindLabel], term) matchesSearchText(item?.[this.bindLabel], term)
@@ -1,6 +1,7 @@
<pngx-widget-frame <pngx-widget-frame
*pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.Document }" *pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.Document }"
[title]="savedView.name" [title]="savedView.name"
[titleIcon]="savedView.icon || 'funnel'"
[loading]="false" [loading]="false"
[draggable]="savedView" [draggable]="savedView"
> >
@@ -8,7 +8,12 @@
<i-bs name="grip-vertical"></i-bs> <i-bs name="grip-vertical"></i-bs>
</div> </div>
} }
<h6 class="card-title mb-0">{{title()}}</h6> <h6 class="card-title mb-0">
@if (titleIcon()) {
<i-bs class="me-2" [name]="titleIcon()"></i-bs>
}
{{title()}}
</h6>
<ng-content select="[title-badge]"></ng-content> <ng-content select="[title-badge]"></ng-content>
@if (badge() !== null && badge() !== undefined) { @if (badge() !== null && badge() !== undefined) {
<span class="badge bg-info text-dark ms-2">{{badge()}}</span> <span class="badge bg-info text-dark ms-2">{{badge()}}</span>
@@ -16,6 +16,8 @@ export class WidgetFrameComponent implements AfterViewInit {
title = input<string>() title = input<string>()
titleIcon = input<string>()
draggable = input<any>() draggable = input<any>()
cardless = input(false) cardless = input(false)
@@ -97,7 +97,9 @@
<div class="dropdown-menu shadow dropdown-menu-right" ngbDropdownMenu> <div class="dropdown-menu shadow dropdown-menu-right" ngbDropdownMenu>
@if (!list.activeSavedViewId) { @if (!list.activeSavedViewId) {
@for (view of savedViewService.allViews; track view) { @for (view of savedViewService.allViews; track view) {
<button ngbDropdownItem (click)="loadViewConfig(view.id)">{{view.name}}</button> <button ngbDropdownItem (click)="loadViewConfig(view.id)">
<i-bs class="me-2" [name]="view.icon || 'funnel'"></i-bs>{{view.name}}
</button>
} }
@if (savedViewService.allViews.length > 0) { @if (savedViewService.allViews.length > 0) {
<div class="dropdown-divider"></div> <div class="dropdown-divider"></div>
@@ -457,6 +457,7 @@ export class DocumentListComponent
modal.componentInstance.buttonsEnabled.set(false) modal.componentInstance.buttonsEnabled.set(false)
let savedView: SavedView = { let savedView: SavedView = {
name: formValue.name, name: formValue.name,
icon: formValue.icon,
filter_rules: this.list.filterRules, filter_rules: this.list.filterRules,
sort_reverse: this.list.sortReverse, sort_reverse: this.list.sortReverse,
sort_field: this.list.sortField, sort_field: this.list.sortField,
@@ -6,6 +6,14 @@
</div> </div>
<div class="modal-body"> <div class="modal-body">
<pngx-input-text i18n-title title="Name" formControlName="name" [error]="error()?.name" autocomplete="off"></pngx-input-text> <pngx-input-text i18n-title title="Name" formControlName="name" [error]="error()?.name" autocomplete="off"></pngx-input-text>
<pngx-input-select
i18n-title
title="Icon"
formControlName="icon"
[items]="savedViewIcons"
iconField="icon"
[error]="error()?.icon">
</pngx-input-select>
<pngx-input-check i18n-title title="Show in sidebar" formControlName="showInSideBar"></pngx-input-check> <pngx-input-check i18n-title title="Show in sidebar" formControlName="showInSideBar"></pngx-input-check>
<pngx-input-check i18n-title title="Show on dashboard" formControlName="showOnDashboard"></pngx-input-check> <pngx-input-check i18n-title title="Show on dashboard" formControlName="showOnDashboard"></pngx-input-check>
<pngx-permissions-form accordion="true" formControlName="permissions_form"></pngx-permissions-form> <pngx-permissions-form accordion="true" formControlName="permissions_form"></pngx-permissions-form>
@@ -9,6 +9,7 @@ import { CheckComponent } from '../../common/input/check/check.component'
import { PermissionsFormComponent } from '../../common/input/permissions/permissions-form/permissions-form.component' import { PermissionsFormComponent } from '../../common/input/permissions/permissions-form/permissions-form.component'
import { PermissionsGroupComponent } from '../../common/input/permissions/permissions-group/permissions-group.component' import { PermissionsGroupComponent } from '../../common/input/permissions/permissions-group/permissions-group.component'
import { PermissionsUserComponent } from '../../common/input/permissions/permissions-user/permissions-user.component' import { PermissionsUserComponent } from '../../common/input/permissions/permissions-user/permissions-user.component'
import { SelectComponent } from '../../common/input/select/select.component'
import { TextComponent } from '../../common/input/text/text.component' import { TextComponent } from '../../common/input/text/text.component'
import { SaveViewConfigDialogComponent } from './save-view-config-dialog.component' import { SaveViewConfigDialogComponent } from './save-view-config-dialog.component'
@@ -40,6 +41,7 @@ describe('SaveViewConfigDialogComponent', () => {
ReactiveFormsModule, ReactiveFormsModule,
SaveViewConfigDialogComponent, SaveViewConfigDialogComponent,
TextComponent, TextComponent,
SelectComponent,
CheckComponent, CheckComponent,
PermissionsFormComponent, PermissionsFormComponent,
PermissionsUserComponent, PermissionsUserComponent,
@@ -63,6 +65,7 @@ describe('SaveViewConfigDialogComponent', () => {
expect(component.defaultName()).toEqual(name) expect(component.defaultName()).toEqual(name)
expect(result).toEqual({ expect(result).toEqual({
name, name,
icon: 'funnel',
showInSideBar: false, showInSideBar: false,
showOnDashboard: false, showOnDashboard: false,
}) })
@@ -94,6 +97,7 @@ describe('SaveViewConfigDialogComponent', () => {
component.save() component.save()
expect(result).toEqual({ expect(result).toEqual({
name, name,
icon: 'funnel',
showInSideBar: true, showInSideBar: true,
showOnDashboard: true, showOnDashboard: true,
}) })
@@ -113,6 +117,7 @@ describe('SaveViewConfigDialogComponent', () => {
component.save() component.save()
expect(result).toEqual({ expect(result).toEqual({
name: '', name: '',
icon: 'funnel',
showInSideBar: false, showInSideBar: false,
showOnDashboard: false, showOnDashboard: false,
permissions_form: permissions, permissions_form: permissions,
@@ -13,9 +13,14 @@ import {
ReactiveFormsModule, ReactiveFormsModule,
} from '@angular/forms' } from '@angular/forms'
import { NgbActiveModal } from '@ng-bootstrap/ng-bootstrap' import { NgbActiveModal } from '@ng-bootstrap/ng-bootstrap'
import {
DEFAULT_SAVED_VIEW_ICON,
SAVED_VIEW_ICONS,
} from 'src/app/data/saved-view-icons'
import { User } from 'src/app/data/user' import { User } from 'src/app/data/user'
import { CheckComponent } from '../../common/input/check/check.component' import { CheckComponent } from '../../common/input/check/check.component'
import { PermissionsFormComponent } from '../../common/input/permissions/permissions-form/permissions-form.component' import { PermissionsFormComponent } from '../../common/input/permissions/permissions-form/permissions-form.component'
import { SelectComponent } from '../../common/input/select/select.component'
import { TextComponent } from '../../common/input/text/text.component' import { TextComponent } from '../../common/input/text/text.component'
@Component({ @Component({
@@ -24,6 +29,7 @@ import { TextComponent } from '../../common/input/text/text.component'
styleUrls: ['./save-view-config-dialog.component.scss'], styleUrls: ['./save-view-config-dialog.component.scss'],
imports: [ imports: [
CheckComponent, CheckComponent,
SelectComponent,
TextComponent, TextComponent,
PermissionsFormComponent, PermissionsFormComponent,
FormsModule, FormsModule,
@@ -41,6 +47,7 @@ export class SaveViewConfigDialogComponent implements OnInit {
public saveClicked = new EventEmitter() public saveClicked = new EventEmitter()
users: User[] users: User[]
readonly savedViewIcons = SAVED_VIEW_ICONS
setDefaultName(value: string) { setDefaultName(value: string) {
this.defaultName.set(value) this.defaultName.set(value)
@@ -49,6 +56,7 @@ export class SaveViewConfigDialogComponent implements OnInit {
saveViewConfigForm = new FormGroup({ saveViewConfigForm = new FormGroup({
name: new FormControl(''), name: new FormControl(''),
icon: new FormControl(DEFAULT_SAVED_VIEW_ICON),
showInSideBar: new FormControl(false), showInSideBar: new FormControl(false),
showOnDashboard: new FormControl(false), showOnDashboard: new FormControl(false),
permissions_form: new FormControl(null), permissions_form: new FormControl(null),
@@ -65,6 +73,7 @@ export class SaveViewConfigDialogComponent implements OnInit {
const formValue = this.saveViewConfigForm.value const formValue = this.saveViewConfigForm.value
const saveViewConfig = { const saveViewConfig = {
name: formValue.name, name: formValue.name,
icon: formValue.icon,
showInSideBar: formValue.showInSideBar, showInSideBar: formValue.showInSideBar,
showOnDashboard: formValue.showOnDashboard, showOnDashboard: formValue.showOnDashboard,
} }
@@ -7,15 +7,24 @@
</pngx-page-header> </pngx-page-header>
<form [formGroup]="savedViewsForm" (ngSubmit)="save()"> <form [formGroup]="savedViewsForm" (ngSubmit)="save()">
<ul class="list-group mb-3" formGroupName="savedViews"> <ul class="list-group mb-3" formGroupName="savedViews">
@for (view of savedViews(); track view) { @for (view of pagedSavedViews(); track view) {
<li class="list-group-item py-3"> <li class="list-group-item py-3">
<div [formGroupName]="view.id"> <div [formGroupName]="view.id">
<div class="row"> <div class="row">
<div class="col"> <div class="col-md">
<pngx-input-text title="Name" formControlName="name"></pngx-input-text> <pngx-input-text title="Name" formControlName="name"></pngx-input-text>
</div> </div>
<div class="col-md">
<pngx-input-select
i18n-title
title="Icon"
formControlName="icon"
[items]="savedViewIcons"
iconField="icon">
</pngx-input-select>
</div>
@if (canSaveSettings) { @if (canSaveSettings) {
<div class="col"> <div class="col-md">
<div class="form-check form-switch mt-3"> <div class="form-check form-switch mt-3">
<input type="checkbox" class="form-check-input" id="show_on_dashboard_{{view.id}}" formControlName="show_on_dashboard"> <input type="checkbox" class="form-check-input" id="show_on_dashboard_{{view.id}}" formControlName="show_on_dashboard">
<label class="form-check-label" for="show_on_dashboard_{{view.id}}" i18n>Show on dashboard</label> <label class="form-check-label" for="show_on_dashboard_{{view.id}}" i18n>Show on dashboard</label>
@@ -81,6 +90,11 @@
} }
</ul> </ul>
<div class="d-flex align-items-center mb-3">
<button type="button" (click)="reset()" class="btn btn-outline-secondary mb-2" [disabled]="(isDirty$ | async) === false" i18n>Cancel</button> <button type="button" (click)="reset()" class="btn btn-outline-secondary mb-2" [disabled]="(isDirty$ | async) === false" i18n>Cancel</button>
<button type="submit" class="btn btn-primary ms-2 mb-2" [disabled]="(isDirty$ | async) === false" i18n>Save</button> <button type="submit" class="btn btn-primary ms-2 mb-2" [disabled]="(isDirty$ | async) === false" i18n>Save</button>
@if (savedViews()?.length > pageSize) {
<ngb-pagination class="ms-auto" [pageSize]="pageSize" [collectionSize]="savedViews().length" [page]="page()" [maxSize]="5" (pageChange)="page.set($event)" size="sm" aria-label="Pagination"></ngb-pagination>
}
</div>
</form> </form>
@@ -4,6 +4,7 @@ import { provideHttpClientTesting } from '@angular/common/http/testing'
import { signal } from '@angular/core' import { signal } from '@angular/core'
import { ComponentFixture, TestBed } from '@angular/core/testing' import { ComponentFixture, TestBed } from '@angular/core/testing'
import { FormsModule, ReactiveFormsModule } from '@angular/forms' import { FormsModule, ReactiveFormsModule } from '@angular/forms'
import { By } from '@angular/platform-browser'
import { NgbModal, NgbModule } from '@ng-bootstrap/ng-bootstrap' import { NgbModal, NgbModule } from '@ng-bootstrap/ng-bootstrap'
import { NgxBootstrapIconsModule, allIcons } from 'ngx-bootstrap-icons' import { NgxBootstrapIconsModule, allIcons } from 'ngx-bootstrap-icons'
import { Subject, of, throwError } from 'rxjs' import { Subject, of, throwError } from 'rxjs'
@@ -25,8 +26,20 @@ import { PageHeaderComponent } from '../../common/page-header/page-header.compon
import { SavedViewsComponent } from './saved-views.component' import { SavedViewsComponent } from './saved-views.component'
const savedViews = [ const savedViews = [
{ id: 1, name: 'view1', show_in_sidebar: true, show_on_dashboard: true }, {
{ id: 2, name: 'view2', show_in_sidebar: false, show_on_dashboard: false }, id: 1,
name: 'view1',
icon: 'archive',
show_in_sidebar: true,
show_on_dashboard: true,
},
{
id: 2,
name: 'view2',
icon: 'funnel',
show_in_sidebar: false,
show_on_dashboard: false,
},
] ]
describe('SavedViewsComponent', () => { describe('SavedViewsComponent', () => {
@@ -157,6 +170,24 @@ describe('SavedViewsComponent', () => {
expect(patchBody.show_in_sidebar).toBeUndefined() expect(patchBody.show_in_sidebar).toBeUndefined()
}) })
it('should persist a changed icon', () => {
const patchSpy = jest.spyOn(savedViewService, 'patchMany')
const view = savedViews[0]
const iconControl = component.savedViewsForm
.get('savedViews')
.get(view.id.toString())
.get('icon')
iconControl.setValue('bell')
iconControl.markAsDirty()
component.save()
expect(patchSpy.mock.calls[0][0][0]).toMatchObject({
id: view.id,
icon: 'bell',
})
})
it('should persist visibility changes to user settings', () => { it('should persist visibility changes to user settings', () => {
const patchSpy = jest.spyOn(savedViewService, 'patchMany') const patchSpy = jest.spyOn(savedViewService, 'patchMany')
const updateVisibilitySpy = jest const updateVisibilitySpy = jest
@@ -222,6 +253,44 @@ describe('SavedViewsComponent', () => {
).toEqual(view.show_on_dashboard) ).toEqual(view.show_on_dashboard)
}) })
it('should page saved views, clamp the page if views are removed', () => {
const manyViews = Array.from({ length: 30 }, (_, i) => ({
id: i + 1,
name: `view${i + 1}`,
})) as SavedView[]
const listSpy = jest.spyOn(savedViewService, 'list').mockReturnValue(
of({
all: manyViews.map((v) => v.id),
count: manyViews.length,
results: manyViews.concat([]),
})
)
component.ngOnInit()
fixture.detectChanges()
expect(listSpy).toHaveBeenCalledWith(1, 100000, null, false, {
full_perms: true,
})
expect(component.pagedSavedViews()).toHaveLength(25)
expect(fixture.debugElement.query(By.css('ngb-pagination'))).not.toBeNull()
// all views have controls, not just the current page
expect(
Object.keys(component.savedViewsForm.get('savedViews').value)
).toHaveLength(30)
component.page.set(2)
expect(component.pagedSavedViews()).toHaveLength(5)
listSpy.mockReturnValue(
of({
all: manyViews.slice(0, 25).map((v) => v.id),
count: 25,
results: manyViews.slice(0, 25),
})
)
component.ngOnInit()
expect(component.page()).toEqual(1)
})
it('should support editing permissions', () => { it('should support editing permissions', () => {
const confirmClicked = new Subject<any>() const confirmClicked = new Subject<any>()
const modalRef = { const modalRef = {
@@ -1,18 +1,29 @@
import { AsyncPipe } from '@angular/common' import { AsyncPipe } from '@angular/common'
import { Component, OnDestroy, OnInit, inject, signal } from '@angular/core' import {
Component,
OnDestroy,
OnInit,
computed,
inject,
signal,
} from '@angular/core'
import { import {
FormControl, FormControl,
FormGroup, FormGroup,
FormsModule, FormsModule,
ReactiveFormsModule, ReactiveFormsModule,
} from '@angular/forms' } from '@angular/forms'
import { NgbModal } from '@ng-bootstrap/ng-bootstrap' import { NgbModal, NgbPaginationModule } from '@ng-bootstrap/ng-bootstrap'
import { dirtyCheck } from '@ngneat/dirty-check-forms' import { dirtyCheck } from '@ngneat/dirty-check-forms'
import { NgxBootstrapIconsModule } from 'ngx-bootstrap-icons' import { NgxBootstrapIconsModule } from 'ngx-bootstrap-icons'
import { BehaviorSubject, Observable, of, switchMap, takeUntil } from 'rxjs' import { BehaviorSubject, Observable, of, switchMap, takeUntil } from 'rxjs'
import { PermissionsDialogComponent } from 'src/app/components/common/permissions-dialog/permissions-dialog.component' import { PermissionsDialogComponent } from 'src/app/components/common/permissions-dialog/permissions-dialog.component'
import { DisplayMode } from 'src/app/data/document' import { DisplayMode } from 'src/app/data/document'
import { SavedView } from 'src/app/data/saved-view' import { SavedView } from 'src/app/data/saved-view'
import {
DEFAULT_SAVED_VIEW_ICON,
SAVED_VIEW_ICONS,
} from 'src/app/data/saved-view-icons'
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive' import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
import { import {
PermissionAction, PermissionAction,
@@ -25,6 +36,7 @@ import { ToastService } from 'src/app/services/toast.service'
import { ConfirmButtonComponent } from '../../common/confirm-button/confirm-button.component' import { ConfirmButtonComponent } from '../../common/confirm-button/confirm-button.component'
import { DragDropSelectComponent } from '../../common/input/drag-drop-select/drag-drop-select.component' import { DragDropSelectComponent } from '../../common/input/drag-drop-select/drag-drop-select.component'
import { NumberComponent } from '../../common/input/number/number.component' import { NumberComponent } from '../../common/input/number/number.component'
import { SelectComponent } from '../../common/input/select/select.component'
import { TextComponent } from '../../common/input/text/text.component' import { TextComponent } from '../../common/input/text/text.component'
import { PageHeaderComponent } from '../../common/page-header/page-header.component' import { PageHeaderComponent } from '../../common/page-header/page-header.component'
import { LoadingComponentWithPermissions } from '../../loading-component/loading.component' import { LoadingComponentWithPermissions } from '../../loading-component/loading.component'
@@ -36,12 +48,14 @@ import { LoadingComponentWithPermissions } from '../../loading-component/loading
PageHeaderComponent, PageHeaderComponent,
ConfirmButtonComponent, ConfirmButtonComponent,
NumberComponent, NumberComponent,
SelectComponent,
TextComponent, TextComponent,
IfPermissionsDirective, IfPermissionsDirective,
DragDropSelectComponent, DragDropSelectComponent,
FormsModule, FormsModule,
ReactiveFormsModule, ReactiveFormsModule,
AsyncPipe, AsyncPipe,
NgbPaginationModule,
NgxBootstrapIconsModule, NgxBootstrapIconsModule,
], ],
}) })
@@ -56,8 +70,17 @@ export class SavedViewsComponent
private readonly modalService = inject(NgbModal) private readonly modalService = inject(NgbModal)
DisplayMode = DisplayMode DisplayMode = DisplayMode
readonly savedViewIcons = SAVED_VIEW_ICONS
readonly savedViews = signal<SavedView[]>(undefined) readonly savedViews = signal<SavedView[]>(undefined)
readonly page = signal(1)
public readonly pageSize = 25
// All views are loaded at init, so paging is only for display
readonly pagedSavedViews = computed(() => {
const start = (this.page() - 1) * this.pageSize
return this.savedViews()?.slice(start, start + this.pageSize)
})
private savedViewsGroup = new FormGroup({}) private savedViewsGroup = new FormGroup({})
public savedViewsForm: FormGroup = new FormGroup({ public savedViewsForm: FormGroup = new FormGroup({
savedViews: this.savedViewsGroup, savedViews: this.savedViewsGroup,
@@ -84,9 +107,11 @@ export class SavedViewsComponent
private reloadViews(): void { private reloadViews(): void {
this.loading.set(true) this.loading.set(true)
this.savedViewService this.savedViewService
.list(null, null, null, false, { full_perms: true }) .list(1, 100000, null, false, { full_perms: true })
.subscribe((r) => { .subscribe((r) => {
this.savedViews.set(r.results) this.savedViews.set(r.results)
const pageCount = Math.ceil(r.results.length / this.pageSize)
this.page.update((page) => Math.min(page, Math.max(1, pageCount)))
this.initialize() this.initialize()
}) })
} }
@@ -110,6 +135,7 @@ export class SavedViewsComponent
storeData.savedViews[view.id.toString()] = { storeData.savedViews[view.id.toString()] = {
id: view.id, id: view.id,
name: view.name, name: view.name,
icon: view.icon ?? DEFAULT_SAVED_VIEW_ICON,
show_on_dashboard: view.show_on_dashboard, show_on_dashboard: view.show_on_dashboard,
show_in_sidebar: view.show_in_sidebar, show_in_sidebar: view.show_in_sidebar,
page_size: view.page_size, page_size: view.page_size,
@@ -122,6 +148,7 @@ export class SavedViewsComponent
new FormGroup({ new FormGroup({
id: new FormControl({ value: null, disabled: !canEdit }), id: new FormControl({ value: null, disabled: !canEdit }),
name: new FormControl({ value: null, disabled: !canEdit }), name: new FormControl({ value: null, disabled: !canEdit }),
icon: new FormControl({ value: null, disabled: !canEdit }),
show_on_dashboard: new FormControl({ show_on_dashboard: new FormControl({
value: null, value: null,
disabled: false, disabled: false,
@@ -200,6 +227,7 @@ export class SavedViewsComponent
const modelFieldsChanged = const modelFieldsChanged =
group.get('name')?.dirty || group.get('name')?.dirty ||
group.get('icon')?.dirty ||
group.get('page_size')?.dirty || group.get('page_size')?.dirty ||
group.get('display_mode')?.dirty || group.get('display_mode')?.dirty ||
group.get('display_fields')?.dirty group.get('display_fields')?.dirty
+89
View File
@@ -0,0 +1,89 @@
export const DEFAULT_SAVED_VIEW_ICON = 'funnel'
export const SAVED_VIEW_ICONS = [
{ id: 'archive', name: $localize`Archive`, icon: 'archive' },
{ id: 'bank', name: $localize`Bank`, icon: 'bank' },
{ id: 'basket', name: $localize`Basket`, icon: 'basket' },
{ id: 'bell', name: $localize`Bell`, icon: 'bell' },
{ id: 'bookmark', name: $localize`Bookmark`, icon: 'bookmark' },
{ id: 'boxes', name: $localize`Boxes`, icon: 'boxes' },
{ id: 'briefcase', name: $localize`Briefcase`, icon: 'briefcase' },
{ id: 'building', name: $localize`Building`, icon: 'building' },
{ id: 'calculator', name: $localize`Calculator`, icon: 'calculator' },
{ id: 'calendar', name: $localize`Calendar`, icon: 'calendar' },
{ id: 'camera', name: $localize`Camera`, icon: 'camera' },
{
id: 'card-checklist',
name: $localize`Checklist`,
icon: 'card-checklist',
},
{ id: 'cash', name: $localize`Cash`, icon: 'cash' },
{ id: 'chat-left-text', name: $localize`Chat`, icon: 'chat-left-text' },
{ id: 'check-circle', name: $localize`Check`, icon: 'check-circle' },
{ id: 'clipboard', name: $localize`Clipboard`, icon: 'clipboard' },
{ id: 'clock-history', name: $localize`Clock`, icon: 'clock-history' },
{ id: 'credit-card', name: $localize`Credit card`, icon: 'credit-card' },
{ id: 'download', name: $localize`Download`, icon: 'download' },
{ id: 'envelope', name: $localize`Envelope`, icon: 'envelope' },
{
id: 'exclamation-triangle',
name: $localize`Warning`,
icon: 'exclamation-triangle',
},
{ id: 'file-earmark', name: $localize`File`, icon: 'file-earmark' },
{
id: 'file-earmark-check',
name: $localize`Checked file`,
icon: 'file-earmark-check',
},
{
id: 'file-earmark-lock',
name: $localize`Locked file`,
icon: 'file-earmark-lock',
},
{
id: 'file-earmark-medical',
name: $localize`Medical file`,
icon: 'file-earmark-medical',
},
{
id: 'file-earmark-person',
name: $localize`Person file`,
icon: 'file-earmark-person',
},
{
id: 'file-earmark-spreadsheet',
name: $localize`Spreadsheet`,
icon: 'file-earmark-spreadsheet',
},
{ id: 'file-text', name: $localize`Text file`, icon: 'file-text' },
{ id: 'files', name: $localize`Files`, icon: 'files' },
{ id: 'folder', name: $localize`Folder`, icon: 'folder' },
{ id: 'funnel', name: $localize`Filter`, icon: 'funnel' },
{ id: 'gear', name: $localize`Gear`, icon: 'gear' },
{ id: 'globe2', name: $localize`Globe`, icon: 'globe2' },
{ id: 'hash', name: $localize`Hash`, icon: 'hash' },
{ id: 'heart', name: $localize`Heart`, icon: 'heart' },
{ id: 'house', name: $localize`House`, icon: 'house' },
{ id: 'inbox', name: $localize`Inbox`, icon: 'inbox' },
{ id: 'journals', name: $localize`Journals`, icon: 'journals' },
{ id: 'list-task', name: $localize`Task list`, icon: 'list-task' },
{ id: 'newspaper', name: $localize`Newspaper`, icon: 'newspaper' },
{ id: 'paperclip', name: $localize`Attachment`, icon: 'paperclip' },
{ id: 'people', name: $localize`People`, icon: 'people' },
{ id: 'person', name: $localize`Person`, icon: 'person' },
{ id: 'printer', name: $localize`Printer`, icon: 'printer' },
{ id: 'receipt', name: $localize`Receipt`, icon: 'receipt' },
{ id: 'safe', name: $localize`Safe`, icon: 'safe' },
{ id: 'search', name: $localize`Search`, icon: 'search' },
{ id: 'send', name: $localize`Send`, icon: 'send' },
{ id: 'shop', name: $localize`Shop`, icon: 'shop' },
{ id: 'stack', name: $localize`Stack`, icon: 'stack' },
{ id: 'stars', name: $localize`Stars`, icon: 'stars' },
{ id: 'tag', name: $localize`Tag`, icon: 'tag' },
{ id: 'tags', name: $localize`Tags`, icon: 'tags' },
{ id: 'telephone', name: $localize`Telephone`, icon: 'telephone' },
{ id: 'truck', name: $localize`Truck`, icon: 'truck' },
{ id: 'upc-scan', name: $localize`Barcode`, icon: 'upc-scan' },
{ id: 'wallet2', name: $localize`Wallet`, icon: 'wallet2' },
]
+2
View File
@@ -5,6 +5,8 @@ import { ObjectWithPermissions } from './object-with-permissions'
export interface SavedView extends ObjectWithPermissions { export interface SavedView extends ObjectWithPermissions {
name?: string name?: string
icon?: string
show_on_dashboard?: boolean show_on_dashboard?: boolean
show_in_sidebar?: boolean show_in_sidebar?: boolean
+46
View File
@@ -35,19 +35,27 @@ import {
arrowRightShort, arrowRightShort,
arrowUpRight, arrowUpRight,
asterisk, asterisk,
bank,
basket,
bell, bell,
bodyText, bodyText,
bookmark,
boxArrowUp, boxArrowUp,
boxArrowUpRight, boxArrowUpRight,
boxes, boxes,
braces, braces,
briefcase,
building,
calculator,
calendar, calendar,
calendarEvent, calendarEvent,
calendarEventFill, calendarEventFill,
camera,
cardChecklist, cardChecklist,
cardHeading, cardHeading,
caretDown, caretDown,
caretUp, caretUp,
cash,
chatLeftText, chatLeftText,
chatSquareDots, chatSquareDots,
check, check,
@@ -65,6 +73,7 @@ import {
clipboardCheckFill, clipboardCheckFill,
clipboardFill, clipboardFill,
clockHistory, clockHistory,
creditCard,
dash, dash,
dashCircle, dashCircle,
diagram3, diagram3,
@@ -83,9 +92,12 @@ import {
fileEarmarkDiff, fileEarmarkDiff,
fileEarmarkFill, fileEarmarkFill,
fileEarmarkLock, fileEarmarkLock,
fileEarmarkMedical,
fileEarmarkMinus, fileEarmarkMinus,
fileEarmarkPerson,
fileEarmarkPlus, fileEarmarkPlus,
fileEarmarkRichtext, fileEarmarkRichtext,
fileEarmarkSpreadsheet,
fileText, fileText,
files, files,
filter, filter,
@@ -93,12 +105,15 @@ import {
folderFill, folderFill,
funnel, funnel,
gear, gear,
globe2,
google, google,
grid, grid,
gripVertical, gripVertical,
hash, hash,
hddStack, hddStack,
heart,
house, house,
inbox,
infoCircle, infoCircle,
journals, journals,
link, link,
@@ -106,7 +121,9 @@ import {
listTask, listTask,
listUl, listUl,
microsoft, microsoft,
newspaper,
nodePlus, nodePlus,
paperclip,
pencil, pencil,
people, people,
peopleFill, peopleFill,
@@ -121,9 +138,12 @@ import {
plusCircle, plusCircle,
printer, printer,
questionCircle, questionCircle,
receipt,
safe,
scissors, scissors,
search, search,
send, send,
shop,
slashCircle, slashCircle,
sliders2Vertical, sliders2Vertical,
sortAlphaDown, sortAlphaDown,
@@ -133,14 +153,17 @@ import {
tag, tag,
tagFill, tagFill,
tags, tags,
telephone,
textIndentLeft, textIndentLeft,
textLeft, textLeft,
threeDots, threeDots,
threeDotsVertical, threeDotsVertical,
trash, trash,
truck,
uiRadios, uiRadios,
unlock, unlock,
upcScan, upcScan,
wallet2,
windowStack, windowStack,
x, x,
xCircle, xCircle,
@@ -258,15 +281,22 @@ const icons = {
arrowRightShort, arrowRightShort,
arrowUpRight, arrowUpRight,
asterisk, asterisk,
bank,
basket,
bell, bell,
braces, braces,
bodyText, bodyText,
bookmark,
boxArrowUp, boxArrowUp,
boxArrowUpRight, boxArrowUpRight,
boxes, boxes,
briefcase,
building,
calculator,
calendar, calendar,
calendarEvent, calendarEvent,
calendarEventFill, calendarEventFill,
camera,
cardChecklist, cardChecklist,
cardHeading, cardHeading,
caretDown, caretDown,
@@ -288,6 +318,8 @@ const icons = {
clipboardCheckFill, clipboardCheckFill,
clipboardFill, clipboardFill,
clockHistory, clockHistory,
cash,
creditCard,
dash, dash,
dashCircle, dashCircle,
diagram3, diagram3,
@@ -306,9 +338,12 @@ const icons = {
fileEarmarkDiff, fileEarmarkDiff,
fileEarmarkFill, fileEarmarkFill,
fileEarmarkLock, fileEarmarkLock,
fileEarmarkMedical,
fileEarmarkMinus, fileEarmarkMinus,
fileEarmarkPerson,
fileEarmarkPlus, fileEarmarkPlus,
fileEarmarkRichtext, fileEarmarkRichtext,
fileEarmarkSpreadsheet,
files, files,
fileText, fileText,
filter, filter,
@@ -316,12 +351,15 @@ const icons = {
folderFill, folderFill,
funnel, funnel,
gear, gear,
globe2,
google, google,
grid, grid,
gripVertical, gripVertical,
hash, hash,
hddStack, hddStack,
heart,
house, house,
inbox,
infoCircle, infoCircle,
journals, journals,
link, link,
@@ -329,8 +367,10 @@ const icons = {
listTask, listTask,
listUl, listUl,
microsoft, microsoft,
newspaper,
nodePlus, nodePlus,
pencil, pencil,
paperclip,
people, people,
peopleFill, peopleFill,
person, person,
@@ -344,10 +384,13 @@ const icons = {
plusCircle, plusCircle,
printer, printer,
questionCircle, questionCircle,
receipt,
safe,
scissors, scissors,
search, search,
send, send,
slashCircle, slashCircle,
shop,
sliders2Vertical, sliders2Vertical,
sortAlphaDown, sortAlphaDown,
sortAlphaUpAlt, sortAlphaUpAlt,
@@ -358,12 +401,15 @@ const icons = {
tags, tags,
textIndentLeft, textIndentLeft,
textLeft, textLeft,
telephone,
threeDots, threeDots,
threeDotsVertical, threeDotsVertical,
trash, trash,
truck,
uiRadios, uiRadios,
unlock, unlock,
upcScan, upcScan,
wallet2,
windowStack, windowStack,
x, x,
xCircle, xCircle,
+106
View File
@@ -0,0 +1,106 @@
from __future__ import annotations
import importlib
import zipfile
# ZIP_ZSTANDARD exists only on Python 3.14+ (PEP 784). None elsewhere.
ZSTD: int | None = getattr(zipfile, "ZIP_ZSTANDARD", None)
# CLI choices are fixed across runtimes so argparse never hides zstd; runtime
# availability is enforced separately in compression_available().
COMPRESSION_CHOICES: tuple[str, ...] = (
"stored",
"deflated",
"bzip2",
"lzma",
"zstd",
)
# Method name -> zipfile compression constant (zstd only when supported).
COMPRESSION_METHODS: dict[str, int] = {
"stored": zipfile.ZIP_STORED,
"deflated": zipfile.ZIP_DEFLATED,
"bzip2": zipfile.ZIP_BZIP2,
"lzma": zipfile.ZIP_LZMA,
}
if ZSTD is not None:
COMPRESSION_METHODS["zstd"] = ZSTD
# Inclusive (min, max) level bounds per method; None => level not applicable.
# Verified on CPython 3.14.3.
#
# zstd's raw library bounds are (-131072, 22)
# (compression.zstd.CompressionParameter.compression_level.bounds()) — the
# minimum is an internal implementation constant (-ZSTD_TARGETLENGTH_MAX),
# not a meaningful distinct "level"; deeper negative values than -22 buy
# nothing over -22 in practice. We expose the conventional zstd CLI range
# instead of the raw library bounds.
LEVEL_BOUNDS: dict[str, tuple[int, int] | None] = {
"stored": None,
"deflated": (0, 9),
"bzip2": (1, 9),
"lzma": None,
"zstd": (-22, 22),
}
# zipfile compress_type id -> method name.
_COMPRESS_TYPE_TO_METHOD: dict[int, str] = {
zipfile.ZIP_STORED: "stored",
zipfile.ZIP_DEFLATED: "deflated",
zipfile.ZIP_BZIP2: "bzip2",
zipfile.ZIP_LZMA: "lzma",
93: "zstd",
}
def compression_available(method: str) -> bool:
"""Whether the running interpreter can actually use the given method."""
if method in ("stored", "deflated"):
# zlib is a hard CPython dependency; stored needs nothing.
return True
if method == "bzip2":
return _module_importable("bz2")
if method == "lzma":
return _module_importable("lzma")
if method == "zstd":
return ZSTD is not None and _module_importable("compression.zstd")
return False # pragma: no cover -- method is always one of COMPRESSION_CHOICES
def _module_importable(name: str) -> bool:
try:
importlib.import_module(name)
except ImportError:
return False
return True
def level_error(method: str, level: int | None) -> str | None:
"""Return a human message if (method, level) is invalid, else None."""
if level is None:
return None
bounds = LEVEL_BOUNDS[method]
if bounds is None:
return f"--zip-compression-level has no effect for '{method}'"
low, high = bounds
if not (low <= level <= high):
return (
f"--zip-compression-level for '{method}' must be between {low} and {high}"
)
return None
def compress_type_readable(compress_type: int) -> bool:
"""Whether this interpreter can decompress an entry of the given type."""
method = _COMPRESS_TYPE_TO_METHOD.get(compress_type)
if method is None:
return False
return compression_available(method)
def unreadable_method_names(compress_types: set[int]) -> set[str]:
"""Map a set of compress_type ids to human method names for error messages."""
names: set[str] = set()
for ct in compress_types:
names.add(_COMPRESS_TYPE_TO_METHOD.get(ct, f"method {ct}"))
return names
+13 -2
View File
@@ -243,11 +243,21 @@ class ZipExportSink(ExportSink):
added as an entry at finalize (a zip entry cannot be interleaved with others). added as an entry at finalize (a zip entry cannot be interleaved with others).
""" """
def __init__(self, target: Path, zip_name: str, *, delete: bool = False) -> None: def __init__(
self,
target: Path,
zip_name: str,
*,
delete: bool = False,
compression: int = zipfile.ZIP_DEFLATED,
compresslevel: int | None = None,
) -> None:
self._target = target.resolve() self._target = target.resolve()
self._zip_path = (self._target / zip_name).with_suffix(".zip") self._zip_path = (self._target / zip_name).with_suffix(".zip")
self._tmp_path = self._zip_path.with_name(self._zip_path.name + ".tmp") self._tmp_path = self._zip_path.with_name(self._zip_path.name + ".tmp")
self._delete = delete self._delete = delete
self._compression = compression
self._compresslevel = compresslevel
self._zip: zipfile.ZipFile | None = None self._zip: zipfile.ZipFile | None = None
self._dirs: set[str] = set() self._dirs: set[str] = set()
self._pending_manifest: tuple[Path, str] | None = None self._pending_manifest: tuple[Path, str] | None = None
@@ -258,7 +268,8 @@ class ZipExportSink(ExportSink):
self._zip = zipfile.ZipFile( self._zip = zipfile.ZipFile(
self._tmp_path, self._tmp_path,
"w", "w",
compression=zipfile.ZIP_DEFLATED, compression=self._compression,
compresslevel=self._compresslevel,
allowZip64=True, allowZip64=True,
) )
@@ -29,6 +29,11 @@ if TYPE_CHECKING:
if settings.AUDIT_LOG_ENABLED: if settings.AUDIT_LOG_ENABLED:
from auditlog.models import LogEntry from auditlog.models import LogEntry
from documents.export.compression import COMPRESSION_CHOICES
from documents.export.compression import COMPRESSION_METHODS
from documents.export.compression import ZSTD
from documents.export.compression import compression_available
from documents.export.compression import level_error
from documents.export.sinks import DirectoryExportSink from documents.export.sinks import DirectoryExportSink
from documents.export.sinks import ExportSink from documents.export.sinks import ExportSink
from documents.export.sinks import StreamingManifestWriter from documents.export.sinks import StreamingManifestWriter
@@ -192,6 +197,28 @@ class Command(CryptMixin, PaperlessCommand):
help="Sets the export zip file name", help="Sets the export zip file name",
) )
parser.add_argument(
"--zip-compression",
choices=COMPRESSION_CHOICES,
default=None,
help=(
"Compression method for the export zip (requires --zip). "
"Default: deflated. 'zstd' requires Python 3.14+ on both the "
"exporting and importing machine."
),
)
parser.add_argument(
"--zip-compression-level",
type=int,
default=None,
help=(
"Compression level for the export zip (requires --zip). "
"deflated: 0-9, bzip2: 1-9, zstd: -22..22; ignored for "
"stored/lzma."
),
)
parser.add_argument( parser.add_argument(
"--data-only", "--data-only",
default=False, default=False,
@@ -247,12 +274,39 @@ class Command(CryptMixin, PaperlessCommand):
if not os.access(self.target, os.W_OK): if not os.access(self.target, os.W_OK):
raise CommandError("That path doesn't appear to be writable") raise CommandError("That path doesn't appear to be writable")
zip_compression: str | None = options["zip_compression"]
zip_compression_level: int | None = options["zip_compression_level"]
if not self.zip_export and (
zip_compression is not None or zip_compression_level is not None
):
raise CommandError(
"--zip-compression and --zip-compression-level require --zip",
)
compression_method = zip_compression or "deflated"
if self.zip_export:
if not compression_available(compression_method):
if compression_method == "zstd" and ZSTD is None:
raise CommandError(
"zstd compression requires Python 3.14 or newer",
)
raise CommandError(
f"Compression method '{compression_method}' is not "
f"available on this Python runtime",
)
level_msg = level_error(compression_method, zip_compression_level)
if level_msg is not None:
raise CommandError(level_msg)
sink: ExportSink sink: ExportSink
if self.zip_export: if self.zip_export:
sink = ZipExportSink( sink = ZipExportSink(
self.target, self.target,
options["zip_name"], options["zip_name"],
delete=self.delete, delete=self.delete,
compression=COMPRESSION_METHODS[compression_method],
compresslevel=zip_compression_level,
) )
else: else:
sink = DirectoryExportSink( sink = DirectoryExportSink(
@@ -55,7 +55,7 @@ class Command(PaperlessCommand):
"--ratio", "--ratio",
default=85.0, default=85.0,
type=float, type=float,
help="Ratio to consider documents a match", help="Ratio to consider documents a match (0.0 - 100.0)",
) )
parser.add_argument( parser.add_argument(
"--delete", "--delete",
@@ -69,6 +69,17 @@ class Command(PaperlessCommand):
action="store_true", action="store_true",
help="Skip the confirmation prompt when used with --delete", help="Skip the confirmation prompt when used with --delete",
) )
parser.add_argument(
"--url",
default=None,
type=str,
help=(
"Base URL of the Paperless instance (e.g. "
"http://localhost:8000 or https://paperless.local). If set, matched "
"documents are shown as clickable (usually ctrl+click) links to "
"<url>/documents/<id>/details instead of by title."
),
)
def _render_results( def _render_results(
self, self,
@@ -76,6 +87,7 @@ class Command(PaperlessCommand):
*, *,
opt_ratio: float, opt_ratio: float,
do_delete: bool, do_delete: bool,
base_url: str | None = None,
) -> list[int]: ) -> list[int]:
"""Render match results as a Rich table. Returns list of PKs to delete.""" """Render match results as a Rich table. Returns list of PKs to delete."""
if not matches: if not matches:
@@ -88,14 +100,23 @@ class Command(PaperlessCommand):
) )
return [] return []
# Fetch titles for matched documents in a single query. # Fetch titles for matched documents in a single query, unless we're
# going to show URLs instead.
titles: dict[int, str] = {}
if not base_url:
all_pks = {pk for m in matches for pk in (m.doc_one_pk, m.doc_two_pk)} all_pks = {pk for m in matches for pk in (m.doc_one_pk, m.doc_two_pk)}
titles: dict[int, str] = dict( titles = dict(
Document.objects.filter(pk__in=all_pks) Document.objects.filter(pk__in=all_pks)
.only("pk", "title") .only("pk", "title")
.values_list("pk", "title"), .values_list("pk", "title"),
) )
def _cell(pk: int) -> str:
if base_url:
doc_url = f"{base_url.rstrip('/')}/documents/{pk}/details"
return f"[link={doc_url}]{doc_url}[/link]"
return f"[dim]#{pk}[/dim] {titles.get(pk, 'Unknown')}"
table = Table( table = Table(
title=f"Fuzzy Matches (threshold: {opt_ratio:.1f}%)", title=f"Fuzzy Matches (threshold: {opt_ratio:.1f}%)",
show_lines=True, show_lines=True,
@@ -124,8 +145,8 @@ class Command(PaperlessCommand):
table.add_row( table.add_row(
str(i), str(i),
f"[dim]#{pk_a}[/dim] {titles.get(pk_a, 'Unknown')}", _cell(pk_a),
f"[dim]#{pk_b}[/dim] {titles.get(pk_b, 'Unknown')}", _cell(pk_b),
Text(f"{ratio:.1f}%", style=ratio_style), Text(f"{ratio:.1f}%", style=ratio_style),
) )
maybe_delete_ids.append(pk_b) maybe_delete_ids.append(pk_b)
@@ -208,6 +229,7 @@ class Command(PaperlessCommand):
matches, matches,
opt_ratio=opt_ratio, opt_ratio=opt_ratio,
do_delete=options["delete"], do_delete=options["delete"],
base_url=options["url"],
) )
if options["delete"] and maybe_delete_ids: if options["delete"] and maybe_delete_ids:
@@ -32,6 +32,8 @@ from django.db.models.signals import post_save
from filelock import FileLock from filelock import FileLock
from guardian.shortcuts import clear_ct_cache from guardian.shortcuts import clear_ct_cache
from documents.export.compression import compress_type_readable
from documents.export.compression import unreadable_method_names
from documents.file_handling import create_source_path_directory from documents.file_handling import create_source_path_directory
from documents.management.commands.base import PaperlessCommand from documents.management.commands.base import PaperlessCommand
from documents.management.commands.mixins import CryptMixin from documents.management.commands.mixins import CryptMixin
@@ -460,6 +462,20 @@ class Command(CryptMixin, PaperlessCommand):
with tempfile.TemporaryDirectory() as tmp_dir: with tempfile.TemporaryDirectory() as tmp_dir:
if is_zipfile(self.source): if is_zipfile(self.source):
with ZipFile(self.source) as zf: with ZipFile(self.source) as zf:
unsupported = {
info.compress_type
for info in zf.infolist()
if not compress_type_readable(info.compress_type)
}
if unsupported:
names = sorted(unreadable_method_names(unsupported))
message = (
f"This archive uses compression this Python cannot "
f"read ({', '.join(names)})."
)
if "zstd" in names:
message += " zstd archives require Python 3.14+."
raise CommandError(message)
zf.extractall(tmp_dir) zf.extractall(tmp_dir)
self.source = Path(tmp_dir) self.source = Path(tmp_dir)
self._run_import() self._run_import()
@@ -0,0 +1,79 @@
from django.db import migrations
from django.db import models
class Migration(migrations.Migration):
dependencies = [
("documents", "0022_add_perf_indexes"),
]
operations = [
migrations.AddField(
model_name="savedview",
name="icon",
field=models.CharField(
choices=[
("archive", "Archive"),
("bank", "Bank"),
("basket", "Basket"),
("bell", "Bell"),
("bookmark", "Bookmark"),
("boxes", "Boxes"),
("briefcase", "Briefcase"),
("building", "Building"),
("calculator", "Calculator"),
("calendar", "Calendar"),
("camera", "Camera"),
("card-checklist", "Checklist"),
("cash", "Cash"),
("chat-left-text", "Chat"),
("check-circle", "Check"),
("clipboard", "Clipboard"),
("clock-history", "Clock"),
("credit-card", "Credit card"),
("download", "Download"),
("envelope", "Envelope"),
("exclamation-triangle", "Warning"),
("file-earmark", "File"),
("file-earmark-check", "Checked file"),
("file-earmark-lock", "Locked file"),
("file-earmark-medical", "Medical file"),
("file-earmark-person", "Person file"),
("file-earmark-spreadsheet", "Spreadsheet"),
("file-text", "Text file"),
("files", "Files"),
("folder", "Folder"),
("funnel", "Filter"),
("gear", "Gear"),
("globe2", "Globe"),
("hash", "Hash"),
("heart", "Heart"),
("house", "House"),
("inbox", "Inbox"),
("journals", "Journals"),
("list-task", "Task list"),
("newspaper", "Newspaper"),
("paperclip", "Attachment"),
("people", "People"),
("person", "Person"),
("printer", "Printer"),
("receipt", "Receipt"),
("safe", "Safe"),
("search", "Search"),
("send", "Send"),
("shop", "Shop"),
("stack", "Stack"),
("stars", "Stars"),
("tag", "Tag"),
("tags", "Tags"),
("telephone", "Telephone"),
("truck", "Truck"),
("upc-scan", "Barcode"),
("wallet2", "Wallet"),
],
default="funnel",
max_length=64,
verbose_name="icon",
),
),
]
+69
View File
@@ -519,6 +519,68 @@ class Document(SoftDeleteModel, ModelWithOwner): # type: ignore[django-manager-
class SavedView(ModelWithOwner): class SavedView(ModelWithOwner):
class Icon(models.TextChoices):
ARCHIVE = ("archive", _("Archive"))
BANK = ("bank", _("Bank"))
BASKET = ("basket", _("Basket"))
BELL = ("bell", _("Bell"))
BOOKMARK = ("bookmark", _("Bookmark"))
BOXES = ("boxes", _("Boxes"))
BRIEFCASE = ("briefcase", _("Briefcase"))
BUILDING = ("building", _("Building"))
CALCULATOR = ("calculator", _("Calculator"))
CALENDAR = ("calendar", _("Calendar"))
CAMERA = ("camera", _("Camera"))
CARD_CHECKLIST = ("card-checklist", _("Checklist"))
CASH = ("cash", _("Cash"))
CHAT_LEFT_TEXT = ("chat-left-text", _("Chat"))
CHECK_CIRCLE = ("check-circle", _("Check"))
CLIPBOARD = ("clipboard", _("Clipboard"))
CLOCK_HISTORY = ("clock-history", _("Clock"))
CREDIT_CARD = ("credit-card", _("Credit card"))
DOWNLOAD = ("download", _("Download"))
ENVELOPE = ("envelope", _("Envelope"))
EXCLAMATION_TRIANGLE = ("exclamation-triangle", _("Warning"))
FILE_EARMARK = ("file-earmark", _("File"))
FILE_EARMARK_CHECK = ("file-earmark-check", _("Checked file"))
FILE_EARMARK_LOCK = ("file-earmark-lock", _("Locked file"))
FILE_EARMARK_MEDICAL = ("file-earmark-medical", _("Medical file"))
FILE_EARMARK_PERSON = ("file-earmark-person", _("Person file"))
FILE_EARMARK_SPREADSHEET = (
"file-earmark-spreadsheet",
_("Spreadsheet"),
)
FILE_TEXT = ("file-text", _("Text file"))
FILES = ("files", _("Files"))
FOLDER = ("folder", _("Folder"))
FUNNEL = ("funnel", _("Filter"))
GEAR = ("gear", _("Gear"))
GLOBE = ("globe2", _("Globe"))
HASH = ("hash", _("Hash"))
HEART = ("heart", _("Heart"))
HOUSE = ("house", _("House"))
INBOX = ("inbox", _("Inbox"))
JOURNALS = ("journals", _("Journals"))
LIST_TASK = ("list-task", _("Task list"))
NEWSPAPER = ("newspaper", _("Newspaper"))
PAPERCLIP = ("paperclip", _("Attachment"))
PEOPLE = ("people", _("People"))
PERSON = ("person", _("Person"))
PRINTER = ("printer", _("Printer"))
RECEIPT = ("receipt", _("Receipt"))
SAFE = ("safe", _("Safe"))
SEARCH = ("search", _("Search"))
SEND = ("send", _("Send"))
SHOP = ("shop", _("Shop"))
STACK = ("stack", _("Stack"))
STARS = ("stars", _("Stars"))
TAG = ("tag", _("Tag"))
TAGS = ("tags", _("Tags"))
TELEPHONE = ("telephone", _("Telephone"))
TRUCK = ("truck", _("Truck"))
UPC_SCAN = ("upc-scan", _("Barcode"))
WALLET = ("wallet2", _("Wallet"))
class DisplayMode(models.TextChoices): class DisplayMode(models.TextChoices):
TABLE = ("table", _("Table")) TABLE = ("table", _("Table"))
SMALL_CARDS = ("smallCards", _("Small Cards")) SMALL_CARDS = ("smallCards", _("Small Cards"))
@@ -541,6 +603,13 @@ class SavedView(ModelWithOwner):
name = models.CharField(_("name"), max_length=128) name = models.CharField(_("name"), max_length=128)
icon = models.CharField(
_("icon"),
max_length=64,
choices=Icon.choices,
default=Icon.FUNNEL,
)
sort_field = models.CharField( sort_field = models.CharField(
_("sort field"), _("sort field"),
max_length=128, max_length=128,
+8
View File
@@ -1383,6 +1383,7 @@ class SavedViewSerializer(OwnedObjectSerializer):
fields = [ fields = [
"id", "id",
"name", "name",
"icon",
"sort_field", "sort_field",
"sort_reverse", "sort_reverse",
"filter_rules", "filter_rules",
@@ -3213,6 +3214,13 @@ class WorkflowActionSerializer(serializers.ModelSerializer[WorkflowAction]):
{"assign_title": f'Invalid f-string detected: "{e.args[0]}"'}, {"assign_title": f'Invalid f-string detected: "{e.args[0]}"'},
) )
if attrs.get("assign_custom_fields_values"):
# Empty strings treated as None to avoid unexpected behavior
attrs["assign_custom_fields_values"] = {
field_id: (None if value == "" else value)
for field_id, value in attrs["assign_custom_fields_values"].items()
}
if ( if (
"type" in attrs "type" in attrs
and attrs["type"] == WorkflowAction.WorkflowActionType.EMAIL and attrs["type"] == WorkflowAction.WorkflowActionType.EMAIL
@@ -0,0 +1,208 @@
import sys
import zipfile
import pytest
import pytest_mock
from documents.export import compression
class TestCompressionMethods:
def test_choices_always_include_zstd(self) -> None:
"""
GIVEN:
- The compression policy module's CLI choices list
WHEN:
- Read on any runtime
THEN:
- zstd is always present; availability is checked separately so
argparse never hides it based on the current Python version
"""
assert compression.COMPRESSION_CHOICES == (
"stored",
"deflated",
"bzip2",
"lzma",
"zstd",
)
@pytest.mark.parametrize(
("name", "constant"),
[
("stored", zipfile.ZIP_STORED),
("deflated", zipfile.ZIP_DEFLATED),
("bzip2", zipfile.ZIP_BZIP2),
("lzma", zipfile.ZIP_LZMA),
],
)
def test_method_maps_to_zipfile_constant(self, name: str, constant: int) -> None:
"""
GIVEN:
- A compression method name
WHEN:
- Looked up in COMPRESSION_METHODS
THEN:
- It maps to the matching zipfile compression constant
"""
assert compression.COMPRESSION_METHODS[name] == constant
def test_stored_and_deflated_always_available(self) -> None:
"""
GIVEN:
- The stored and deflated compression methods
WHEN:
- Checked with compression_available()
THEN:
- Both are always available (zlib is a hard CPython dependency)
"""
assert compression.compression_available("stored")
assert compression.compression_available("deflated")
def test_zstd_availability_tracks_runtime(self) -> None:
"""
GIVEN:
- The zstd compression method
WHEN:
- Checked with compression_available() on this runtime
THEN:
- Availability matches whether Python is 3.14+
"""
expected: bool = sys.version_info >= (3, 14)
assert compression.compression_available("zstd") == expected
def test_unimportable_module_reports_unavailable(
self,
mocker: pytest_mock.MockerFixture,
) -> None:
"""
GIVEN:
- A compression method whose backing module fails to import
(e.g. a minimal Python build without bz2/lzma compiled in)
WHEN:
- Checked with compression_available()
THEN:
- False is returned rather than the ImportError propagating
"""
mocker.patch(
"documents.export.compression.importlib.import_module",
side_effect=ImportError,
)
assert not compression.compression_available("bzip2")
class TestLevelError:
@pytest.mark.parametrize(
("method", "level"),
[
("deflated", 0),
("deflated", 9),
("bzip2", 1),
("bzip2", 9),
("zstd", -22),
("zstd", 22),
("deflated", None),
("stored", None),
],
)
def test_valid_levels_return_none(self, method: str, level: int | None) -> None:
"""
GIVEN:
- A method and a level within its valid bounds (or no level)
WHEN:
- Checked with level_error()
THEN:
- No error message is returned
"""
assert compression.level_error(method, level) is None
@pytest.mark.parametrize(
("method", "level"),
[
("deflated", 10),
("deflated", -1),
("bzip2", 0),
("bzip2", 10),
("zstd", -23),
("zstd", 23),
],
)
def test_out_of_range_levels_return_message(
self,
method: str,
level: int,
) -> None:
"""
GIVEN:
- A method and a level outside its valid bounds
WHEN:
- Checked with level_error()
THEN:
- An error message naming the valid range is returned
"""
msg: str | None = compression.level_error(method, level)
assert msg is not None
assert "between" in msg
@pytest.mark.parametrize("method", ["stored", "lzma"])
def test_level_on_levelless_method_is_rejected(self, method: str) -> None:
"""
GIVEN:
- A method that ignores compression level (stored, lzma)
WHEN:
- A level is passed to level_error() anyway
THEN:
- An error message noting the level has no effect is returned
"""
msg: str | None = compression.level_error(method, 5)
assert msg is not None
assert "no effect" in msg
class TestCompressTypeReadable:
@pytest.mark.parametrize("ct", [zipfile.ZIP_STORED, zipfile.ZIP_DEFLATED])
def test_stored_and_deflated_always_readable(self, ct: int) -> None:
"""
GIVEN:
- A stored or deflated compress_type id
WHEN:
- Checked with compress_type_readable()
THEN:
- It is always readable
"""
assert compression.compress_type_readable(ct)
def test_zstd_compress_type_readability_tracks_runtime(self) -> None:
"""
GIVEN:
- The zstd compress_type id (93, ZIP_ZSTANDARD)
WHEN:
- Checked with compress_type_readable() on this runtime
THEN:
- Readability matches whether Python is 3.14+
"""
expected: bool = sys.version_info >= (3, 14)
assert compression.compress_type_readable(93) == expected
def test_unknown_compress_type_is_unreadable(self) -> None:
"""
GIVEN:
- An unrecognized compress_type id
WHEN:
- Checked with compress_type_readable()
THEN:
- It is reported as unreadable
"""
assert not compression.compress_type_readable(9999)
def test_unreadable_method_names_lists_methods(self) -> None:
"""
GIVEN:
- A set containing an unknown compress_type id
WHEN:
- Passed to unreadable_method_names()
THEN:
- It is reported generically as "method <id>"
"""
# An unknown method id maps to no name and is reported generically.
names: set[str] = compression.unreadable_method_names({9999})
assert names == {"method 9999"}
+43
View File
@@ -5,6 +5,7 @@ import zipfile
from pathlib import Path from pathlib import Path
import pytest import pytest
import pytest_mock
from pytest_django.fixtures import SettingsWrapper from pytest_django.fixtures import SettingsWrapper
from documents.export.sinks import DirectoryExportSink from documents.export.sinks import DirectoryExportSink
@@ -305,6 +306,48 @@ class TestZipExportSink:
assert not (target / "export.zip").exists() assert not (target / "export.zip").exists()
class TestZipExportSinkCompression:
@pytest.mark.parametrize(
("method", "constant"),
[
("stored", zipfile.ZIP_STORED),
("deflated", zipfile.ZIP_DEFLATED),
("bzip2", zipfile.ZIP_BZIP2),
("lzma", zipfile.ZIP_LZMA),
],
)
def test_compression_and_level_forwarded_to_zipfile(
self,
mocker: pytest_mock.MockerFixture,
tmp_path: Path,
method: str,
constant: int,
) -> None:
"""
GIVEN:
- A ZipExportSink constructed with a compression method and level
WHEN:
- The sink is opened
THEN:
- zipfile.ZipFile is constructed with those values forwarded
unchanged (whether ZipFile actually compresses is Python's own
contract, not ours, so this checks the call args, not a real
archive)
"""
target: Path = tmp_path / "out"
target.mkdir()
zip_cls = mocker.patch("documents.export.sinks.zipfile.ZipFile")
sink = ZipExportSink(target, "export", compression=constant, compresslevel=5)
sink._open()
zip_cls.assert_called_once_with(
mocker.ANY,
"w",
compression=constant,
compresslevel=5,
allowZip64=True,
)
class TestStreamContract: class TestStreamContract:
@pytest.fixture(params=["dir", "zip"]) @pytest.fixture(params=["dir", "zip"])
def sink(self, request: pytest.FixtureRequest, tmp_path: Path) -> ExportSink: def sink(self, request: pytest.FixtureRequest, tmp_path: Path) -> ExportSink:
+10 -1
View File
@@ -2905,18 +2905,20 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
v1 = SavedView.objects.get(name="test") v1 = SavedView.objects.get(name="test")
self.assertEqual(v1.sort_field, "created2") self.assertEqual(v1.sort_field, "created2")
self.assertEqual(v1.icon, SavedView.Icon.FUNNEL)
self.assertEqual(v1.filter_rules.count(), 1) self.assertEqual(v1.filter_rules.count(), 1)
self.assertEqual(v1.owner, self.user) self.assertEqual(v1.owner, self.user)
response = self.client.patch( response = self.client.patch(
f"/api/saved_views/{v1.id}/", f"/api/saved_views/{v1.id}/",
{"sort_reverse": True}, {"sort_reverse": True, "icon": SavedView.Icon.RECEIPT},
format="json", format="json",
) )
v1 = SavedView.objects.get(id=v1.id) v1 = SavedView.objects.get(id=v1.id)
self.assertEqual(response.status_code, status.HTTP_200_OK) self.assertEqual(response.status_code, status.HTTP_200_OK)
self.assertTrue(v1.sort_reverse) self.assertTrue(v1.sort_reverse)
self.assertEqual(v1.icon, SavedView.Icon.RECEIPT)
self.assertEqual(v1.filter_rules.count(), 1) self.assertEqual(v1.filter_rules.count(), 1)
view["filter_rules"] = [{"rule_type": 12, "value": "secret"}] view["filter_rules"] = [{"rule_type": 12, "value": "secret"}]
@@ -2936,6 +2938,13 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
v1 = SavedView.objects.get(id=v1.id) v1 = SavedView.objects.get(id=v1.id)
self.assertEqual(v1.filter_rules.count(), 0) self.assertEqual(v1.filter_rules.count(), 0)
response = self.client.patch(
f"/api/saved_views/{v1.id}/",
{"icon": "not-an-icon"},
format="json",
)
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
def test_saved_view_display_options(self) -> None: def test_saved_view_display_options(self) -> None:
""" """
GIVEN: GIVEN:
@@ -422,6 +422,11 @@ class TestApiWorkflows(DirectoriesMixin, APITestCase):
json.dumps( json.dumps(
{ {
"assign_title": "", "assign_title": "",
"assign_custom_fields": [self.cf1.id, self.cf2.id],
"assign_custom_fields_values": {
str(self.cf1.id): "",
str(self.cf2.id): 0,
},
}, },
), ),
content_type="application/json", content_type="application/json",
@@ -429,6 +434,10 @@ class TestApiWorkflows(DirectoriesMixin, APITestCase):
self.assertEqual(response.status_code, status.HTTP_201_CREATED) self.assertEqual(response.status_code, status.HTTP_201_CREATED)
action = WorkflowAction.objects.get(id=response.data["id"]) action = WorkflowAction.objects.get(id=response.data["id"])
self.assertIsNone(action.assign_title) self.assertIsNone(action.assign_title)
self.assertEqual(
action.assign_custom_fields_values,
{str(self.cf1.id): None, str(self.cf2.id): 0},
)
response = self.client.post( response = self.client.post(
self.ENDPOINT_TRIGGERS, self.ENDPOINT_TRIGGERS,
@@ -6,6 +6,8 @@ from datetime import timedelta
from io import StringIO from io import StringIO
from pathlib import Path from pathlib import Path
from unittest import mock from unittest import mock
from zipfile import ZIP_DEFLATED
from zipfile import ZIP_LZMA
from zipfile import ZipFile from zipfile import ZipFile
import pytest import pytest
@@ -1078,6 +1080,197 @@ class TestExportImport(
skip_checks=True, skip_checks=True,
) )
def test_compression_flags_require_zip(self) -> None:
"""
GIVEN:
- A request to export without --zip
WHEN:
- --zip-compression or --zip-compression-level is passed anyway
THEN:
- A CommandError is raised (the flags are meaningless without --zip)
"""
cases = {
"zip-compression": ["--zip-compression", "lzma"],
"zip-compression-level": ["--zip-compression-level", "5"],
}
for case_id, args in cases.items():
with self.subTest(case_id), self.assertRaises(CommandError):
call_command(
"document_exporter",
self.target,
*args,
skip_checks=True,
)
def test_zip_compression_level_out_of_range_raises(self) -> None:
"""
GIVEN:
- A request to export to a zip file
WHEN:
- --zip-compression-level is outside the chosen method's valid range
THEN:
- A CommandError is raised
"""
with self.assertRaises(CommandError):
call_command(
"document_exporter",
self.target,
"--zip",
"--zip-compression",
"deflated",
"--zip-compression-level",
"99",
skip_checks=True,
)
def test_zip_compression_level_rejected_for_levelless_method(self) -> None:
"""
GIVEN:
- A request to export to a zip file with a compression method
that ignores level entirely (stored, lzma)
WHEN:
- --zip-compression-level is also passed
THEN:
- A CommandError is raised
"""
for method in ("stored", "lzma"):
with self.subTest(method), self.assertRaises(CommandError):
call_command(
"document_exporter",
self.target,
"--zip",
"--zip-compression",
method,
"--zip-compression-level",
"5",
skip_checks=True,
)
def test_zstd_unavailable_raises_friendly_error(self) -> None:
"""
GIVEN:
- A Python runtime without zstd support (< 3.14)
WHEN:
- --zip-compression zstd is requested
THEN:
- A CommandError naming the Python version requirement is raised
zstd availability is mocked rather than relying on the actual
runtime: on a Python 3.14+ CI leg, ZSTD is not None, so without the
mock this check is skipped and the command falls through into the
real export, which fails on missing document files instead of
raising the expected CommandError.
"""
with (
mock.patch(
"documents.management.commands.document_exporter.ZSTD",
None,
),
mock.patch(
"documents.management.commands.document_exporter.compression_available",
return_value=False,
),
self.assertRaises(CommandError) as e,
):
call_command(
"document_exporter",
self.target,
"--zip",
"--zip-compression",
"zstd",
skip_checks=True,
)
self.assertIn("3.14", str(e.exception))
def test_non_zstd_unavailable_raises_generic_error(self) -> None:
"""
GIVEN:
- A Python runtime missing the module backing a non-zstd method
(e.g. bz2/lzma not compiled in on a minimal build)
WHEN:
- That method is requested via --zip-compression
THEN:
- A CommandError is raised naming the method, not the
zstd-specific "requires 3.14" message
"""
with (
mock.patch(
"documents.management.commands.document_exporter.compression_available",
return_value=False,
),
self.assertRaises(CommandError) as e,
):
call_command(
"document_exporter",
self.target,
"--zip",
"--zip-compression",
"bzip2",
skip_checks=True,
)
self.assertIn("bzip2", str(e.exception))
self.assertNotIn("3.14", str(e.exception))
def test_zip_compression_flag_resolves_to_sink_constant(self) -> None:
"""
GIVEN:
- A request to export to a zip file with --zip-compression lzma
WHEN:
- The export runs
THEN:
- ZipExportSink is constructed with the resolved ZIP_LZMA constant
(whether zipfile actually compresses with the chosen method is
Python's own contract, and ZipExportSink's own tests already
cover the forwarding; what this command owns is resolving the
CLI string to the right constant, so assert that resolution
directly)
"""
with mock.patch(
"documents.management.commands.document_exporter.ZipExportSink",
) as sink_cls:
call_command(
"document_exporter",
self.target,
"--zip",
"--zip-compression",
"lzma",
skip_checks=True,
)
sink_cls.assert_called_once_with(
mock.ANY,
mock.ANY,
delete=False,
compression=ZIP_LZMA,
compresslevel=None,
)
def test_default_zip_compression_resolves_to_deflate(self) -> None:
"""
GIVEN:
- A request to export to a zip file with no --zip-compression flag
WHEN:
- The export runs
THEN:
- ZipExportSink is constructed with the default ZIP_DEFLATED
constant and compresslevel=None, matching pre-existing behavior
"""
with mock.patch(
"documents.management.commands.document_exporter.ZipExportSink",
) as sink_cls:
call_command(
"document_exporter",
self.target,
"--zip",
skip_checks=True,
)
sink_cls.assert_called_once_with(
mock.ANY,
mock.ANY,
delete=False,
compression=ZIP_DEFLATED,
compresslevel=None,
)
@pytest.mark.management @pytest.mark.management
class TestCryptExportImport( class TestCryptExportImport(
+41 -1
View File
@@ -1,3 +1,4 @@
import os
from io import StringIO from io import StringIO
from unittest.mock import patch from unittest.mock import patch
@@ -41,7 +42,7 @@ class TestFuzzyMatchCommand(TestCase):
def test_invalid_ratio_upper_limit(self) -> None: def test_invalid_ratio_upper_limit(self) -> None:
""" """
GIVEN:s GIVEN:
- Invalid ratio above upper - Invalid ratio above upper
WHEN: WHEN:
- Command is called - Command is called
@@ -108,6 +109,45 @@ class TestFuzzyMatchCommand(TestCase):
stdout, _ = self.call_command("--processes", "1") stdout, _ = self.call_command("--processes", "1")
self.assertIn("Found 1 matching pair(s)", stdout) self.assertIn("Found 1 matching pair(s)", stdout)
def test_with_matches_and_url(self) -> None:
"""
GIVEN:
- 2 documents exist
- Similarity between content is 86.667
- --url is provided
WHEN:
- Command is called with --url
THEN:
- 1 match is returned from doc 1 to doc 2
- No match from doc 2 to doc 1 reported
- Output contains clickable links to the documents instead of titles
"""
# Content similarity is 86.667
Document.objects.create(
checksum="BEEFCAFE",
title="A",
content="first document scanned by bob",
mime_type="application/pdf",
filename="test.pdf",
)
Document.objects.create(
checksum="DEADBEAF",
title="A",
content="first document scanned by alice",
mime_type="application/pdf",
filename="other_test.pdf",
)
with patch.dict(os.environ, {"COLUMNS": "200"}):
stdout, _ = self.call_command(
"--processes",
"1",
"--url",
"http://localhost:8000",
)
self.assertIn("Found 1 matching pair(s)", stdout)
self.assertIn("http://localhost:8000/documents/1/details", stdout)
self.assertIn("http://localhost:8000/documents/2/details", stdout)
def test_with_3_matches(self) -> None: def test_with_3_matches(self) -> None:
""" """
GIVEN: GIVEN:
@@ -525,6 +525,71 @@ class TestCommandImport(
self.assertEqual(doc.tags.count(), 1) self.assertEqual(doc.tags.count(), 1)
self.assertEqual(doc.tags.first().name, "batch-flush-tag") self.assertEqual(doc.tags.first().name, "batch-flush-tag")
def test_import_rejects_unreadable_compression(self) -> None:
"""
GIVEN:
- A zip archive with an entry whose compression this Python can't read
WHEN:
- Import is attempted
THEN:
- A CommandError naming the issue is raised, before extraction
"""
import zipfile
from unittest import mock
archive = Path(self.dirs.scratch_dir) / "export.zip"
with zipfile.ZipFile(archive, "w") as zf:
zf.writestr("manifest.json", "[]")
with mock.patch(
"documents.management.commands.document_importer.compress_type_readable",
return_value=False,
):
with self.assertRaises(CommandError) as e:
call_command(
"document_importer",
str(archive),
"--no-progress-bar",
skip_checks=True,
)
self.assertIn("compression", str(e.exception))
def test_import_rejects_unreadable_zstd_with_version_hint(self) -> None:
"""
GIVEN:
- A zip archive with an entry compressed with zstd
WHEN:
- Import is attempted on a Python runtime that can't read zstd
THEN:
- The CommandError names the 3.14+ requirement, not just the
generic "can't read" message
"""
import zipfile
from unittest import mock
archive = Path(self.dirs.scratch_dir) / "export.zip"
with zipfile.ZipFile(archive, "w") as zf:
zf.writestr("manifest.json", "[]")
with (
mock.patch(
"documents.management.commands.document_importer.compress_type_readable",
return_value=False,
),
mock.patch(
"documents.management.commands.document_importer.unreadable_method_names",
return_value={"zstd"},
),
):
with self.assertRaises(CommandError) as e:
call_command(
"document_importer",
str(archive),
"--no-progress-bar",
skip_checks=True,
)
self.assertIn("3.14", str(e.exception))
@pytest.mark.management @pytest.mark.management
@pytest.mark.django_db @pytest.mark.django_db
+49
View File
@@ -2000,6 +2000,55 @@ class TestWorkflows(
r"Doc added in \w{3,}", r"Doc added in \w{3,}",
) # Match any 3-letter month name ) # Match any 3-letter month name
def test_document_updated_workflow_existing_custom_field_empty_value(self) -> None:
"""
GIVEN:
- Existing workflow with UPDATED trigger and action that assigns a custom field
with an empty value
WHEN:
- Document is updated that already contains the field with a value
THEN:
- The existing value is left untouched, see GH #13627
"""
trigger = WorkflowTrigger.objects.create(
type=WorkflowTrigger.WorkflowTriggerType.DOCUMENT_UPDATED,
filter_has_document_type=self.dt,
)
action = WorkflowAction.objects.create()
action.assign_custom_fields.add(self.cf1)
action.assign_custom_fields_values = {self.cf1.pk: ""}
action.save()
w = Workflow.objects.create(
name="Workflow 1",
order=0,
)
w.triggers.add(trigger)
w.actions.add(action)
w.save()
doc = Document.objects.create(
title="sample test",
correspondent=self.c,
original_filename="sample.pdf",
)
CustomFieldInstance.objects.create(
document=doc,
field=self.cf1,
value_text="existing value",
)
superuser = User.objects.create_superuser("superuser")
self.client.force_authenticate(user=superuser)
self.client.patch(
f"/api/documents/{doc.id}/",
{"document_type": self.dt.id},
format="json",
)
doc.refresh_from_db()
self.assertEqual(doc.custom_fields.get(field=self.cf1).value, "existing value")
def test_document_updated_workflow_existing_custom_field(self) -> None: def test_document_updated_workflow_existing_custom_field(self) -> None:
""" """
GIVEN: GIVEN:
+1 -1
View File
@@ -2267,7 +2267,7 @@ class ChatStreamingView(GenericAPIView[Any]):
if not has_perms_owner_aware(request.user, "view_document", document): if not has_perms_owner_aware(request.user, "view_document", document):
return HttpResponseForbidden("Insufficient permissions") return HttpResponseForbidden("Insufficient permissions")
documents = [document] documents = Document.objects.filter(pk=document.pk)
else: else:
documents = Document.objects.filter( documents = Document.objects.filter(
id__in=permitted_document_ids(request.user), id__in=permitted_document_ids(request.user),
+2 -1
View File
@@ -105,7 +105,8 @@ def apply_assignment_to_document(
field=field, field=field,
document=document, document=document,
).first() ).first()
if instance and args[value_field_name] is not None: # empty string is indistinguishable from no value in the UI
if instance and args[value_field_name] not in (None, ""):
setattr(instance, value_field_name, args[value_field_name]) setattr(instance, value_field_name, args[value_field_name])
instance.save() instance.save()
elif not instance: elif not instance:
File diff suppressed because it is too large Load Diff
+33 -7
View File
@@ -3,7 +3,9 @@ Built-in remote-OCR document parser.
Handles documents by sending them to a configured remote OCR engine Handles documents by sending them to a configured remote OCR engine
(currently Azure AI Vision / Document Intelligence) and retrieving both (currently Azure AI Vision / Document Intelligence) and retrieving both
the extracted text and a searchable PDF with an embedded text layer. the extracted text and a searchable PDF with an embedded text layer. For
born-digital PDFs that need no archive copy, the remote call is skipped
entirely in favor of locally-extracted text (see ``RemoteDocumentParser.parse``).
When no engine is configured, ``score()`` returns ``None`` so the parser When no engine is configured, ``score()`` returns ``None`` so the parser
is effectively invisible to the registry the tesseract parser handles is effectively invisible to the registry the tesseract parser handles
@@ -22,6 +24,8 @@ from typing import Self
from django.conf import settings from django.conf import settings
from documents.parsers import ParseError from documents.parsers import ParseError
from paperless.parsers.utils import extract_pdf_text
from paperless.parsers.utils import post_process_text
from paperless.version import __full_version_str__ from paperless.version import __full_version_str__
if TYPE_CHECKING: if TYPE_CHECKING:
@@ -70,8 +74,11 @@ class RemoteDocumentParser:
"""Parse documents via a remote OCR API (currently Azure AI Vision). """Parse documents via a remote OCR API (currently Azure AI Vision).
This parser sends documents to a remote engine that returns both This parser sends documents to a remote engine that returns both
extracted text and a searchable PDF with an embedded text layer. extracted text and a searchable PDF with an embedded text layer,
It does not depend on Tesseract or ocrmypdf. except when ``parse()`` is called with ``produce_archive=False`` for
a PDF, in which case the remote call is skipped and only locally
extracted text is returned (no archive). It does not depend on
Tesseract or ocrmypdf.
Class attributes Class attributes
---------------- ----------------
@@ -160,8 +167,11 @@ class RemoteDocumentParser:
Returns Returns
------- -------
bool bool
Always True the remote engine always returns a PDF with an Always True the remote engine is capable of returning a PDF
embedded text layer that serves as the archive copy. with an embedded text layer to serve as the archive copy.
Whether it actually does so for a given document depends on
``produce_archive`` passed to :meth:`parse` (see there for when
the remote engine call, and thus archive generation, is skipped).
""" """
return True return True
@@ -218,6 +228,12 @@ class RemoteDocumentParser:
) -> None: ) -> None:
"""Send the document to the remote engine and store results. """Send the document to the remote engine and store results.
When *produce_archive* is False for a PDF, the caller (via
``documents.consumer.should_produce_archive``) has already determined
that the document is born-digital and needs no archive skip the
remote engine entirely rather than re-OCRing it and creating a
duplicate text layer.
Parameters Parameters
---------- ----------
document_path: document_path:
@@ -225,8 +241,8 @@ class RemoteDocumentParser:
mime_type: mime_type:
Detected MIME type of the document. Detected MIME type of the document.
produce_archive: produce_archive:
Ignored the remote engine always returns a searchable PDF, Whether an archive copy is wanted. For PDFs, False skips the
which is stored as the archive copy regardless of this flag. remote engine and uses locally-extracted text instead.
""" """
config = RemoteEngineConfig( config = RemoteEngineConfig(
engine=settings.REMOTE_OCR_ENGINE, engine=settings.REMOTE_OCR_ENGINE,
@@ -241,6 +257,16 @@ class RemoteDocumentParser:
self._text = "" self._text = ""
return return
if not produce_archive and mime_type == "application/pdf":
logger.debug(
"Remote OCR: skipped — no archive requested, "
"using locally-extracted text",
)
self._text = (
post_process_text(extract_pdf_text(document_path, log=logger)) or ""
)
return
if config.engine == "azureai": if config.engine == "azureai":
self._text = self._azure_ai_vision_parse(document_path, config) self._text = self._azure_ai_vision_parse(document_path, config)
@@ -337,6 +337,117 @@ class TestRemoteParserParse:
assert remote_parser.get_date() is None assert remote_parser.get_date() is None
# ---------------------------------------------------------------------------
# parse() — produce_archive=False skips the remote engine (PDFs only)
# ---------------------------------------------------------------------------
class TestRemoteParserSkipsWhenNoArchiveWanted:
"""When the caller has already decided no archive is needed for a PDF
(documents.consumer.should_produce_archive), the remote engine call is
skipped entirely in favor of locally-extracted text.
"""
def test_pdf_skips_azure_when_no_archive_requested(
self,
remote_parser: RemoteDocumentParser,
simple_digital_pdf_file: Path,
azure_client: Mock,
) -> None:
"""
GIVEN: produce_archive=False for a PDF
WHEN: parse() is called
THEN: Azure is never invoked, no archive is produced, and text
comes from local pdftotext extraction
"""
remote_parser.parse(
simple_digital_pdf_file,
"application/pdf",
produce_archive=False,
)
azure_client.begin_analyze_document.assert_not_called()
assert remote_parser.get_archive_path() is None
assert remote_parser.get_text() != ""
def test_pdf_no_archive_requested_text_matches_local_extraction(
self,
remote_parser: RemoteDocumentParser,
simple_digital_pdf_file: Path,
azure_client: Mock,
mocker: MockerFixture,
) -> None:
"""
GIVEN: produce_archive=False for a PDF
WHEN: parse() is called
THEN: the returned text is exactly the locally-extracted text,
not anything from the (unused) Azure mock
"""
mocker.patch(
"paperless.parsers.remote.extract_pdf_text",
return_value="Local digital text.",
)
remote_parser.parse(
simple_digital_pdf_file,
"application/pdf",
produce_archive=False,
)
assert remote_parser.get_text() == "Local digital text."
def test_pdf_no_archive_requested_closes_no_client(
self,
remote_parser: RemoteDocumentParser,
simple_digital_pdf_file: Path,
azure_client: Mock,
) -> None:
remote_parser.parse(
simple_digital_pdf_file,
"application/pdf",
produce_archive=False,
)
azure_client.close.assert_not_called()
def test_non_pdf_still_calls_azure_when_no_archive_requested(
self,
remote_parser: RemoteDocumentParser,
simple_digital_pdf_file: Path,
azure_client: Mock,
) -> None:
"""
Images have no local-text fallback, so produce_archive=False does
not skip the remote engine for non-PDF MIME types.
"""
remote_parser.parse(
simple_digital_pdf_file,
"image/png",
produce_archive=False,
)
azure_client.begin_analyze_document.assert_called_once()
assert remote_parser.get_text() == _DEFAULT_TEXT
@pytest.mark.usefixtures("no_engine_settings")
def test_unconfigured_engine_takes_precedence_over_skip(
self,
remote_parser: RemoteDocumentParser,
simple_digital_pdf_file: Path,
) -> None:
"""An unconfigured engine still short-circuits before the
produce_archive check, returning empty text as before.
"""
remote_parser.parse(
simple_digital_pdf_file,
"application/pdf",
produce_archive=False,
)
assert remote_parser.get_text() == ""
assert remote_parser.get_archive_path() is None
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# parse() — Azure failure path # parse() — Azure failure path
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
+21 -6
View File
@@ -2,6 +2,8 @@ import json
import logging import logging
import sys import sys
from django.db.models import QuerySet
from documents.models import Document from documents.models import Document
from paperless.config import AIConfig from paperless.config import AIConfig
from paperless_ai.client import AIClient from paperless_ai.client import AIClient
@@ -82,10 +84,21 @@ def _build_document_reference(
def _get_document_references( def _get_document_references(
documents: list[Document], documents: QuerySet[Document],
top_nodes: list, top_nodes: list,
) -> list[dict[str, int | str]]: ) -> list[dict[str, int | str]]:
allowed_documents = {doc.pk: doc for doc in documents} candidate_ids: set[int] = set()
for node in top_nodes:
try:
candidate_ids.add(int(node.metadata["document_id"]))
except (KeyError, TypeError, ValueError): # pragma: no cover
continue
if not candidate_ids:
return []
allowed_documents = {doc.pk: doc for doc in documents.filter(pk__in=candidate_ids)}
references: list[dict[str, int | str]] = [] references: list[dict[str, int | str]] = []
seen_document_ids: set[int] = set() seen_document_ids: set[int] = set()
@@ -119,7 +132,7 @@ def _format_chat_metadata_trailer(references: list[dict[str, int | str]]) -> str
def stream_chat_with_documents( def stream_chat_with_documents(
query_str: str, query_str: str,
documents: list[Document], documents: QuerySet[Document],
output_language: str | None = None, output_language: str | None = None,
): ):
try: try:
@@ -135,10 +148,10 @@ def stream_chat_with_documents(
def _stream_chat_with_documents( def _stream_chat_with_documents(
query_str: str, query_str: str,
documents: list[Document], documents: QuerySet[Document],
output_language: str | None = None, output_language: str | None = None,
): ):
if not documents: if not documents.exists():
yield CHAT_NO_CONTENT_MESSAGE yield CHAT_NO_CONTENT_MESSAGE
return return
@@ -148,7 +161,9 @@ def _stream_chat_with_documents(
from llama_index.core.retrievers import VectorIndexRetriever from llama_index.core.retrievers import VectorIndexRetriever
config = AIConfig() config = AIConfig()
filters = _document_id_filters(str(doc.pk) for doc in documents) filters = _document_id_filters(
str(pk) for pk in documents.values_list("pk", flat=True)
)
# Hold the shared read lock for the whole operation: the query engine # Hold the shared read lock for the whole operation: the query engine
# retrieves from the vector store again during synthesis, so the connection # retrieves from the vector store again during synthesis, so the connection
+101 -46
View File
@@ -3,10 +3,12 @@ from unittest.mock import MagicMock
from unittest.mock import patch from unittest.mock import patch
import pytest import pytest
from django.db.models.signals import post_init
from llama_index.core import settings as llama_settings from llama_index.core import settings as llama_settings
from llama_index.core.embeddings.mock_embed_model import MockEmbedding from llama_index.core.embeddings.mock_embed_model import MockEmbedding
from llama_index.core.schema import TextNode from llama_index.core.schema import TextNode
from documents.models import Document
from documents.tests.factories import DocumentFactory from documents.tests.factories import DocumentFactory
from paperless_ai import chat from paperless_ai import chat
from paperless_ai import indexing from paperless_ai import indexing
@@ -36,16 +38,6 @@ def patch_embed_nodes():
yield mock_embed_nodes yield mock_embed_nodes
@pytest.fixture
def mock_document():
doc = MagicMock()
doc.pk = 1
doc.title = "Test Document"
doc.filename = "test_file.pdf"
doc.content = "This is the document content."
return doc
def assert_chat_output( def assert_chat_output(
output: list[str], output: list[str],
*, *,
@@ -61,6 +53,13 @@ def assert_chat_output(
} }
def _fake_documents_queryset(pks: list[int]) -> MagicMock:
qs = MagicMock()
qs.exists.return_value = bool(pks)
qs.values_list.return_value = pks
return qs
@pytest.mark.parametrize( @pytest.mark.parametrize(
("output_language", "expected_language_line"), ("output_language", "expected_language_line"),
[ [
@@ -107,9 +106,10 @@ def test_build_refine_prompt(
@pytest.mark.django_db @pytest.mark.django_db
def test_stream_chat_with_one_document_retrieval( def test_stream_chat_with_one_document_retrieval(
mock_document,
patch_embed_nodes, patch_embed_nodes,
) -> None: ) -> None:
document = DocumentFactory.create(title="Test Document", content="ignored")
documents = Document.objects.filter(pk=document.pk)
with ( with (
patch("paperless_ai.chat.AIClient") as mock_client_cls, patch("paperless_ai.chat.AIClient") as mock_client_cls,
patch("paperless_ai.chat.load_or_build_index") as mock_load_index, patch("paperless_ai.chat.load_or_build_index") as mock_load_index,
@@ -124,22 +124,19 @@ def test_stream_chat_with_one_document_retrieval(
mock_client_cls.return_value = mock_client mock_client_cls.return_value = mock_client
mock_client.llm = MagicMock() mock_client.llm = MagicMock()
mock_node = TextNode(
text="This is node content.",
metadata={"document_id": str(mock_document.pk), "title": "Test Document"},
)
mock_index = MagicMock() mock_index = MagicMock()
# Simulate get_nodes returning nodes (content exists) mock_index.vector_store.get_nodes.return_value = [
mock_index.vector_store.get_nodes.return_value = [mock_node] TextNode(
text="This is node content.",
metadata={"document_id": str(document.pk), "title": "Test Document"},
),
]
mock_load_index.return_value = mock_index mock_load_index.return_value = mock_index
mock_retriever_instance = MagicMock() mock_retriever_instance = MagicMock()
mock_retriever_instance.retrieve.return_value = [ mock_retriever_instance.retrieve.return_value = [
MagicMock( MagicMock(
metadata={ metadata={"document_id": str(document.pk), "title": "Test Document"},
"document_id": str(mock_document.pk),
"title": "Test Document",
},
), ),
] ]
@@ -153,7 +150,7 @@ def test_stream_chat_with_one_document_retrieval(
"llama_index.core.retrievers.VectorIndexRetriever", "llama_index.core.retrievers.VectorIndexRetriever",
return_value=mock_retriever_instance, return_value=mock_retriever_instance,
): ):
output = list(stream_chat_with_documents("What is this?", [mock_document])) output = list(stream_chat_with_documents("What is this?", documents))
mock_query_engine.query.assert_called_once_with("What is this?") mock_query_engine.query.assert_called_once_with("What is this?")
synthesizer_kwargs = mock_get_response_synthesizer.call_args.kwargs synthesizer_kwargs = mock_get_response_synthesizer.call_args.kwargs
@@ -166,13 +163,16 @@ def test_stream_chat_with_one_document_retrieval(
output, output,
expected_chunks=["chunk1", "chunk2"], expected_chunks=["chunk1", "chunk2"],
expected_references=[ expected_references=[
{"id": mock_document.pk, "title": "Test Document"}, {"id": document.pk, "title": "Test Document"},
], ],
) )
@pytest.mark.django_db @pytest.mark.django_db
def test_stream_chat_with_multiple_documents_retrieval(patch_embed_nodes) -> None: def test_stream_chat_with_multiple_documents_retrieval(patch_embed_nodes) -> None:
doc1 = DocumentFactory.create(title="Document 1", content="ignored")
doc2 = DocumentFactory.create(title="Document 2", content="ignored")
documents = Document.objects.filter(pk__in=[doc1.pk, doc2.pk])
with ( with (
patch("paperless_ai.chat.AIClient") as mock_client_cls, patch("paperless_ai.chat.AIClient") as mock_client_cls,
patch("paperless_ai.chat.load_or_build_index") as mock_load_index, patch("paperless_ai.chat.load_or_build_index") as mock_load_index,
@@ -184,23 +184,23 @@ def test_stream_chat_with_multiple_documents_retrieval(patch_embed_nodes) -> Non
mock_client_cls.return_value = mock_client mock_client_cls.return_value = mock_client
mock_client.llm = MagicMock() mock_client.llm = MagicMock()
mock_node1 = TextNode(
text="Content for doc 1.",
metadata={"document_id": "1", "title": "Document 1"},
)
mock_node2 = TextNode(
text="Content for doc 2.",
metadata={"document_id": "2", "title": "Document 2"},
)
mock_index = MagicMock() mock_index = MagicMock()
# Simulate get_nodes returning nodes (content exists) mock_index.vector_store.get_nodes.return_value = [
mock_index.vector_store.get_nodes.return_value = [mock_node1, mock_node2] TextNode(
text="Content for doc 1.",
metadata={"document_id": str(doc1.pk), "title": "Document 1"},
),
TextNode(
text="Content for doc 2.",
metadata={"document_id": str(doc2.pk), "title": "Document 2"},
),
]
mock_load_index.return_value = mock_index mock_load_index.return_value = mock_index
mock_retriever_instance = MagicMock() mock_retriever_instance = MagicMock()
mock_retriever_instance.retrieve.return_value = [ mock_retriever_instance.retrieve.return_value = [
MagicMock(metadata={"document_id": "1", "title": "Document 1"}), MagicMock(metadata={"document_id": str(doc1.pk), "title": "Document 1"}),
MagicMock(metadata={"document_id": "2", "title": "Document 2"}), MagicMock(metadata={"document_id": str(doc2.pk), "title": "Document 2"}),
] ]
mock_response_stream = MagicMock() mock_response_stream = MagicMock()
@@ -210,14 +210,11 @@ def test_stream_chat_with_multiple_documents_retrieval(patch_embed_nodes) -> Non
mock_query_engine_cls.return_value = mock_query_engine mock_query_engine_cls.return_value = mock_query_engine
mock_query_engine.query.return_value = mock_response_stream mock_query_engine.query.return_value = mock_response_stream
doc1 = MagicMock(pk=1, title="Document 1", filename="doc1.pdf")
doc2 = MagicMock(pk=2, title="Document 2", filename="doc2.pdf")
with patch( with patch(
"llama_index.core.retrievers.VectorIndexRetriever", "llama_index.core.retrievers.VectorIndexRetriever",
return_value=mock_retriever_instance, return_value=mock_retriever_instance,
): ):
output = list(stream_chat_with_documents("What's up?", [doc1, doc2])) output = list(stream_chat_with_documents("What's up?", documents))
mock_query_engine.query.assert_called_once_with("What's up?") mock_query_engine.query.assert_called_once_with("What's up?")
patch_embed_nodes.assert_not_called() patch_embed_nodes.assert_not_called()
@@ -225,15 +222,15 @@ def test_stream_chat_with_multiple_documents_retrieval(patch_embed_nodes) -> Non
output, output,
expected_chunks=["chunk1", "chunk2"], expected_chunks=["chunk1", "chunk2"],
expected_references=[ expected_references=[
{"id": 1, "title": "Document 1"}, {"id": doc1.pk, "title": "Document 1"},
{"id": 2, "title": "Document 2"}, {"id": doc2.pk, "title": "Document 2"},
], ],
) )
def test_stream_chat_empty_document_list() -> None: def test_stream_chat_empty_document_list() -> None:
with patch("paperless_ai.chat.load_or_build_index") as mock_load_index: with patch("paperless_ai.chat.load_or_build_index") as mock_load_index:
output = list(stream_chat_with_documents("Any info?", [])) output = list(stream_chat_with_documents("Any info?", Document.objects.none()))
mock_load_index.assert_not_called() mock_load_index.assert_not_called()
assert output == ["Sorry, I couldn't find any content to answer your question."] assert output == ["Sorry, I couldn't find any content to answer your question."]
@@ -253,7 +250,9 @@ def test_stream_chat_no_matching_nodes() -> None:
mock_index.vector_store.get_nodes.return_value = [] mock_index.vector_store.get_nodes.return_value = []
mock_load_index.return_value = mock_index mock_load_index.return_value = mock_index
output = list(stream_chat_with_documents("Any info?", [MagicMock(pk=1)])) output = list(
stream_chat_with_documents("Any info?", _fake_documents_queryset([1])),
)
assert output == ["Sorry, I couldn't find any content to answer your question."] assert output == ["Sorry, I couldn't find any content to answer your question."]
@@ -282,7 +281,9 @@ def test_stream_chat_unexpected_failure_returns_generic_error(caplog) -> None:
) )
mock_retriever_cls.return_value = mock_retriever mock_retriever_cls.return_value = mock_retriever
output = list(stream_chat_with_documents("Any info?", [MagicMock(pk=1)])) output = list(
stream_chat_with_documents("Any info?", _fake_documents_queryset([1])),
)
assert output == [CHAT_ERROR_MESSAGE] assert output == [CHAT_ERROR_MESSAGE]
assert "Failed to stream document chat response" in caplog.text assert "Failed to stream document chat response" in caplog.text
@@ -298,7 +299,12 @@ class TestStreamChatRetrieval:
) -> None: ) -> None:
doc = DocumentFactory.create(content="hello world") doc = DocumentFactory.create(content="hello world")
# Nothing indexed for this document yet. # Nothing indexed for this document yet.
out = list(chat.stream_chat_with_documents("question?", [doc])) out = list(
chat.stream_chat_with_documents(
"question?",
Document.objects.filter(pk=doc.pk),
),
)
assert chat.CHAT_NO_CONTENT_MESSAGE in out assert chat.CHAT_NO_CONTENT_MESSAGE in out
def test_chat_filter_contains_only_requested_document_ids( def test_chat_filter_contains_only_requested_document_ids(
@@ -332,7 +338,12 @@ class TestStreamChatRetrieval:
side_effect=capture_retriever, side_effect=capture_retriever,
) )
list(chat.stream_chat_with_documents("question?", [included])) list(
chat.stream_chat_with_documents(
"question?",
Document.objects.filter(pk=included.pk),
),
)
assert captured_filters, "VectorIndexRetriever was never constructed" assert captured_filters, "VectorIndexRetriever was never constructed"
filt = captured_filters[0] filt = captured_filters[0]
@@ -340,3 +351,47 @@ class TestStreamChatRetrieval:
filter_values = filt.filters[0].value filter_values = filt.filters[0].value
assert str(included.pk) in filter_values assert str(included.pk) in filter_values
assert str(excluded.pk) not in filter_values assert str(excluded.pk) not in filter_values
@pytest.mark.django_db
def test_get_document_references_only_queries_referenced_documents(
self,
django_assert_num_queries,
) -> None:
"""Building references must not hydrate every document the caller is
permitted to see -- only the (<= CHAT_RETRIEVER_TOP_K) documents that
the retriever actually returned nodes for.
"""
referenced = DocumentFactory.create(title="Referenced Document")
# Many more documents are "accessible" but never referenced by a node.
DocumentFactory.create_batch(200)
documents = Document.objects.all()
top_nodes = [
MagicMock(
metadata={
"document_id": str(referenced.pk),
"title": "Referenced Document",
},
),
]
hydrated_count = 0
def _count_hydration(sender, instance, **kwargs):
nonlocal hydrated_count
hydrated_count += 1
post_init.connect(_count_hydration, sender=Document)
try:
# One query: `documents.filter(pk__in=candidate_ids)` for the single
# referenced id. No query should scale with the 200 unreferenced documents.
with django_assert_num_queries(1):
references = chat._get_document_references(documents, top_nodes)
finally:
post_init.disconnect(_count_hydration, sender=Document)
# The bug this guards against: the old code hydrated all 201 accessible
# documents via `{doc.pk: doc for doc in documents}` before filtering by
# top_nodes. Only the referenced document should ever be constructed.
assert hydrated_count == 1
assert references == [{"id": referenced.pk, "title": "Referenced Document"}]