mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-09-10 19:58:01 +00:00
Compare commits
19
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f60e851c08 | ||
|
|
9a8163fbbb | ||
|
|
d5f9605daf | ||
|
|
8593f84cae | ||
|
|
ca512af5ec | ||
|
|
a00755907e | ||
|
|
abdf15466c | ||
|
|
54a6f0fd2b | ||
|
|
2d64684043 | ||
|
|
60709b8319 | ||
|
|
1c96819625 | ||
|
|
3e56dace73 | ||
|
|
310628699d | ||
|
|
aff0f9cf41 | ||
|
|
bf716ebfd1 | ||
|
|
7d67a10a35 | ||
|
|
8d1bc5dd24 | ||
|
|
43a8d7d412 | ||
|
|
c40922440b |
@@ -0,0 +1,58 @@
|
|||||||
|
#!/command/with-contenv /usr/bin/bash
|
||||||
|
# shellcheck shell=bash
|
||||||
|
declare -r log_prefix="[init-compile-bytecode]"
|
||||||
|
|
||||||
|
# PYTHONDONTWRITEBYTECODE=1 is set for the whole container. This unit compiles a
|
||||||
|
# scoped set of libraries anyway, to speed up startup without bloating image size.
|
||||||
|
|
||||||
|
# Handle the people using a read only file system
|
||||||
|
if [[ "${S6_READ_ONLY_ROOT}" == "1" ]]; then
|
||||||
|
echo "${log_prefix} S6_READ_ONLY_ROOT=1, skipping (nothing to write bytecode to)"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
# When running as a non-root user, site-packages is still root-owned and unwritable,
|
||||||
|
# so this step would just fail loudly on every container start. Skip it.
|
||||||
|
if [[ -n "${USER_IS_NON_ROOT}" ]]; then
|
||||||
|
echo "${log_prefix} USER_IS_NON_ROOT is set, skipping (site-packages is not writable)"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
declare -r site_packages="$(python3 -c 'import site; print(site.getsitepackages()[0])')"
|
||||||
|
|
||||||
|
# Deliberately scoped to packages that paperless.settings/paperless/__init__.py import
|
||||||
|
# unconditionally on every manage.py invocation (Django itself, the always-loaded
|
||||||
|
# INSTALLED_APPS, and celery). This is NOT "compile everything" - the optional AI stack
|
||||||
|
# (torch, llama-index, sentence-transformers, ...) is intentionally excluded since it is
|
||||||
|
# lazy-imported and large.
|
||||||
|
declare -a scope=(
|
||||||
|
"${PAPERLESS_SRC_DIR}"
|
||||||
|
"${site_packages}/django"
|
||||||
|
"${site_packages}/celery"
|
||||||
|
"${site_packages}/kombu"
|
||||||
|
"${site_packages}/rest_framework"
|
||||||
|
"${site_packages}/django_filters"
|
||||||
|
"${site_packages}/whitenoise"
|
||||||
|
"${site_packages}/corsheaders"
|
||||||
|
"${site_packages}/django_extensions"
|
||||||
|
"${site_packages}/guardian"
|
||||||
|
"${site_packages}/allauth"
|
||||||
|
"${site_packages}/drf_spectacular"
|
||||||
|
"${site_packages}/drf_spectacular_sidecar"
|
||||||
|
"${site_packages}/treenode"
|
||||||
|
"${site_packages}/compression_middleware"
|
||||||
|
)
|
||||||
|
|
||||||
|
declare -a existing_scope=()
|
||||||
|
for path in "${scope[@]}"; do
|
||||||
|
[[ -d "${path}" ]] && existing_scope+=("${path}")
|
||||||
|
done
|
||||||
|
|
||||||
|
echo "${log_prefix} Compiling bytecode for: ${existing_scope[*]}"
|
||||||
|
declare -r start_seconds=${SECONDS}
|
||||||
|
|
||||||
|
if ! PYTHONDONTWRITEBYTECODE= python3 -m compileall -q "${existing_scope[@]}"; then
|
||||||
|
echo "${log_prefix} WARNING: compileall reported errors (read-only filesystem or unwritable site-packages?); continuing without a bytecode cache"
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "${log_prefix} Done in $((SECONDS - start_seconds))s"
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
oneshot
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
/etc/s6-overlay/s6-rc.d/init-compile-bytecode/run
|
||||||
@@ -1200,6 +1200,15 @@ still perform some basic text pre-processing before matching.
|
|||||||
|
|
||||||
Defaults to true, enabling the feature.
|
Defaults to true, enabling the feature.
|
||||||
|
|
||||||
|
#### [`PAPERLESS_CLASSIFIER_MATCH_THRESHOLD=<float>`](#PAPERLESS_CLASSIFIER_MATCH_THRESHOLD) {#PAPERLESS_CLASSIFIER_MATCH_THRESHOLD}
|
||||||
|
|
||||||
|
: Sets the minimum confidence score (0.0-1.0) required for the automatic
|
||||||
|
classifier to assign a correspondent, document type, or storage path to a
|
||||||
|
document. Predictions below this threshold are discarded and the field is
|
||||||
|
left unassigned, preventing low-confidence guesses from being applied.
|
||||||
|
|
||||||
|
Defaults to 0.6.
|
||||||
|
|
||||||
#### [`PAPERLESS_DATE_PARSER_LANGUAGES=<lang>`](#PAPERLESS_DATE_PARSER_LANGUAGES) {#PAPERLESS_DATE_PARSER_LANGUAGES}
|
#### [`PAPERLESS_DATE_PARSER_LANGUAGES=<lang>`](#PAPERLESS_DATE_PARSER_LANGUAGES) {#PAPERLESS_DATE_PARSER_LANGUAGES}
|
||||||
|
|
||||||
: Specifies which language Paperless should use when parsing dates from documents.
|
: Specifies which language Paperless should use when parsing dates from documents.
|
||||||
|
|||||||
+250
-203
File diff suppressed because it is too large
Load Diff
@@ -112,6 +112,22 @@
|
|||||||
|
|
||||||
<pngx-input-check i18n-title title="Use 'slim' sidebar (icons only)" formControlName="slimSidebarEnabled"></pngx-input-check>
|
<pngx-input-check i18n-title title="Use 'slim' sidebar (icons only)" formControlName="slimSidebarEnabled"></pngx-input-check>
|
||||||
|
|
||||||
|
<p class="mb-2 mt-3" i18n>Sidebar items to show:</p>
|
||||||
|
@for (option of sidebarItemOptions; track option.id) {
|
||||||
|
<div class="form-check">
|
||||||
|
<input
|
||||||
|
class="form-check-input"
|
||||||
|
type="checkbox"
|
||||||
|
[id]="'sidebar-item-setting-' + option.id"
|
||||||
|
[checked]="isSidebarItemShown(option.id)"
|
||||||
|
(change)="toggleSidebarItem(option.id, $event.target.checked)"
|
||||||
|
/>
|
||||||
|
<label class="form-check-label" [for]="'sidebar-item-setting-' + option.id">
|
||||||
|
{{ option.label }}
|
||||||
|
</label>
|
||||||
|
</div>
|
||||||
|
}
|
||||||
|
|
||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
|
|||||||
@@ -24,7 +24,7 @@ import {
|
|||||||
SystemStatus,
|
SystemStatus,
|
||||||
SystemStatusItemStatus,
|
SystemStatusItemStatus,
|
||||||
} from 'src/app/data/system-status'
|
} from 'src/app/data/system-status'
|
||||||
import { SETTINGS_KEYS } from 'src/app/data/ui-settings'
|
import { HideableSidebarItemID, SETTINGS_KEYS } from 'src/app/data/ui-settings'
|
||||||
import { IfOwnerDirective } from 'src/app/directives/if-owner.directive'
|
import { IfOwnerDirective } from 'src/app/directives/if-owner.directive'
|
||||||
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
|
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
|
||||||
import { PermissionsGuard } from 'src/app/guards/permissions.guard'
|
import { PermissionsGuard } from 'src/app/guards/permissions.guard'
|
||||||
@@ -209,6 +209,45 @@ describe('SettingsComponent', () => {
|
|||||||
fixture.detectChanges()
|
fixture.detectChanges()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
it('supports configuring sidebar items and canceling changes', () => {
|
||||||
|
completeSetup()
|
||||||
|
|
||||||
|
component.toggleSidebarItem(HideableSidebarItemID.Workflows, false)
|
||||||
|
fixture.detectChanges()
|
||||||
|
|
||||||
|
expect(component.settingsForm.value.sidebarHiddenItems).toContain(
|
||||||
|
HideableSidebarItemID.Workflows
|
||||||
|
)
|
||||||
|
|
||||||
|
settingsService.updateSidebarItemVisibility(
|
||||||
|
HideableSidebarItemID.Mail,
|
||||||
|
false
|
||||||
|
)
|
||||||
|
|
||||||
|
expect(component.settingsForm.value.sidebarHiddenItems).toContain(
|
||||||
|
HideableSidebarItemID.Mail
|
||||||
|
)
|
||||||
|
|
||||||
|
component.reset()
|
||||||
|
|
||||||
|
expect(component.settingsForm.value.sidebarHiddenItems).not.toContain(
|
||||||
|
HideableSidebarItemID.Workflows
|
||||||
|
)
|
||||||
|
expect(component.settingsForm.value.sidebarHiddenItems).not.toContain(
|
||||||
|
HideableSidebarItemID.Mail
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('enables sidebar item controls on general settings until destroyed', () => {
|
||||||
|
completeSetup()
|
||||||
|
|
||||||
|
expect(settingsService.organizingSidebarItems()).toBe(true)
|
||||||
|
|
||||||
|
component.ngOnDestroy()
|
||||||
|
|
||||||
|
expect(settingsService.organizingSidebarItems()).toBe(false)
|
||||||
|
})
|
||||||
|
|
||||||
it('should support tabbed settings & change URL, prevent navigation if dirty confirmation rejected', async () => {
|
it('should support tabbed settings & change URL, prevent navigation if dirty confirmation rejected', async () => {
|
||||||
completeSetup()
|
completeSetup()
|
||||||
const navigateSpy = jest.spyOn(router, 'navigate')
|
const navigateSpy = jest.spyOn(router, 'navigate')
|
||||||
@@ -249,6 +288,7 @@ describe('SettingsComponent', () => {
|
|||||||
|
|
||||||
it('should support save local settings updating appearance settings and calling API, show error', () => {
|
it('should support save local settings updating appearance settings and calling API, show error', () => {
|
||||||
completeSetup()
|
completeSetup()
|
||||||
|
component.toggleSidebarItem(HideableSidebarItemID.Workflows, false)
|
||||||
const toastErrorSpy = jest.spyOn(toastService, 'showError')
|
const toastErrorSpy = jest.spyOn(toastService, 'showError')
|
||||||
const toastSpy = jest.spyOn(toastService, 'show')
|
const toastSpy = jest.spyOn(toastService, 'show')
|
||||||
const storeSpy = jest.spyOn(settingsService, 'storeSettings')
|
const storeSpy = jest.spyOn(settingsService, 'storeSettings')
|
||||||
@@ -267,7 +307,10 @@ describe('SettingsComponent', () => {
|
|||||||
expect(toastErrorSpy).toHaveBeenCalled()
|
expect(toastErrorSpy).toHaveBeenCalled()
|
||||||
expect(storeSpy).toHaveBeenCalled()
|
expect(storeSpy).toHaveBeenCalled()
|
||||||
expect(appearanceSettingsSpy).not.toHaveBeenCalled()
|
expect(appearanceSettingsSpy).not.toHaveBeenCalled()
|
||||||
expect(setSpy).toHaveBeenCalledTimes(33)
|
expect(setSpy).toHaveBeenCalledTimes(34)
|
||||||
|
expect(setSpy).toHaveBeenCalledWith(SETTINGS_KEYS.SIDEBAR_HIDDEN_ITEMS, [
|
||||||
|
HideableSidebarItemID.Workflows,
|
||||||
|
])
|
||||||
|
|
||||||
// succeed
|
// succeed
|
||||||
storeSpy.mockReturnValueOnce(of(true))
|
storeSpy.mockReturnValueOnce(of(true))
|
||||||
|
|||||||
@@ -39,7 +39,12 @@ import {
|
|||||||
SystemStatus,
|
SystemStatus,
|
||||||
SystemStatusItemStatus,
|
SystemStatusItemStatus,
|
||||||
} from 'src/app/data/system-status'
|
} from 'src/app/data/system-status'
|
||||||
import { GlobalSearchType, SETTINGS_KEYS } from 'src/app/data/ui-settings'
|
import {
|
||||||
|
GlobalSearchType,
|
||||||
|
HIDEABLE_SIDEBAR_ITEM_IDS,
|
||||||
|
HideableSidebarItemID,
|
||||||
|
SETTINGS_KEYS,
|
||||||
|
} from 'src/app/data/ui-settings'
|
||||||
import { User } from 'src/app/data/user'
|
import { User } from 'src/app/data/user'
|
||||||
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
|
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
|
||||||
import { CustomDatePipe } from 'src/app/pipes/custom-date.pipe'
|
import { CustomDatePipe } from 'src/app/pipes/custom-date.pipe'
|
||||||
@@ -102,6 +107,14 @@ const documentDetailFieldOptions = [
|
|||||||
{ id: DocumentDetailFieldID.Tags, label: $localize`Tags` },
|
{ id: DocumentDetailFieldID.Tags, label: $localize`Tags` },
|
||||||
]
|
]
|
||||||
|
|
||||||
|
const sidebarItemLabels: Record<HideableSidebarItemID, string> = {
|
||||||
|
[HideableSidebarItemID.Dashboard]: $localize`Dashboard`,
|
||||||
|
[HideableSidebarItemID.SavedViews]: $localize`Saved Views`,
|
||||||
|
[HideableSidebarItemID.Workflows]: $localize`Workflows`,
|
||||||
|
[HideableSidebarItemID.Mail]: $localize`Mail`,
|
||||||
|
[HideableSidebarItemID.Documentation]: $localize`Documentation`,
|
||||||
|
}
|
||||||
|
|
||||||
@Component({
|
@Component({
|
||||||
selector: 'pngx-settings',
|
selector: 'pngx-settings',
|
||||||
templateUrl: './settings.component.html',
|
templateUrl: './settings.component.html',
|
||||||
@@ -149,6 +162,7 @@ export class SettingsComponent
|
|||||||
bulkEditApplyOnClose: new FormControl(null),
|
bulkEditApplyOnClose: new FormControl(null),
|
||||||
documentListItemPerPage: new FormControl(null),
|
documentListItemPerPage: new FormControl(null),
|
||||||
slimSidebarEnabled: new FormControl(null),
|
slimSidebarEnabled: new FormControl(null),
|
||||||
|
sidebarHiddenItems: new FormControl<HideableSidebarItemID[]>([]),
|
||||||
darkModeUseSystem: new FormControl(null),
|
darkModeUseSystem: new FormControl(null),
|
||||||
darkModeEnabled: new FormControl(null),
|
darkModeEnabled: new FormControl(null),
|
||||||
darkModeInvertThumbs: new FormControl(null),
|
darkModeInvertThumbs: new FormControl(null),
|
||||||
@@ -186,6 +200,7 @@ export class SettingsComponent
|
|||||||
|
|
||||||
store: BehaviorSubject<any>
|
store: BehaviorSubject<any>
|
||||||
storeSub: Subscription
|
storeSub: Subscription
|
||||||
|
sidebarItemsSub: Subscription
|
||||||
isDirty$: Observable<boolean>
|
isDirty$: Observable<boolean>
|
||||||
isDirty: boolean = false
|
isDirty: boolean = false
|
||||||
unsubscribeNotifier: Subject<any> = new Subject()
|
unsubscribeNotifier: Subject<any> = new Subject()
|
||||||
@@ -203,6 +218,10 @@ export class SettingsComponent
|
|||||||
public readonly PdfEditorEditMode = PdfEditorEditMode
|
public readonly PdfEditorEditMode = PdfEditorEditMode
|
||||||
|
|
||||||
public readonly documentDetailFieldOptions = documentDetailFieldOptions
|
public readonly documentDetailFieldOptions = documentDetailFieldOptions
|
||||||
|
public readonly sidebarItemOptions = HIDEABLE_SIDEBAR_ITEM_IDS.map((id) => ({
|
||||||
|
id,
|
||||||
|
label: sidebarItemLabels[id],
|
||||||
|
}))
|
||||||
|
|
||||||
get systemStatusHasErrors(): boolean {
|
get systemStatusHasErrors(): boolean {
|
||||||
const status = this.systemStatus()
|
const status = this.systemStatus()
|
||||||
@@ -230,6 +249,10 @@ export class SettingsComponent
|
|||||||
|
|
||||||
constructor() {
|
constructor() {
|
||||||
super()
|
super()
|
||||||
|
this.sidebarItemsSub =
|
||||||
|
this.settings.sidebarHiddenItemsEditingChanged.subscribe((hiddenItems) =>
|
||||||
|
this.settingsForm.controls.sidebarHiddenItems.setValue(hiddenItems)
|
||||||
|
)
|
||||||
this.settings.settingsSaved.subscribe(() => {
|
this.settings.settingsSaved.subscribe(() => {
|
||||||
if (!this.savePending) this.initialize()
|
if (!this.savePending) this.initialize()
|
||||||
this.savedViewsService.maybeRefreshDocumentCounts()
|
this.savedViewsService.maybeRefreshDocumentCounts()
|
||||||
@@ -279,14 +302,21 @@ export class SettingsComponent
|
|||||||
|
|
||||||
this.activatedRoute.paramMap.subscribe((paramMap) => {
|
this.activatedRoute.paramMap.subscribe((paramMap) => {
|
||||||
const section = paramMap.get('section')
|
const section = paramMap.get('section')
|
||||||
|
let navID = SettingsNavIDs.General
|
||||||
if (section) {
|
if (section) {
|
||||||
const navIDKey: string = Object.keys(SettingsNavIDs).find(
|
const navIDKey: string = Object.keys(SettingsNavIDs).find(
|
||||||
(navID) => navID.toLowerCase() == section
|
(navID) => navID.toLowerCase() == section
|
||||||
)
|
)
|
||||||
if (navIDKey) {
|
if (navIDKey) {
|
||||||
this.activeNavID.set(SettingsNavIDs[navIDKey])
|
navID = SettingsNavIDs[navIDKey]
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
this.activeNavID.set(navID)
|
||||||
|
this.settings.sidebarHiddenItemsEditing.set(
|
||||||
|
navID === SettingsNavIDs.General
|
||||||
|
? [...this.settingsForm.controls.sidebarHiddenItems.value]
|
||||||
|
: null
|
||||||
|
)
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -310,6 +340,7 @@ export class SettingsComponent
|
|||||||
SETTINGS_KEYS.DOCUMENT_LIST_SIZE
|
SETTINGS_KEYS.DOCUMENT_LIST_SIZE
|
||||||
),
|
),
|
||||||
slimSidebarEnabled: this.settings.get(SETTINGS_KEYS.SLIM_SIDEBAR),
|
slimSidebarEnabled: this.settings.get(SETTINGS_KEYS.SLIM_SIDEBAR),
|
||||||
|
sidebarHiddenItems: this.settings.get(SETTINGS_KEYS.SIDEBAR_HIDDEN_ITEMS),
|
||||||
darkModeUseSystem: this.settings.get(SETTINGS_KEYS.DARK_MODE_USE_SYSTEM),
|
darkModeUseSystem: this.settings.get(SETTINGS_KEYS.DARK_MODE_USE_SYSTEM),
|
||||||
darkModeEnabled: this.settings.get(SETTINGS_KEYS.DARK_MODE_ENABLED),
|
darkModeEnabled: this.settings.get(SETTINGS_KEYS.DARK_MODE_ENABLED),
|
||||||
darkModeInvertThumbs: this.settings.get(
|
darkModeInvertThumbs: this.settings.get(
|
||||||
@@ -436,6 +467,12 @@ export class SettingsComponent
|
|||||||
this.settingsForm.patchValue(currentFormValue)
|
this.settingsForm.patchValue(currentFormValue)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (this.settings.organizingSidebarItems()) {
|
||||||
|
this.settings.sidebarHiddenItemsEditing.set([
|
||||||
|
...this.settingsForm.controls.sidebarHiddenItems.value,
|
||||||
|
])
|
||||||
|
}
|
||||||
|
|
||||||
if (this.canViewSystemStatus) {
|
if (this.canViewSystemStatus) {
|
||||||
this.systemStatusService.get().subscribe((status) => {
|
this.systemStatusService.get().subscribe((status) => {
|
||||||
this.systemStatus.set(status)
|
this.systemStatus.set(status)
|
||||||
@@ -444,8 +481,18 @@ export class SettingsComponent
|
|||||||
}
|
}
|
||||||
|
|
||||||
ngOnDestroy() {
|
ngOnDestroy() {
|
||||||
|
this.settings.sidebarHiddenItemsEditing.set(null)
|
||||||
if (this.isDirty) this.settings.updateAppearanceSettings() // in case user changed appearance but didn't save
|
if (this.isDirty) this.settings.updateAppearanceSettings() // in case user changed appearance but didn't save
|
||||||
this.storeSub && this.storeSub.unsubscribe()
|
this.storeSub && this.storeSub.unsubscribe()
|
||||||
|
this.sidebarItemsSub.unsubscribe()
|
||||||
|
}
|
||||||
|
|
||||||
|
isSidebarItemShown(item: HideableSidebarItemID): boolean {
|
||||||
|
return !(this.settingsForm.value.sidebarHiddenItems || []).includes(item)
|
||||||
|
}
|
||||||
|
|
||||||
|
toggleSidebarItem(item: HideableSidebarItemID, checked: boolean): void {
|
||||||
|
this.settings.updateSidebarItemVisibility(item, checked)
|
||||||
}
|
}
|
||||||
|
|
||||||
public saveSettings() {
|
public saveSettings() {
|
||||||
@@ -473,6 +520,10 @@ export class SettingsComponent
|
|||||||
SETTINGS_KEYS.SLIM_SIDEBAR,
|
SETTINGS_KEYS.SLIM_SIDEBAR,
|
||||||
this.settingsForm.value.slimSidebarEnabled
|
this.settingsForm.value.slimSidebarEnabled
|
||||||
)
|
)
|
||||||
|
this.settings.set(
|
||||||
|
SETTINGS_KEYS.SIDEBAR_HIDDEN_ITEMS,
|
||||||
|
this.settingsForm.value.sidebarHiddenItems
|
||||||
|
)
|
||||||
this.settings.set(
|
this.settings.set(
|
||||||
SETTINGS_KEYS.DARK_MODE_USE_SYSTEM,
|
SETTINGS_KEYS.DARK_MODE_USE_SYSTEM,
|
||||||
this.settingsForm.value.darkModeUseSystem
|
this.settingsForm.value.darkModeUseSystem
|
||||||
@@ -632,6 +683,11 @@ export class SettingsComponent
|
|||||||
|
|
||||||
reset() {
|
reset() {
|
||||||
this.settingsForm.patchValue(this.store.getValue())
|
this.settingsForm.patchValue(this.store.getValue())
|
||||||
|
if (this.settings.organizingSidebarItems()) {
|
||||||
|
this.settings.sidebarHiddenItemsEditing.set([
|
||||||
|
...this.settingsForm.controls.sidebarHiddenItems.value,
|
||||||
|
])
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
clearThemeColor() {
|
clearThemeColor() {
|
||||||
|
|||||||
@@ -86,12 +86,15 @@
|
|||||||
}
|
}
|
||||||
<div class="sidebar-sticky pt-3 pb-1 d-flex flex-column justify-space-around">
|
<div class="sidebar-sticky pt-3 pb-1 d-flex flex-column justify-space-around">
|
||||||
<ul class="nav flex-column">
|
<ul class="nav flex-column">
|
||||||
<li class="nav-item app-link">
|
<li class="nav-item app-link position-relative" [class.d-none]="settingsService.sidebarItemIsHidden(HideableSidebarItemID.Dashboard) && !settingsService.organizingSidebarItems()">
|
||||||
<a class="nav-link" routerLink="dashboard" routerLinkActive="active" (click)="closeMenu()"
|
<a class="nav-link" [class.opacity-50]="settingsService.sidebarItemIsHidden(HideableSidebarItemID.Dashboard)" [class.pe-5]="settingsService.organizingSidebarItems() && !slimSidebarEnabled && !slimSidebarAnimating()" routerLink="dashboard" routerLinkActive="active" (click)="closeMenu()"
|
||||||
ngbPopover="Dashboard" i18n-ngbPopover [disablePopover]="!slimSidebarPopoversEnabled" placement="end"
|
ngbPopover="Dashboard" i18n-ngbPopover [disablePopover]="!slimSidebarPopoversEnabled" placement="end"
|
||||||
container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
||||||
<i-bs class="me-2" name="house"></i-bs><span class="nav-link-label"><ng-container i18n>Dashboard</ng-container></span>
|
<i-bs class="me-2" name="house"></i-bs><span class="nav-link-label"><ng-container i18n>Dashboard</ng-container></span>
|
||||||
</a>
|
</a>
|
||||||
|
@if (settingsService.organizingSidebarItems()) {
|
||||||
|
<pngx-input-switch class="position-absolute top-50 end-0 translate-middle-y me-1" [class.d-none]="slimSidebarEnabled || slimSidebarAnimating()" [compact]="true" title="Dashboard" i18n-title [ngModel]="!settingsService.sidebarItemIsHidden(HideableSidebarItemID.Dashboard)" (ngModelChange)="toggleSidebarItem(HideableSidebarItemID.Dashboard, $event)"></pngx-input-switch>
|
||||||
|
}
|
||||||
</li>
|
</li>
|
||||||
<li class="nav-item app-link" *pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.Document }">
|
<li class="nav-item app-link" *pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.Document }">
|
||||||
<a class="nav-link" routerLink="documents" routerLinkActive="active"
|
<a class="nav-link" routerLink="documents" routerLinkActive="active"
|
||||||
@@ -237,29 +240,38 @@
|
|||||||
</div>
|
</div>
|
||||||
</li>
|
</li>
|
||||||
}
|
}
|
||||||
<li class="nav-item app-link" *pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.SavedView }">
|
<li class="nav-item app-link position-relative" [class.d-none]="settingsService.sidebarItemIsHidden(HideableSidebarItemID.SavedViews) && !settingsService.organizingSidebarItems()" *pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.SavedView }">
|
||||||
<a class="nav-link" routerLink="savedviews" routerLinkActive="active" (click)="closeMenu()"
|
<a class="nav-link" [class.opacity-50]="settingsService.sidebarItemIsHidden(HideableSidebarItemID.SavedViews)" [class.pe-5]="settingsService.organizingSidebarItems() && !slimSidebarEnabled && !slimSidebarAnimating()" routerLink="savedviews" routerLinkActive="active" (click)="closeMenu()"
|
||||||
ngbPopover="Saved Views" i18n-ngbPopover [disablePopover]="!slimSidebarPopoversEnabled" placement="end"
|
ngbPopover="Saved Views" i18n-ngbPopover [disablePopover]="!slimSidebarPopoversEnabled" placement="end"
|
||||||
container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
||||||
<i-bs class="me-2" name="window-stack"></i-bs><span class="nav-link-label"><ng-container i18n>Saved Views</ng-container></span>
|
<i-bs class="me-2" name="window-stack"></i-bs><span class="nav-link-label"><ng-container i18n>Saved Views</ng-container></span>
|
||||||
</a>
|
</a>
|
||||||
|
@if (settingsService.organizingSidebarItems()) {
|
||||||
|
<pngx-input-switch class="position-absolute top-50 end-0 translate-middle-y me-1" [class.d-none]="slimSidebarEnabled || slimSidebarAnimating()" [compact]="true" title="Saved Views" i18n-title [ngModel]="!settingsService.sidebarItemIsHidden(HideableSidebarItemID.SavedViews)" (ngModelChange)="toggleSidebarItem(HideableSidebarItemID.SavedViews, $event)"></pngx-input-switch>
|
||||||
|
}
|
||||||
</li>
|
</li>
|
||||||
<li class="nav-item app-link"
|
<li class="nav-item app-link position-relative" [class.d-none]="settingsService.sidebarItemIsHidden(HideableSidebarItemID.Workflows) && !settingsService.organizingSidebarItems()"
|
||||||
*pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.Workflow }"
|
*pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.Workflow }"
|
||||||
tourAnchor="tour.workflows">
|
tourAnchor="tour.workflows">
|
||||||
<a class="nav-link" routerLink="workflows" routerLinkActive="active" (click)="closeMenu()"
|
<a class="nav-link" [class.opacity-50]="settingsService.sidebarItemIsHidden(HideableSidebarItemID.Workflows)" [class.pe-5]="settingsService.organizingSidebarItems() && !slimSidebarEnabled && !slimSidebarAnimating()" routerLink="workflows" routerLinkActive="active" (click)="closeMenu()"
|
||||||
ngbPopover="Workflows" i18n-ngbPopover [disablePopover]="!slimSidebarPopoversEnabled" placement="end"
|
ngbPopover="Workflows" i18n-ngbPopover [disablePopover]="!slimSidebarPopoversEnabled" placement="end"
|
||||||
container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
||||||
<i-bs class="me-2" name="boxes"></i-bs><span class="nav-link-label"><ng-container i18n>Workflows</ng-container></span>
|
<i-bs class="me-2" name="boxes"></i-bs><span class="nav-link-label"><ng-container i18n>Workflows</ng-container></span>
|
||||||
</a>
|
</a>
|
||||||
|
@if (settingsService.organizingSidebarItems()) {
|
||||||
|
<pngx-input-switch class="position-absolute top-50 end-0 translate-middle-y me-1" [class.d-none]="slimSidebarEnabled || slimSidebarAnimating()" [compact]="true" title="Workflows" i18n-title [ngModel]="!settingsService.sidebarItemIsHidden(HideableSidebarItemID.Workflows)" (ngModelChange)="toggleSidebarItem(HideableSidebarItemID.Workflows, $event)"></pngx-input-switch>
|
||||||
|
}
|
||||||
</li>
|
</li>
|
||||||
<li class="nav-item app-link" *pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.MailAccount }"
|
<li class="nav-item app-link position-relative" [class.d-none]="settingsService.sidebarItemIsHidden(HideableSidebarItemID.Mail) && !settingsService.organizingSidebarItems()" *pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.MailAccount }"
|
||||||
tourAnchor="tour.mail">
|
tourAnchor="tour.mail">
|
||||||
<a class="nav-link" routerLink="mail" routerLinkActive="active" (click)="closeMenu()" ngbPopover="Mail"
|
<a class="nav-link" [class.opacity-50]="settingsService.sidebarItemIsHidden(HideableSidebarItemID.Mail)" [class.pe-5]="settingsService.organizingSidebarItems() && !slimSidebarEnabled && !slimSidebarAnimating()" routerLink="mail" routerLinkActive="active" (click)="closeMenu()" ngbPopover="Mail"
|
||||||
i18n-ngbPopover [disablePopover]="!slimSidebarPopoversEnabled" placement="end" container="body"
|
i18n-ngbPopover [disablePopover]="!slimSidebarPopoversEnabled" placement="end" container="body"
|
||||||
triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
||||||
<i-bs class="me-2" name="envelope"></i-bs><span class="nav-link-label"><ng-container i18n>Mail</ng-container></span>
|
<i-bs class="me-2" name="envelope"></i-bs><span class="nav-link-label"><ng-container i18n>Mail</ng-container></span>
|
||||||
</a>
|
</a>
|
||||||
|
@if (settingsService.organizingSidebarItems()) {
|
||||||
|
<pngx-input-switch class="position-absolute top-50 end-0 translate-middle-y me-1" [class.d-none]="slimSidebarEnabled || slimSidebarAnimating()" [compact]="true" title="Mail" i18n-title [ngModel]="!settingsService.sidebarItemIsHidden(HideableSidebarItemID.Mail)" (ngModelChange)="toggleSidebarItem(HideableSidebarItemID.Mail, $event)"></pngx-input-switch>
|
||||||
|
}
|
||||||
</li>
|
</li>
|
||||||
<li class="nav-item app-link" *pngxIfPermissions="{ action: PermissionAction.Delete, type: PermissionType.Document }">
|
<li class="nav-item app-link" *pngxIfPermissions="{ action: PermissionAction.Delete, type: PermissionType.Document }">
|
||||||
<a class="nav-link" routerLink="trash" routerLinkActive="active" (click)="closeMenu()" ngbPopover="Trash"
|
<a class="nav-link" routerLink="trash" routerLinkActive="active" (click)="closeMenu()" ngbPopover="Trash"
|
||||||
@@ -322,13 +334,16 @@
|
|||||||
</a>
|
</a>
|
||||||
</li>
|
</li>
|
||||||
}
|
}
|
||||||
<li class="nav-item mt-2" tourAnchor="tour.outro">
|
<li class="nav-item mt-2 position-relative" [class.d-none]="settingsService.sidebarItemIsHidden(HideableSidebarItemID.Documentation) && !settingsService.organizingSidebarItems()" tourAnchor="tour.outro">
|
||||||
<a class="text-muted small d-flex align-items-center flex-wrap text-decoration-none nav-anchor"
|
<a class="text-muted small d-flex align-items-center flex-wrap text-decoration-none nav-anchor" [class.opacity-50]="settingsService.sidebarItemIsHidden(HideableSidebarItemID.Documentation)" [class.pe-5]="settingsService.organizingSidebarItems() && !slimSidebarEnabled && !slimSidebarAnimating()"
|
||||||
target="_blank" rel="noopener noreferrer" href="https://docs.paperless-ngx.com" ngbPopover="Documentation"
|
target="_blank" rel="noopener noreferrer" href="https://docs.paperless-ngx.com" ngbPopover="Documentation"
|
||||||
i18n-ngbPopover [disablePopover]="!slimSidebarPopoversEnabled" placement="end" container="body"
|
i18n-ngbPopover [disablePopover]="!slimSidebarPopoversEnabled" placement="end" container="body"
|
||||||
triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
||||||
<i-bs class="d-flex me-2" name="question-circle"></i-bs><span><ng-container i18n>Documentation</ng-container></span>
|
<i-bs class="d-flex me-2" name="question-circle"></i-bs><span><ng-container i18n>Documentation</ng-container></span>
|
||||||
</a>
|
</a>
|
||||||
|
@if (settingsService.organizingSidebarItems()) {
|
||||||
|
<pngx-input-switch class="position-absolute top-50 end-0 translate-middle-y me-1" [class.d-none]="slimSidebarEnabled || slimSidebarAnimating()" [compact]="true" title="Documentation" i18n-title [ngModel]="!settingsService.sidebarItemIsHidden(HideableSidebarItemID.Documentation)" (ngModelChange)="toggleSidebarItem(HideableSidebarItemID.Documentation, $event)"></pngx-input-switch>
|
||||||
|
}
|
||||||
</li>
|
</li>
|
||||||
<li class="nav-item" [class.visually-hidden]="slimSidebarEnabled">
|
<li class="nav-item" [class.visually-hidden]="slimSidebarEnabled">
|
||||||
<div class="text-muted small d-flex align-items-center flex-wrap nav-label">
|
<div class="text-muted small d-flex align-items-center flex-wrap nav-label">
|
||||||
|
|||||||
@@ -15,7 +15,7 @@ import { provideUiTour } from 'ngx-ui-tour-ng-bootstrap'
|
|||||||
import { of, throwError } from 'rxjs'
|
import { of, throwError } from 'rxjs'
|
||||||
import { routes } from 'src/app/app-routing.module'
|
import { routes } from 'src/app/app-routing.module'
|
||||||
import { SavedView } from 'src/app/data/saved-view'
|
import { SavedView } from 'src/app/data/saved-view'
|
||||||
import { SETTINGS_KEYS } from 'src/app/data/ui-settings'
|
import { HideableSidebarItemID, SETTINGS_KEYS } from 'src/app/data/ui-settings'
|
||||||
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
|
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
|
||||||
import { PermissionsGuard } from 'src/app/guards/permissions.guard'
|
import { PermissionsGuard } from 'src/app/guards/permissions.guard'
|
||||||
import {
|
import {
|
||||||
@@ -287,6 +287,82 @@ describe('AppFrameComponent', () => {
|
|||||||
jest.useRealTimers()
|
jest.useRealTimers()
|
||||||
})
|
})
|
||||||
|
|
||||||
|
it('should hide configured sidebar items', () => {
|
||||||
|
settingsService.set(SETTINGS_KEYS.SIDEBAR_HIDDEN_ITEMS, [
|
||||||
|
HideableSidebarItemID.Dashboard,
|
||||||
|
HideableSidebarItemID.Workflows,
|
||||||
|
])
|
||||||
|
fixture.detectChanges()
|
||||||
|
|
||||||
|
expect(
|
||||||
|
fixture.nativeElement.querySelector('[routerLink="dashboard"]')
|
||||||
|
.parentElement.classList
|
||||||
|
).toContain('d-none')
|
||||||
|
expect(
|
||||||
|
fixture.nativeElement.querySelector('[routerLink="workflows"]')
|
||||||
|
.parentElement.classList
|
||||||
|
).toContain('d-none')
|
||||||
|
expect(
|
||||||
|
fixture.nativeElement.querySelector('[routerLink="mail"]').parentElement
|
||||||
|
.classList
|
||||||
|
).not.toContain('d-none')
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should show hidden items and visibility switches while customizing', () => {
|
||||||
|
settingsService.set(SETTINGS_KEYS.SIDEBAR_HIDDEN_ITEMS, [
|
||||||
|
HideableSidebarItemID.Dashboard,
|
||||||
|
])
|
||||||
|
settingsService.sidebarHiddenItemsEditing.set([
|
||||||
|
HideableSidebarItemID.Dashboard,
|
||||||
|
])
|
||||||
|
fixture.detectChanges()
|
||||||
|
|
||||||
|
expect(
|
||||||
|
fixture.nativeElement.querySelectorAll('pngx-input-switch').length
|
||||||
|
).toBe(5)
|
||||||
|
expect(
|
||||||
|
fixture.nativeElement.querySelector('[routerLink="dashboard"]')
|
||||||
|
.parentElement.classList
|
||||||
|
).not.toContain('d-none')
|
||||||
|
expect(
|
||||||
|
fixture.nativeElement.querySelector('[routerLink="dashboard"]').classList
|
||||||
|
).toContain('opacity-50')
|
||||||
|
|
||||||
|
settingsService.set(SETTINGS_KEYS.SLIM_SIDEBAR, true)
|
||||||
|
fixture.detectChanges()
|
||||||
|
|
||||||
|
expect(
|
||||||
|
Array.from(
|
||||||
|
fixture.nativeElement.querySelectorAll('pngx-input-switch')
|
||||||
|
).every((toggle: HTMLElement) => toggle.classList.contains('d-none'))
|
||||||
|
).toBe(true)
|
||||||
|
expect(
|
||||||
|
fixture.nativeElement.querySelector('[routerLink="dashboard"]').classList
|
||||||
|
).not.toContain('pe-5')
|
||||||
|
|
||||||
|
settingsService.set(SETTINGS_KEYS.SLIM_SIDEBAR, false)
|
||||||
|
component.slimSidebarAnimating.set(true)
|
||||||
|
fixture.detectChanges()
|
||||||
|
|
||||||
|
expect(
|
||||||
|
Array.from(
|
||||||
|
fixture.nativeElement.querySelectorAll('pngx-input-switch')
|
||||||
|
).every((toggle: HTMLElement) => toggle.classList.contains('d-none'))
|
||||||
|
).toBe(true)
|
||||||
|
|
||||||
|
component.slimSidebarAnimating.set(false)
|
||||||
|
fixture.detectChanges()
|
||||||
|
|
||||||
|
expect(
|
||||||
|
Array.from(
|
||||||
|
fixture.nativeElement.querySelectorAll('pngx-input-switch')
|
||||||
|
).every((toggle: HTMLElement) => !toggle.classList.contains('d-none'))
|
||||||
|
).toBe(true)
|
||||||
|
expect(
|
||||||
|
fixture.nativeElement.querySelector('[routerLink="dashboard"]').classList
|
||||||
|
).toContain('pe-5')
|
||||||
|
})
|
||||||
|
|
||||||
it('should show error on toggle slim sidebar if store settings fails', () => {
|
it('should show error on toggle slim sidebar if store settings fails', () => {
|
||||||
jest.spyOn(console, 'warn').mockImplementation(() => {})
|
jest.spyOn(console, 'warn').mockImplementation(() => {})
|
||||||
const toastSpy = jest.spyOn(toastService, 'showError')
|
const toastSpy = jest.spyOn(toastService, 'showError')
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ import {
|
|||||||
} from '@angular/cdk/drag-drop'
|
} from '@angular/cdk/drag-drop'
|
||||||
import { NgClass } from '@angular/common'
|
import { NgClass } from '@angular/common'
|
||||||
import { Component, HostListener, inject, OnInit, signal } from '@angular/core'
|
import { Component, HostListener, inject, OnInit, signal } from '@angular/core'
|
||||||
|
import { FormsModule } from '@angular/forms'
|
||||||
import { ActivatedRoute, Router, RouterModule } from '@angular/router'
|
import { ActivatedRoute, Router, RouterModule } from '@angular/router'
|
||||||
import {
|
import {
|
||||||
NgbCollapseModule,
|
NgbCollapseModule,
|
||||||
@@ -21,7 +22,11 @@ import { Observable } from 'rxjs'
|
|||||||
import { first } from 'rxjs/operators'
|
import { first } from 'rxjs/operators'
|
||||||
import { Document } from 'src/app/data/document'
|
import { Document } from 'src/app/data/document'
|
||||||
import { SavedView } from 'src/app/data/saved-view'
|
import { SavedView } from 'src/app/data/saved-view'
|
||||||
import { CollapsibleSection, SETTINGS_KEYS } from 'src/app/data/ui-settings'
|
import {
|
||||||
|
CollapsibleSection,
|
||||||
|
HideableSidebarItemID,
|
||||||
|
SETTINGS_KEYS,
|
||||||
|
} from 'src/app/data/ui-settings'
|
||||||
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
|
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
|
||||||
import { ComponentCanDeactivate } from 'src/app/guards/dirty-doc.guard'
|
import { ComponentCanDeactivate } from 'src/app/guards/dirty-doc.guard'
|
||||||
import { DocumentTitlePipe } from 'src/app/pipes/document-title.pipe'
|
import { DocumentTitlePipe } from 'src/app/pipes/document-title.pipe'
|
||||||
@@ -48,6 +53,7 @@ import { ChatComponent } from '../chat/chat/chat.component'
|
|||||||
import { BrandMarkComponent } from '../common/logo/brand-mark/brand-mark.component'
|
import { BrandMarkComponent } from '../common/logo/brand-mark/brand-mark.component'
|
||||||
import { LogoComponent } from '../common/logo/logo.component'
|
import { LogoComponent } from '../common/logo/logo.component'
|
||||||
import { ProfileEditDialogComponent } from '../common/profile-edit-dialog/profile-edit-dialog.component'
|
import { ProfileEditDialogComponent } from '../common/profile-edit-dialog/profile-edit-dialog.component'
|
||||||
|
import { SwitchComponent } from '../common/input/switch/switch.component'
|
||||||
import { DocumentDetailComponent } from '../document-detail/document-detail.component'
|
import { DocumentDetailComponent } from '../document-detail/document-detail.component'
|
||||||
import { ComponentWithPermissions } from '../with-permissions/with-permissions.component'
|
import { ComponentWithPermissions } from '../with-permissions/with-permissions.component'
|
||||||
import { GlobalSearchComponent } from './global-search/global-search.component'
|
import { GlobalSearchComponent } from './global-search/global-search.component'
|
||||||
@@ -76,6 +82,8 @@ const SCROLL_THRESHOLD = 16
|
|||||||
NgxBootstrapIconsModule,
|
NgxBootstrapIconsModule,
|
||||||
DragDropModule,
|
DragDropModule,
|
||||||
TourNgBootstrap,
|
TourNgBootstrap,
|
||||||
|
FormsModule,
|
||||||
|
SwitchComponent,
|
||||||
],
|
],
|
||||||
})
|
})
|
||||||
export class AppFrameComponent
|
export class AppFrameComponent
|
||||||
@@ -98,6 +106,7 @@ export class AppFrameComponent
|
|||||||
readonly isMenuCollapsed = signal(true)
|
readonly isMenuCollapsed = signal(true)
|
||||||
readonly slimSidebarAnimating = signal(false)
|
readonly slimSidebarAnimating = signal(false)
|
||||||
readonly mobileSearchHidden = signal(false)
|
readonly mobileSearchHidden = signal(false)
|
||||||
|
readonly HideableSidebarItemID = HideableSidebarItemID
|
||||||
private readonly versionSetting = this.settingsService.getSignal<string>(
|
private readonly versionSetting = this.settingsService.getSignal<string>(
|
||||||
SETTINGS_KEYS.VERSION
|
SETTINGS_KEYS.VERSION
|
||||||
)
|
)
|
||||||
@@ -195,6 +204,10 @@ export class AppFrameComponent
|
|||||||
}, 200) // slightly longer than css animation for slim sidebar
|
}, 200) // slightly longer than css animation for slim sidebar
|
||||||
}
|
}
|
||||||
|
|
||||||
|
toggleSidebarItem(item: HideableSidebarItemID, visible: boolean): void {
|
||||||
|
this.settingsService.updateSidebarItemVisibility(item, visible)
|
||||||
|
}
|
||||||
|
|
||||||
toggleAttributesSections(event?: Event): void {
|
toggleAttributesSections(event?: Event): void {
|
||||||
event?.preventDefault()
|
event?.preventDefault()
|
||||||
event?.stopPropagation()
|
event?.stopPropagation()
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
<div class="mb-3">
|
<div [class.mb-3]="!compact">
|
||||||
<div class="row">
|
<div [class.row]="!compact">
|
||||||
@if (!horizontal) {
|
@if (!horizontal && !compact) {
|
||||||
<div class="d-flex align-items-center position-relative hidden-button-container col-md-3">
|
<div class="d-flex align-items-center position-relative hidden-button-container col-md-3">
|
||||||
<label class="form-label" [for]="inputId" [ngbTooltip]="showUnsetNote && isUnset ? tipContent: null" placement="end">
|
<label class="form-label" [for]="inputId" [ngbTooltip]="showUnsetNote && isUnset ? tipContent: null" placement="end">
|
||||||
{{title}}
|
{{title}}
|
||||||
@@ -17,8 +17,8 @@
|
|||||||
}
|
}
|
||||||
<div [ngClass]="{'align-items-center': horizontal, 'd-flex': horizontal}">
|
<div [ngClass]="{'align-items-center': horizontal, 'd-flex': horizontal}">
|
||||||
<div class="form-check form-switch">
|
<div class="form-check form-switch">
|
||||||
<input #inputField type="checkbox" class="form-check-input" [id]="inputId" [(ngModel)]="value" [ngModelOptions]="{standalone: true}" (change)="onChange(value)" (blur)="onTouched()" [disabled]="disabled">
|
<input #inputField type="checkbox" class="form-check-input" [id]="inputId" [(ngModel)]="value" [ngModelOptions]="{standalone: true}" (change)="onChange(value)" (blur)="onTouched()" [disabled]="disabled" [attr.aria-label]="compact ? title : null">
|
||||||
@if (horizontal) {
|
@if (horizontal && !compact) {
|
||||||
<label class="form-check-label" [class.text-muted]="showUnsetNote && isUnset" [for]="inputId" [ngbTooltip]="showUnsetNote && isUnset ? tipContent: null" placement="end">
|
<label class="form-check-label" [class.text-muted]="showUnsetNote && isUnset" [for]="inputId" [ngbTooltip]="showUnsetNote && isUnset ? tipContent: null" placement="end">
|
||||||
{{title}}
|
{{title}}
|
||||||
@if (showUnsetNote && isUnset) {
|
@if (showUnsetNote && isUnset) {
|
||||||
|
|||||||
@@ -48,4 +48,14 @@ describe('SwitchComponent', () => {
|
|||||||
component.value = undefined
|
component.value = undefined
|
||||||
expect(component.isUnset).toBeTruthy()
|
expect(component.isUnset).toBeTruthy()
|
||||||
})
|
})
|
||||||
|
|
||||||
|
it('should support a compact layout', () => {
|
||||||
|
component.compact = true
|
||||||
|
component.title = 'Test switch'
|
||||||
|
fixture.detectChanges()
|
||||||
|
|
||||||
|
expect(fixture.nativeElement.querySelector('.mb-3')).toBeNull()
|
||||||
|
expect(fixture.nativeElement.querySelector('.row')).toBeNull()
|
||||||
|
expect(input.getAttribute('aria-label')).toEqual('Test switch')
|
||||||
|
})
|
||||||
})
|
})
|
||||||
|
|||||||
@@ -25,6 +25,9 @@ export class SwitchComponent extends AbstractInputComponent<boolean> {
|
|||||||
@Input()
|
@Input()
|
||||||
showUnsetNote: boolean = false
|
showUnsetNote: boolean = false
|
||||||
|
|
||||||
|
@Input()
|
||||||
|
compact: boolean = false
|
||||||
|
|
||||||
constructor() {
|
constructor() {
|
||||||
super()
|
super()
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -24,6 +24,16 @@ export enum CollapsibleSection {
|
|||||||
ATTRIBUTES = 'attributes',
|
ATTRIBUTES = 'attributes',
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export enum HideableSidebarItemID {
|
||||||
|
Dashboard = 'dashboard',
|
||||||
|
SavedViews = 'saved_views',
|
||||||
|
Workflows = 'workflows',
|
||||||
|
Mail = 'mail',
|
||||||
|
Documentation = 'documentation',
|
||||||
|
}
|
||||||
|
|
||||||
|
export const HIDEABLE_SIDEBAR_ITEM_IDS = Object.values(HideableSidebarItemID)
|
||||||
|
|
||||||
export const PAPERLESS_GREEN_HEX = '#17541f'
|
export const PAPERLESS_GREEN_HEX = '#17541f'
|
||||||
|
|
||||||
export const SETTINGS_KEYS = {
|
export const SETTINGS_KEYS = {
|
||||||
@@ -56,6 +66,7 @@ export const SETTINGS_KEYS = {
|
|||||||
NOTES_ENABLED: 'general-settings:notes-enabled',
|
NOTES_ENABLED: 'general-settings:notes-enabled',
|
||||||
AUDITLOG_ENABLED: 'general-settings:auditlog-enabled',
|
AUDITLOG_ENABLED: 'general-settings:auditlog-enabled',
|
||||||
SLIM_SIDEBAR: 'general-settings:slim-sidebar',
|
SLIM_SIDEBAR: 'general-settings:slim-sidebar',
|
||||||
|
SIDEBAR_HIDDEN_ITEMS: 'general-settings:sidebar:hidden-items',
|
||||||
ATTRIBUTES_SECTIONS_COLLAPSED:
|
ATTRIBUTES_SECTIONS_COLLAPSED:
|
||||||
'general-settings:attributes-sections-collapsed',
|
'general-settings:attributes-sections-collapsed',
|
||||||
UPDATE_CHECKING_ENABLED: 'general-settings:update-checking:enabled',
|
UPDATE_CHECKING_ENABLED: 'general-settings:update-checking:enabled',
|
||||||
@@ -127,6 +138,11 @@ export const SETTINGS: UiSetting[] = [
|
|||||||
type: 'boolean',
|
type: 'boolean',
|
||||||
default: false,
|
default: false,
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
key: SETTINGS_KEYS.SIDEBAR_HIDDEN_ITEMS,
|
||||||
|
type: 'array',
|
||||||
|
default: [],
|
||||||
|
},
|
||||||
{
|
{
|
||||||
key: SETTINGS_KEYS.ATTRIBUTES_SECTIONS_COLLAPSED,
|
key: SETTINGS_KEYS.ATTRIBUTES_SECTIONS_COLLAPSED,
|
||||||
type: 'array',
|
type: 'array',
|
||||||
|
|||||||
@@ -14,7 +14,11 @@ import { CustomFieldDataType } from '../data/custom-field'
|
|||||||
import { DEFAULT_DISPLAY_FIELDS, DisplayField } from '../data/document'
|
import { DEFAULT_DISPLAY_FIELDS, DisplayField } from '../data/document'
|
||||||
import { SavedView } from '../data/saved-view'
|
import { SavedView } from '../data/saved-view'
|
||||||
import { RemoteOCRModeConfig } from '../data/paperless-config'
|
import { RemoteOCRModeConfig } from '../data/paperless-config'
|
||||||
import { SETTINGS_KEYS, UiSettings } from '../data/ui-settings'
|
import {
|
||||||
|
HideableSidebarItemID,
|
||||||
|
SETTINGS_KEYS,
|
||||||
|
UiSettings,
|
||||||
|
} from '../data/ui-settings'
|
||||||
import { PermissionsService } from './permissions.service'
|
import { PermissionsService } from './permissions.service'
|
||||||
import { CustomFieldsService } from './rest/custom-fields.service'
|
import { CustomFieldsService } from './rest/custom-fields.service'
|
||||||
import { SettingsService } from './settings.service'
|
import { SettingsService } from './settings.service'
|
||||||
@@ -230,6 +234,35 @@ describe('SettingsService', () => {
|
|||||||
expect(notesEnabled()).toBeFalsy()
|
expect(notesEnabled()).toBeFalsy()
|
||||||
})
|
})
|
||||||
|
|
||||||
|
it('updates sidebar item visibility', () => {
|
||||||
|
httpTestingController
|
||||||
|
.expectOne(`${environment.apiBaseUrl}ui_settings/`)
|
||||||
|
.flush(ui_settings)
|
||||||
|
|
||||||
|
expect(
|
||||||
|
settingsService.sidebarItemIsHidden(HideableSidebarItemID.Workflows)
|
||||||
|
).toBe(false)
|
||||||
|
|
||||||
|
settingsService.updateSidebarItemVisibility(
|
||||||
|
HideableSidebarItemID.Workflows,
|
||||||
|
false
|
||||||
|
)
|
||||||
|
|
||||||
|
expect(
|
||||||
|
settingsService.sidebarItemIsHidden(HideableSidebarItemID.Workflows)
|
||||||
|
).toBe(true)
|
||||||
|
expect(settingsService.get(SETTINGS_KEYS.SIDEBAR_HIDDEN_ITEMS)).toEqual([])
|
||||||
|
|
||||||
|
settingsService.updateSidebarItemVisibility(
|
||||||
|
HideableSidebarItemID.Workflows,
|
||||||
|
true
|
||||||
|
)
|
||||||
|
|
||||||
|
expect(
|
||||||
|
settingsService.sidebarItemIsHidden(HideableSidebarItemID.Workflows)
|
||||||
|
).toBe(false)
|
||||||
|
})
|
||||||
|
|
||||||
it('updates setting signals when settings are reinitialized', () => {
|
it('updates setting signals when settings are reinitialized', () => {
|
||||||
let req = httpTestingController.expectOne(
|
let req = httpTestingController.expectOne(
|
||||||
`${environment.apiBaseUrl}ui_settings/`
|
`${environment.apiBaseUrl}ui_settings/`
|
||||||
|
|||||||
@@ -24,6 +24,7 @@ import { DEFAULT_DISPLAY_FIELDS, DisplayField } from '../data/document'
|
|||||||
import { RemoteOCRModeConfig } from '../data/paperless-config'
|
import { RemoteOCRModeConfig } from '../data/paperless-config'
|
||||||
import { SavedView } from '../data/saved-view'
|
import { SavedView } from '../data/saved-view'
|
||||||
import {
|
import {
|
||||||
|
HideableSidebarItemID,
|
||||||
PAPERLESS_GREEN_HEX,
|
PAPERLESS_GREEN_HEX,
|
||||||
SETTINGS,
|
SETTINGS,
|
||||||
SETTINGS_KEYS,
|
SETTINGS_KEYS,
|
||||||
@@ -313,6 +314,18 @@ export class SettingsService {
|
|||||||
readonly globalDropzoneEnabled = signal(true)
|
readonly globalDropzoneEnabled = signal(true)
|
||||||
readonly globalDropzoneActive = signal(false)
|
readonly globalDropzoneActive = signal(false)
|
||||||
readonly organizingSidebarSavedViews = signal(false)
|
readonly organizingSidebarSavedViews = signal(false)
|
||||||
|
readonly sidebarHiddenItemsEditing = signal<HideableSidebarItemID[] | null>(
|
||||||
|
null
|
||||||
|
)
|
||||||
|
readonly organizingSidebarItems = computed(
|
||||||
|
() => this.sidebarHiddenItemsEditing() !== null
|
||||||
|
)
|
||||||
|
readonly sidebarHiddenItemsEditingChanged = new EventEmitter<
|
||||||
|
HideableSidebarItemID[]
|
||||||
|
>()
|
||||||
|
readonly hiddenSidebarItems = this.getSignal<HideableSidebarItemID[]>(
|
||||||
|
SETTINGS_KEYS.SIDEBAR_HIDDEN_ITEMS
|
||||||
|
)
|
||||||
|
|
||||||
readonly allDisplayFields = signal<Array<{ id: DisplayField; name: string }>>(
|
readonly allDisplayFields = signal<Array<{ id: DisplayField; name: string }>>(
|
||||||
DEFAULT_DISPLAY_FIELDS
|
DEFAULT_DISPLAY_FIELDS
|
||||||
@@ -749,6 +762,29 @@ export class SettingsService {
|
|||||||
return this.storeSettings()
|
return this.storeSettings()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
sidebarItemIsHidden(item: HideableSidebarItemID): boolean {
|
||||||
|
return (
|
||||||
|
this.sidebarHiddenItemsEditing() ?? this.hiddenSidebarItems()
|
||||||
|
).includes(item)
|
||||||
|
}
|
||||||
|
|
||||||
|
updateSidebarItemVisibility(
|
||||||
|
item: HideableSidebarItemID,
|
||||||
|
visible: boolean
|
||||||
|
): void {
|
||||||
|
const hiddenItems = new Set(
|
||||||
|
this.sidebarHiddenItemsEditing() ?? this.hiddenSidebarItems()
|
||||||
|
)
|
||||||
|
if (visible) {
|
||||||
|
hiddenItems.delete(item)
|
||||||
|
} else {
|
||||||
|
hiddenItems.add(item)
|
||||||
|
}
|
||||||
|
const updatedHiddenItems = [...hiddenItems]
|
||||||
|
this.sidebarHiddenItemsEditing.set(updatedHiddenItems)
|
||||||
|
this.sidebarHiddenItemsEditingChanged.emit(updatedHiddenItems)
|
||||||
|
}
|
||||||
|
|
||||||
updateSavedViewsVisibility(
|
updateSavedViewsVisibility(
|
||||||
dashboardVisibleViewIds: number[],
|
dashboardVisibleViewIds: number[],
|
||||||
sidebarVisibleViewIds: number[]
|
sidebarVisibleViewIds: number[]
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -28,6 +28,7 @@ class DocumentsConfig(AppConfig):
|
|||||||
document_consumption_finished.connect(set_storage_path)
|
document_consumption_finished.connect(set_storage_path)
|
||||||
document_consumption_finished.connect(add_to_index)
|
document_consumption_finished.connect(add_to_index)
|
||||||
document_consumption_finished.connect(run_workflows_added)
|
document_consumption_finished.connect(run_workflows_added)
|
||||||
|
document_consumption_finished.connect(add_to_index)
|
||||||
document_consumption_finished.connect(add_or_update_document_in_llm_index)
|
document_consumption_finished.connect(add_or_update_document_in_llm_index)
|
||||||
document_updated.connect(run_workflows_updated)
|
document_updated.connect(run_workflows_updated)
|
||||||
document_updated.connect(send_websocket_document_updated)
|
document_updated.connect(send_websocket_document_updated)
|
||||||
|
|||||||
+38
-32
@@ -2,6 +2,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import logging
|
import logging
|
||||||
import tempfile
|
import tempfile
|
||||||
|
import uuid
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import TYPE_CHECKING
|
from typing import TYPE_CHECKING
|
||||||
from typing import Literal
|
from typing import Literal
|
||||||
@@ -298,53 +299,55 @@ def modify_custom_fields(
|
|||||||
) -> Literal["OK"]:
|
) -> Literal["OK"]:
|
||||||
qs = Document.objects.filter(id__in=doc_ids).only("pk")
|
qs = Document.objects.filter(id__in=doc_ids).only("pk")
|
||||||
affected_docs = list(qs.values_list("pk", flat=True))
|
affected_docs = list(qs.values_list("pk", flat=True))
|
||||||
# Ensure add_custom_fields is a list of tuples, supports old API
|
# Ensure add_custom_fields is a list of (int, value) tuples, supports old API
|
||||||
add_custom_fields = (
|
add_custom_fields = (
|
||||||
add_custom_fields.items()
|
[(int(field), value) for field, value in add_custom_fields.items()]
|
||||||
if isinstance(add_custom_fields, dict)
|
if isinstance(add_custom_fields, dict)
|
||||||
else [(field, None) for field in add_custom_fields]
|
else [(int(field), None) for field in add_custom_fields]
|
||||||
)
|
)
|
||||||
|
|
||||||
custom_fields = CustomField.objects.filter(
|
# Resolved once, instead of re-querying the same field for every document
|
||||||
id__in=[int(field) for field, _ in add_custom_fields],
|
custom_fields_by_id: dict[int, CustomField] = CustomField.objects.in_bulk(
|
||||||
).distinct()
|
[field_id for field_id, _ in add_custom_fields],
|
||||||
|
)
|
||||||
|
# Passed to update_or_create() below rather than a bare id, so the FK is
|
||||||
|
# cached on the created instance and auditlog's post_save receiver does
|
||||||
|
# not reload it per row. Only needed for additions. content is deferred:
|
||||||
|
# the one field here that is both large and unused.
|
||||||
|
docs_by_id: dict[int, Document] = (
|
||||||
|
Document.objects.defer("content").in_bulk(affected_docs)
|
||||||
|
if add_custom_fields
|
||||||
|
else {}
|
||||||
|
)
|
||||||
for field_id, value in add_custom_fields:
|
for field_id, value in add_custom_fields:
|
||||||
for doc_id in affected_docs:
|
custom_field = custom_fields_by_id[field_id]
|
||||||
defaults = {}
|
|
||||||
custom_field = custom_fields.get(id=field_id)
|
|
||||||
if custom_field:
|
|
||||||
value_field = CustomFieldInstance.TYPE_TO_DATA_STORE_NAME_MAP[
|
value_field = CustomFieldInstance.TYPE_TO_DATA_STORE_NAME_MAP[
|
||||||
custom_field.data_type
|
custom_field.data_type
|
||||||
]
|
]
|
||||||
defaults[value_field] = value
|
is_doclink = custom_field.data_type == CustomField.FieldDataType.DOCUMENTLINK
|
||||||
if (
|
for doc_id in affected_docs:
|
||||||
custom_field.data_type == CustomField.FieldDataType.DOCUMENTLINK
|
if is_doclink and value and doc_id in value:
|
||||||
and value
|
|
||||||
and doc_id in value
|
|
||||||
):
|
|
||||||
# Prevent self-linking
|
# Prevent self-linking
|
||||||
continue
|
continue
|
||||||
CustomFieldInstance.objects.update_or_create(
|
CustomFieldInstance.objects.update_or_create(
|
||||||
document_id=doc_id,
|
document=docs_by_id[doc_id],
|
||||||
field_id=field_id,
|
field=custom_field,
|
||||||
defaults=defaults,
|
defaults={value_field: value},
|
||||||
)
|
)
|
||||||
if custom_field.data_type == CustomField.FieldDataType.DOCUMENTLINK:
|
if is_doclink:
|
||||||
doc = Document.objects.get(id=doc_id)
|
reflect_doclinks(docs_by_id[doc_id], custom_field, value)
|
||||||
reflect_doclinks(doc, custom_field, value)
|
|
||||||
|
|
||||||
# For doc link fields that are being removed, remove symmetrical links
|
# For doc link fields that are being removed, remove symmetrical links.
|
||||||
|
# select_related avoids a per-instance reload of the document and field.
|
||||||
for doclink_being_removed_instance in CustomFieldInstance.objects.filter(
|
for doclink_being_removed_instance in CustomFieldInstance.objects.filter(
|
||||||
document_id__in=affected_docs,
|
document_id__in=affected_docs,
|
||||||
field__id__in=remove_custom_fields,
|
field__id__in=remove_custom_fields,
|
||||||
field__data_type=CustomField.FieldDataType.DOCUMENTLINK,
|
field__data_type=CustomField.FieldDataType.DOCUMENTLINK,
|
||||||
value_document_ids__isnull=False,
|
value_document_ids__isnull=False,
|
||||||
):
|
).select_related("field", "document"):
|
||||||
for target_doc_id in doclink_being_removed_instance.value:
|
for target_doc_id in doclink_being_removed_instance.value:
|
||||||
remove_doclink(
|
remove_doclink(
|
||||||
document=Document.objects.get(
|
document=doclink_being_removed_instance.document,
|
||||||
id=doclink_being_removed_instance.document.id,
|
|
||||||
),
|
|
||||||
field=doclink_being_removed_instance.field,
|
field=doclink_being_removed_instance.field,
|
||||||
target_doc_id=target_doc_id,
|
target_doc_id=target_doc_id,
|
||||||
)
|
)
|
||||||
@@ -379,7 +382,7 @@ def delete(doc_ids: list[int]) -> Literal["OK"]:
|
|||||||
)
|
)
|
||||||
delete_ids = list({*doc_ids, *version_ids})
|
delete_ids = list({*doc_ids, *version_ids})
|
||||||
|
|
||||||
Document.objects.filter(id__in=delete_ids).delete()
|
Document.objects.filter(id__in=delete_ids).delete(transaction_id=uuid.uuid4())
|
||||||
|
|
||||||
from documents.search import get_backend
|
from documents.search import get_backend
|
||||||
|
|
||||||
@@ -1177,10 +1180,13 @@ def remove_doclink(
|
|||||||
"""
|
"""
|
||||||
Removes a 'symmetrical' link to `document` from the target document's existing custom field instance
|
Removes a 'symmetrical' link to `document` from the target document's existing custom field instance
|
||||||
"""
|
"""
|
||||||
target_doc_field_instance = CustomFieldInstance.objects.filter(
|
# select_related: a signal receiver (auditlog) touches .document/.field on
|
||||||
document_id=target_doc_id,
|
# the save() below, without this that is a per-call reload query
|
||||||
field=field,
|
target_doc_field_instance = (
|
||||||
).first()
|
CustomFieldInstance.objects.filter(document_id=target_doc_id, field=field)
|
||||||
|
.select_related("document", "field")
|
||||||
|
.first()
|
||||||
|
)
|
||||||
if (
|
if (
|
||||||
target_doc_field_instance is not None
|
target_doc_field_instance is not None
|
||||||
and document.id in target_doc_field_instance.value
|
and document.id in target_doc_field_instance.value
|
||||||
|
|||||||
+63
-25
@@ -34,6 +34,27 @@ from paperless.signed_pickle import signed_pickle_loads
|
|||||||
|
|
||||||
logger = logging.getLogger("paperless.classifier")
|
logger = logging.getLogger("paperless.classifier")
|
||||||
|
|
||||||
|
|
||||||
|
def _predict_with_threshold(classifier, X, threshold: float) -> int | None:
|
||||||
|
"""
|
||||||
|
Return the predicted class id, or None if:
|
||||||
|
- the prediction is -1 (no match), or
|
||||||
|
- the winning class probability is below the configured threshold.
|
||||||
|
|
||||||
|
Using predict_proba() instead of predict() lets us apply a minimum-confidence
|
||||||
|
cutoff so that uncertain predictions are discarded rather than assigned.
|
||||||
|
"""
|
||||||
|
probas = classifier.predict_proba(X)[0]
|
||||||
|
best_idx = int(probas.argmax())
|
||||||
|
best_class = int(classifier.classes_[best_idx])
|
||||||
|
|
||||||
|
if best_class == -1:
|
||||||
|
return None
|
||||||
|
if threshold > 0.0 and probas[best_idx] < threshold:
|
||||||
|
return None
|
||||||
|
return best_class
|
||||||
|
|
||||||
|
|
||||||
ADVANCED_TEXT_PROCESSING_ENABLED = (
|
ADVANCED_TEXT_PROCESSING_ENABLED = (
|
||||||
settings.NLTK_LANGUAGE is not None and settings.NLTK_ENABLED
|
settings.NLTK_LANGUAGE is not None and settings.NLTK_ENABLED
|
||||||
)
|
)
|
||||||
@@ -102,7 +123,8 @@ class DocumentClassifier:
|
|||||||
# v8 - Added storage path classifier
|
# v8 - Added storage path classifier
|
||||||
# v9 - Changed from hashing to time/ids for re-train check
|
# v9 - Changed from hashing to time/ids for re-train check
|
||||||
# v10 - HMAC-signed model file
|
# v10 - HMAC-signed model file
|
||||||
FORMAT_VERSION = 10
|
# v11 - Use sample_weight for balanced training; predict_proba with threshold
|
||||||
|
FORMAT_VERSION = 11
|
||||||
|
|
||||||
HMAC_SIZE = 32 # SHA-256 digest length
|
HMAC_SIZE = 32 # SHA-256 digest length
|
||||||
|
|
||||||
@@ -324,6 +346,13 @@ class DocumentClassifier:
|
|||||||
from sklearn.preprocessing import LabelBinarizer
|
from sklearn.preprocessing import LabelBinarizer
|
||||||
from sklearn.preprocessing import MultiLabelBinarizer
|
from sklearn.preprocessing import MultiLabelBinarizer
|
||||||
|
|
||||||
|
# MLPClassifier does not support class_weight directly
|
||||||
|
# (https://github.com/scikit-learn/scikit-learn/issues/9113), so we use
|
||||||
|
# compute_sample_weight to balance classes during training and prevent
|
||||||
|
# over-represented correspondents from dominating predictions.
|
||||||
|
# https://scikit-learn.org/stable/modules/generated/sklearn.utils.class_weight.compute_sample_weight.html
|
||||||
|
from sklearn.utils.class_weight import compute_sample_weight
|
||||||
|
|
||||||
# Step 2: vectorize data
|
# Step 2: vectorize data
|
||||||
logger.debug("Vectorizing data...")
|
logger.debug("Vectorizing data...")
|
||||||
notify("Vectorizing document content...")
|
notify("Vectorizing document content...")
|
||||||
@@ -369,7 +398,7 @@ class DocumentClassifier:
|
|||||||
self.tags_binarizer = MultiLabelBinarizer()
|
self.tags_binarizer = MultiLabelBinarizer()
|
||||||
labels_tags_vectorized = self.tags_binarizer.fit_transform(labels_tags)
|
labels_tags_vectorized = self.tags_binarizer.fit_transform(labels_tags)
|
||||||
|
|
||||||
self.tags_classifier = MLPClassifier(tol=0.01)
|
self.tags_classifier = MLPClassifier(tol=0.01, random_state=0)
|
||||||
self.tags_classifier.fit(data_vectorized, labels_tags_vectorized)
|
self.tags_classifier.fit(data_vectorized, labels_tags_vectorized)
|
||||||
else:
|
else:
|
||||||
self.tags_classifier = None
|
self.tags_classifier = None
|
||||||
@@ -380,8 +409,12 @@ class DocumentClassifier:
|
|||||||
notify(
|
notify(
|
||||||
f"Training correspondent classifier ({num_correspondents} correspondent(s))...",
|
f"Training correspondent classifier ({num_correspondents} correspondent(s))...",
|
||||||
)
|
)
|
||||||
self.correspondent_classifier = MLPClassifier(tol=0.01)
|
self.correspondent_classifier = MLPClassifier(tol=0.01, random_state=0)
|
||||||
self.correspondent_classifier.fit(data_vectorized, labels_correspondent)
|
self.correspondent_classifier.fit(
|
||||||
|
data_vectorized,
|
||||||
|
labels_correspondent,
|
||||||
|
sample_weight=compute_sample_weight("balanced", labels_correspondent),
|
||||||
|
)
|
||||||
else:
|
else:
|
||||||
self.correspondent_classifier = None
|
self.correspondent_classifier = None
|
||||||
logger.debug(
|
logger.debug(
|
||||||
@@ -393,8 +426,12 @@ class DocumentClassifier:
|
|||||||
notify(
|
notify(
|
||||||
f"Training document type classifier ({num_document_types} type(s))...",
|
f"Training document type classifier ({num_document_types} type(s))...",
|
||||||
)
|
)
|
||||||
self.document_type_classifier = MLPClassifier(tol=0.01)
|
self.document_type_classifier = MLPClassifier(tol=0.01, random_state=0)
|
||||||
self.document_type_classifier.fit(data_vectorized, labels_document_type)
|
self.document_type_classifier.fit(
|
||||||
|
data_vectorized,
|
||||||
|
labels_document_type,
|
||||||
|
sample_weight=compute_sample_weight("balanced", labels_document_type),
|
||||||
|
)
|
||||||
else:
|
else:
|
||||||
self.document_type_classifier = None
|
self.document_type_classifier = None
|
||||||
logger.debug(
|
logger.debug(
|
||||||
@@ -406,10 +443,11 @@ class DocumentClassifier:
|
|||||||
"Training storage paths classifier...",
|
"Training storage paths classifier...",
|
||||||
)
|
)
|
||||||
notify(f"Training storage path classifier ({num_storage_paths} path(s))...")
|
notify(f"Training storage path classifier ({num_storage_paths} path(s))...")
|
||||||
self.storage_path_classifier = MLPClassifier(tol=0.01)
|
self.storage_path_classifier = MLPClassifier(tol=0.01, random_state=0)
|
||||||
self.storage_path_classifier.fit(
|
self.storage_path_classifier.fit(
|
||||||
data_vectorized,
|
data_vectorized,
|
||||||
labels_storage_path,
|
labels_storage_path,
|
||||||
|
sample_weight=compute_sample_weight("balanced", labels_storage_path),
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
self.storage_path_classifier = None
|
self.storage_path_classifier = None
|
||||||
@@ -546,23 +584,23 @@ class DocumentClassifier:
|
|||||||
def predict_correspondent(self, content: str) -> int | None:
|
def predict_correspondent(self, content: str) -> int | None:
|
||||||
if self.correspondent_classifier:
|
if self.correspondent_classifier:
|
||||||
X = self._vectorize(content)
|
X = self._vectorize(content)
|
||||||
correspondent_id = self.correspondent_classifier.predict(X)
|
predicted_id = _predict_with_threshold(
|
||||||
if correspondent_id != -1:
|
self.correspondent_classifier,
|
||||||
return correspondent_id
|
X,
|
||||||
else:
|
settings.CLASSIFIER_MATCH_THRESHOLD,
|
||||||
return None
|
)
|
||||||
else:
|
return predicted_id
|
||||||
return None
|
return None
|
||||||
|
|
||||||
def predict_document_type(self, content: str) -> int | None:
|
def predict_document_type(self, content: str) -> int | None:
|
||||||
if self.document_type_classifier:
|
if self.document_type_classifier:
|
||||||
X = self._vectorize(content)
|
X = self._vectorize(content)
|
||||||
document_type_id = self.document_type_classifier.predict(X)
|
predicted_id = _predict_with_threshold(
|
||||||
if document_type_id != -1:
|
self.document_type_classifier,
|
||||||
return document_type_id
|
X,
|
||||||
else:
|
settings.CLASSIFIER_MATCH_THRESHOLD,
|
||||||
return None
|
)
|
||||||
else:
|
return predicted_id
|
||||||
return None
|
return None
|
||||||
|
|
||||||
def predict_tags(self, content: str) -> list[int]:
|
def predict_tags(self, content: str) -> list[int]:
|
||||||
@@ -589,10 +627,10 @@ class DocumentClassifier:
|
|||||||
def predict_storage_path(self, content: str) -> int | None:
|
def predict_storage_path(self, content: str) -> int | None:
|
||||||
if self.storage_path_classifier:
|
if self.storage_path_classifier:
|
||||||
X = self._vectorize(content)
|
X = self._vectorize(content)
|
||||||
storage_path_id = self.storage_path_classifier.predict(X)
|
predicted_id = _predict_with_threshold(
|
||||||
if storage_path_id != -1:
|
self.storage_path_classifier,
|
||||||
return storage_path_id
|
X,
|
||||||
else:
|
settings.CLASSIFIER_MATCH_THRESHOLD,
|
||||||
return None
|
)
|
||||||
else:
|
return predicted_id
|
||||||
return None
|
return None
|
||||||
|
|||||||
@@ -52,7 +52,6 @@ from documents.templating.workflows import parse_w_workflow_placeholders
|
|||||||
from documents.utils import compute_checksum
|
from documents.utils import compute_checksum
|
||||||
from documents.utils import copy_basic_file_stats
|
from documents.utils import copy_basic_file_stats
|
||||||
from documents.utils import copy_file_with_basic_stats
|
from documents.utils import copy_file_with_basic_stats
|
||||||
from documents.utils import normalize_unicode
|
|
||||||
from documents.utils import run_subprocess
|
from documents.utils import run_subprocess
|
||||||
from paperless.config import OcrConfig
|
from paperless.config import OcrConfig
|
||||||
from paperless.config import RemoteOCRConfig
|
from paperless.config import RemoteOCRConfig
|
||||||
@@ -202,9 +201,7 @@ class ConsumerPluginMixin:
|
|||||||
|
|
||||||
self.renew_logging_group()
|
self.renew_logging_group()
|
||||||
|
|
||||||
self.filename = normalize_unicode(
|
self.filename = self.metadata.filename or self.input_doc.original_file.name
|
||||||
self.metadata.filename or self.input_doc.original_file.name,
|
|
||||||
)
|
|
||||||
|
|
||||||
def _send_progress(
|
def _send_progress(
|
||||||
self,
|
self,
|
||||||
|
|||||||
@@ -156,6 +156,15 @@ class FileStabilityTracker:
|
|||||||
logger.debug(f"File disappeared during stability check: {path}")
|
logger.debug(f"File disappeared during stability check: {path}")
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
# Stable, but empty: some scanners create a zero byte placeholder
|
||||||
|
# and only write the page some time later. Consuming it now can
|
||||||
|
# only fail so drop it and let the writer's next event
|
||||||
|
# (or the periodic rescan) bring it back once it has content
|
||||||
|
if not tracked.last_size:
|
||||||
|
to_remove.append(path)
|
||||||
|
logger.debug("Ignoring stable but empty file: %s", path)
|
||||||
|
continue
|
||||||
|
|
||||||
# File is stable, we can return it
|
# File is stable, we can return it
|
||||||
to_yield.append(path)
|
to_yield.append(path)
|
||||||
logger.info(f"File is stable: {path}")
|
logger.info(f"File is stable: {path}")
|
||||||
|
|||||||
@@ -21,7 +21,6 @@ from documents.models import Workflow
|
|||||||
from documents.models import WorkflowTrigger
|
from documents.models import WorkflowTrigger
|
||||||
from documents.permissions import permitted_object_ids
|
from documents.permissions import permitted_object_ids
|
||||||
from documents.regex import safe_regex_search
|
from documents.regex import safe_regex_search
|
||||||
from documents.utils import normalize_unicode
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
from django.db.models import QuerySet
|
from django.db.models import QuerySet
|
||||||
@@ -312,12 +311,11 @@ def consumable_document_matches_workflow(
|
|||||||
trigger_matched = False
|
trigger_matched = False
|
||||||
|
|
||||||
# Document filename vs trigger filename
|
# Document filename vs trigger filename
|
||||||
document_filename = normalize_unicode(document.original_file.name)
|
|
||||||
if (
|
if (
|
||||||
trigger.filter_filename is not None
|
trigger.filter_filename is not None
|
||||||
and len(trigger.filter_filename) > 0
|
and len(trigger.filter_filename) > 0
|
||||||
and not fnmatch(
|
and not fnmatch(
|
||||||
document_filename.lower(),
|
document.original_file.name.lower(),
|
||||||
trigger.filter_filename.lower(),
|
trigger.filter_filename.lower(),
|
||||||
)
|
)
|
||||||
):
|
):
|
||||||
@@ -330,12 +328,10 @@ def consumable_document_matches_workflow(
|
|||||||
# Document path vs trigger path
|
# Document path vs trigger path
|
||||||
|
|
||||||
# Use the original_path if set, else us the original_file
|
# Use the original_path if set, else us the original_file
|
||||||
match_against = normalize_unicode(
|
match_against = (
|
||||||
str(
|
|
||||||
document.original_path
|
document.original_path
|
||||||
if document.original_path is not None
|
if document.original_path is not None
|
||||||
else document.original_file,
|
else document.original_file
|
||||||
),
|
|
||||||
)
|
)
|
||||||
|
|
||||||
if (
|
if (
|
||||||
@@ -540,7 +536,7 @@ def existing_document_matches_workflow(
|
|||||||
and len(trigger.filter_filename) > 0
|
and len(trigger.filter_filename) > 0
|
||||||
and document.original_filename is not None
|
and document.original_filename is not None
|
||||||
and not fnmatch(
|
and not fnmatch(
|
||||||
normalize_unicode(document.original_filename).lower(),
|
document.original_filename.lower(),
|
||||||
trigger.filter_filename.lower(),
|
trigger.filter_filename.lower(),
|
||||||
)
|
)
|
||||||
):
|
):
|
||||||
|
|||||||
+25
-4
@@ -1,4 +1,5 @@
|
|||||||
import datetime
|
import datetime
|
||||||
|
import uuid
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Final
|
from typing import Final
|
||||||
|
|
||||||
@@ -27,7 +28,6 @@ from django_softdelete.models import SoftDeleteModel
|
|||||||
|
|
||||||
from documents.data_models import DocumentSource
|
from documents.data_models import DocumentSource
|
||||||
from documents.parsers import get_default_file_extension
|
from documents.parsers import get_default_file_extension
|
||||||
from documents.utils import normalize_unicode
|
|
||||||
|
|
||||||
|
|
||||||
class ModelWithOwner(models.Model):
|
class ModelWithOwner(models.Model):
|
||||||
@@ -375,6 +375,7 @@ class Document(SoftDeleteModel, ModelWithOwner): # type: ignore[django-manager-
|
|||||||
If the queryset already annotated ``effective_content``, that value is used.
|
If the queryset already annotated ``effective_content``, that value is used.
|
||||||
"""
|
"""
|
||||||
# Here to avoid circular import
|
# Here to avoid circular import
|
||||||
|
from documents.versioning import LATEST_VERSION_CONTENT_PREFETCH_ATTR
|
||||||
from documents.versioning import sort_versions_newest_first
|
from documents.versioning import sort_versions_newest_first
|
||||||
from documents.versioning import versions_newest_first
|
from documents.versioning import versions_newest_first
|
||||||
|
|
||||||
@@ -384,6 +385,19 @@ class Document(SoftDeleteModel, ModelWithOwner): # type: ignore[django-manager-
|
|||||||
if self.root_document_id is not None or self.pk is None:
|
if self.root_document_id is not None or self.pk is None:
|
||||||
return self.content
|
return self.content
|
||||||
|
|
||||||
|
latest_version_prefetch = getattr(
|
||||||
|
self,
|
||||||
|
LATEST_VERSION_CONTENT_PREFETCH_ATTR,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
if latest_version_prefetch is not None:
|
||||||
|
# Empty list means prefetch ran and found no versions — use own content.
|
||||||
|
return (
|
||||||
|
latest_version_prefetch[0].content
|
||||||
|
if latest_version_prefetch
|
||||||
|
else self.content
|
||||||
|
)
|
||||||
|
|
||||||
prefetched_cache = getattr(self, "_prefetched_objects_cache", None)
|
prefetched_cache = getattr(self, "_prefetched_objects_cache", None)
|
||||||
prefetched_versions = (
|
prefetched_versions = (
|
||||||
prefetched_cache.get("versions")
|
prefetched_cache.get("versions")
|
||||||
@@ -468,7 +482,7 @@ class Document(SoftDeleteModel, ModelWithOwner): # type: ignore[django-manager-
|
|||||||
context_document = (
|
context_document = (
|
||||||
self.root_document if self.root_document_id is not None else self
|
self.root_document if self.root_document_id is not None else self
|
||||||
)
|
)
|
||||||
result = normalize_unicode(str(context_document))
|
result = str(context_document)
|
||||||
|
|
||||||
if counter:
|
if counter:
|
||||||
result += f"_{counter:02}"
|
result += f"_{counter:02}"
|
||||||
@@ -515,13 +529,20 @@ class Document(SoftDeleteModel, ModelWithOwner): # type: ignore[django-manager-
|
|||||||
def delete(
|
def delete(
|
||||||
self,
|
self,
|
||||||
*args,
|
*args,
|
||||||
|
transaction_id=None,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
# If deleting a root document, move all its versions to trash as well.
|
# Versions must share the root's transaction ID so they are restored
|
||||||
|
# together by django-softdelete.
|
||||||
|
if transaction_id is None:
|
||||||
|
transaction_id = uuid.uuid4()
|
||||||
if self.root_document_id is None:
|
if self.root_document_id is None:
|
||||||
Document.objects.filter(root_document=self).delete()
|
Document.objects.filter(root_document=self).delete(
|
||||||
|
transaction_id=transaction_id,
|
||||||
|
)
|
||||||
return super().delete(
|
return super().delete(
|
||||||
*args,
|
*args,
|
||||||
|
transaction_id=transaction_id,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -87,9 +87,9 @@ from documents.regex import validate_regex_pattern
|
|||||||
from documents.templating.filepath import validate_filepath_template_and_render
|
from documents.templating.filepath import validate_filepath_template_and_render
|
||||||
from documents.templating.utils import convert_format_str_to_template_format
|
from documents.templating.utils import convert_format_str_to_template_format
|
||||||
from documents.templating.workflows import validate_workflow_template
|
from documents.templating.workflows import validate_workflow_template
|
||||||
from documents.utils import normalize_unicode
|
|
||||||
from documents.validators import uri_validator
|
from documents.validators import uri_validator
|
||||||
from documents.validators import url_validator
|
from documents.validators import url_validator
|
||||||
|
from documents.versioning import has_prefetched_effective_content
|
||||||
from documents.versioning import sort_versions_newest_first
|
from documents.versioning import sort_versions_newest_first
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
@@ -1153,8 +1153,14 @@ class DocumentSerializer(
|
|||||||
|
|
||||||
def to_representation(self, instance):
|
def to_representation(self, instance):
|
||||||
doc = super().to_representation(instance)
|
doc = super().to_representation(instance)
|
||||||
if "content" in self.fields and hasattr(instance, "effective_content"):
|
if "content" in self.fields and has_prefetched_effective_content(instance):
|
||||||
doc["content"] = getattr(instance, "effective_content") or ""
|
# Only resolve version-aware content when it's cheap: an SQL
|
||||||
|
# annotation or a versions prefetch is already on the instance.
|
||||||
|
# A caller that set up neither (e.g. TrashView, GlobalSearchView,
|
||||||
|
# which build their own querysets) gets the document's own,
|
||||||
|
# unresolved content instead of paying for an extra per-instance
|
||||||
|
# query -- same as before effective_content resolution existed.
|
||||||
|
doc["content"] = instance.get_effective_content() or ""
|
||||||
if self.truncate_content and "content" in self.fields:
|
if self.truncate_content and "content" in self.fields:
|
||||||
doc["content"] = doc.get("content")[0:550]
|
doc["content"] = doc.get("content")[0:550]
|
||||||
return doc
|
return doc
|
||||||
@@ -1248,30 +1254,31 @@ class DocumentSerializer(
|
|||||||
|
|
||||||
validated_data["tags"] = list(final_tags)
|
validated_data["tags"] = list(final_tags)
|
||||||
if validated_data.get("remove_inbox_tags"):
|
if validated_data.get("remove_inbox_tags"):
|
||||||
tag_ids_being_added = (
|
current_tag_ids = {t.pk for t in instance.tags.all()}
|
||||||
[
|
tags = (
|
||||||
tag.id
|
validated_data["tags"]
|
||||||
for tag in validated_data["tags"]
|
|
||||||
if tag not in instance.tags.all()
|
|
||||||
]
|
|
||||||
if "tags" in validated_data
|
if "tags" in validated_data
|
||||||
else []
|
else list(instance.tags.all())
|
||||||
)
|
)
|
||||||
inbox_tags_not_being_added = Tag.objects.filter(is_inbox_tag=True).exclude(
|
|
||||||
id__in=tag_ids_being_added,
|
# Tags newly added in this update, plus their ancestors, are kept
|
||||||
)
|
keep_ids: set[int] = set()
|
||||||
if "tags" in validated_data:
|
for tag in tags:
|
||||||
validated_data["tags"] = [
|
if tag.pk not in current_tag_ids:
|
||||||
tag
|
keep_ids.add(tag.pk)
|
||||||
for tag in validated_data["tags"]
|
keep_ids.update(int(pk) for pk in tag.get_ancestors_pks())
|
||||||
if tag not in inbox_tags_not_being_added
|
|
||||||
]
|
# Remove inbox tags and their descendants, except those being kept
|
||||||
else:
|
remove_ids: set[int] = set()
|
||||||
validated_data["tags"] = [
|
for inbox_tag in (
|
||||||
tag
|
Tag.objects.filter(is_inbox_tag=True)
|
||||||
for tag in instance.tags.all()
|
.exclude(pk__in=keep_ids)
|
||||||
if tag not in inbox_tags_not_being_added
|
.only("pk", "tn_descendants_pks")
|
||||||
]
|
):
|
||||||
|
remove_ids.add(inbox_tag.pk)
|
||||||
|
remove_ids.update(int(pk) for pk in inbox_tag.get_descendants_pks())
|
||||||
|
|
||||||
|
validated_data["tags"] = [t for t in tags if t.pk not in remove_ids]
|
||||||
|
|
||||||
if settings.AUDIT_LOG_ENABLED:
|
if settings.AUDIT_LOG_ENABLED:
|
||||||
with set_actor(self.user):
|
with set_actor(self.user):
|
||||||
@@ -3121,13 +3128,6 @@ class WorkflowTriggerSerializer(serializers.ModelSerializer[WorkflowTrigger]):
|
|||||||
):
|
):
|
||||||
attrs["filter_path"] = None
|
attrs["filter_path"] = None
|
||||||
|
|
||||||
# Normalize once at write time, since these are matched against many
|
|
||||||
# documents but edited rarely
|
|
||||||
if attrs.get("filter_filename") is not None:
|
|
||||||
attrs["filter_filename"] = normalize_unicode(attrs["filter_filename"])
|
|
||||||
if attrs.get("filter_path") is not None:
|
|
||||||
attrs["filter_path"] = normalize_unicode(attrs["filter_path"])
|
|
||||||
|
|
||||||
if (
|
if (
|
||||||
"filter_custom_field_query" in attrs
|
"filter_custom_field_query" in attrs
|
||||||
and attrs["filter_custom_field_query"] is not None
|
and attrs["filter_custom_field_query"] is not None
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
|
import unicodedata
|
||||||
from collections.abc import Iterable
|
from collections.abc import Iterable
|
||||||
from pathlib import PurePath
|
from pathlib import PurePath
|
||||||
|
|
||||||
@@ -25,7 +26,6 @@ from documents.templating.environment import _template_environment
|
|||||||
from documents.templating.filters import format_datetime
|
from documents.templating.filters import format_datetime
|
||||||
from documents.templating.filters import get_cf_value
|
from documents.templating.filters import get_cf_value
|
||||||
from documents.templating.filters import localize_date
|
from documents.templating.filters import localize_date
|
||||||
from documents.utils import normalize_unicode
|
|
||||||
|
|
||||||
logger = logging.getLogger("paperless.templating")
|
logger = logging.getLogger("paperless.templating")
|
||||||
|
|
||||||
@@ -42,7 +42,7 @@ class FilePathTemplate(Template):
|
|||||||
3. Removing extra spaces before and after forward slashes
|
3. Removing extra spaces before and after forward slashes
|
||||||
4. Preserving spaces in other parts of the path
|
4. Preserving spaces in other parts of the path
|
||||||
"""
|
"""
|
||||||
value = normalize_unicode(value)
|
value = unicodedata.normalize("NFC", value)
|
||||||
value = value.replace("\n", "").replace("\r", "")
|
value = value.replace("\n", "").replace("\r", "")
|
||||||
value = re.sub(r"\s*/\s*", "/", value)
|
value = re.sub(r"\s*/\s*", "/", value)
|
||||||
|
|
||||||
@@ -184,17 +184,17 @@ def get_basic_metadata_context(
|
|||||||
"""
|
"""
|
||||||
return {
|
return {
|
||||||
"title": pathvalidate.sanitize_filename(
|
"title": pathvalidate.sanitize_filename(
|
||||||
normalize_unicode(document.title),
|
unicodedata.normalize("NFC", document.title),
|
||||||
replacement_text="-",
|
replacement_text="-",
|
||||||
),
|
),
|
||||||
"correspondent": pathvalidate.sanitize_filename(
|
"correspondent": pathvalidate.sanitize_filename(
|
||||||
normalize_unicode(document.correspondent.name),
|
unicodedata.normalize("NFC", document.correspondent.name),
|
||||||
replacement_text="-",
|
replacement_text="-",
|
||||||
)
|
)
|
||||||
if document.correspondent
|
if document.correspondent
|
||||||
else no_value_default,
|
else no_value_default,
|
||||||
"document_type": pathvalidate.sanitize_filename(
|
"document_type": pathvalidate.sanitize_filename(
|
||||||
normalize_unicode(document.document_type.name),
|
unicodedata.normalize("NFC", document.document_type.name),
|
||||||
replacement_text="-",
|
replacement_text="-",
|
||||||
)
|
)
|
||||||
if document.document_type
|
if document.document_type
|
||||||
@@ -205,7 +205,8 @@ def get_basic_metadata_context(
|
|||||||
"owner_username": document.owner.username
|
"owner_username": document.owner.username
|
||||||
if document.owner
|
if document.owner
|
||||||
else no_value_default,
|
else no_value_default,
|
||||||
"original_name": normalize_unicode(
|
"original_name": unicodedata.normalize(
|
||||||
|
"NFC",
|
||||||
PurePath(document.original_filename).with_suffix("").name,
|
PurePath(document.original_filename).with_suffix("").name,
|
||||||
)
|
)
|
||||||
if document.original_filename
|
if document.original_filename
|
||||||
@@ -274,12 +275,12 @@ def get_tags_context(tags: Iterable[Tag]) -> dict[str, str | list[str]]:
|
|||||||
return {
|
return {
|
||||||
"tag_list": pathvalidate.sanitize_filename(
|
"tag_list": pathvalidate.sanitize_filename(
|
||||||
",".join(
|
",".join(
|
||||||
sorted(normalize_unicode(tag.name) for tag in tags),
|
sorted(unicodedata.normalize("NFC", tag.name) for tag in tags),
|
||||||
),
|
),
|
||||||
replacement_text="-",
|
replacement_text="-",
|
||||||
),
|
),
|
||||||
# Assumed to be ordered, but a template could loop through to find what they want
|
# Assumed to be ordered, but a template could loop through to find what they want
|
||||||
"tag_name_list": [normalize_unicode(x.name) for x in tags],
|
"tag_name_list": [unicodedata.normalize("NFC", x.name) for x in tags],
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -306,7 +307,7 @@ def get_custom_fields_context(
|
|||||||
CustomField.FieldDataType.LONG_TEXT,
|
CustomField.FieldDataType.LONG_TEXT,
|
||||||
}:
|
}:
|
||||||
value = pathvalidate.sanitize_filename(
|
value = pathvalidate.sanitize_filename(
|
||||||
normalize_unicode(field_instance.value),
|
unicodedata.normalize("NFC", field_instance.value),
|
||||||
replacement_text="-",
|
replacement_text="-",
|
||||||
)
|
)
|
||||||
elif (
|
elif (
|
||||||
@@ -315,7 +316,8 @@ def get_custom_fields_context(
|
|||||||
):
|
):
|
||||||
options = field_instance.field.extra_data["select_options"]
|
options = field_instance.field.extra_data["select_options"]
|
||||||
value = pathvalidate.sanitize_filename(
|
value = pathvalidate.sanitize_filename(
|
||||||
normalize_unicode(
|
unicodedata.normalize(
|
||||||
|
"NFC",
|
||||||
next(
|
next(
|
||||||
option["label"]
|
option["label"]
|
||||||
for option in options
|
for option in options
|
||||||
@@ -328,7 +330,7 @@ def get_custom_fields_context(
|
|||||||
value = field_instance.value
|
value = field_instance.value
|
||||||
field_data["custom_fields"][
|
field_data["custom_fields"][
|
||||||
pathvalidate.sanitize_filename(
|
pathvalidate.sanitize_filename(
|
||||||
normalize_unicode(field_instance.field.name),
|
unicodedata.normalize("NFC", field_instance.field.name),
|
||||||
replacement_text="-",
|
replacement_text="-",
|
||||||
)
|
)
|
||||||
] = {
|
] = {
|
||||||
|
|||||||
@@ -207,3 +207,65 @@ class TestTrashAPI(DirectoriesMixin, APITestCase):
|
|||||||
)
|
)
|
||||||
self.assertEqual(resp.status_code, status.HTTP_400_BAD_REQUEST)
|
self.assertEqual(resp.status_code, status.HTTP_400_BAD_REQUEST)
|
||||||
self.assertIn("have not yet been deleted", resp.data["documents"][0])
|
self.assertIn("have not yet been deleted", resp.data["documents"][0])
|
||||||
|
|
||||||
|
def _make_versioned_document(self) -> tuple[Document, list[Document]]:
|
||||||
|
root = Document.objects.create(
|
||||||
|
title="root",
|
||||||
|
content="root-content",
|
||||||
|
checksum="root",
|
||||||
|
mime_type="application/pdf",
|
||||||
|
)
|
||||||
|
versions = [
|
||||||
|
Document.objects.create(
|
||||||
|
title=f"v{index}",
|
||||||
|
content=f"v{index}-content",
|
||||||
|
checksum=f"v{index}",
|
||||||
|
mime_type="application/pdf",
|
||||||
|
root_document=root,
|
||||||
|
version_index=index,
|
||||||
|
)
|
||||||
|
for index in range(1, 3)
|
||||||
|
]
|
||||||
|
return root, versions
|
||||||
|
|
||||||
|
def test_api_trash_restore_document_restores_its_versions(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- Existing document with two versions
|
||||||
|
WHEN:
|
||||||
|
- API request to delete the document
|
||||||
|
- API request to restore it from the trash
|
||||||
|
THEN:
|
||||||
|
- Only the document itself is listed in the trash
|
||||||
|
- A version cannot be restored without its root
|
||||||
|
- The document is restored together with all of its versions
|
||||||
|
"""
|
||||||
|
root, versions = self._make_versioned_document()
|
||||||
|
|
||||||
|
self.client.force_login(user=self.user)
|
||||||
|
self.client.delete(f"/api/documents/{root.pk}/")
|
||||||
|
self.assertEqual(Document.deleted_objects.count(), 3)
|
||||||
|
|
||||||
|
resp = self.client.get("/api/trash/")
|
||||||
|
self.assertEqual(resp.status_code, status.HTTP_200_OK)
|
||||||
|
self.assertEqual(resp.data["count"], 1)
|
||||||
|
self.assertEqual(resp.data["results"][0]["id"], root.pk)
|
||||||
|
|
||||||
|
# A version cannot be restored while its root remains in the trash.
|
||||||
|
resp = self.client.post(
|
||||||
|
"/api/trash/",
|
||||||
|
{"action": "restore", "documents": [versions[0].pk]},
|
||||||
|
)
|
||||||
|
self.assertEqual(resp.status_code, status.HTTP_400_BAD_REQUEST)
|
||||||
|
self.assertIn("Restore the root document", resp.data["documents"][0])
|
||||||
|
|
||||||
|
resp = self.client.post(
|
||||||
|
"/api/trash/",
|
||||||
|
{"action": "restore", "documents": [root.pk]},
|
||||||
|
)
|
||||||
|
self.assertEqual(resp.status_code, status.HTTP_200_OK)
|
||||||
|
self.assertEqual(Document.deleted_objects.count(), 0)
|
||||||
|
self.assertCountEqual(
|
||||||
|
Document.objects.filter(root_document=root).values_list("id", flat=True),
|
||||||
|
[version.pk for version in versions],
|
||||||
|
)
|
||||||
|
|||||||
@@ -1,69 +0,0 @@
|
|||||||
import unicodedata
|
|
||||||
from typing import TYPE_CHECKING
|
|
||||||
from unittest import mock
|
|
||||||
|
|
||||||
import celery.result
|
|
||||||
import pytest
|
|
||||||
from django.core.files.uploadedfile import SimpleUploadedFile
|
|
||||||
|
|
||||||
from documents.models import Document
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
|
||||||
from documents.data_models import ConsumableDocument
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture()
|
|
||||||
def consume_file_mock():
|
|
||||||
with mock.patch("documents.tasks.consume_file.apply_async") as m:
|
|
||||||
m.return_value = celery.result.AsyncResult(id="test-task-id")
|
|
||||||
yield m
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture()
|
|
||||||
def directories(tmp_path, settings, _media_settings):
|
|
||||||
scratch = tmp_path / "scratch"
|
|
||||||
scratch.mkdir()
|
|
||||||
settings.SCRATCH_DIR = scratch
|
|
||||||
return scratch
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
|
||||||
class TestUpdateVersionNFCNormalization:
|
|
||||||
def test_nfd_filename_normalized_to_nfc(
|
|
||||||
self,
|
|
||||||
admin_client,
|
|
||||||
consume_file_mock: mock.MagicMock,
|
|
||||||
directories,
|
|
||||||
):
|
|
||||||
"""Uploaded new-version file with NFD filename must have its temp name stored as NFC."""
|
|
||||||
document = Document.objects.create(
|
|
||||||
title="Test",
|
|
||||||
content="content",
|
|
||||||
checksum="checksum",
|
|
||||||
mime_type="application/pdf",
|
|
||||||
)
|
|
||||||
|
|
||||||
nfd = unicodedata.normalize("NFD", "Rechnung März.pdf")
|
|
||||||
nfc = unicodedata.normalize("NFC", "Rechnung März.pdf")
|
|
||||||
|
|
||||||
assert nfd != nfc
|
|
||||||
|
|
||||||
uploaded = SimpleUploadedFile(
|
|
||||||
nfd,
|
|
||||||
b"%PDF-1.4 test",
|
|
||||||
content_type="application/pdf",
|
|
||||||
)
|
|
||||||
response = admin_client.post(
|
|
||||||
f"/api/documents/{document.pk}/update_version/",
|
|
||||||
{"document": uploaded},
|
|
||||||
)
|
|
||||||
|
|
||||||
assert response.status_code == 200
|
|
||||||
|
|
||||||
task_kwargs = consume_file_mock.call_args.kwargs["kwargs"]
|
|
||||||
input_doc: ConsumableDocument = task_kwargs["input_doc"]
|
|
||||||
|
|
||||||
assert input_doc.original_file.name == nfc, (
|
|
||||||
f"Expected NFC filename {nfc!r}, got {input_doc.original_file.name!r}"
|
|
||||||
)
|
|
||||||
assert unicodedata.is_normalized("NFC", input_doc.original_file.name)
|
|
||||||
@@ -392,6 +392,11 @@ class TestBulkEdit(DirectoriesMixin, TestCase):
|
|||||||
self.assertFalse(Document.objects.filter(id=self.doc1.id).exists())
|
self.assertFalse(Document.objects.filter(id=self.doc1.id).exists())
|
||||||
self.assertFalse(Document.objects.filter(id=version.id).exists())
|
self.assertFalse(Document.objects.filter(id=version.id).exists())
|
||||||
|
|
||||||
|
Document.deleted_objects.get(id=self.doc1.id).restore(strict=False)
|
||||||
|
|
||||||
|
self.assertTrue(Document.objects.filter(id=self.doc1.id).exists())
|
||||||
|
self.assertTrue(Document.objects.filter(id=version.id).exists())
|
||||||
|
|
||||||
def test_delete_version_document_keeps_root(self) -> None:
|
def test_delete_version_document_keeps_root(self) -> None:
|
||||||
version = Document.objects.create(
|
version = Document.objects.create(
|
||||||
checksum="A-v1",
|
checksum="A-v1",
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ import warnings
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from unittest import mock
|
from unittest import mock
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
import pytest
|
import pytest
|
||||||
from django.conf import settings
|
from django.conf import settings
|
||||||
from django.test import TestCase
|
from django.test import TestCase
|
||||||
@@ -11,6 +12,7 @@ from django.test import override_settings
|
|||||||
from documents.classifier import ClassifierModelCorruptError
|
from documents.classifier import ClassifierModelCorruptError
|
||||||
from documents.classifier import DocumentClassifier
|
from documents.classifier import DocumentClassifier
|
||||||
from documents.classifier import IncompatibleClassifierVersionError
|
from documents.classifier import IncompatibleClassifierVersionError
|
||||||
|
from documents.classifier import _predict_with_threshold
|
||||||
from documents.classifier import load_classifier
|
from documents.classifier import load_classifier
|
||||||
from documents.models import Correspondent
|
from documents.models import Correspondent
|
||||||
from documents.models import Document
|
from documents.models import Document
|
||||||
@@ -625,6 +627,103 @@ class TestClassifier(DirectoriesMixin, TestCase):
|
|||||||
self.assertEqual(self.classifier.predict_storage_path(doc1.content), sp.pk)
|
self.assertEqual(self.classifier.predict_storage_path(doc1.content), sp.pk)
|
||||||
self.assertIsNone(self.classifier.predict_storage_path(doc2.content))
|
self.assertIsNone(self.classifier.predict_storage_path(doc2.content))
|
||||||
|
|
||||||
|
def test_predict_rejects_prediction_below_match_threshold(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- Classifiers trained against test data with confident predictions
|
||||||
|
WHEN:
|
||||||
|
- CLASSIFIER_MATCH_THRESHOLD exceeds the model's confidence
|
||||||
|
THEN:
|
||||||
|
- Every predict_* method discards the match in favor of no match
|
||||||
|
"""
|
||||||
|
c1 = Correspondent.objects.create(
|
||||||
|
name="c1",
|
||||||
|
matching_algorithm=Correspondent.MATCH_AUTO,
|
||||||
|
)
|
||||||
|
dt1 = DocumentType.objects.create(
|
||||||
|
name="dt1",
|
||||||
|
matching_algorithm=DocumentType.MATCH_AUTO,
|
||||||
|
)
|
||||||
|
sp1 = StoragePath.objects.create(
|
||||||
|
name="sp1",
|
||||||
|
matching_algorithm=StoragePath.MATCH_AUTO,
|
||||||
|
)
|
||||||
|
|
||||||
|
doc1 = Document.objects.create(
|
||||||
|
title="doc1",
|
||||||
|
content="this is a document from c1",
|
||||||
|
correspondent=c1,
|
||||||
|
document_type=dt1,
|
||||||
|
storage_path=sp1,
|
||||||
|
checksum="A",
|
||||||
|
)
|
||||||
|
Document.objects.create(
|
||||||
|
title="doc2",
|
||||||
|
content="this is a document from no one",
|
||||||
|
checksum="B",
|
||||||
|
)
|
||||||
|
|
||||||
|
self.classifier.train()
|
||||||
|
|
||||||
|
predictors = {
|
||||||
|
"correspondent": self.classifier.predict_correspondent,
|
||||||
|
"document_type": self.classifier.predict_document_type,
|
||||||
|
"storage_path": self.classifier.predict_storage_path,
|
||||||
|
}
|
||||||
|
# No real prediction can reach a confidence this high, so this
|
||||||
|
# isolates the threshold check from the model's actual output.
|
||||||
|
with override_settings(CLASSIFIER_MATCH_THRESHOLD=0.999999):
|
||||||
|
for name, predict in predictors.items():
|
||||||
|
with self.subTest(field=name):
|
||||||
|
self.assertIsNone(predict(doc1.content))
|
||||||
|
|
||||||
|
def test_train_uses_balanced_sample_weight(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A training set with correspondents, document types and storage paths
|
||||||
|
WHEN:
|
||||||
|
- The classifier is trained
|
||||||
|
THEN:
|
||||||
|
- Each MLP classifier is fit with balanced sample weights, so that
|
||||||
|
over-represented classes don't dominate predictions
|
||||||
|
"""
|
||||||
|
c1 = Correspondent.objects.create(
|
||||||
|
name="c1",
|
||||||
|
matching_algorithm=Correspondent.MATCH_AUTO,
|
||||||
|
)
|
||||||
|
dt1 = DocumentType.objects.create(
|
||||||
|
name="dt1",
|
||||||
|
matching_algorithm=DocumentType.MATCH_AUTO,
|
||||||
|
)
|
||||||
|
sp1 = StoragePath.objects.create(
|
||||||
|
name="sp1",
|
||||||
|
matching_algorithm=StoragePath.MATCH_AUTO,
|
||||||
|
)
|
||||||
|
|
||||||
|
Document.objects.create(
|
||||||
|
title="doc1",
|
||||||
|
content="this is a document from c1",
|
||||||
|
correspondent=c1,
|
||||||
|
document_type=dt1,
|
||||||
|
storage_path=sp1,
|
||||||
|
checksum="A",
|
||||||
|
)
|
||||||
|
Document.objects.create(
|
||||||
|
title="doc2",
|
||||||
|
content="this is a document from no one",
|
||||||
|
checksum="B",
|
||||||
|
)
|
||||||
|
|
||||||
|
with mock.patch(
|
||||||
|
"sklearn.utils.class_weight.compute_sample_weight",
|
||||||
|
return_value=None,
|
||||||
|
) as mocked_compute_sample_weight:
|
||||||
|
self.classifier.train()
|
||||||
|
|
||||||
|
self.assertEqual(mocked_compute_sample_weight.call_count, 3)
|
||||||
|
for call in mocked_compute_sample_weight.call_args_list:
|
||||||
|
self.assertEqual(call.args[0], "balanced")
|
||||||
|
|
||||||
def test_one_tag_predict(self) -> None:
|
def test_one_tag_predict(self) -> None:
|
||||||
t1 = Tag.objects.create(name="t1", matching_algorithm=Tag.MATCH_AUTO, pk=12)
|
t1 = Tag.objects.create(name="t1", matching_algorithm=Tag.MATCH_AUTO, pk=12)
|
||||||
|
|
||||||
@@ -810,6 +909,52 @@ class TestClassifier(DirectoriesMixin, TestCase):
|
|||||||
load_classifier(raise_exception=True)
|
load_classifier(raise_exception=True)
|
||||||
|
|
||||||
|
|
||||||
|
class _StubProbaClassifier:
|
||||||
|
"""
|
||||||
|
A fake scikit-learn classifier exposing just enough of the API for
|
||||||
|
`_predict_with_threshold`: `classes_` and `predict_proba`.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, classes: list[int], probabilities: list[float]) -> None:
|
||||||
|
self.classes_ = np.array(classes)
|
||||||
|
self._probabilities = np.array([probabilities])
|
||||||
|
|
||||||
|
def predict_proba(self, X) -> np.ndarray:
|
||||||
|
return self._probabilities
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("classes", "probabilities", "threshold", "expected"),
|
||||||
|
[
|
||||||
|
# confident prediction above the threshold is returned
|
||||||
|
([-1, 3], [0.1, 0.9], 0.6, 3),
|
||||||
|
# prediction below the threshold is discarded
|
||||||
|
([-1, 3], [0.45, 0.55], 0.6, None),
|
||||||
|
# boundary: exactly at the threshold is accepted, not discarded
|
||||||
|
([-1, 3], [0.4, 0.6], 0.6, 3),
|
||||||
|
# the winning class is the "no match" pseudo-class, regardless of its
|
||||||
|
# own confidence
|
||||||
|
([-1, 3], [0.99, 0.01], 0.0, None),
|
||||||
|
# threshold of 0.0 disables the confidence check entirely
|
||||||
|
([-1, 3], [0.45, 0.55], 0.0, 3),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_predict_with_threshold(classes, probabilities, threshold, expected) -> None:
|
||||||
|
classifier = _StubProbaClassifier(classes, probabilities)
|
||||||
|
result = _predict_with_threshold(classifier, X=None, threshold=threshold)
|
||||||
|
assert result == expected
|
||||||
|
|
||||||
|
|
||||||
|
def test_classifier_match_threshold_default() -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- No PAPERLESS_CLASSIFIER_MATCH_THRESHOLD environment variable is set
|
||||||
|
THEN:
|
||||||
|
- The classifier match threshold defaults to 0.6
|
||||||
|
"""
|
||||||
|
assert settings.CLASSIFIER_MATCH_THRESHOLD == 0.6
|
||||||
|
|
||||||
|
|
||||||
def test_preprocess_content() -> None:
|
def test_preprocess_content() -> None:
|
||||||
"""
|
"""
|
||||||
GIVEN:
|
GIVEN:
|
||||||
|
|||||||
@@ -0,0 +1,457 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from types import SimpleNamespace
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from django.db import connection
|
||||||
|
from django.test.utils import CaptureQueriesContext
|
||||||
|
from rest_framework import status
|
||||||
|
|
||||||
|
from documents.models import Document
|
||||||
|
from documents.tests.factories import DocumentFactory
|
||||||
|
from documents.versioning import LATEST_VERSION_CONTENT_PREFETCH_ATTR
|
||||||
|
from documents.versioning import has_prefetched_effective_content
|
||||||
|
from documents.versioning import latest_version_content_prefetch
|
||||||
|
from documents.views import DocumentViewSet
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from rest_framework.test import APIClient
|
||||||
|
|
||||||
|
|
||||||
|
class TestNeedsEffectiveContentAnnotation:
|
||||||
|
"""
|
||||||
|
DocumentViewSet._needs_effective_content_annotation() decides whether
|
||||||
|
the effective_content correlated subquery is worth attaching to the
|
||||||
|
queryset at all -- see TestDocumentListEffectiveContentAnnotation below
|
||||||
|
for why. This only checks that decision's own logic (a plain query-param
|
||||||
|
membership test), not that Django/DRF's filtering machinery works.
|
||||||
|
"""
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("params", "expected"),
|
||||||
|
[
|
||||||
|
({}, False),
|
||||||
|
({"ordering": "-added"}, False),
|
||||||
|
({"tags__id__in": "1,2"}, False),
|
||||||
|
({"search": ""}, False),
|
||||||
|
({"search": " "}, False),
|
||||||
|
({"content__icontains": ""}, False),
|
||||||
|
({"search": "foo"}, True),
|
||||||
|
({"title_content": "foo"}, True),
|
||||||
|
({"content__istartswith": "foo"}, True),
|
||||||
|
({"content__iendswith": "foo"}, True),
|
||||||
|
({"content__icontains": "foo"}, True),
|
||||||
|
({"content__iexact": "foo"}, True),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_detects_content_filter_params(
|
||||||
|
self,
|
||||||
|
params: dict[str, str],
|
||||||
|
expected: bool, # noqa: FBT001
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A view bound to a request carrying the given query params
|
||||||
|
WHEN:
|
||||||
|
- Checking whether the effective_content annotation is needed
|
||||||
|
THEN:
|
||||||
|
- It is needed only for requests that actually filter on it
|
||||||
|
"""
|
||||||
|
view = DocumentViewSet()
|
||||||
|
view.request = SimpleNamespace(query_params=params)
|
||||||
|
|
||||||
|
assert view._needs_effective_content_annotation() is expected
|
||||||
|
|
||||||
|
|
||||||
|
class TestNeedsEffectiveContentPrefetch:
|
||||||
|
"""
|
||||||
|
DocumentViewSet._needs_effective_content_prefetch() decides whether the
|
||||||
|
single-version content prefetch is worth attaching. It has to read the
|
||||||
|
`fields` param exactly the way get_serializer() does, or a request whose
|
||||||
|
response includes content ends up without the prefetch and pays
|
||||||
|
get_effective_content()'s per-instance fallback instead.
|
||||||
|
"""
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("params", "expected"),
|
||||||
|
[
|
||||||
|
pytest.param({}, True, id="no-fields-param-keeps-every-field"),
|
||||||
|
pytest.param({"fields": ""}, True, id="blank-fields-keeps-every-field"),
|
||||||
|
pytest.param(
|
||||||
|
{"fields": "id,content"},
|
||||||
|
True,
|
||||||
|
id="content-among-requested-fields",
|
||||||
|
),
|
||||||
|
pytest.param({"fields": "content"}, True, id="content-only"),
|
||||||
|
pytest.param({"fields": "id"}, False, id="content-not-requested"),
|
||||||
|
pytest.param(
|
||||||
|
{"fields": "id,title"},
|
||||||
|
False,
|
||||||
|
id="several-fields-without-content",
|
||||||
|
),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_detects_whether_content_can_reach_the_response(
|
||||||
|
self,
|
||||||
|
params: dict[str, str],
|
||||||
|
expected: bool, # noqa: FBT001
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A view bound to a request carrying the given query params
|
||||||
|
WHEN:
|
||||||
|
- Checking whether the content prefetch is needed
|
||||||
|
THEN:
|
||||||
|
- It is needed exactly when get_serializer() would emit content,
|
||||||
|
which treats a blank `fields` the same as an absent one
|
||||||
|
"""
|
||||||
|
view = DocumentViewSet()
|
||||||
|
view.request = SimpleNamespace(query_params=params)
|
||||||
|
|
||||||
|
assert view._needs_effective_content_prefetch() is expected
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.django_db
|
||||||
|
class TestDocumentListEffectiveContentAnnotation:
|
||||||
|
"""
|
||||||
|
DocumentViewSet.get_queryset() only attaches the effective_content
|
||||||
|
correlated subquery when a request actually filters on it. Attaching it
|
||||||
|
unconditionally re-executes it once per candidate row before the page's
|
||||||
|
LIMIT is applied -- fine on SQLite/Postgres, but pathological on
|
||||||
|
MariaDB's default cardinality estimation for the root_document_id
|
||||||
|
self-join once candidate counts get large (see the root_document_id /
|
||||||
|
effective_content perf investigation).
|
||||||
|
"""
|
||||||
|
|
||||||
|
def test_list_without_content_filter_skips_annotation_but_returns_latest_content(
|
||||||
|
self,
|
||||||
|
admin_client: APIClient,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document whose latest version has different content
|
||||||
|
WHEN:
|
||||||
|
- Listing documents with no search/content-filter param
|
||||||
|
THEN:
|
||||||
|
- The response still reflects the latest version's content
|
||||||
|
- The database never evaluates effective_content per row
|
||||||
|
"""
|
||||||
|
root = DocumentFactory(content="old-root-content")
|
||||||
|
DocumentFactory(
|
||||||
|
root_document=root,
|
||||||
|
version_index=1,
|
||||||
|
content="new-version-content",
|
||||||
|
)
|
||||||
|
|
||||||
|
with CaptureQueriesContext(connection) as ctx:
|
||||||
|
response = admin_client.get("/api/documents/?fields=id,content")
|
||||||
|
|
||||||
|
assert response.status_code == status.HTTP_200_OK
|
||||||
|
assert response.data["results"] == [
|
||||||
|
{"id": root.id, "content": "new-version-content"},
|
||||||
|
]
|
||||||
|
assert not any(
|
||||||
|
"effective_content" in query["sql"] for query in ctx.captured_queries
|
||||||
|
)
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"fields_param",
|
||||||
|
[
|
||||||
|
pytest.param("", id="blank-fields"),
|
||||||
|
pytest.param("id,content", id="content-requested"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_content_resolves_without_a_query_per_document(
|
||||||
|
self,
|
||||||
|
admin_client: APIClient,
|
||||||
|
fields_param: str,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- One versioned root document, then two more
|
||||||
|
WHEN:
|
||||||
|
- Listing documents with a `fields` param that keeps content
|
||||||
|
THEN:
|
||||||
|
- Every root's content resolves to its latest version's
|
||||||
|
- The query count does not grow with the number of documents,
|
||||||
|
i.e. a blank `fields` does not skip the prefetch and fall back
|
||||||
|
to loading each root's deferred version content
|
||||||
|
"""
|
||||||
|
first = DocumentFactory(content="first-root-content")
|
||||||
|
DocumentFactory(
|
||||||
|
root_document=first,
|
||||||
|
version_index=1,
|
||||||
|
content="first-version-content",
|
||||||
|
)
|
||||||
|
|
||||||
|
with CaptureQueriesContext(connection) as one_document:
|
||||||
|
response = admin_client.get(f"/api/documents/?fields={fields_param}")
|
||||||
|
|
||||||
|
assert response.status_code == status.HTTP_200_OK
|
||||||
|
assert [r["content"] for r in response.data["results"]] == [
|
||||||
|
"first-version-content",
|
||||||
|
]
|
||||||
|
|
||||||
|
for index in range(2):
|
||||||
|
root = DocumentFactory(content=f"root-content-{index}")
|
||||||
|
DocumentFactory(
|
||||||
|
root_document=root,
|
||||||
|
version_index=1,
|
||||||
|
content=f"version-content-{index}",
|
||||||
|
)
|
||||||
|
with CaptureQueriesContext(connection) as three_documents:
|
||||||
|
response = admin_client.get(f"/api/documents/?fields={fields_param}")
|
||||||
|
|
||||||
|
assert response.status_code == status.HTTP_200_OK
|
||||||
|
assert sorted(r["content"] for r in response.data["results"]) == [
|
||||||
|
"first-version-content",
|
||||||
|
"version-content-0",
|
||||||
|
"version-content-1",
|
||||||
|
]
|
||||||
|
assert len(_get_document_queries(three_documents)) == len(
|
||||||
|
_get_document_queries(one_document),
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_list_without_content_field_skips_prefetch_and_omits_content(
|
||||||
|
self,
|
||||||
|
admin_client: APIClient,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A versioned root document
|
||||||
|
WHEN:
|
||||||
|
- Listing documents without asking for content
|
||||||
|
THEN:
|
||||||
|
- Content is neither serialized nor resolved
|
||||||
|
- Nothing pays for the prefetch or the per-instance fallback
|
||||||
|
"""
|
||||||
|
root = DocumentFactory(content="root-content")
|
||||||
|
DocumentFactory(
|
||||||
|
root_document=root,
|
||||||
|
version_index=1,
|
||||||
|
content="version-content",
|
||||||
|
)
|
||||||
|
|
||||||
|
with CaptureQueriesContext(connection) as ctx:
|
||||||
|
response = admin_client.get("/api/documents/?fields=id")
|
||||||
|
|
||||||
|
assert response.status_code == status.HTTP_200_OK
|
||||||
|
assert response.data["results"] == [{"id": root.id}]
|
||||||
|
assert _get_effective_content_fallback_queries(ctx) == []
|
||||||
|
# Only the list query itself reads a content column: no extra query
|
||||||
|
# for the skipped prefetch, none for a per-instance fallback
|
||||||
|
content_queries = [
|
||||||
|
query
|
||||||
|
for query in ctx.captured_queries
|
||||||
|
if '"documents_document"."content"' in query["sql"]
|
||||||
|
]
|
||||||
|
assert len(content_queries) == 1
|
||||||
|
|
||||||
|
def test_latest_version_content_prefetch_carries_only_the_newest_version(
|
||||||
|
self,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with two versions
|
||||||
|
WHEN:
|
||||||
|
- Fetching the root through latest_version_content_prefetch()
|
||||||
|
THEN:
|
||||||
|
- The prefetch carries only the single newest version, not every
|
||||||
|
historical version's content (the whole point of not reusing
|
||||||
|
the metadata-only "versions" prefetch for this)
|
||||||
|
"""
|
||||||
|
root = DocumentFactory(content="root-content")
|
||||||
|
DocumentFactory(
|
||||||
|
root_document=root,
|
||||||
|
version_index=1,
|
||||||
|
content="older-version-content",
|
||||||
|
)
|
||||||
|
DocumentFactory(
|
||||||
|
root_document=root,
|
||||||
|
version_index=2,
|
||||||
|
content="newest-version-content",
|
||||||
|
)
|
||||||
|
|
||||||
|
fetched_root = (
|
||||||
|
Document.objects.filter(pk=root.pk)
|
||||||
|
.prefetch_related(
|
||||||
|
latest_version_content_prefetch(),
|
||||||
|
)
|
||||||
|
.get()
|
||||||
|
)
|
||||||
|
|
||||||
|
latest = getattr(fetched_root, LATEST_VERSION_CONTENT_PREFETCH_ATTR)
|
||||||
|
assert [v.content for v in latest] == ["newest-version-content"]
|
||||||
|
|
||||||
|
|
||||||
|
class TestHasPrefetchedEffectiveContent:
|
||||||
|
"""
|
||||||
|
DocumentSerializer.to_representation() only calls get_effective_content()
|
||||||
|
when has_prefetched_effective_content() says it's cheap -- otherwise a
|
||||||
|
caller that never set up an annotation or prefetch (TrashView,
|
||||||
|
GlobalSearchView, which build their own querysets and don't display
|
||||||
|
content at all) would pay for a per-instance query nobody asked for.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def test_false_with_no_annotation_or_prefetch(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A document the ORM never annotated or prefetched for
|
||||||
|
WHEN:
|
||||||
|
- Asking whether its effective content is already resolved
|
||||||
|
THEN:
|
||||||
|
- It is not, so the serializer must leave it alone
|
||||||
|
"""
|
||||||
|
document = DocumentFactory.build()
|
||||||
|
|
||||||
|
assert has_prefetched_effective_content(document) is False
|
||||||
|
|
||||||
|
def test_true_with_effective_content_annotation(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A document carrying the queryset's effective_content annotation
|
||||||
|
WHEN:
|
||||||
|
- Asking whether its effective content is already resolved
|
||||||
|
THEN:
|
||||||
|
- It is, straight off the annotation
|
||||||
|
"""
|
||||||
|
document = DocumentFactory.build()
|
||||||
|
document.effective_content = "resolved"
|
||||||
|
|
||||||
|
assert has_prefetched_effective_content(document) is True
|
||||||
|
|
||||||
|
def test_true_with_lean_prefetch_attr_even_when_empty(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A document the lean content prefetch ran for, finding no versions
|
||||||
|
WHEN:
|
||||||
|
- Asking whether its effective content is already resolved
|
||||||
|
THEN:
|
||||||
|
- It is: an empty prefetch is an answer, not a missing one
|
||||||
|
"""
|
||||||
|
document = DocumentFactory.build()
|
||||||
|
setattr(document, LATEST_VERSION_CONTENT_PREFETCH_ATTR, [])
|
||||||
|
|
||||||
|
assert has_prefetched_effective_content(document) is True
|
||||||
|
|
||||||
|
def test_true_with_metadata_versions_prefetch_cache(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A document carrying only the metadata "versions" prefetch
|
||||||
|
WHEN:
|
||||||
|
- Asking whether its effective content is already resolved
|
||||||
|
THEN:
|
||||||
|
- It is, via get_effective_content()'s prefetch-cache branch
|
||||||
|
"""
|
||||||
|
document = DocumentFactory.build()
|
||||||
|
document._prefetched_objects_cache = {"versions": []}
|
||||||
|
|
||||||
|
assert has_prefetched_effective_content(document) is True
|
||||||
|
|
||||||
|
|
||||||
|
def _get_document_queries(
|
||||||
|
ctx: CaptureQueriesContext,
|
||||||
|
) -> list[dict[str, str]]:
|
||||||
|
"""
|
||||||
|
The queries a list request spends on the documents themselves, i.e.
|
||||||
|
everything but the one-time django_content_type lookup guardian's
|
||||||
|
permission filtering makes. That lookup is process-cached, and the
|
||||||
|
autouse fixture in conftest clears the cache before every test, so it
|
||||||
|
lands in whichever request happens to run first and never repeats --
|
||||||
|
counting it makes a request look like it costs one query more than the
|
||||||
|
identical request after it.
|
||||||
|
"""
|
||||||
|
return [q for q in ctx.captured_queries if '"django_content_type"' not in q["sql"]]
|
||||||
|
|
||||||
|
|
||||||
|
def _get_effective_content_fallback_queries(
|
||||||
|
ctx: CaptureQueriesContext,
|
||||||
|
) -> list[dict[str, str]]:
|
||||||
|
"""
|
||||||
|
Document.get_effective_content()'s per-instance fallback (no annotation,
|
||||||
|
no prefetch) is a `.values_list("content", flat=True).first()` query --
|
||||||
|
a SELECT of just the content column. Distinct from get_versions()'s own,
|
||||||
|
unrelated per-instance metadata query (id/checksum/added/etc, no
|
||||||
|
content) run to build the "versions" response field, which isn't part
|
||||||
|
of what this test file covers.
|
||||||
|
"""
|
||||||
|
return [
|
||||||
|
q
|
||||||
|
for q in ctx.captured_queries
|
||||||
|
if q["sql"].startswith('SELECT "documents_document"."content" FROM')
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.django_db
|
||||||
|
class TestTrashAndGlobalSearchEffectiveContentIsNeverPerInstance:
|
||||||
|
"""
|
||||||
|
TrashView and GlobalSearchView serialize Document instances with
|
||||||
|
DocumentSerializer too, but build their querysets independently of
|
||||||
|
DocumentViewSet.get_queryset(). TrashView doesn't display content at all,
|
||||||
|
so it keeps the document's own unresolved content; GlobalSearchView
|
||||||
|
annotates effective_content itself, so it shows the latest version's.
|
||||||
|
Neither should ever fall back to a per-instance query.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def test_trash_list_shows_unresolved_content_with_no_extra_query(
|
||||||
|
self,
|
||||||
|
admin_client: APIClient,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A trashed root document whose own content differs from what a
|
||||||
|
version would have had (also trashed, deletion cascades)
|
||||||
|
WHEN:
|
||||||
|
- Listing trash
|
||||||
|
THEN:
|
||||||
|
- The response shows the document's own content
|
||||||
|
- Nothing ever queries for versions to resolve it
|
||||||
|
"""
|
||||||
|
root = DocumentFactory(content="own-content")
|
||||||
|
DocumentFactory(
|
||||||
|
root_document=root,
|
||||||
|
version_index=1,
|
||||||
|
content="version-content",
|
||||||
|
)
|
||||||
|
root.delete()
|
||||||
|
|
||||||
|
with CaptureQueriesContext(connection) as ctx:
|
||||||
|
response = admin_client.get("/api/trash/")
|
||||||
|
|
||||||
|
assert response.status_code == status.HTTP_200_OK
|
||||||
|
[result] = [r for r in response.data["results"] if r["id"] == root.id]
|
||||||
|
assert result["content"] == "own-content"
|
||||||
|
assert _get_effective_content_fallback_queries(ctx) == []
|
||||||
|
|
||||||
|
def test_global_search_db_only_shows_latest_version_content_with_no_extra_query(
|
||||||
|
self,
|
||||||
|
admin_client: APIClient,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document, findable by title, whose own content differs
|
||||||
|
from its latest version's
|
||||||
|
WHEN:
|
||||||
|
- Using the global search endpoint's db_only mode
|
||||||
|
THEN:
|
||||||
|
- The response shows the latest version's content, resolved by
|
||||||
|
GlobalSearchView's own effective_content annotation
|
||||||
|
- There is no per-instance fallback query
|
||||||
|
"""
|
||||||
|
root = DocumentFactory(title="findme", content="own-content")
|
||||||
|
DocumentFactory(
|
||||||
|
root_document=root,
|
||||||
|
version_index=1,
|
||||||
|
content="version-content",
|
||||||
|
)
|
||||||
|
|
||||||
|
with CaptureQueriesContext(connection) as ctx:
|
||||||
|
response = admin_client.get(
|
||||||
|
"/api/search/?query=findme&db_only=true",
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == status.HTTP_200_OK
|
||||||
|
[result] = [d for d in response.data["documents"] if d["id"] == root.id]
|
||||||
|
assert result["content"] == "version-content"
|
||||||
|
assert _get_effective_content_fallback_queries(ctx) == []
|
||||||
@@ -110,7 +110,7 @@ class TestDocument(TestCase):
|
|||||||
checksum="checksum",
|
checksum="checksum",
|
||||||
mime_type="application/pdf",
|
mime_type="application/pdf",
|
||||||
)
|
)
|
||||||
Document.objects.create(
|
version = Document.objects.create(
|
||||||
root_document=root,
|
root_document=root,
|
||||||
correspondent=root.correspondent,
|
correspondent=root.correspondent,
|
||||||
title="Version",
|
title="Version",
|
||||||
@@ -124,6 +124,10 @@ class TestDocument(TestCase):
|
|||||||
self.assertEqual(Document.objects.count(), 0)
|
self.assertEqual(Document.objects.count(), 0)
|
||||||
self.assertEqual(Document.deleted_objects.count(), 2)
|
self.assertEqual(Document.deleted_objects.count(), 2)
|
||||||
|
|
||||||
|
root.restore(strict=False)
|
||||||
|
|
||||||
|
self.assertTrue(Document.objects.filter(pk=version.pk).exists())
|
||||||
|
|
||||||
def test_file_name(self) -> None:
|
def test_file_name(self) -> None:
|
||||||
doc = Document(
|
doc = Document(
|
||||||
mime_type="application/pdf",
|
mime_type="application/pdf",
|
||||||
|
|||||||
@@ -1,48 +0,0 @@
|
|||||||
import unicodedata
|
|
||||||
from datetime import date
|
|
||||||
|
|
||||||
import pytest
|
|
||||||
|
|
||||||
from documents.models import Correspondent
|
|
||||||
from documents.models import Document
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
|
||||||
class TestGetPublicFilenameNfc:
|
|
||||||
def test_normalizes_nfd_title_to_nfc(self) -> None:
|
|
||||||
nfd_title = unicodedata.normalize("NFD", "Gehaltserhöhung")
|
|
||||||
assert not unicodedata.is_normalized("NFC", nfd_title)
|
|
||||||
|
|
||||||
doc = Document(
|
|
||||||
mime_type="application/pdf",
|
|
||||||
title=nfd_title,
|
|
||||||
created=date(2025, 10, 17),
|
|
||||||
)
|
|
||||||
|
|
||||||
result = doc.get_public_filename()
|
|
||||||
|
|
||||||
assert unicodedata.is_normalized("NFC", result)
|
|
||||||
assert (
|
|
||||||
result
|
|
||||||
== "2025-10-17 "
|
|
||||||
+ unicodedata.normalize(
|
|
||||||
"NFC",
|
|
||||||
nfd_title,
|
|
||||||
)
|
|
||||||
+ ".pdf"
|
|
||||||
)
|
|
||||||
|
|
||||||
def test_normalizes_nfd_correspondent_name_to_nfc(self) -> None:
|
|
||||||
nfd_name = unicodedata.normalize("NFD", "Müller GmbH")
|
|
||||||
correspondent = Correspondent.objects.create(name=nfd_name)
|
|
||||||
|
|
||||||
doc = Document.objects.create(
|
|
||||||
mime_type="application/pdf",
|
|
||||||
title="Rechnung",
|
|
||||||
created=date(2025, 10, 17),
|
|
||||||
correspondent=correspondent,
|
|
||||||
)
|
|
||||||
|
|
||||||
result = doc.get_public_filename()
|
|
||||||
|
|
||||||
assert unicodedata.is_normalized("NFC", result)
|
|
||||||
@@ -136,6 +136,23 @@ def wait_for_mock_call(
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def sleep_past_stability(
|
||||||
|
owner: FileStabilityTracker | ConsumerThread,
|
||||||
|
*,
|
||||||
|
windows: float = 1.5,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
Block until a tracked file's stability window has certainly elapsed.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
owner: The tracker, or the consumer thread running one, whose
|
||||||
|
configured stability delay sets the wait.
|
||||||
|
windows: How many stability windows to wait, giving slop for a slow
|
||||||
|
or loaded test runner.
|
||||||
|
"""
|
||||||
|
sleep(owner.stability_delay * windows)
|
||||||
|
|
||||||
|
|
||||||
class TestTrackedFile:
|
class TestTrackedFile:
|
||||||
"""Tests for the TrackedFile dataclass."""
|
"""Tests for the TrackedFile dataclass."""
|
||||||
|
|
||||||
@@ -261,6 +278,56 @@ class TestFileStabilityTracker:
|
|||||||
assert len(stable) == 0
|
assert len(stable) == 0
|
||||||
assert stability_tracker.pending_count == 1
|
assert stability_tracker.pending_count == 1
|
||||||
|
|
||||||
|
def test_get_stable_files_skips_empty_file(
|
||||||
|
self,
|
||||||
|
stability_tracker: FileStabilityTracker,
|
||||||
|
tmp_path: Path,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A zero byte file, tracked and past its stability delay
|
||||||
|
WHEN:
|
||||||
|
- Stable files are collected
|
||||||
|
THEN:
|
||||||
|
- The file is not yielded for consumption
|
||||||
|
- The file is dropped from tracking rather than held, so an
|
||||||
|
abandoned placeholder does not keep the watch loop awake
|
||||||
|
"""
|
||||||
|
empty = tmp_path / "scan.pdf"
|
||||||
|
empty.write_bytes(b"")
|
||||||
|
stability_tracker.track(empty, Change.added)
|
||||||
|
sleep_past_stability(stability_tracker)
|
||||||
|
|
||||||
|
stable = list(stability_tracker.get_stable_files())
|
||||||
|
|
||||||
|
assert stable == []
|
||||||
|
assert stability_tracker.pending_count == 0
|
||||||
|
|
||||||
|
def test_empty_file_is_yielded_once_content_arrives(
|
||||||
|
self,
|
||||||
|
stability_tracker: FileStabilityTracker,
|
||||||
|
tmp_path: Path,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A zero byte file which was dropped from tracking while empty
|
||||||
|
WHEN:
|
||||||
|
- The writer fills the file and a new event re-tracks it
|
||||||
|
THEN:
|
||||||
|
- The file is yielded for consumption once it is stable
|
||||||
|
"""
|
||||||
|
target = tmp_path / "scan.pdf"
|
||||||
|
target.write_bytes(b"")
|
||||||
|
stability_tracker.track(target, Change.added)
|
||||||
|
sleep_past_stability(stability_tracker)
|
||||||
|
assert list(stability_tracker.get_stable_files()) == []
|
||||||
|
|
||||||
|
target.write_bytes(b"%PDF-1.4 content")
|
||||||
|
stability_tracker.track(target, Change.modified)
|
||||||
|
sleep_past_stability(stability_tracker)
|
||||||
|
|
||||||
|
assert list(stability_tracker.get_stable_files()) == [target]
|
||||||
|
|
||||||
def test_get_stable_files_deleted_during_check(self, temp_file: Path) -> None:
|
def test_get_stable_files_deleted_during_check(self, temp_file: Path) -> None:
|
||||||
"""Test deleted file is not returned during stability check."""
|
"""Test deleted file is not returned during stability check."""
|
||||||
tracker = FileStabilityTracker(stability_delay=0.1)
|
tracker = FileStabilityTracker(stability_delay=0.1)
|
||||||
@@ -879,6 +946,51 @@ class TestCommandWatch:
|
|||||||
|
|
||||||
mock_consume_file_delay.apply_async.assert_called()
|
mock_consume_file_delay.apply_async.assert_called()
|
||||||
|
|
||||||
|
def test_scanner_placeholder_is_not_consumed_while_empty(
|
||||||
|
self,
|
||||||
|
consumption_dir: Path,
|
||||||
|
sample_pdf: Path,
|
||||||
|
mock_consume_file_delay: MagicMock,
|
||||||
|
start_consumer: Callable[..., ConsumerThread],
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A scanner which creates a zero byte placeholder and only writes
|
||||||
|
the page some time later (GH discussion #13969)
|
||||||
|
WHEN:
|
||||||
|
- The placeholder sits untouched well past the stability delay
|
||||||
|
- The scanner then writes the real content
|
||||||
|
THEN:
|
||||||
|
- The empty placeholder is never queued, as it could only fail
|
||||||
|
with "Unsupported mime type inode/x-empty"
|
||||||
|
- The file is queued exactly once, when the content lands
|
||||||
|
"""
|
||||||
|
thread = start_consumer(stability_delay=0.2)
|
||||||
|
|
||||||
|
target = consumption_dir / "scan.pdf"
|
||||||
|
target.write_bytes(b"") # the scanner's placeholder
|
||||||
|
|
||||||
|
# Well past the stability delay: the old behaviour queued it here.
|
||||||
|
sleep_past_stability(thread, windows=5)
|
||||||
|
if thread.exception:
|
||||||
|
raise thread.exception
|
||||||
|
assert mock_consume_file_delay.apply_async.call_count == 0
|
||||||
|
|
||||||
|
shutil.copy(sample_pdf, target) # the scanner finishes the page
|
||||||
|
|
||||||
|
assert wait_for_mock_call(
|
||||||
|
mock_consume_file_delay.apply_async,
|
||||||
|
timeout_s=5.0,
|
||||||
|
)
|
||||||
|
if thread.exception:
|
||||||
|
raise thread.exception
|
||||||
|
|
||||||
|
assert mock_consume_file_delay.apply_async.call_count == 1
|
||||||
|
queued_doc = mock_consume_file_delay.apply_async.call_args.kwargs["kwargs"][
|
||||||
|
"input_doc"
|
||||||
|
]
|
||||||
|
assert queued_doc.original_file.name == "scan.pdf"
|
||||||
|
|
||||||
def test_ignores_macos_files(
|
def test_ignores_macos_files(
|
||||||
self,
|
self,
|
||||||
consumption_dir: Path,
|
consumption_dir: Path,
|
||||||
|
|||||||
@@ -1,80 +0,0 @@
|
|||||||
import unicodedata
|
|
||||||
|
|
||||||
import pytest
|
|
||||||
|
|
||||||
from documents.data_models import ConsumableDocument
|
|
||||||
from documents.data_models import DocumentSource
|
|
||||||
from documents.matching import consumable_document_matches_workflow
|
|
||||||
from documents.matching import existing_document_matches_workflow
|
|
||||||
from documents.models import Document
|
|
||||||
from documents.models import Workflow
|
|
||||||
from documents.models import WorkflowTrigger
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
|
||||||
class TestMatchingNfcNormalization:
|
|
||||||
def test_consumable_document_filename_nfd_matches_nfc_pattern(
|
|
||||||
self,
|
|
||||||
tmp_path,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A file on disk whose name is NFD-normalized
|
|
||||||
- A workflow trigger filename filter typed as NFC
|
|
||||||
WHEN:
|
|
||||||
- The consumable document is checked against the trigger
|
|
||||||
THEN:
|
|
||||||
- It matches, because both sides are normalized before comparing
|
|
||||||
"""
|
|
||||||
nfd_name = unicodedata.normalize("NFD", "Gehaltserhöhung.pdf")
|
|
||||||
nfc_pattern = unicodedata.normalize("NFC", "*Gehaltserhöhung*")
|
|
||||||
assert nfd_name != unicodedata.normalize("NFC", nfd_name)
|
|
||||||
|
|
||||||
file_path = tmp_path / nfd_name
|
|
||||||
file_path.write_bytes(b"%PDF-1.4 test")
|
|
||||||
|
|
||||||
document = ConsumableDocument(
|
|
||||||
source=DocumentSource.ConsumeFolder,
|
|
||||||
original_file=file_path,
|
|
||||||
)
|
|
||||||
trigger = WorkflowTrigger(
|
|
||||||
type=WorkflowTrigger.WorkflowTriggerType.CONSUMPTION,
|
|
||||||
filter_filename=nfc_pattern,
|
|
||||||
sources=[],
|
|
||||||
)
|
|
||||||
|
|
||||||
matched, reason = consumable_document_matches_workflow(document, trigger)
|
|
||||||
|
|
||||||
assert matched, reason
|
|
||||||
|
|
||||||
def test_existing_document_filename_nfd_matches_nfc_pattern(self) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A Document whose original_filename is NFD-normalized (e.g. from
|
|
||||||
before normalization was applied at consumption time)
|
|
||||||
- A workflow trigger filename filter typed as NFC
|
|
||||||
WHEN:
|
|
||||||
- The document is checked against the trigger
|
|
||||||
THEN:
|
|
||||||
- It matches, because both sides are normalized before comparing
|
|
||||||
"""
|
|
||||||
nfd_name = unicodedata.normalize("NFD", "Gehaltserhöhung.pdf")
|
|
||||||
nfc_pattern = unicodedata.normalize("NFC", "*Gehaltserhöhung*")
|
|
||||||
|
|
||||||
document = Document.objects.create(
|
|
||||||
title="Test",
|
|
||||||
content="content",
|
|
||||||
checksum="checksum",
|
|
||||||
mime_type="application/pdf",
|
|
||||||
original_filename=nfd_name,
|
|
||||||
)
|
|
||||||
workflow = Workflow.objects.create(name="Test workflow", order=0)
|
|
||||||
trigger = WorkflowTrigger.objects.create(
|
|
||||||
type=WorkflowTrigger.WorkflowTriggerType.DOCUMENT_ADDED,
|
|
||||||
filter_filename=nfc_pattern,
|
|
||||||
)
|
|
||||||
workflow.triggers.add(trigger)
|
|
||||||
|
|
||||||
matched, reason = existing_document_matches_workflow(document, trigger)
|
|
||||||
|
|
||||||
assert matched, reason
|
|
||||||
@@ -1,6 +0,0 @@
|
|||||||
from documents.utils import normalize_unicode
|
|
||||||
|
|
||||||
|
|
||||||
class TestNormalizeUnicode:
|
|
||||||
def test_none_passes_through(self) -> None:
|
|
||||||
assert normalize_unicode(None) is None
|
|
||||||
@@ -2,6 +2,7 @@ from unittest import mock
|
|||||||
|
|
||||||
from django.contrib.auth.models import Permission
|
from django.contrib.auth.models import Permission
|
||||||
from django.contrib.auth.models import User
|
from django.contrib.auth.models import User
|
||||||
|
from rest_framework import status
|
||||||
from rest_framework.test import APITestCase
|
from rest_framework.test import APITestCase
|
||||||
|
|
||||||
from documents import bulk_edit
|
from documents import bulk_edit
|
||||||
@@ -108,6 +109,44 @@ class TestTagHierarchy(DirectoriesMixin, APITestCase):
|
|||||||
self.document.refresh_from_db()
|
self.document.refresh_from_db()
|
||||||
assert self.document.tags.count() == 0
|
assert self.document.tags.count() == 0
|
||||||
|
|
||||||
|
def test_remove_inbox_tags_removes_nested_children(self) -> None:
|
||||||
|
inbox = Tag.objects.create(name="Inbox", is_inbox_tag=True)
|
||||||
|
nested = Tag.objects.create(name="Nested", tn_parent=inbox)
|
||||||
|
self.document.add_nested_tags([nested])
|
||||||
|
|
||||||
|
resp = self.client.patch(
|
||||||
|
f"/api/documents/{self.document.pk}/",
|
||||||
|
{"title": "new title", "remove_inbox_tags": True},
|
||||||
|
format="json",
|
||||||
|
)
|
||||||
|
assert resp.status_code == status.HTTP_200_OK
|
||||||
|
self.document.refresh_from_db()
|
||||||
|
assert self.document.tags.count() == 0
|
||||||
|
|
||||||
|
# A subsequent save must not re-add the inbox tag as an ancestor
|
||||||
|
resp = self.client.patch(
|
||||||
|
f"/api/documents/{self.document.pk}/",
|
||||||
|
{"title": "another title", "tags": [], "remove_inbox_tags": True},
|
||||||
|
format="json",
|
||||||
|
)
|
||||||
|
assert resp.status_code == status.HTTP_200_OK
|
||||||
|
self.document.refresh_from_db()
|
||||||
|
assert self.document.tags.count() == 0
|
||||||
|
|
||||||
|
def test_remove_inbox_tags_keeps_inbox_when_nested_child_added(self) -> None:
|
||||||
|
inbox = Tag.objects.create(name="Inbox", is_inbox_tag=True)
|
||||||
|
nested = Tag.objects.create(name="Nested", tn_parent=inbox)
|
||||||
|
self.document.add_nested_tags([inbox])
|
||||||
|
|
||||||
|
self.client.patch(
|
||||||
|
f"/api/documents/{self.document.pk}/",
|
||||||
|
{"tags": [nested.pk], "remove_inbox_tags": True},
|
||||||
|
format="json",
|
||||||
|
)
|
||||||
|
self.document.refresh_from_db()
|
||||||
|
tags = set(self.document.tags.values_list("pk", flat=True))
|
||||||
|
assert tags == {inbox.pk, nested.pk}
|
||||||
|
|
||||||
def test_bulk_edit_respects_hierarchy(self) -> None:
|
def test_bulk_edit_respects_hierarchy(self) -> None:
|
||||||
bulk_edit.add_tag([self.document.pk], self.child.pk)
|
bulk_edit.add_tag([self.document.pk], self.child.pk)
|
||||||
self.document.refresh_from_db()
|
self.document.refresh_from_db()
|
||||||
|
|||||||
@@ -32,6 +32,7 @@ from documents.signals.handlers import update_llm_suggestions_cache
|
|||||||
from documents.tests.utils import DirectoriesMixin
|
from documents.tests.utils import DirectoriesMixin
|
||||||
from documents.tests.utils import read_streaming_response
|
from documents.tests.utils import read_streaming_response
|
||||||
from paperless.models import ApplicationConfiguration
|
from paperless.models import ApplicationConfiguration
|
||||||
|
from paperless_ai.exceptions import LLMProviderError
|
||||||
from paperless_ai.exceptions import LLMTimeoutError
|
from paperless_ai.exceptions import LLMTimeoutError
|
||||||
|
|
||||||
|
|
||||||
@@ -737,6 +738,38 @@ class TestAISuggestions(DirectoriesMixin, TestCase):
|
|||||||
get_llm_suggestion_cache(self.document.pk, backend="openai-like"),
|
get_llm_suggestion_cache(self.document.pk, backend="openai-like"),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
@patch("documents.views.get_ai_document_classification")
|
||||||
|
@override_settings(
|
||||||
|
AI_ENABLED=True,
|
||||||
|
LLM_BACKEND="openai-like",
|
||||||
|
)
|
||||||
|
def test_ai_suggestions_with_llm_provider_error(
|
||||||
|
self,
|
||||||
|
mock_get_ai_classification,
|
||||||
|
) -> None:
|
||||||
|
mock_get_ai_classification.side_effect = LLMProviderError(
|
||||||
|
"confidential provider response",
|
||||||
|
)
|
||||||
|
|
||||||
|
self.client.force_login(user=self.user)
|
||||||
|
response = self.client.get(
|
||||||
|
f"/api/documents/{self.document.pk}/ai_suggestions/",
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_502_BAD_GATEWAY)
|
||||||
|
self.assertEqual(
|
||||||
|
response.json(),
|
||||||
|
{
|
||||||
|
"ai": [
|
||||||
|
"AI backend rejected the request. Check logs for details.",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
)
|
||||||
|
self.assertNotIn("confidential provider response", response.content.decode())
|
||||||
|
self.assertIsNone(
|
||||||
|
get_llm_suggestion_cache(self.document.pk, backend="openai-like"),
|
||||||
|
)
|
||||||
|
|
||||||
@patch("documents.views.get_ai_document_classification")
|
@patch("documents.views.get_ai_document_classification")
|
||||||
@override_settings(
|
@override_settings(
|
||||||
AI_ENABLED=True,
|
AI_ENABLED=True,
|
||||||
|
|||||||
@@ -1,7 +1,6 @@
|
|||||||
import hashlib
|
import hashlib
|
||||||
import logging
|
import logging
|
||||||
import shutil
|
import shutil
|
||||||
import unicodedata
|
|
||||||
from collections.abc import Callable
|
from collections.abc import Callable
|
||||||
from collections.abc import Iterable
|
from collections.abc import Iterable
|
||||||
from collections.abc import Iterator
|
from collections.abc import Iterator
|
||||||
@@ -32,25 +31,6 @@ def identity(iterable: Iterable[_T]) -> Iterable[_T]:
|
|||||||
return iterable
|
return iterable
|
||||||
|
|
||||||
|
|
||||||
def normalize_unicode(value: str | None) -> str | None:
|
|
||||||
"""
|
|
||||||
Normalize a string to Unicode NFC form, or return None unchanged.
|
|
||||||
|
|
||||||
This is the single normalization pass for any user- or filesystem-supplied
|
|
||||||
text that ends up in a filename, path, or is compared/matched against one
|
|
||||||
(titles, correspondent/tag/type names, uploaded filenames, workflow and
|
|
||||||
mail rule filename/path filters). Composed (NFC) and decomposed (NFD)
|
|
||||||
forms of the same visible text are different byte sequences, which breaks
|
|
||||||
exact comparisons and filesystem lookups even though the text looks
|
|
||||||
identical. Always normalize through this function rather than calling
|
|
||||||
unicodedata.normalize() directly, so every call site agrees on the same
|
|
||||||
form.
|
|
||||||
"""
|
|
||||||
if value is None:
|
|
||||||
return None
|
|
||||||
return unicodedata.normalize("NFC", value)
|
|
||||||
|
|
||||||
|
|
||||||
class QuerySetStream(Generic[_M]):
|
class QuerySetStream(Generic[_M]):
|
||||||
"""Stream a QuerySet via .iterator(chunk_size=...) instead of
|
"""Stream a QuerySet via .iterator(chunk_size=...) instead of
|
||||||
materializing it (plus any prefetch caches) all at once, while still
|
materializing it (plus any prefetch caches) all at once, while still
|
||||||
|
|||||||
@@ -7,9 +7,12 @@ from typing import Any
|
|||||||
|
|
||||||
from django.db.models import F
|
from django.db.models import F
|
||||||
from django.db.models import OuterRef
|
from django.db.models import OuterRef
|
||||||
|
from django.db.models import Prefetch
|
||||||
from django.db.models import QuerySet
|
from django.db.models import QuerySet
|
||||||
from django.db.models import Subquery
|
from django.db.models import Subquery
|
||||||
|
from django.db.models import Window
|
||||||
from django.db.models.functions import Coalesce
|
from django.db.models.functions import Coalesce
|
||||||
|
from django.db.models.functions import RowNumber
|
||||||
|
|
||||||
from documents.models import Document
|
from documents.models import Document
|
||||||
|
|
||||||
@@ -46,6 +49,68 @@ def annotate_effective_content(documents: QuerySet[Document]) -> QuerySet[Docume
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
LATEST_VERSION_CONTENT_PREFETCH_ATTR = "_latest_version_content_prefetch"
|
||||||
|
|
||||||
|
|
||||||
|
def latest_version_content_prefetch() -> Prefetch:
|
||||||
|
"""
|
||||||
|
A Prefetch for Document.versions scoped to just the newest version's
|
||||||
|
content, for get_effective_content()'s fallback when no SQL annotation
|
||||||
|
is present.
|
||||||
|
|
||||||
|
Deliberately not merged into a metadata-only "versions" prefetch (the one
|
||||||
|
used for the serialized versions list): that one fetches every historical
|
||||||
|
version of every document, and pulling full OCR content for versions
|
||||||
|
nobody will read wastes DB transfer/memory at scale. This one is windowed
|
||||||
|
down to a single row per root, then bounded by Prefetch's own IN-list to
|
||||||
|
whatever page/result set it's attached to -- one cheap bulk query total,
|
||||||
|
not one per document and not one per version.
|
||||||
|
"""
|
||||||
|
return Prefetch(
|
||||||
|
"versions",
|
||||||
|
queryset=(
|
||||||
|
Document.objects.filter(
|
||||||
|
root_document_id__isnull=False,
|
||||||
|
deleted_at__isnull=True,
|
||||||
|
)
|
||||||
|
.annotate(
|
||||||
|
rn=Window(
|
||||||
|
RowNumber(),
|
||||||
|
partition_by=F("root_document_id"),
|
||||||
|
order_by=[
|
||||||
|
F("version_index").desc(nulls_last=True),
|
||||||
|
F("id").desc(),
|
||||||
|
],
|
||||||
|
),
|
||||||
|
)
|
||||||
|
.filter(rn=1)
|
||||||
|
.only("id", "root_document_id", "content")
|
||||||
|
),
|
||||||
|
to_attr=LATEST_VERSION_CONTENT_PREFETCH_ATTR,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def has_prefetched_effective_content(document: Document) -> bool:
|
||||||
|
"""
|
||||||
|
True if document.get_effective_content() can answer without an extra
|
||||||
|
per-instance query -- an SQL ``effective_content`` annotation, the lean
|
||||||
|
latest_version_content_prefetch(), or the metadata-only "versions"
|
||||||
|
prefetch is already present on the instance.
|
||||||
|
|
||||||
|
Callers that haven't set any of those up (e.g. views that build their
|
||||||
|
own querysets independently of DocumentViewSet.get_queryset(), like
|
||||||
|
TrashView or GlobalSearchView) intentionally don't pay for version-aware
|
||||||
|
content resolution -- see DocumentSerializer.to_representation(), which
|
||||||
|
uses this to decide whether to call get_effective_content() at all.
|
||||||
|
"""
|
||||||
|
if hasattr(document, "effective_content"):
|
||||||
|
return True
|
||||||
|
if getattr(document, LATEST_VERSION_CONTENT_PREFETCH_ATTR, None) is not None:
|
||||||
|
return True
|
||||||
|
prefetched_cache = getattr(document, "_prefetched_objects_cache", None)
|
||||||
|
return isinstance(prefetched_cache, dict) and "versions" in prefetched_cache
|
||||||
|
|
||||||
|
|
||||||
def sort_versions_newest_first(documents: list[Document]) -> list[Document]:
|
def sort_versions_newest_first(documents: list[Document]) -> list[Document]:
|
||||||
"""
|
"""
|
||||||
Same sorting as versions_newest_first()
|
Same sorting as versions_newest_first()
|
||||||
|
|||||||
+100
-22
@@ -36,7 +36,6 @@ from django.db.migrations.recorder import MigrationRecorder
|
|||||||
from django.db.models import Avg
|
from django.db.models import Avg
|
||||||
from django.db.models import Case
|
from django.db.models import Case
|
||||||
from django.db.models import Count
|
from django.db.models import Count
|
||||||
from django.db.models import F
|
|
||||||
from django.db.models import IntegerField
|
from django.db.models import IntegerField
|
||||||
from django.db.models import Max
|
from django.db.models import Max
|
||||||
from django.db.models import Model
|
from django.db.models import Model
|
||||||
@@ -137,12 +136,14 @@ from documents.filters import CustomFieldFilterSet
|
|||||||
from documents.filters import DocumentFilterSet
|
from documents.filters import DocumentFilterSet
|
||||||
from documents.filters import DocumentsOrderingFilter
|
from documents.filters import DocumentsOrderingFilter
|
||||||
from documents.filters import DocumentTypeFilterSet
|
from documents.filters import DocumentTypeFilterSet
|
||||||
|
from documents.filters import EffectiveContentFilter
|
||||||
from documents.filters import PaperlessTaskFilterSet
|
from documents.filters import PaperlessTaskFilterSet
|
||||||
from documents.filters import PermittedObjectsFilter
|
from documents.filters import PermittedObjectsFilter
|
||||||
from documents.filters import ShareLinkBundleFilterSet
|
from documents.filters import ShareLinkBundleFilterSet
|
||||||
from documents.filters import ShareLinkFilterSet
|
from documents.filters import ShareLinkFilterSet
|
||||||
from documents.filters import StoragePathFilterSet
|
from documents.filters import StoragePathFilterSet
|
||||||
from documents.filters import TagFilterSet
|
from documents.filters import TagFilterSet
|
||||||
|
from documents.filters import TitleContentFilter
|
||||||
from documents.mail import EmailAttachment
|
from documents.mail import EmailAttachment
|
||||||
from documents.mail import send_email
|
from documents.mail import send_email
|
||||||
from documents.matching import match_correspondents
|
from documents.matching import match_correspondents
|
||||||
@@ -231,12 +232,12 @@ from documents.tasks import sanity_check
|
|||||||
from documents.tasks import train_classifier
|
from documents.tasks import train_classifier
|
||||||
from documents.tasks import update_document_parent_tags
|
from documents.tasks import update_document_parent_tags
|
||||||
from documents.utils import get_boolean
|
from documents.utils import get_boolean
|
||||||
from documents.utils import normalize_unicode
|
|
||||||
from documents.versioning import VersionResolutionError
|
from documents.versioning import VersionResolutionError
|
||||||
from documents.versioning import annotate_effective_content
|
from documents.versioning import annotate_effective_content
|
||||||
from documents.versioning import get_latest_version_for_root
|
from documents.versioning import get_latest_version_for_root
|
||||||
from documents.versioning import get_request_version_param
|
from documents.versioning import get_request_version_param
|
||||||
from documents.versioning import get_root_document
|
from documents.versioning import get_root_document
|
||||||
|
from documents.versioning import latest_version_content_prefetch
|
||||||
from documents.versioning import resolve_requested_version_for_root
|
from documents.versioning import resolve_requested_version_for_root
|
||||||
from documents.versioning import versions_newest_first
|
from documents.versioning import versions_newest_first
|
||||||
from paperless import version
|
from paperless import version
|
||||||
@@ -253,6 +254,7 @@ from paperless.views import StandardPagination
|
|||||||
from paperless_ai.ai_classifier import get_ai_document_classification
|
from paperless_ai.ai_classifier import get_ai_document_classification
|
||||||
from paperless_ai.ai_classifier import get_llm_output_language
|
from paperless_ai.ai_classifier import get_llm_output_language
|
||||||
from paperless_ai.chat import stream_chat_with_documents
|
from paperless_ai.chat import stream_chat_with_documents
|
||||||
|
from paperless_ai.exceptions import LLMProviderError
|
||||||
from paperless_ai.exceptions import LLMTimeoutError
|
from paperless_ai.exceptions import LLMTimeoutError
|
||||||
from paperless_ai.matching import extract_unmatched_names
|
from paperless_ai.matching import extract_unmatched_names
|
||||||
from paperless_ai.matching import match_correspondents_by_name
|
from paperless_ai.matching import match_correspondents_by_name
|
||||||
@@ -1084,12 +1086,59 @@ class DocumentViewSet(
|
|||||||
],
|
],
|
||||||
}
|
}
|
||||||
|
|
||||||
def get_queryset(self):
|
@classmethod
|
||||||
latest_version_content = Subquery(
|
def _content_filter_params(cls) -> tuple[str, ...]:
|
||||||
versions_newest_first(
|
"""
|
||||||
Document.objects.filter(root_document=OuterRef("pk")),
|
Query params whose filtering needs effective_content evaluated in SQL
|
||||||
).values("content")[:1],
|
against every candidate row -- see
|
||||||
|
_needs_effective_content_annotation(). Derived rather than
|
||||||
|
hand-maintained so a new content-filtering param counts automatically.
|
||||||
|
"""
|
||||||
|
params = [
|
||||||
|
name
|
||||||
|
for name, f in DocumentFilterSet.declared_filters.items()
|
||||||
|
if isinstance(f, (TitleContentFilter, EffectiveContentFilter))
|
||||||
|
]
|
||||||
|
if "effective_content" in cls.search_fields:
|
||||||
|
params.append(SearchFilter().search_param)
|
||||||
|
return tuple(params)
|
||||||
|
|
||||||
|
def _needs_effective_content_annotation(self) -> bool:
|
||||||
|
# effective_content is a per-row correlated subquery resolving each
|
||||||
|
# document's latest version. Filtering *on* it forces the database to
|
||||||
|
# evaluate it for every candidate row before reaching the LIMIT, which
|
||||||
|
# the root_document_id self-join makes pathological on MariaDB
|
||||||
|
# specifically once real candidate counts get large; otherwise the
|
||||||
|
# "versions" prefetch + Document.get_effective_content() resolves only
|
||||||
|
# the page that survives pagination. Every param here is deprecated in
|
||||||
|
# favor of the Tantivy-backed search endpoint (see filters.py's
|
||||||
|
# TitleContentFilter/EffectiveContentFilter docs), so pay that cost
|
||||||
|
# only when one is actually used. Blank values don't count, matching
|
||||||
|
# how those filters themselves no-op on them -- an empty `?search=`
|
||||||
|
# applies no predicate.
|
||||||
|
params = self.request.query_params
|
||||||
|
return any(
|
||||||
|
params.get(param, "").strip() for param in self._content_filter_params()
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def _requested_fields(self) -> list[str] | None:
|
||||||
|
# The sparse-fieldset `fields` param, as DynamicFieldsModelSerializer
|
||||||
|
# wants it: None means "no restriction, serialize everything", which
|
||||||
|
# a blank value means too. get_queryset() and get_serializer() both
|
||||||
|
# branch on this, and they have to read it identically -- a queryset
|
||||||
|
# that skips the content prefetch for a response that still
|
||||||
|
# serializes content reintroduces get_effective_content()'s
|
||||||
|
# per-instance fallback.
|
||||||
|
fields_param = self.request.query_params.get("fields")
|
||||||
|
return fields_param.split(",") if fields_param else None
|
||||||
|
|
||||||
|
def _needs_effective_content_prefetch(self) -> bool:
|
||||||
|
# The prefetch spares get_effective_content() a per-instance fallback
|
||||||
|
# query, but only earns itself when content can reach the response.
|
||||||
|
fields = self._requested_fields()
|
||||||
|
return fields is None or "content" in fields
|
||||||
|
|
||||||
|
def get_queryset(self):
|
||||||
# A correlated subquery avoids the LEFT JOIN + Count() this used to
|
# A correlated subquery avoids the LEFT JOIN + Count() this used to
|
||||||
# be, which forced a GROUP BY aggregate over every matching document
|
# be, which forced a GROUP BY aggregate over every matching document
|
||||||
# before the query could even be sorted or limited.
|
# before the query could even be sorted or limited.
|
||||||
@@ -1109,13 +1158,7 @@ class DocumentViewSet(
|
|||||||
# ObjectFilter.filter(). A blanket .distinct() here forces the
|
# ObjectFilter.filter(). A blanket .distinct() here forces the
|
||||||
# database to fully sort and dedupe every visible document before
|
# database to fully sort and dedupe every visible document before
|
||||||
# it can apply LIMIT, which is disastrous at scale.
|
# it can apply LIMIT, which is disastrous at scale.
|
||||||
return (
|
prefetches = [
|
||||||
Document.objects.filter(root_document__isnull=True)
|
|
||||||
.order_by("-created", "-id")
|
|
||||||
.annotate(effective_content=Coalesce(latest_version_content, F("content")))
|
|
||||||
.annotate(num_notes=Coalesce(note_count, 0))
|
|
||||||
.select_related("correspondent", "storage_path", "document_type", "owner")
|
|
||||||
.prefetch_related(
|
|
||||||
Prefetch(
|
Prefetch(
|
||||||
"versions",
|
"versions",
|
||||||
queryset=Document.objects.only(
|
queryset=Document.objects.only(
|
||||||
@@ -1134,15 +1177,24 @@ class DocumentViewSet(
|
|||||||
),
|
),
|
||||||
# NotesSerializer nests the author, this avoids query per note
|
# NotesSerializer nests the author, this avoids query per note
|
||||||
Prefetch("notes", queryset=Note.objects.select_related("user")),
|
Prefetch("notes", queryset=Note.objects.select_related("user")),
|
||||||
|
]
|
||||||
|
if self._needs_effective_content_prefetch():
|
||||||
|
prefetches.append(latest_version_content_prefetch())
|
||||||
|
queryset = (
|
||||||
|
Document.objects.filter(root_document__isnull=True)
|
||||||
|
.order_by("-created", "-id")
|
||||||
|
.annotate(num_notes=Coalesce(note_count, 0))
|
||||||
|
.select_related("correspondent", "storage_path", "document_type", "owner")
|
||||||
|
.prefetch_related(*prefetches)
|
||||||
)
|
)
|
||||||
)
|
if self._needs_effective_content_annotation():
|
||||||
|
queryset = annotate_effective_content(queryset)
|
||||||
|
return queryset
|
||||||
|
|
||||||
def get_serializer(self, *args, **kwargs):
|
def get_serializer(self, *args, **kwargs):
|
||||||
fields_param = self.request.query_params.get("fields", None)
|
|
||||||
fields = fields_param.split(",") if fields_param else None
|
|
||||||
truncate_content = self.request.query_params.get("truncate_content", "False")
|
truncate_content = self.request.query_params.get("truncate_content", "False")
|
||||||
kwargs.setdefault("context", self.get_serializer_context())
|
kwargs.setdefault("context", self.get_serializer_context())
|
||||||
kwargs.setdefault("fields", fields)
|
kwargs.setdefault("fields", self._requested_fields())
|
||||||
kwargs.setdefault("truncate_content", truncate_content.lower() in ["true", "1"])
|
kwargs.setdefault("truncate_content", truncate_content.lower() in ["true", "1"])
|
||||||
try:
|
try:
|
||||||
full_perms = get_boolean(
|
full_perms = get_boolean(
|
||||||
@@ -1604,6 +1656,22 @@ class DocumentViewSet(
|
|||||||
{"ai": [_("AI backend request timed out.")]},
|
{"ai": [_("AI backend request timed out.")]},
|
||||||
status=status.HTTP_503_SERVICE_UNAVAILABLE,
|
status=status.HTTP_503_SERVICE_UNAVAILABLE,
|
||||||
)
|
)
|
||||||
|
except LLMProviderError:
|
||||||
|
logger.exception(
|
||||||
|
"AI backend rejected the request for document %s",
|
||||||
|
doc.pk,
|
||||||
|
)
|
||||||
|
return Response(
|
||||||
|
{
|
||||||
|
"ai": [
|
||||||
|
_(
|
||||||
|
"AI backend rejected the request. "
|
||||||
|
"Check logs for details.",
|
||||||
|
),
|
||||||
|
],
|
||||||
|
},
|
||||||
|
status=status.HTTP_502_BAD_GATEWAY,
|
||||||
|
)
|
||||||
set_llm_suggestions_cache(
|
set_llm_suggestions_cache(
|
||||||
doc.pk,
|
doc.pk,
|
||||||
llm_suggestions,
|
llm_suggestions,
|
||||||
@@ -2069,7 +2137,6 @@ class DocumentViewSet(
|
|||||||
|
|
||||||
try:
|
try:
|
||||||
doc_name, doc_data = serializer.validated_data.get("document")
|
doc_name, doc_data = serializer.validated_data.get("document")
|
||||||
doc_name = normalize_unicode(doc_name)
|
|
||||||
version_label = serializer.validated_data.get("version_label")
|
version_label = serializer.validated_data.get("version_label")
|
||||||
|
|
||||||
t = int(mktime(datetime.now().timetuple()))
|
t = int(mktime(datetime.now().timetuple()))
|
||||||
@@ -3336,7 +3403,7 @@ class PostDocumentView(GenericAPIView[Any]):
|
|||||||
serializer.is_valid(raise_exception=True)
|
serializer.is_valid(raise_exception=True)
|
||||||
|
|
||||||
doc_name, doc_data = serializer.validated_data.get("document")
|
doc_name, doc_data = serializer.validated_data.get("document")
|
||||||
doc_name = normalize_unicode(doc_name)
|
doc_name = normalize("NFC", doc_name)
|
||||||
correspondent_id = serializer.validated_data.get("correspondent")
|
correspondent_id = serializer.validated_data.get("correspondent")
|
||||||
document_type_id = serializer.validated_data.get("document_type")
|
document_type_id = serializer.validated_data.get("document_type")
|
||||||
storage_path_id = serializer.validated_data.get("storage_path")
|
storage_path_id = serializer.validated_data.get("storage_path")
|
||||||
@@ -5439,7 +5506,10 @@ class TrashView(ListModelMixin, PassUserMixin):
|
|||||||
|
|
||||||
model = Document
|
model = Document
|
||||||
|
|
||||||
queryset = Document.deleted_objects.all()
|
# A version is listed separately only when its root is not in the trash.
|
||||||
|
queryset = Document.deleted_objects.exclude(
|
||||||
|
root_document_id__in=Document.deleted_objects.values("id"),
|
||||||
|
)
|
||||||
|
|
||||||
def get(self, request: Request, format: str | None = None) -> Response:
|
def get(self, request: Request, format: str | None = None) -> Response:
|
||||||
self.serializer_class = DocumentSerializer
|
self.serializer_class = DocumentSerializer
|
||||||
@@ -5470,7 +5540,15 @@ class TrashView(ListModelMixin, PassUserMixin):
|
|||||||
return HttpResponseForbidden("Insufficient permissions")
|
return HttpResponseForbidden("Insufficient permissions")
|
||||||
action = serializer.validated_data.get("action")
|
action = serializer.validated_data.get("action")
|
||||||
if action == "restore":
|
if action == "restore":
|
||||||
restored = list(Document.deleted_objects.filter(id__in=doc_ids))
|
restored = list(self.get_queryset().filter(id__in=doc_ids))
|
||||||
|
if len(restored) != len(doc_ids):
|
||||||
|
raise ValidationError(
|
||||||
|
{
|
||||||
|
"documents": [
|
||||||
|
"Restore the root document instead of one of its versions.",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
)
|
||||||
for doc in restored:
|
for doc in restored:
|
||||||
doc.restore(strict=False)
|
doc.restore(strict=False)
|
||||||
if restored:
|
if restored:
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user