Compare commits

..
147 changed files with 4463 additions and 7968 deletions
+4 -2
View File
@@ -72,9 +72,11 @@ jobs:
'You are welcome to open a new issue that describes the problem you observed in your own words.'
: 'This issue was automatically closed because it was not opened using our bug report form. ' +
'Issues have to be created through the form so that the details we need to investigate are included.\n\n' +
`If the problem is still there, please [open a new issue](${newIssue}) using the form. No other action is needed here.\n\n` +
`If the problem is still there, please [open a new issue](${newIssue}) using the form — that is all it takes ` +
'to get it looked at, and no other action is needed here.\n\n' +
'If any part of your report was written by an AI tool or agent, you must say so: undisclosed AI-generated ' +
`contributions are a violation of our [Code of Conduct](${codeOfConduct}).`;
`contributions are a violation of our [Code of Conduct](${codeOfConduct}), and such reports must describe the ` +
`behavior you observed only, without code analysis or suggested fixes. See our [contributing guidelines](${contributing}).`;
await github.rest.issues.createComment({ ...common, body });
await github.rest.issues.addLabels({ ...common, labels: ['ai'] });
+2 -23
View File
@@ -25,10 +25,6 @@ jobs:
pr-bot:
name: Automated PR Bot
runs-on: ubuntu-latest
# Runs after Anti-slop so the welcome comment can see whether the PR was closed
# instead of racing it. Still runs if that job fails, so labeling is not lost.
needs: Anti-slop
if: ${{ !cancelled() }}
permissions:
contents: read
pull-requests: write
@@ -103,25 +99,8 @@ jobs:
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
with:
script: |
const user = context.payload.pull_request.user.login;
// Re-read the PR: Anti-slop may have closed and labeled it after the webhook
const { data: pr } = await github.rest.pulls.get({
owner: context.repo.owner,
repo: context.repo.repo,
pull_number: context.payload.pull_request.number,
});
if (pr.state === 'closed') {
core.info('Skipping comment: PR is already closed');
return;
}
const labels = pr.labels.map((label) => (typeof label === 'string' ? label : label.name));
if (labels.includes('ai')) {
core.info('Skipping comment: PR is labeled ai');
return;
}
const pr = context.payload.pull_request;
const user = pr.user.login;
const { data: members } = await github.rest.orgs.listMembers({
org: 'paperless-ngx',
-3
View File
@@ -115,6 +115,3 @@ celerybeat-schedule*
# Git worktree local folder
.worktrees
# Agent workflow scratch (ledgers, briefs, review packages)
.superpowers/
+1 -2
View File
@@ -521,8 +521,7 @@ Pass `--recreate` to wipe the existing index before rebuilding. Use this when th
index is corrupted or you want a fully clean rebuild.
Pass `--if-needed` to skip the rebuild if the index is already up to date (schema
version, schema fingerprint and search language all match). Safe to run on every
startup or upgrade.
version and search language match). Safe to run on every startup or upgrade.
Specify `optimize` to optimize the index. This command is regularly invoked by the
task scheduler.
+1 -3
View File
@@ -138,9 +138,7 @@ for suggested generation and embedding models.
With AI enabled, Paperless-ngx can suggest a title, tags, correspondent, document type,
storage path and dates by sending the document to the LLM. This is **opt-in per request**
and surfaces through the "Suggest" control on the document detail page, alongside the
classic classifier-based suggestions — it does not disable them. Suggestions are requested
automatically when you open a document that carries an inbox tag unless "Automatically request
suggestions for inbox documents" under Settings > Documents is disabled. Suggestion output
classic classifier-based suggestions — it does not disable them. Suggestion output
language can be steered with
[`PAPERLESS_AI_LLM_OUTPUT_LANGUAGE`](configuration.md#PAPERLESS_AI_LLM_OUTPUT_LANGUAGE)
(otherwise it follows the user's UI language).
+5 -94
View File
@@ -317,8 +317,6 @@ a "document already exists" message.
Paperless-ngx can suggest tags, correspondents, document types and storage paths for documents based on the content of the document. This is done using a (non-LLM) machine learning model that is trained on the documents in your database. The suggestions are shown in the document detail page and can be accepted or rejected by the user.
Suggestions are requested automatically when you open a document that still has an inbox tag. To only request them by pressing the "Suggest" button instead, turn off "Automatically request suggestions for inbox documents" under Settings > Documents.
## AI Features
Paperless-ngx includes several features that use AI to enhance the document management experience. These features are optional and can be enabled or disabled in the settings. If you are using the AI features, you may want to also enable the "LLM index" feature, which supports Retrieval-Augmented Generation (RAG) designed to improve the quality of AI responses. The LLM index feature is not enabled by default and requires additional configuration.
@@ -686,8 +684,7 @@ It requires [AI features](configuration.md#ai) to be enabled. You can specify:
never replace the document's existing tags.
The action works with every trigger **except Consumption Started**, because suggestions are made from
the document's text, which does not exist until after the document has been processed. Documents whose
processed text is empty or contains only whitespace are skipped.
the document's text, which does not exist until after the document has been processed.
Because the query to the AI service is slow, the action is queued and runs in the background rather
than as part of the workflow run itself. The document is updated once the suggestions come back.
@@ -931,19 +928,6 @@ Matching documents with logical expressions:
```
shopname AND (product1 OR product2)
invoice NOT draft
```
`AND`, `OR` and `NOT` must be written in capitals, and parentheses group sub-expressions. Terms written next to each other with no operator between them are combined with `AND`.
!!! warning
A leading `-` does **not** exclude a term. Separators are stripped during indexing, so `invoice -secret` searches for `invoice` and `secret`, which is the opposite of what you probably intended. Use `NOT` to exclude a term: `invoice NOT secret`.
Matching an exact phrase, in order, by quoting it:
```
"quick brown fox"
```
Matching specific tags, correspondents or types:
@@ -951,12 +935,8 @@ Matching specific tags, correspondents or types:
```
type:invoice tag:unpaid
correspondent:university certificate
tag:bills,unpaid
```
- `document_type` may be abbreviated to `type`, and `storage_path` to `path`.
- A comma-separated list after `tag:` requires **all** of the listed tags, so `tag:bills,unpaid` matches only documents tagged both `bills` and `unpaid`.
Matching dates:
```
@@ -965,58 +945,14 @@ added:yesterday
modified:today
```
Matching by archive metadata:
```
asn:100
page_count:12
num_notes:0
checksum:9f86d081884c7d659a2feaa0c55ad015a3bf4f1b2b0b822cd15d6c15b0f00a08
original_filename:invoice.pdf
```
- `asn` matches a document's Archive Serial Number.
- `page_count` matches a document's page count.
- `num_notes` matches how many notes a document has.
- `checksum` matches the checksum of the original document file (not the archived/processed version). Unlike the text fields, this one is stored verbatim rather than tokenized, so only a complete, lowercase checksum matches. To search by the first few characters instead, use a wildcard: `checksum:9f86d081*`. Wildcard patterns on the text fields are also tried stemmed, to line up with the stemmed index, but `checksum` is indexed without stemming, so its patterns are not stemmed either: a wildcard prefix is matched literally, apart from being lowercased first. `checksum:9F86D081*` therefore does find the document, even though the plain uppercase term does not.
- `original_filename` matches the filename of the document as originally consumed.
`asn`, `page_count` and `num_notes` are numeric and also accept ranges, for example `asn:[50 to 150]`.
Matching inexact words:
```
invoice*
title:Invoice*
produ*name
```
Wildcards are matched against the _stemmed_ terms stored in the index, not
against the words as they appear in the document. Each literal part of a
pattern is tried both as you typed it and in its stemmed form, so a trailing
`*` matches a word and its inflections (`invoice*` finds "invoice", "invoices"
and "invoiced") as well as longer words whose stored term still begins with
what you typed (`copy*` finds "copyright" alongside "copy" and "copies").
It is still not a plain prefix search over the original text. A trailing `*`
matches a stored term when either the run you typed or its stemmed form is a
prefix of that term, so a fragment that stops part-way between the two matches
neither: `universities*` finds "university" and "universities", which are both
stored as `univers`, while the shorter `universit*` finds nothing at all. For
the same reason `happine*` does not find "happiness", which is stored as
`happi`. And a pattern that requires letters after the wildcard which stemming
has removed cannot match either: `productname` is stored as `productnam`, so
`produ*name` finds nothing.
Matching natural date keywords:
The multi-word date keywords listed below work quoted or unquoted after a
date field (`added:"previous month"` and `added:previous month` are
equivalent); elsewhere in a query the same words are treated as ordinary
search text. Other date expressions the parser accepts (relative offsets
like `-1 week`, or specific dates like `12 december 2019`) must be quoted when
they stand alone as a value; inside a range's brackets they work unquoted, as
in `added:[-1 week to now]`.
```
added:today
modified:yesterday
@@ -1029,30 +965,6 @@ Supported date keywords: `today`, `yesterday`, `previous week`,
`this month`, `previous month`, `this year`, `previous year`,
`previous quarter`.
These other date forms also work after a date field:
```
added:tomorrow
created:2005-03-04
added:january
modified:"next monday"
added:"last monday"
added:"2005-01-01T00:00:00Z"
created:[2005-01-01 to 2005-01-31]
added:[2005-06-15T09:00:00Z to 2005-06-15T17:00:00Z]
```
- `tomorrow`, like `today` and `yesterday`, covers that whole day.
- An ISO date such as `2005-03-04` covers that whole day, and `2005-01` covers that whole month.
- A month name such as `january` covers that whole month in the current year.
- `next <weekday>` and `last <weekday>` each cover that whole day and must be quoted. A bare weekday name such as `monday` is not accepted.
- A full timestamp such as `2005-01-01T00:00:00Z` matches that exact instant. Like the other expressions above, it has to be quoted when it stands on its own: `added:"2005-01-01T00:00:00Z"`. The unquoted spelling is rejected with an error rather than searched, because only part of it can be read as a date.
- A range takes two of the above as its bounds, for example `created:[2005 to 2009]` or `added:[2005-01-01 to 2005-01-31]`. Bounds may carry a time of day. A bound is normally written without quotes; if you do quote one, use single quotes (`added:['-1 week' to now]`), because a double-quoted bound is rejected with an error.
!!! warning
As a value on its own, `now`, `noon`, `midnight` and relative offsets such as `"-3 days"` or `"-1 week"` are accepted by the parser but resolve to a single instant rather than to a span of time, so they match only a document whose timestamp is exactly that instant, which in practice means no documents at all. Quoting does not change this. As a *range bound* they are the opposite of a trap and are what you want: `added:['-1 week' to now]` covers the whole of the last seven days. Spellings like `now-3days` and `"3 days ago"` are rejected outright wherever they appear.
#### Searching custom fields
Custom field names and values are included in the full-text index, but they
@@ -1068,7 +980,6 @@ custom_fields.name:Insurance custom_fields.value:policy
- `custom_fields.value` matches against the value of any custom field.
- `custom_fields.name` matches the name of the field (use quotes for multi-word names).
- Combine both to find documents where a specific named field contains a specific value.
- The bare `custom_fields:` prefix is shorthand for `custom_fields.value:`.
Because separators are stripped during indexing, individual parts of formatted
codes are searchable on their own. A value stored as `A-1312/99.50` produces the
@@ -1096,9 +1007,9 @@ notes.note:reminder
notes.user:alice notes.note:insurance
```
The bare `notes:` prefix is shorthand for `notes.note:`.
All of these constructs can be combined as you see fit. What is described above is the whole of the query language paperless supports. It resembles other search query languages without being identical to any of them, so a construct that is not documented here is most likely treated as ordinary search text rather than as syntax, and an unrecognized field name is searched as text too.
All of these constructs can be combined as you see fit. If you want to
learn more about the query language used by paperless, see the
[Tantivy query language documentation](https://docs.rs/tantivy/latest/tantivy/query/struct.QueryParser.html).
!!! note
+7 -8
View File
@@ -32,21 +32,21 @@ dependencies = [
"django-cors-headers~=4.9.0",
"django-extensions~=4.1",
"django-filter~=25.1",
"django-guardian>=3.3.3,<3.5",
"django-guardian~=3.3.3",
"django-multiselectfield~=1.0.1",
"django-rich~=2.2.0",
"django-soft-delete~=1.0.18",
"django-treenode>=0.24",
"djangorestframework~=3.16",
"drf-spectacular~=0.30",
"drf-spectacular-sidecar>=2026.7.1,<2026.9",
"drf-spectacular-sidecar~=2026.7.1",
"drf-writable-nested~=0.7.1",
"filelock~=3.32.0",
"flower>=2.0.1,<2.2",
"gotenberg-client[httpx]~=1.0",
"httpx-oauth~=0.17",
"ijson>=3.5.1",
"imap-tools>=1.14,<1.16",
"imap-tools~=1.14.0",
"jinja2~=3.1.6",
"langdetect~=1.0.9",
"llama-index-core>=0.14.23",
@@ -56,7 +56,7 @@ dependencies = [
"llama-index-llms-ollama>=0.9.1",
"llama-index-llms-openai-like>=0.7.1",
"nltk~=3.10.0",
"ocrmypdf>=17.7,<17.12",
"ocrmypdf>=17.7,<17.11",
"openai>=2.48",
"pathvalidate~=3.3.1",
"pdf2image~=1.17.0",
@@ -77,7 +77,6 @@ dependencies = [
"torch~=2.13.0",
"watchfiles>=1.2",
"whitenoise~=6.11",
"whoosh-compat[tantivy]==0.1",
"zxing-cpp~=3.1.0",
]
[project.optional-dependencies]
@@ -104,17 +103,17 @@ docs = [
"zensical>=0.0.51",
]
lint = [
"prek>=0.4.11,<0.6",
"prek~=0.4.11",
"ruff~=0.16.1",
]
testing = [
"daphne",
"factory-boy~=3.3.1",
"faker>=40.36,<40.38",
"faker~=40.36.0",
"imagehash",
"pytest~=9.1.1",
"pytest-cov~=7.1.0",
"pytest-django>=4.12,<4.15",
"pytest-django~=4.12.0",
"pytest-env~=1.7.0",
"pytest-httpx",
"pytest-mock~=3.15.1",
+1 -3
View File
@@ -71,10 +71,8 @@
"tsConfig": "tsconfig.app.json",
"localize": true,
"assets": [
"src/favicon.ico",
"src/apple-touch-icon.png",
"src/icon-192.png",
"src/icon-512.png",
"src/icon-512-maskable.png",
"src/assets",
"src/manifest.webmanifest",
{
+206 -324
View File
File diff suppressed because it is too large Load Diff
+19
View File
@@ -14,6 +14,7 @@ import { DocumentListComponent } from './components/document-list/document-list.
import { DocumentAttributesComponent } from './components/manage/document-attributes/document-attributes.component'
import { MailComponent } from './components/manage/mail/mail.component'
import { SavedViewsComponent } from './components/manage/saved-views/saved-views.component'
import { ShareLinksComponent } from './components/manage/share-links/share-links.component'
import { WorkflowsComponent } from './components/manage/workflows/workflows.component'
import { NotFoundComponent } from './components/not-found/not-found.component'
import { DirtyDocGuard } from './guards/dirty-doc.guard'
@@ -310,6 +311,24 @@ export const routes: Routes = [
componentName: 'SavedViewsComponent',
},
},
{
path: 'share-links',
component: ShareLinksComponent,
canActivate: [PermissionsGuard],
data: {
requiredPermissionAny: [
{
action: PermissionAction.View,
type: PermissionType.ShareLink,
},
{
action: PermissionAction.View,
type: PermissionType.ShareLinkBundle,
},
],
componentName: 'ShareLinksComponent',
},
},
],
},
@@ -23,30 +23,17 @@
<div class="col">
<div class="card bg-light">
<div class="card-body">
<div class="card-title d-flex align-items-center flex-wrap">
<div class="card-title d-flex align-items-center">
<h6 class="mb-0">
{{option.title}}
</h6>
<a class="btn btn-sm btn-link" title="Read the documentation about this setting" i18n-title [href]="getDocsUrl(option.config_key)" target="_blank" referrerpolicy="no-referrer">
<i-bs name="info-circle"></i-bs>
</a>
@if (isExternallyConfigured(option.config_key)) {
@if (isSet(option.key)) {
<span class="badge rounded-pill bg-body-secondary text-dark fw-normal" title="This value overrides {{option.config_key}}, which is set outside Paperless." i18n-title>Overrides external</span>
} @else {
<span class="badge rounded-pill bg-body-secondary text-dark fw-normal" title="{{option.config_key}} is set outside Paperless. Enter a value here to override it." i18n-title>Set externally</span>
}
}
@if (isSet(option.key)) {
@if (isExternallyConfigured(option.config_key)) {
<button type="button" class="btn btn-sm btn-link text-danger ms-auto pe-0" title="Use the externally configured value" i18n-title (click)="resetOption(option.key)">
<i-bs class="me-1" name="x"></i-bs><ng-container i18n>Reset to external</ng-container>
</button>
} @else {
<button type="button" class="btn btn-sm btn-link text-danger ms-auto pe-0" title="Reset" i18n-title (click)="resetOption(option.key)">
<i-bs class="me-1" name="x"></i-bs><ng-container i18n>Reset</ng-container>
</button>
}
<button type="button" class="btn btn-sm btn-link text-danger ms-auto pe-0" title="Reset" i18n-title (click)="resetOption(option.key)">
<i-bs class="me-1" name="x"></i-bs><ng-container i18n>Reset</ng-container>
</button>
}
</div>
<div class="mb-n3">
@@ -163,19 +163,6 @@ describe('ConfigComponent', () => {
expect(component.configForm.get('barcodes_enabled').value).toBeNull()
})
it('should identify externally configured options', () => {
component.externallyConfiguredVariables = new Set([
'PAPERLESS_OCR_LANGUAGE',
])
expect(
component.isExternallyConfigured('PAPERLESS_OCR_LANGUAGE')
).toBeTruthy()
expect(
component.isExternallyConfigured('PAPERLESS_OCR_OUTPUT_TYPE')
).toBeFalsy()
})
it('should group options into sections within a category, or not', () => {
const sections = component.getCategorySections(ConfigCategory.OCR)
expect(sections).toEqual([null, ConfigSection.RemoteOCR])
@@ -69,7 +69,6 @@ export class ConfigComponent
public configForm = new FormGroup({})
public errors = {}
public externallyConfiguredVariables = new Set<string>()
get optionCategories(): string[] {
return Object.values(ConfigCategory)
@@ -153,9 +152,6 @@ export class ConfigComponent
}
private initialize(config: PaperlessConfig) {
this.externallyConfiguredVariables = new Set(
config.externally_configured_variables ?? []
)
if (!this.store) {
this.store = new BehaviorSubject(config)
@@ -166,9 +162,7 @@ export class ConfigComponent
this.configForm.patchValue(state, { emitEvent: false })
})
this.isDirty$ = dirtyCheck(this.configForm, this.store.asObservable(), {
excludeKeys: ['externally_configured_variables'],
})
this.isDirty$ = dirtyCheck(this.configForm, this.store.asObservable())
}
this.configForm.patchValue(config)
@@ -233,10 +227,6 @@ export class ConfigComponent
return this.configForm.get(key).value != null
}
public isExternallyConfigured(configKey: string): boolean {
return this.externallyConfiguredVariables.has(configKey)
}
public resetOption(key: string) {
this.configForm.get(key).setValue(null)
}
@@ -237,12 +237,6 @@
</div>
</div>
<div class="row">
<div class="col">
<pngx-input-check i18n-title title="Automatically request suggestions for inbox documents" i18n-hint hint="If un-checked, suggestions must be requested via the Suggest button." formControlName="documentEditingAutoSuggest"></pngx-input-check>
</div>
</div>
<div class="row">
<div class="col">
<pngx-input-check i18n-title title="Show document thumbnail during loading" formControlName="documentEditingOverlayThumbnail"></pngx-input-check>
@@ -267,7 +267,7 @@ describe('SettingsComponent', () => {
expect(toastErrorSpy).toHaveBeenCalled()
expect(storeSpy).toHaveBeenCalled()
expect(appearanceSettingsSpy).not.toHaveBeenCalled()
expect(setSpy).toHaveBeenCalledTimes(33)
expect(setSpy).toHaveBeenCalledTimes(32)
// succeed
storeSpy.mockReturnValueOnce(of(true))
@@ -168,7 +168,6 @@ export class SettingsComponent
pdfEditorDefaultEditMode: new FormControl(null),
documentEditingRemoveInboxTags: new FormControl(null),
documentEditingOverlayThumbnail: new FormControl(null),
documentEditingAutoSuggest: new FormControl(null),
documentDetailsHiddenFields: new FormControl([]),
searchDbOnly: new FormControl(null),
searchLink: new FormControl(null),
@@ -369,9 +368,6 @@ export class SettingsComponent
documentEditingOverlayThumbnail: this.settings.get(
SETTINGS_KEYS.DOCUMENT_EDITING_OVERLAY_THUMBNAIL
),
documentEditingAutoSuggest: this.settings.get(
SETTINGS_KEYS.DOCUMENT_EDITING_AUTO_SUGGEST
),
documentDetailsHiddenFields: this.settings.get(
SETTINGS_KEYS.DOCUMENT_DETAILS_HIDDEN_FIELDS
),
@@ -569,10 +565,6 @@ export class SettingsComponent
SETTINGS_KEYS.DOCUMENT_EDITING_OVERLAY_THUMBNAIL,
this.settingsForm.value.documentEditingOverlayThumbnail
)
this.settings.set(
SETTINGS_KEYS.DOCUMENT_EDITING_AUTO_SUGGEST,
this.settingsForm.value.documentEditingAutoSuggest
)
this.settings.set(
SETTINGS_KEYS.DOCUMENT_DETAILS_HIDDEN_FIELDS,
this.settingsForm.value.documentDetailsHiddenFields
@@ -244,6 +244,15 @@
<i-bs class="me-2" name="window-stack"></i-bs><span class="nav-link-label"><ng-container i18n>Saved Views</ng-container></span>
</a>
</li>
@if (canManageShareLinks) {
<li class="nav-item app-link">
<a class="nav-link" routerLink="share-links" routerLinkActive="active" (click)="closeMenu()"
ngbPopover="Share links" i18n-ngbPopover [disablePopover]="!slimSidebarPopoversEnabled" placement="end"
container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim">
<i-bs class="me-2" name="link"></i-bs><span class="nav-link-label"><ng-container i18n>Share links</ng-container></span>
</a>
</li>
}
<li class="nav-item app-link"
*pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.Workflow }"
tourAnchor="tour.workflows">
@@ -221,6 +221,19 @@ export class AppFrameComponent
return this.appTitleSetting() || environment.appTitle
}
get canManageShareLinks(): boolean {
return (
this.permissionsService.currentUserCan(
PermissionAction.View,
PermissionType.ShareLink
) ||
this.permissionsService.currentUserCan(
PermissionAction.View,
PermissionType.ShareLinkBundle
)
)
}
get customAppTitle(): string {
return this.appTitleSetting()
}
@@ -4,7 +4,7 @@ import { ComponentFixture, TestBed } from '@angular/core/testing'
import { FormsModule, ReactiveFormsModule } from '@angular/forms'
import { NgbActiveModal, NgbModule } from '@ng-bootstrap/ng-bootstrap'
import { NgSelectModule } from '@ng-select/ng-select'
import { of, throwError } from 'rxjs'
import { of } from 'rxjs'
import {
MailAction,
MailMetadataCorrespondentOption,
@@ -15,7 +15,6 @@ import { CorrespondentService } from 'src/app/services/rest/correspondent.servic
import { DocumentTypeService } from 'src/app/services/rest/document-type.service'
import { MailAccountService } from 'src/app/services/rest/mail-account.service'
import { SettingsService } from 'src/app/services/settings.service'
import { ToastService } from 'src/app/services/toast.service'
import { CheckComponent } from '../../input/check/check.component'
import { NumberComponent } from '../../input/number/number.component'
import { PermissionsFormComponent } from '../../input/permissions/permissions-form/permissions-form.component'
@@ -82,41 +81,6 @@ describe('MailRuleEditDialogComponent', () => {
fixture.detectChanges()
})
it('should use empty related object lists when retrieval fails', () => {
const failed = () => throwError(() => new Error('Forbidden'))
const toastSpy = jest.spyOn(TestBed.inject(ToastService), 'showError')
jest
.spyOn(TestBed.inject(MailAccountService), 'listAll')
.mockReturnValue(failed())
jest
.spyOn(TestBed.inject(CorrespondentService), 'listAll')
.mockReturnValue(failed())
jest
.spyOn(TestBed.inject(DocumentTypeService), 'listAll')
.mockReturnValue(failed())
const failedFixture = TestBed.createComponent(MailRuleEditDialogComponent)
const failedComponent = failedFixture.componentInstance
expect(failedComponent.accounts()).toEqual([])
expect(failedComponent.correspondents()).toEqual([])
expect(failedComponent.documentTypes()).toEqual([])
expect(() => failedFixture.detectChanges()).not.toThrow()
expect(toastSpy).toHaveBeenCalledTimes(3)
expect(toastSpy).toHaveBeenCalledWith(
'Error retrieving mail accounts',
expect.any(Error)
)
expect(toastSpy).toHaveBeenCalledWith(
'Error retrieving correspondents',
expect.any(Error)
)
expect(toastSpy).toHaveBeenCalledWith(
'Error retrieving document types',
expect.any(Error)
)
})
it('should support create and edit modes', () => {
component.dialogMode.set(EditDialogMode.CREATE)
const createTitleSpy = jest.spyOn(component, 'getCreateTitle')
@@ -6,7 +6,7 @@ import {
FormsModule,
ReactiveFormsModule,
} from '@angular/forms'
import { catchError, map, of } from 'rxjs'
import { map } from 'rxjs'
import { EditDialogComponent } from 'src/app/components/common/edit-dialog/edit-dialog.component'
import { Correspondent } from 'src/app/data/correspondent'
import { DocumentType } from 'src/app/data/document-type'
@@ -26,7 +26,6 @@ import { MailAccountService } from 'src/app/services/rest/mail-account.service'
import { MailRuleService } from 'src/app/services/rest/mail-rule.service'
import { UserService } from 'src/app/services/rest/user.service'
import { SettingsService } from 'src/app/services/settings.service'
import { ToastService } from 'src/app/services/toast.service'
import { CheckComponent } from '../../input/check/check.component'
import { NumberComponent } from '../../input/number/number.component'
import { SelectComponent } from '../../input/select/select.component'
@@ -159,45 +158,17 @@ export class MailRuleEditDialogComponent extends EditDialogComponent<MailRule> {
private readonly accountService = inject(MailAccountService)
private readonly correspondentService = inject(CorrespondentService)
private readonly documentTypeService = inject(DocumentTypeService)
private readonly toastService = inject(ToastService)
readonly accounts = toSignal(
this.accountService.listAll().pipe(
map((result) => result.results),
catchError((error) => {
this.toastService.showError(
$localize`Error retrieving mail accounts`,
error
)
return of([])
})
),
this.accountService.listAll().pipe(map((result) => result.results)),
{ initialValue: undefined as MailAccount[] }
)
readonly correspondents = toSignal(
this.correspondentService.listAll().pipe(
map((result) => result.results),
catchError((error) => {
this.toastService.showError(
$localize`Error retrieving correspondents`,
error
)
return of([])
})
),
this.correspondentService.listAll().pipe(map((result) => result.results)),
{ initialValue: undefined as Correspondent[] }
)
readonly documentTypes = toSignal(
this.documentTypeService.listAll().pipe(
map((result) => result.results),
catchError((error) => {
this.toastService.showError(
$localize`Error retrieving document types`,
error
)
return of([])
})
),
this.documentTypeService.listAll().pipe(map((result) => result.results)),
{ initialValue: undefined as DocumentType[] }
)
@@ -81,23 +81,6 @@ describe('UserEditDialogComponent', () => {
fixture.detectChanges()
})
it('should use an empty group list when retrieval fails', () => {
const toastSpy = jest.spyOn(toastService, 'showError')
jest
.spyOn(TestBed.inject(GroupService), 'listAll')
.mockReturnValue(throwError(() => new Error('Forbidden')))
const failedFixture = TestBed.createComponent(UserEditDialogComponent)
const failedComponent = failedFixture.componentInstance
expect(failedComponent.groups()).toEqual([])
expect(() => failedFixture.detectChanges()).not.toThrow()
expect(toastSpy).toHaveBeenCalledWith(
'Error retrieving groups',
expect.any(Error)
)
})
it('should support create and edit modes', () => {
component.dialogMode.set(EditDialogMode.CREATE)
const createTitleSpy = jest.spyOn(component, 'getCreateTitle')
@@ -6,7 +6,7 @@ import {
FormsModule,
ReactiveFormsModule,
} from '@angular/forms'
import { catchError, first, map, of } from 'rxjs'
import { first, map } from 'rxjs'
import { EditDialogComponent } from 'src/app/components/common/edit-dialog/edit-dialog.component'
import { Group } from 'src/app/data/group'
import { User } from 'src/app/data/user'
@@ -42,13 +42,7 @@ export class UserEditDialogComponent
private readonly groupsService = inject(GroupService)
readonly groups = toSignal(
this.groupsService.listAll().pipe(
map((result) => result.results),
catchError((error) => {
this.toastService.showError($localize`Error retrieving groups`, error)
return of([])
})
),
this.groupsService.listAll().pipe(map((result) => result.results)),
{ initialValue: undefined as Group[] }
)
readonly passwordIsSet = signal(false)
@@ -11,7 +11,7 @@ import {
} from '@angular/forms'
import { NgbActiveModal, NgbModule } from '@ng-bootstrap/ng-bootstrap'
import { NgSelectModule } from '@ng-select/ng-select'
import { of, throwError } from 'rxjs'
import { of } from 'rxjs'
import { CustomFieldQueriesModel } from 'src/app/components/common/custom-fields-query-dropdown/custom-fields-query-dropdown.component'
import { CustomFieldDataType } from 'src/app/data/custom-field'
import { CustomFieldQueryLogicalOperator } from 'src/app/data/custom-field-query'
@@ -39,7 +39,6 @@ import { DocumentTypeService } from 'src/app/services/rest/document-type.service
import { MailRuleService } from 'src/app/services/rest/mail-rule.service'
import { StoragePathService } from 'src/app/services/rest/storage-path.service'
import { SettingsService } from 'src/app/services/settings.service'
import { ToastService } from 'src/app/services/toast.service'
import { CustomFieldQueryExpression } from 'src/app/utils/custom-field-query-element'
import { ConfirmButtonComponent } from '../../confirm-button/confirm-button.component'
import { NumberComponent } from '../../input/number/number.component'
@@ -207,44 +206,6 @@ describe('WorkflowEditDialogComponent', () => {
settingsService.set(SETTINGS_KEYS.AI_ENABLED, ai)
}
it('should use empty related object lists when access is forbidden', () => {
const forbidden = () => throwError(() => new Error('Forbidden'))
const toastSpy = jest.spyOn(TestBed.inject(ToastService), 'showError')
jest
.spyOn(TestBed.inject(CorrespondentService), 'listAll')
.mockReturnValue(forbidden())
jest
.spyOn(TestBed.inject(DocumentTypeService), 'listAll')
.mockReturnValue(forbidden())
jest
.spyOn(TestBed.inject(StoragePathService), 'listAll')
.mockReturnValue(forbidden())
jest
.spyOn(TestBed.inject(MailRuleService), 'listAll')
.mockReturnValue(forbidden())
jest
.spyOn(TestBed.inject(CustomFieldsService), 'listAll')
.mockReturnValue(forbidden())
const forbiddenFixture = TestBed.createComponent(
WorkflowEditDialogComponent
)
const forbiddenComponent = forbiddenFixture.componentInstance
expect(forbiddenComponent.correspondents()).toEqual([])
expect(forbiddenComponent.documentTypes()).toEqual([])
expect(forbiddenComponent.storagePaths()).toEqual([])
expect(forbiddenComponent.mailRules()).toEqual([])
expect(forbiddenComponent.customFields()).toEqual([])
expect(forbiddenComponent.dateCustomFields()).toEqual([])
expect(() => forbiddenFixture.detectChanges()).not.toThrow()
expect(toastSpy).toHaveBeenCalledTimes(1)
expect(toastSpy).toHaveBeenCalledWith(
'Some workflow options could not be loaded.',
expect.any(Error)
)
})
it('should support create and edit modes, support adding triggers and actions on new workflow', () => {
component.dialogMode.set(EditDialogMode.CREATE)
const createTitleSpy = jest.spyOn(component, 'getCreateTitle')
@@ -16,7 +16,7 @@ import {
} from '@angular/forms'
import { NgbAccordionModule } from '@ng-bootstrap/ng-bootstrap'
import { NgxBootstrapIconsModule } from 'ngx-bootstrap-icons'
import { Subscription, catchError, map, of, takeUntil } from 'rxjs'
import { Subscription, map, takeUntil } from 'rxjs'
import { Correspondent } from 'src/app/data/correspondent'
import { CustomField, CustomFieldDataType } from 'src/app/data/custom-field'
import { DocumentType } from 'src/app/data/document-type'
@@ -48,7 +48,6 @@ import { StoragePathService } from 'src/app/services/rest/storage-path.service'
import { UserService } from 'src/app/services/rest/user.service'
import { WorkflowService } from 'src/app/services/rest/workflow.service'
import { SettingsService } from 'src/app/services/settings.service'
import { ToastService } from 'src/app/services/toast.service'
import { CustomFieldQueryExpression } from 'src/app/utils/custom-field-query-element'
import { ConfirmButtonComponent } from '../../confirm-button/confirm-button.component'
import {
@@ -513,43 +512,26 @@ export class WorkflowEditDialogComponent
private readonly storagePathService = inject(StoragePathService)
private readonly mailRuleService = inject(MailRuleService)
private readonly customFieldsService = inject(CustomFieldsService)
private readonly toastService = inject(ToastService)
private relatedObjectLoadErrorShown = false
readonly templates = signal<Workflow[]>(undefined)
readonly correspondents = toSignal(
this.correspondentService.listAll().pipe(
map((result) => result.results),
catchError((error) => this.handleRelatedObjectLoadError(error))
),
this.correspondentService.listAll().pipe(map((result) => result.results)),
{ initialValue: undefined as Correspondent[] }
)
readonly documentTypes = toSignal(
this.documentTypeService.listAll().pipe(
map((result) => result.results),
catchError((error) => this.handleRelatedObjectLoadError(error))
),
this.documentTypeService.listAll().pipe(map((result) => result.results)),
{ initialValue: undefined as DocumentType[] }
)
readonly storagePaths = toSignal(
this.storagePathService.listAll().pipe(
map((result) => result.results),
catchError((error) => this.handleRelatedObjectLoadError(error))
),
this.storagePathService.listAll().pipe(map((result) => result.results)),
{ initialValue: undefined as StoragePath[] }
)
readonly mailRules = toSignal(
this.mailRuleService.listAll().pipe(
map((result) => result.results),
catchError((error) => this.handleRelatedObjectLoadError(error))
),
this.mailRuleService.listAll().pipe(map((result) => result.results)),
{ initialValue: undefined as MailRule[] }
)
readonly customFields = toSignal(
this.customFieldsService.listAll().pipe(
map((result) => result.results),
catchError((error) => this.handleRelatedObjectLoadError(error))
),
this.customFieldsService.listAll().pipe(map((result) => result.results)),
{ initialValue: undefined as CustomField[] }
)
readonly dateCustomFields = computed(() =>
@@ -563,17 +545,6 @@ export class WorkflowEditDialogComponent
SETTINGS_KEYS.AI_ENABLED
)
private handleRelatedObjectLoadError(error) {
if (!this.relatedObjectLoadErrorShown) {
this.relatedObjectLoadErrorShown = true
this.toastService.showError(
$localize`Some workflow options could not be loaded.`,
error
)
}
return of([])
}
expandedItem: number = null
private readonly triggerFilterOptionsMap = new WeakMap<
@@ -7,9 +7,8 @@ import {
ReactiveFormsModule,
} from '@angular/forms'
import { NgSelectModule } from '@ng-select/ng-select'
import { of, throwError } from 'rxjs'
import { of } from 'rxjs'
import { GroupService } from 'src/app/services/rest/group.service'
import { ToastService } from 'src/app/services/toast.service'
import { PermissionsGroupComponent } from './permissions-group.component'
describe('PermissionsGroupComponent', () => {
@@ -61,19 +60,4 @@ describe('PermissionsGroupComponent', () => {
expect(component.value).toEqual({ id: 2, name: 'Group 2' })
expect(groupServiceSpy).toHaveBeenCalled()
})
it('should use an empty group list when retrieval fails', () => {
const toastSpy = jest.spyOn(TestBed.inject(ToastService), 'showError')
groupServiceSpy.mockReturnValue(throwError(() => new Error('Forbidden')))
const failedFixture = TestBed.createComponent(PermissionsGroupComponent)
const failedComponent = failedFixture.componentInstance
expect(failedComponent.groups()).toEqual([])
expect(() => failedFixture.detectChanges()).not.toThrow()
expect(toastSpy).toHaveBeenCalledWith(
'Error retrieving groups',
expect.any(Error)
)
})
})
@@ -6,10 +6,9 @@ import {
ReactiveFormsModule,
} from '@angular/forms'
import { NgSelectComponent } from '@ng-select/ng-select'
import { catchError, map, of } from 'rxjs'
import { map } from 'rxjs/operators'
import { Group } from 'src/app/data/group'
import { GroupService } from 'src/app/services/rest/group.service'
import { ToastService } from 'src/app/services/toast.service'
import { AbstractInputComponent } from '../../abstract-input'
@Component({
@@ -27,15 +26,8 @@ import { AbstractInputComponent } from '../../abstract-input'
})
export class PermissionsGroupComponent extends AbstractInputComponent<Group> {
private readonly groupService = inject(GroupService)
private readonly toastService = inject(ToastService)
readonly groups = toSignal(
this.groupService.listAll().pipe(
map((result) => result.results),
catchError((error) => {
this.toastService.showError($localize`Error retrieving groups`, error)
return of([])
})
),
this.groupService.listAll().pipe(map((result) => result.results)),
{ initialValue: undefined as Group[] }
)
}
@@ -7,9 +7,8 @@ import {
ReactiveFormsModule,
} from '@angular/forms'
import { NgSelectModule } from '@ng-select/ng-select'
import { of, throwError } from 'rxjs'
import { of } from 'rxjs'
import { UserService } from 'src/app/services/rest/user.service'
import { ToastService } from 'src/app/services/toast.service'
import { PermissionsUserComponent } from './permissions-user.component'
describe('PermissionsUserComponent', () => {
@@ -61,19 +60,4 @@ describe('PermissionsUserComponent', () => {
expect(component.value).toEqual({ id: 2, name: 'User 2' })
expect(userServiceSpy).toHaveBeenCalled()
})
it('should use an empty user list when retrieval fails', () => {
const toastSpy = jest.spyOn(TestBed.inject(ToastService), 'showError')
userServiceSpy.mockReturnValue(throwError(() => new Error('Forbidden')))
const failedFixture = TestBed.createComponent(PermissionsUserComponent)
const failedComponent = failedFixture.componentInstance
expect(failedComponent.users()).toEqual([])
expect(() => failedFixture.detectChanges()).not.toThrow()
expect(toastSpy).toHaveBeenCalledWith(
'Error retrieving users',
expect.any(Error)
)
})
})
@@ -6,10 +6,9 @@ import {
ReactiveFormsModule,
} from '@angular/forms'
import { NgSelectComponent } from '@ng-select/ng-select'
import { catchError, map, of } from 'rxjs'
import { map } from 'rxjs/operators'
import { User } from 'src/app/data/user'
import { UserService } from 'src/app/services/rest/user.service'
import { ToastService } from 'src/app/services/toast.service'
import { AbstractInputComponent } from '../../abstract-input'
@Component({
@@ -27,15 +26,8 @@ import { AbstractInputComponent } from '../../abstract-input'
})
export class PermissionsUserComponent extends AbstractInputComponent<User[]> {
private readonly userService = inject(UserService)
private readonly toastService = inject(ToastService)
readonly users = toSignal(
this.userService.listAll().pipe(
map((result) => result.results),
catchError((error) => {
this.toastService.showError($localize`Error retrieving users`, error)
return of([])
})
),
this.userService.listAll().pipe(map((result) => result.results)),
{ initialValue: undefined as User[] }
)
}
@@ -4,9 +4,8 @@ import { ComponentFixture, TestBed } from '@angular/core/testing'
import { FormsModule, ReactiveFormsModule } from '@angular/forms'
import { NgbActiveModal, NgbModule } from '@ng-bootstrap/ng-bootstrap'
import { NgSelectModule } from '@ng-select/ng-select'
import { of, throwError } from 'rxjs'
import { of } from 'rxjs'
import { UserService } from 'src/app/services/rest/user.service'
import { ToastService } from 'src/app/services/toast.service'
import { PermissionsFormComponent } from '../input/permissions/permissions-form/permissions-form.component'
import { PermissionsGroupComponent } from '../input/permissions/permissions-group/permissions-group.component'
import { PermissionsUserComponent } from '../input/permissions/permissions-user/permissions-user.component'
@@ -78,23 +77,6 @@ describe('PermissionsDialogComponent', () => {
fixture.detectChanges()
})
it('should use an empty user list when retrieval fails', () => {
const toastSpy = jest.spyOn(TestBed.inject(ToastService), 'showError')
jest
.spyOn(TestBed.inject(UserService), 'listAll')
.mockReturnValue(throwError(() => new Error('Forbidden')))
const failedFixture = TestBed.createComponent(PermissionsDialogComponent)
const failedComponent = failedFixture.componentInstance
expect(failedComponent.users()).toEqual([])
expect(() => failedFixture.detectChanges()).not.toThrow()
expect(toastSpy).toHaveBeenCalledWith(
'Error retrieving users',
expect.any(Error)
)
})
it('should return permissions', () => {
expect(component.permissions).toEqual({
owner: null,
@@ -14,11 +14,10 @@ import {
ReactiveFormsModule,
} from '@angular/forms'
import { NgbActiveModal } from '@ng-bootstrap/ng-bootstrap'
import { catchError, map, of } from 'rxjs'
import { map } from 'rxjs'
import { ObjectWithPermissions } from 'src/app/data/object-with-permissions'
import { User } from 'src/app/data/user'
import { UserService } from 'src/app/services/rest/user.service'
import { ToastService } from 'src/app/services/toast.service'
import { PermissionsFormComponent } from '../input/permissions/permissions-form/permissions-form.component'
import { SwitchComponent } from '../input/switch/switch.component'
@@ -36,16 +35,9 @@ import { SwitchComponent } from '../input/switch/switch.component'
export class PermissionsDialogComponent {
activeModal = inject(NgbActiveModal)
private userService = inject(UserService)
private toastService = inject(ToastService)
readonly users = toSignal(
this.userService.listAll().pipe(
map((r) => r.results),
catchError((error) => {
this.toastService.showError($localize`Error retrieving users`, error)
return of([])
})
),
this.userService.listAll().pipe(map((r) => r.results)),
{ initialValue: undefined as User[] }
)
readonly title = signal($localize`Set permissions`)
@@ -1473,35 +1473,6 @@ describe('DocumentDetailComponent', () => {
})
})
it('should not automatically get suggestions if auto-suggest is disabled', () => {
settingsService.set(SETTINGS_KEYS.DOCUMENT_EDITING_AUTO_SUGGEST, false)
const suggestionsSpy = jest.spyOn(documentService, 'getSuggestions')
suggestionsSpy.mockReturnValue(of({ tags: [42] }))
initNormally()
expect(suggestionsSpy).not.toHaveBeenCalled()
// still available on demand
component.getSuggestions()
expect(suggestionsSpy).toHaveBeenCalled()
})
it('should not automatically get AI suggestions if auto-suggest is disabled', () => {
settingsService.set(SETTINGS_KEYS.DOCUMENT_EDITING_AUTO_SUGGEST, false)
const getSetting = settingsService.get.bind(settingsService)
jest
.spyOn(settingsService, 'get')
.mockImplementation((key) =>
key === SETTINGS_KEYS.AI_ENABLED ? true : getSetting(key)
)
const aiSuggestionsSpy = jest.spyOn(documentService, 'getAiSuggestions')
aiSuggestionsSpy.mockReturnValue(of({ tags: [42] }))
initNormally()
expect(aiSuggestionsSpy).not.toHaveBeenCalled()
component.getSuggestions()
expect(aiSuggestionsSpy).toHaveBeenCalled()
})
it('should reset the suggestions loading state if the document changes mid-request', () => {
const getSetting = settingsService.get.bind(settingsService)
jest
@@ -237,9 +237,6 @@ export class DocumentDetailComponent
this.settings.getSignal<boolean>(
SETTINGS_KEYS.DOCUMENT_EDITING_OVERLAY_THUMBNAIL
)
private readonly autoSuggestSetting = this.settings.getSignal<boolean>(
SETTINGS_KEYS.DOCUMENT_EDITING_AUTO_SUGGEST
)
private readonly hiddenFieldsSetting = this.settings.getSignal<
DocumentDetailFieldID[]
>(SETTINGS_KEYS.DOCUMENT_DETAILS_HIDDEN_FIELDS)
@@ -360,10 +357,6 @@ export class DocumentDetailComponent
return this.aiEnabledSetting()
}
get autoSuggest(): boolean {
return this.autoSuggestSetting()
}
get archiveContentRenderType(): ContentRenderType {
const hasArchiveVersion =
this.metadata()?.has_archive_version ??
@@ -911,7 +904,6 @@ export class DocumentDetailComponent
this.updateFormForCustomFields()
this.loadMetadataForSelectedVersion()
if (
this.autoSuggest &&
this.permissionsService.currentUserHasObjectPermissions(
PermissionAction.Change,
doc
@@ -7,6 +7,7 @@ import {
import { EventEmitter, signal } from '@angular/core'
import { ComponentFixture, TestBed } from '@angular/core/testing'
import { By } from '@angular/platform-browser'
import { Router } from '@angular/router'
import { NgbModal, NgbModalRef } from '@ng-bootstrap/ng-bootstrap'
import { NgxBootstrapIconsModule, allIcons } from 'ngx-bootstrap-icons'
import { of, throwError } from 'rxjs'
@@ -46,7 +47,6 @@ import { StoragePathEditDialogComponent } from '../../common/edit-dialog/storage
import { TagEditDialogComponent } from '../../common/edit-dialog/tag-edit-dialog/tag-edit-dialog.component'
import { FilterableDropdownComponent } from '../../common/filterable-dropdown/filterable-dropdown.component'
import { ShareLinkBundleDialogComponent } from '../../common/share-link-bundle-dialog/share-link-bundle-dialog.component'
import { ShareLinkBundleManageDialogComponent } from '../../common/share-link-bundle-manage-dialog/share-link-bundle-manage-dialog.component'
import { BulkEditorComponent } from './bulk-editor.component'
const selectionData: SelectionData = {
@@ -82,6 +82,7 @@ describe('BulkEditorComponent', () => {
let customFieldsService: CustomFieldsService
let httpTestingController: HttpTestingController
let shareLinkBundleService: ShareLinkBundleService
let router: Router
beforeEach(async () => {
TestBed.configureTestingModule({
@@ -167,11 +168,14 @@ describe('BulkEditorComponent', () => {
provide: ShareLinkBundleService,
useValue: {
createBundle: jest.fn(),
listAllBundles: jest.fn(),
rebuildBundle: jest.fn(),
delete: jest.fn(),
},
},
{
provide: Router,
useValue: { navigate: jest.fn().mockResolvedValue(true) },
},
provideHttpClient(withInterceptorsFromDi()),
provideHttpClientTesting(),
],
@@ -189,6 +193,7 @@ describe('BulkEditorComponent', () => {
customFieldsService = TestBed.inject(CustomFieldsService)
httpTestingController = TestBed.inject(HttpTestingController)
shareLinkBundleService = TestBed.inject(ShareLinkBundleService)
router = TestBed.inject(Router)
fixture = TestBed.createComponent(BulkEditorComponent)
component = fixture.componentInstance
@@ -1824,9 +1829,9 @@ describe('BulkEditorComponent', () => {
},
}
const openSpy = jest.spyOn(modalService, 'open')
openSpy.mockReturnValueOnce(modalRef as NgbModalRef)
openSpy.mockReturnValueOnce({} as NgbModalRef)
const openSpy = jest
.spyOn(modalService, 'open')
.mockReturnValueOnce(modalRef as NgbModalRef)
;(shareLinkBundleService.createBundle as jest.Mock).mockReturnValueOnce(
of({ id: 42 })
)
@@ -1860,11 +1865,9 @@ describe('BulkEditorComponent', () => {
dialogInstance.onOpenManage()
expect(modalRef.close).toHaveBeenCalled()
expect(openSpy).toHaveBeenNthCalledWith(
2,
ShareLinkBundleManageDialogComponent,
expect.objectContaining({ backdrop: 'static', size: 'lg' })
)
expect(router.navigate).toHaveBeenCalledWith(['/share-links'], {
queryParams: { type: 'bundles' },
})
openSpy.mockRestore()
})
@@ -1917,13 +1920,10 @@ describe('BulkEditorComponent', () => {
openSpy.mockRestore()
})
it('should open share link bundle management dialog', () => {
const openSpy = jest.spyOn(modalService, 'open')
it('should navigate to share link bundle management', () => {
component.manageShareLinkBundles()
expect(openSpy).toHaveBeenCalledWith(
ShareLinkBundleManageDialogComponent,
expect.objectContaining({ backdrop: 'static', size: 'lg' })
)
openSpy.mockRestore()
expect(router.navigate).toHaveBeenCalledWith(['/share-links'], {
queryParams: { type: 'bundles' },
})
})
})
@@ -12,6 +12,7 @@ import {
FormsModule,
ReactiveFormsModule,
} from '@angular/forms'
import { Router } from '@angular/router'
import {
NgbDropdownModule,
NgbModal,
@@ -69,7 +70,6 @@ import {
import { ToggleableItemState } from '../../common/filterable-dropdown/toggleable-dropdown-button/toggleable-dropdown-button.component'
import { PermissionsDialogComponent } from '../../common/permissions-dialog/permissions-dialog.component'
import { ShareLinkBundleDialogComponent } from '../../common/share-link-bundle-dialog/share-link-bundle-dialog.component'
import { ShareLinkBundleManageDialogComponent } from '../../common/share-link-bundle-manage-dialog/share-link-bundle-manage-dialog.component'
import { ComponentWithPermissions } from '../../with-permissions/with-permissions.component'
import { CustomFieldsBulkEditDialogComponent } from './custom-fields-bulk-edit-dialog/custom-fields-bulk-edit-dialog.component'
@@ -104,6 +104,7 @@ export class BulkEditorComponent
public readonly permissionService = inject(PermissionsService)
private savedViewService = inject(SavedViewService)
private readonly shareLinkBundleService = inject(ShareLinkBundleService)
private readonly router = inject(Router)
tagSelectionModel = new FilterableDropdownSelectionModel(true)
correspondentSelectionModel = new FilterableDropdownSelectionModel()
@@ -1135,9 +1136,8 @@ export class BulkEditorComponent
}
manageShareLinkBundles() {
this.modalService.open(ShareLinkBundleManageDialogComponent, {
backdrop: 'static',
size: 'lg',
void this.router.navigate(['/share-links'], {
queryParams: { type: 'bundles' },
})
}
@@ -10,7 +10,7 @@
}
</div>
</div>
@if (textFilterTarget === 'asn' || textFilterTarget === 'duplicates') {
@if (textFilterTarget === 'asn') {
<select class="form-select flex-grow-0 w-auto" [(ngModel)]="textFilterModifier" (change)="textFilterModifierChange()">
@for (m of textFilterModifiers; track m) {
<option ngbDropdownItem [value]="m.id">{{m.label}}</option>
@@ -23,7 +23,7 @@
</button>
}
<input #textFilterInput class="form-control form-control-sm" type="text"
[disabled]="textFilterInputDisabled"
[disabled]="textFilterModifierIsNull"
[(ngModel)]="textFilter"
(keydown)="textFilterKeydown($event)"
[ngbTypeahead]="searchAutoComplete"
@@ -53,7 +53,6 @@ import {
FILTER_HAS_CUSTOM_FIELDS_ALL,
FILTER_HAS_CUSTOM_FIELDS_ANY,
FILTER_HAS_DOCUMENT_TYPE_ANY,
FILTER_HAS_DUPLICATES,
FILTER_HAS_STORAGE_PATH_ANY,
FILTER_HAS_TAGS_ALL,
FILTER_HAS_TAGS_ANY,
@@ -428,38 +427,6 @@ describe('FilterEditorComponent', () => {
expect(component.textFilterTarget).toEqual('mime-type') // TEXT_FILTER_TARGET_MIME_TYPE
})
it('should ingest filter rules for documents with duplicates', () => {
component.filterRules = [
{
rule_type: FILTER_HAS_DUPLICATES,
value: 'true',
},
]
fixture.detectChanges()
expect(component.textFilterTarget).toEqual('duplicates')
expect(component.textFilterModifier).toEqual('has-duplicates')
expect(component.textFilterInputDisabled).toBeTruthy()
})
it('should ingest filter rules for documents without duplicates', () => {
component.filterRules = [
{
rule_type: FILTER_HAS_DUPLICATES,
value: 'false',
},
]
expect(component.textFilterTarget).toEqual('duplicates')
expect(component.textFilterModifier).toEqual('does-not-have-duplicates')
expect(component.filterRules).toEqual([
{
rule_type: FILTER_HAS_DUPLICATES,
value: 'false',
},
])
})
it('should ingest text filter rules for fulltext query', () => {
expect(component.textFilter).toEqual(null)
component.filterRules = [
@@ -1423,33 +1390,6 @@ describe('FilterEditorComponent', () => {
])
})
it('should convert duplicate target input to the correct filter rule', () => {
const textFieldTargetDropdown = fixture.debugElement.queryAll(
By.directive(NgbDropdownItem)
)[5]
textFieldTargetDropdown.triggerEventHandler('click')
fixture.detectChanges()
expect(component.textFilterTarget).toEqual('duplicates')
expect(component.filterRules).toEqual([
{
rule_type: FILTER_HAS_DUPLICATES,
value: 'true',
},
])
const textFieldModifierSelect = fixture.debugElement.query(By.css('select'))
textFieldModifierSelect.nativeElement.value = 'does-not-have-duplicates'
textFieldModifierSelect.nativeElement.dispatchEvent(new Event('change'))
fixture.detectChanges()
expect(component.filterRules).toEqual([
{
rule_type: FILTER_HAS_DUPLICATES,
value: 'false',
},
])
})
it('should convert user input to correct filter rules on full text query', () => {
component.textFilterInput.nativeElement.value = 'foo'
component.textFilterInput.nativeElement.dispatchEvent(new Event('input'))
@@ -2238,22 +2178,6 @@ describe('FilterEditorComponent', () => {
]
expect(component.generateFilterName()).toEqual('Without any tag')
component.filterRules = [
{
rule_type: FILTER_HAS_DUPLICATES,
value: 'true',
},
]
expect(component.generateFilterName()).toEqual('With duplicates')
component.filterRules = [
{
rule_type: FILTER_HAS_DUPLICATES,
value: 'false',
},
]
expect(component.generateFilterName()).toEqual('Without duplicates')
component.filterRules = [
{
rule_type: FILTER_CUSTOM_FIELDS_QUERY,
@@ -65,7 +65,6 @@ import {
FILTER_HAS_CUSTOM_FIELDS_ALL,
FILTER_HAS_CUSTOM_FIELDS_ANY,
FILTER_HAS_DOCUMENT_TYPE_ANY,
FILTER_HAS_DUPLICATES,
FILTER_HAS_STORAGE_PATH_ANY,
FILTER_HAS_TAGS_ALL,
FILTER_HAS_TAGS_ANY,
@@ -130,15 +129,12 @@ const TEXT_FILTER_TARGET_FULLTEXT_QUERY = 'fulltext-query'
const TEXT_FILTER_TARGET_FULLTEXT_MORELIKE = 'fulltext-morelike'
const TEXT_FILTER_TARGET_CUSTOM_FIELDS = 'custom-fields'
const TEXT_FILTER_TARGET_MIME_TYPE = 'mime-type'
const TEXT_FILTER_TARGET_DUPLICATES = 'duplicates'
const TEXT_FILTER_MODIFIER_EQUALS = 'equals'
const TEXT_FILTER_MODIFIER_NULL = 'is null'
const TEXT_FILTER_MODIFIER_NOTNULL = 'not null'
const TEXT_FILTER_MODIFIER_GT = 'greater'
const TEXT_FILTER_MODIFIER_LT = 'less'
const TEXT_FILTER_MODIFIER_HAS_DUPLICATES = 'has-duplicates'
const TEXT_FILTER_MODIFIER_DOES_NOT_HAVE_DUPLICATES = 'does-not-have-duplicates'
const RELATIVE_DATE_QUERY_REGEXP_CREATED = /created:[\["]([^\]]+)[\]"]/g
const RELATIVE_DATE_QUERY_REGEXP_ADDED = /added:[\["]([^\]]+)[\]"]/g
@@ -209,7 +205,6 @@ const DEFAULT_TEXT_FILTER_TARGET_OPTIONS = [
id: TEXT_FILTER_TARGET_FULLTEXT_QUERY,
name: $localize`Advanced search`,
},
{ id: TEXT_FILTER_TARGET_DUPLICATES, name: $localize`Duplicates` },
]
const DEPRECATED_CUSTOM_FIELDS_TEXT_FILTER_TARGET_OPTION = {
@@ -246,17 +241,6 @@ const DEFAULT_TEXT_FILTER_MODIFIER_OPTIONS = [
},
]
const DUPLICATES_FILTER_MODIFIER_OPTIONS = [
{
id: TEXT_FILTER_MODIFIER_HAS_DUPLICATES,
label: $localize`exist`,
},
{
id: TEXT_FILTER_MODIFIER_DOES_NOT_HAVE_DUPLICATES,
label: $localize`do not exist`,
},
]
@Component({
selector: 'pngx-filter-editor',
templateUrl: './filter-editor.component.html',
@@ -336,12 +320,6 @@ export class FilterEditorComponent
if (rule.value == 'false') {
return $localize`Without any tag`
}
break
case FILTER_HAS_DUPLICATES:
return rule.value == 'false'
? $localize`Without duplicates`
: $localize`With duplicates`
case FILTER_CUSTOM_FIELDS_QUERY:
return $localize`Custom fields query`
@@ -412,9 +390,7 @@ export class FilterEditorComponent
public textFilterModifier: string
get textFilterModifiers() {
return this.textFilterTarget === TEXT_FILTER_TARGET_DUPLICATES
? DUPLICATES_FILTER_MODIFIER_OPTIONS
: DEFAULT_TEXT_FILTER_MODIFIER_OPTIONS
return DEFAULT_TEXT_FILTER_MODIFIER_OPTIONS
}
get textFilterModifierIsNull(): boolean {
@@ -423,13 +399,6 @@ export class FilterEditorComponent
)
}
get textFilterInputDisabled(): boolean {
return (
this.textFilterModifierIsNull ||
this.textFilterTarget === TEXT_FILTER_TARGET_DUPLICATES
)
}
tagSelectionModel = new FilterableDropdownSelectionModel(true)
correspondentSelectionModel = new FilterableDropdownSelectionModel()
documentTypeSelectionModel = new FilterableDropdownSelectionModel()
@@ -475,7 +444,6 @@ export class FilterEditorComponent
this.customFieldQueriesModel.clear(false)
this._textFilter = null
this._moreLikeId = null
this.textFilterTarget = TEXT_FILTER_TARGET_TITLE_CONTENT
this.dateAddedTo = null
this.dateAddedFrom = null
this.dateCreatedTo = null
@@ -509,13 +477,6 @@ export class FilterEditorComponent
this.textFilterTarget = TEXT_FILTER_TARGET_MIME_TYPE
this._textFilter = rule.value
break
case FILTER_HAS_DUPLICATES:
this.textFilterTarget = TEXT_FILTER_TARGET_DUPLICATES
this.textFilterModifier =
rule.value == 'false' || rule.value == '0'
? TEXT_FILTER_MODIFIER_DOES_NOT_HAVE_DUPLICATES
: TEXT_FILTER_MODIFIER_HAS_DUPLICATES
break
case FILTER_FULLTEXT_QUERY:
let allQueryArgs = rule.value.split(',')
let textQueryArgs = []
@@ -839,14 +800,6 @@ export class FilterEditorComponent
value: this._textFilter.trim(),
})
}
if (this.textFilterTarget == TEXT_FILTER_TARGET_DUPLICATES) {
filterRules.push({
rule_type: FILTER_HAS_DUPLICATES,
value: (
this.textFilterModifier == TEXT_FILTER_MODIFIER_HAS_DUPLICATES
).toString(),
})
}
if (this._textFilter && this.textFilterTarget == TEXT_FILTER_TARGET_TITLE) {
filterRules.push({
rule_type: FILTER_SIMPLE_TITLE,
@@ -1210,7 +1163,7 @@ export class FilterEditorComponent
}
get textFilter() {
return this.textFilterInputDisabled ? '' : this._textFilter
return this.textFilterModifierIsNull ? '' : this._textFilter
}
set textFilter(value) {
@@ -1410,24 +1363,12 @@ export class FilterEditorComponent
this._textFilter = ''
}
this.textFilterTarget = target
if (target == TEXT_FILTER_TARGET_DUPLICATES) {
this._textFilter = ''
this.textFilterModifier = TEXT_FILTER_MODIFIER_HAS_DUPLICATES
} else if (
[
TEXT_FILTER_MODIFIER_HAS_DUPLICATES,
TEXT_FILTER_MODIFIER_DOES_NOT_HAVE_DUPLICATES,
].includes(this.textFilterModifier)
) {
this.textFilterModifier = TEXT_FILTER_MODIFIER_EQUALS
}
this.textFilterInput.nativeElement.focus()
this.updateRules()
}
textFilterModifierChange() {
if (
this.textFilterTarget == TEXT_FILTER_TARGET_DUPLICATES ||
this.textFilterModifierIsNull ||
([
TEXT_FILTER_MODIFIER_EQUALS,
@@ -1,38 +1,22 @@
<div class="modal-header">
<h4 class="modal-title">{{ title }}</h4>
<button type="button" class="btn-close" aria-label="Close" (click)="close()"></button>
</div>
<div class="modal-body">
@if (loading()) {
<div class="d-flex align-items-center gap-2">
<div class="spinner-border spinner-border-sm" role="status"></div>
<span i18n>Loading share link bundles…</span>
</div>
}
<div class="border border-top-0 rounded-bottom p-3">
@if (!loading() && error()) {
<div class="alert alert-danger mb-0" role="alert">
{{ error() }}
</div>
}
@if (!loading() && !error()) {
<div class="d-flex justify-content-between align-items-center mb-2">
<p class="mb-0 text-muted small">
<ng-container i18n>Status updates every few seconds while bundles are being prepared.</ng-container>
</p>
</div>
@if (bundles().length === 0) {
<p class="mb-0 text-muted fst-italic" i18n>No share link bundles currently exist.</p>
}
@if (bundles().length > 0) {
<div class="table-responsive">
<table class="table table-sm align-middle mb-0">
<table class="table table-sm align-middle mb-0 bg-body">
<thead>
<tr>
<th scope="col" i18n>Created</th>
<th scope="col" i18n>Status</th>
<th scope="col" class="fw-normal" pngxSortable="created" [currentSortField]="sortField()" [currentSortReverse]="sortReverse()" (sort)="onSort($event)" i18n>Created</th>
<th scope="col" class="fw-normal" pngxSortable="status" [currentSortField]="sortField()" [currentSortReverse]="sortReverse()" (sort)="onSort($event)" i18n>Status</th>
<th scope="col" i18n>Size</th>
<th scope="col" i18n>Expires</th>
<th scope="col" class="fw-normal" pngxSortable="expiration" [currentSortField]="sortField()" [currentSortReverse]="sortReverse()" (sort)="onSort($event)" i18n>Expires</th>
<th scope="col" i18n>Documents</th>
<th scope="col" i18n>File version</th>
<th scope="col" class="text-end" i18n>Actions</th>
@@ -96,6 +80,9 @@
<td>
@if (bundle.expiration) {
{{ bundle.expiration | date: 'short' }}
@if (isExpired(bundle.expiration)) {
<span class="badge text-bg-danger ms-2" i18n>Expired</span>
}
}
@if (!bundle.expiration) {
<span i18n>Never</span>
@@ -104,42 +91,49 @@
<td>{{ bundle.document_count }}</td>
<td>{{ fileVersionLabel(bundle.file_version) }}</td>
<td class="text-end">
<div class="btn-group btn-group-sm">
<button
type="button"
class="btn btn-outline-primary"
[disabled]="bundle.status !== statuses.Ready"
(click)="copy(bundle)"
title="Copy share link"
i18n-title
>
@if (copiedSlug() === bundle.slug) {
<i-bs name="clipboard-check"></i-bs>
}
@if (copiedSlug() !== bundle.slug) {
<i-bs name="clipboard"></i-bs>
}
<span class="visually-hidden" i18n>Copy share link</span>
</button>
@if (bundle.status === statuses.Failed) {
<div class="d-inline-block position-relative">
<span
class="badge bg-primary small fade position-absolute top-50 end-100 translate-middle-y me-2 pe-none z-3 text-nowrap"
[class.show]="copiedSlug() === bundle.slug"
i18n
>Copied!</span>
<div class="btn-group btn-group-sm">
<button
type="button"
class="btn btn-outline-warning"
[disabled]="loading()"
(click)="retry(bundle)"
class="btn btn-outline-primary"
[disabled]="bundle.status !== statuses.Ready"
(click)="copy(bundle)"
title="Copy share link"
i18n-title
>
<i-bs name="arrow-clockwise"></i-bs>
<span class="visually-hidden" i18n>Retry</span>
@if (copiedSlug() === bundle.slug) {
<i-bs name="clipboard-check"></i-bs>
}
@if (copiedSlug() !== bundle.slug) {
<i-bs name="clipboard"></i-bs>
}
<span class="visually-hidden" i18n>Copy share link</span>
</button>
}
<pngx-confirm-button
buttonClasses="btn btn-sm btn-outline-danger"
[disabled]="loading()"
(confirm)="delete(bundle)"
iconName="trash"
>
<span class="visually-hidden" i18n>Delete share link bundle</span>
</pngx-confirm-button>
@if (bundle.status === statuses.Failed) {
<button
type="button"
class="btn btn-outline-warning"
[disabled]="loading()"
(click)="retry(bundle)"
>
<i-bs name="arrow-clockwise"></i-bs>
<span class="visually-hidden" i18n>Retry</span>
</button>
}
<pngx-confirm-button
buttonClasses="btn btn-sm btn-outline-danger"
[disabled]="loading()"
(confirm)="delete(bundle)"
iconName="trash"
>
<span class="visually-hidden" i18n>Delete share link bundle</span>
</pngx-confirm-button>
</div>
</div>
</td>
</tr>
@@ -147,10 +141,32 @@
</tbody>
</table>
</div>
<div class="d-flex flex-wrap justify-content-end align-items-center gap-3 mt-3 ms-auto">
<div class="d-flex flex-wrap justify-content-end align-items-center gap-3">
<div class="d-flex align-items-center">
<label class="small text-muted me-2" for="shareLinkBundlePageSize" i18n>Show:</label>
<select id="shareLinkBundlePageSize" class="form-select form-select-sm w-auto" [(ngModel)]="pageSize">
<option [ngValue]="25">25</option>
<option [ngValue]="50">50</option>
<option [ngValue]="100">100</option>
</select>
<span class="small text-muted ms-2 d-none d-md-inline" i18n>per page</span>
</div>
@if (total() > pageSize) {
<ngb-pagination
class="mb-0"
[pageSize]="pageSize"
[collectionSize]="total()"
[page]="page()"
[maxSize]="5"
(pageChange)="setPage($event)"
size="sm"
aria-label="Share link bundles pagination"
i18n-aria-label
></ngb-pagination>
}
</div>
</div>
}
}
</div>
<div class="modal-footer">
<button type="button" class="btn btn-outline-secondary btn-sm" (click)="close()" i18n>Close</button>
</div>
@@ -1,6 +1,5 @@
import { Clipboard } from '@angular/cdk/clipboard'
import { ComponentFixture, TestBed } from '@angular/core/testing'
import { NgbActiveModal } from '@ng-bootstrap/ng-bootstrap'
import { NgxBootstrapIconsModule, allIcons } from 'ngx-bootstrap-icons'
import { of, throwError } from 'rxjs'
import { FileVersion } from 'src/app/data/share-link'
@@ -8,13 +7,15 @@ import {
ShareLinkBundleStatus,
ShareLinkBundleSummary,
} from 'src/app/data/share-link-bundle'
import { SETTINGS_KEYS } from 'src/app/data/ui-settings'
import { ShareLinkBundleService } from 'src/app/services/rest/share-link-bundle.service'
import { SettingsService } from 'src/app/services/settings.service'
import { ToastService } from 'src/app/services/toast.service'
import { environment } from 'src/environments/environment'
import { ShareLinkBundleManageDialogComponent } from './share-link-bundle-manage-dialog.component'
import { ShareLinkBundleListComponent } from './share-link-bundle-list.component'
class MockShareLinkBundleService {
listAllBundles = jest.fn()
list = jest.fn()
delete = jest.fn()
rebuildBundle = jest.fn()
}
@@ -24,13 +25,12 @@ class MockToastService {
showError = jest.fn()
}
describe('ShareLinkBundleManageDialogComponent', () => {
let component: ShareLinkBundleManageDialogComponent
let fixture: ComponentFixture<ShareLinkBundleManageDialogComponent>
describe('ShareLinkBundleListComponent', () => {
let component: ShareLinkBundleListComponent
let fixture: ComponentFixture<ShareLinkBundleListComponent>
let service: MockShareLinkBundleService
let toastService: MockToastService
let clipboard: Clipboard
let activeModal: NgbActiveModal
let originalApiBaseUrl: string
beforeEach(() => {
@@ -38,26 +38,24 @@ describe('ShareLinkBundleManageDialogComponent', () => {
toastService = new MockToastService()
originalApiBaseUrl = environment.apiBaseUrl
service.listAllBundles.mockReturnValue(of([]))
service.list.mockReturnValue(of({ count: 0, results: [] }))
service.delete.mockReturnValue(of(true))
service.rebuildBundle.mockReturnValue(of(sampleBundle()))
TestBed.configureTestingModule({
imports: [
ShareLinkBundleManageDialogComponent,
ShareLinkBundleListComponent,
NgxBootstrapIconsModule.pick(allIcons),
],
providers: [
NgbActiveModal,
{ provide: ShareLinkBundleService, useValue: service },
{ provide: ToastService, useValue: toastService },
],
})
fixture = TestBed.createComponent(ShareLinkBundleManageDialogComponent)
fixture = TestBed.createComponent(ShareLinkBundleListComponent)
component = fixture.componentInstance
clipboard = TestBed.inject(Clipboard)
activeModal = TestBed.inject(NgbActiveModal)
})
afterEach(() => {
@@ -84,28 +82,28 @@ describe('ShareLinkBundleManageDialogComponent', () => {
it('loads bundles on init and polls periodically', () => {
jest.useFakeTimers()
const bundles = [sampleBundle({ status: ShareLinkBundleStatus.Ready })]
service.listAllBundles.mockReset()
service.listAllBundles
.mockReturnValueOnce(of(bundles))
.mockReturnValue(of(bundles))
service.list.mockReset()
service.list
.mockReturnValueOnce(of({ count: bundles.length, results: bundles }))
.mockReturnValue(of({ count: bundles.length, results: bundles }))
fixture.detectChanges()
expect(service.listAllBundles).toHaveBeenCalledTimes(1)
expect(service.list).toHaveBeenCalledWith(1, 25, 'created', true)
expect(component.bundles()).toEqual(bundles)
expect(component.loading()).toBe(false)
expect(component.error()).toBeNull()
jest.advanceTimersByTime(5000)
expect(service.listAllBundles).toHaveBeenCalledTimes(2)
expect(service.list).toHaveBeenCalledTimes(2)
})
it('handles errors when loading bundles', () => {
jest.useFakeTimers()
service.listAllBundles.mockReset()
service.listAllBundles
service.list.mockReset()
service.list
.mockReturnValueOnce(throwError(() => new Error('load fail')))
.mockReturnValue(of([]))
.mockReturnValue(of({ count: 0, results: [] }))
fixture.detectChanges()
@@ -114,7 +112,57 @@ describe('ShareLinkBundleManageDialogComponent', () => {
expect(component.loading()).toBe(false)
jest.advanceTimersByTime(5000)
expect(service.listAllBundles).toHaveBeenCalledTimes(2)
expect(service.list).toHaveBeenCalledTimes(2)
})
it('loads another page', () => {
fixture.detectChanges()
component.setPage(2)
expect(service.list).toHaveBeenLastCalledWith(2, 25, 'created', true)
})
it('sorts bundles and returns to the first page', () => {
fixture.detectChanges()
component.page.set(2)
component.onSort({ column: 'status', reverse: false })
expect(component.page()).toBe(1)
expect(service.list).toHaveBeenLastCalledWith(1, 25, 'status', false)
})
it('marks expired share link bundles', () => {
service.list.mockReturnValue(
of({
count: 1,
results: [sampleBundle({ expiration: '2000-01-01T00:00:00.000Z' })],
})
)
fixture.detectChanges()
expect(fixture.nativeElement.textContent).toContain('Expired')
})
it('stores a changed page size and reloads from the first page', () => {
fixture.detectChanges()
const settingsService = TestBed.inject(SettingsService)
jest
.spyOn(settingsService, 'get')
.mockReturnValueOnce({ share_link_bundles: 25 })
const setSpy = jest.spyOn(settingsService, 'set')
jest.spyOn(settingsService, 'storeSettings').mockReturnValue(of({}))
component.page.set(2)
component.pageSize = 100
expect(setSpy).toHaveBeenCalledWith(SETTINGS_KEYS.OBJECT_LIST_SIZES, {
share_link_bundles: 100,
})
expect(component.page()).toBe(1)
expect(service.list).toHaveBeenLastCalledWith(1, 100, 'created', true)
})
it('copies bundle links when ready', () => {
@@ -126,16 +174,24 @@ describe('ShareLinkBundleManageDialogComponent', () => {
slug: 'ready-slug',
status: ShareLinkBundleStatus.Ready,
})
component.bundles.set([readyBundle])
fixture.detectChanges()
component.copy(readyBundle)
expect(clipboard.copy).toHaveBeenCalledWith(
component.getShareUrl(readyBundle)
)
expect(component.copiedSlug()).toBe('ready-slug')
expect(toastService.showInfo).toHaveBeenCalled()
expect(toastService.showInfo).not.toHaveBeenCalled()
fixture.detectChanges()
expect(
fixture.nativeElement.querySelector('.badge.show').textContent
).toContain('Copied!')
jest.advanceTimersByTime(3000)
expect(component.copiedSlug()).toBeNull()
fixture.detectChanges()
expect(fixture.nativeElement.querySelector('.badge.show')).toBeNull()
})
it('ignores copy requests for non-ready bundles', () => {
@@ -146,7 +202,7 @@ describe('ShareLinkBundleManageDialogComponent', () => {
})
it('deletes bundles and refreshes list', () => {
service.listAllBundles.mockReturnValue(of([]))
service.list.mockReturnValue(of({ count: 0, results: [] }))
service.delete.mockReturnValue(of(true))
fixture.detectChanges()
@@ -157,12 +213,12 @@ describe('ShareLinkBundleManageDialogComponent', () => {
expect(toastService.showInfo).toHaveBeenCalledWith(
expect.stringContaining('deleted.')
)
expect(service.listAllBundles).toHaveBeenCalledTimes(2)
expect(service.list).toHaveBeenCalledTimes(2)
expect(component.loading()).toBe(false)
})
it('handles delete errors gracefully', () => {
service.listAllBundles.mockReturnValue(of([]))
service.list.mockReturnValue(of({ count: 0, results: [] }))
service.delete.mockReturnValue(throwError(() => new Error('delete fail')))
fixture.detectChanges()
@@ -174,7 +230,7 @@ describe('ShareLinkBundleManageDialogComponent', () => {
})
it('retries bundle build and replaces existing entry', () => {
service.listAllBundles.mockReturnValue(of([]))
service.list.mockReturnValue(of({ count: 0, results: [] }))
const updated = sampleBundle({ status: ShareLinkBundleStatus.Ready })
service.rebuildBundle.mockReturnValue(of(updated))
@@ -189,7 +245,7 @@ describe('ShareLinkBundleManageDialogComponent', () => {
})
it('adds new bundle when retry returns unknown entry', () => {
service.listAllBundles.mockReturnValue(of([]))
service.list.mockReturnValue(of({ count: 0, results: [] }))
service.rebuildBundle.mockReturnValue(
of(sampleBundle({ id: 99, slug: 'new-slug' }))
)
@@ -203,7 +259,7 @@ describe('ShareLinkBundleManageDialogComponent', () => {
})
it('handles retry errors', () => {
service.listAllBundles.mockReturnValue(of([]))
service.list.mockReturnValue(of({ count: 0, results: [] }))
service.rebuildBundle.mockReturnValue(throwError(() => new Error('fail')))
fixture.detectChanges()
@@ -213,8 +269,8 @@ describe('ShareLinkBundleManageDialogComponent', () => {
expect(toastService.showError).toHaveBeenCalled()
})
it('maps helpers and closes dialog', () => {
service.listAllBundles.mockReturnValue(of([]))
it('maps status and file version helpers', () => {
service.list.mockReturnValue(of({ count: 0, results: [] }))
fixture.detectChanges()
expect(component.statusLabel(ShareLinkBundleStatus.Processing)).toContain(
@@ -227,9 +283,5 @@ describe('ShareLinkBundleManageDialogComponent', () => {
environment.apiBaseUrl = 'https://example.com/api/'
const url = component.getShareUrl(sampleBundle({ slug: 'sluggy' }))
expect(url).toBe('https://example.com/share/sluggy')
const closeSpy = jest.spyOn(activeModal, 'close')
component.close()
expect(closeSpy).toHaveBeenCalled()
})
})
@@ -1,7 +1,11 @@
import { Clipboard } from '@angular/cdk/clipboard'
import { CommonModule } from '@angular/common'
import { Component, OnDestroy, OnInit, inject, signal } from '@angular/core'
import { NgbActiveModal, NgbPopoverModule } from '@ng-bootstrap/ng-bootstrap'
import { FormsModule } from '@angular/forms'
import {
NgbPaginationModule,
NgbPopoverModule,
} from '@ng-bootstrap/ng-bootstrap'
import { NgxBootstrapIconsModule } from 'ngx-bootstrap-icons'
import { Subject, catchError, of, switchMap, takeUntil, timer } from 'rxjs'
import { FileVersion } from 'src/app/data/share-link'
@@ -11,42 +15,77 @@ import {
ShareLinkBundleStatus,
ShareLinkBundleSummary,
} from 'src/app/data/share-link-bundle'
import { SETTINGS_KEYS } from 'src/app/data/ui-settings'
import {
SortEvent,
SortableDirective,
} from 'src/app/directives/sortable.directive'
import { FileSizePipe } from 'src/app/pipes/file-size.pipe'
import { ShareLinkBundleService } from 'src/app/services/rest/share-link-bundle.service'
import { SettingsService } from 'src/app/services/settings.service'
import { ToastService } from 'src/app/services/toast.service'
import { environment } from 'src/environments/environment'
import { LoadingComponentWithPermissions } from '../../loading-component/loading.component'
import { ConfirmButtonComponent } from '../confirm-button/confirm-button.component'
import { ConfirmButtonComponent } from 'src/app/components/common/confirm-button/confirm-button.component'
import { LoadingComponentWithPermissions } from 'src/app/components/loading-component/loading.component'
@Component({
selector: 'pngx-share-link-bundle-manage-dialog',
templateUrl: './share-link-bundle-manage-dialog.component.html',
styleUrls: ['./share-link-bundle-manage-dialog.component.scss'],
selector: 'pngx-share-link-bundle-list',
templateUrl: './share-link-bundle-list.component.html',
styleUrls: ['./share-link-bundle-list.component.scss'],
imports: [
ConfirmButtonComponent,
CommonModule,
FormsModule,
NgbPaginationModule,
NgbPopoverModule,
NgxBootstrapIconsModule,
SortableDirective,
FileSizePipe,
],
})
export class ShareLinkBundleManageDialogComponent
export class ShareLinkBundleListComponent
extends LoadingComponentWithPermissions
implements OnInit, OnDestroy
{
private readonly activeModal = inject(NgbActiveModal)
private readonly shareLinkBundleService = inject(ShareLinkBundleService)
private readonly settingsService = inject(SettingsService)
private readonly toastService = inject(ToastService)
private readonly clipboard = inject(Clipboard)
title = $localize`Share link bundles`
readonly bundles = signal<ShareLinkBundleSummary[]>([])
readonly error = signal<string | null>(null)
readonly copiedSlug = signal<string | null>(null)
readonly total = signal(0)
readonly page = signal(1)
readonly sortField = signal('created')
readonly sortReverse = signal(true)
readonly statuses = ShareLinkBundleStatus
readonly fileVersions = FileVersion
get pageSize(): number {
return (
this.settingsService.get(SETTINGS_KEYS.OBJECT_LIST_SIZES)
?.share_link_bundles || 25
)
}
set pageSize(pageSize: number) {
this.settingsService.set(SETTINGS_KEYS.OBJECT_LIST_SIZES, {
...this.settingsService.get(SETTINGS_KEYS.OBJECT_LIST_SIZES),
share_link_bundles: pageSize,
})
this.settingsService.storeSettings().subscribe({
next: () => {
this.page.set(1)
this.triggerRefresh(false)
},
error: (error) => {
this.toastService.showError($localize`Error saving settings`, error)
},
})
}
private readonly refresh$ = new Subject<boolean>()
ngOnInit(): void {
@@ -57,25 +96,33 @@ export class ShareLinkBundleManageDialogComponent
this.loading.set(true)
}
this.error.set(null)
return this.shareLinkBundleService.listAllBundles().pipe(
catchError((error) => {
if (!silent) {
this.loading.set(false)
}
this.error.set($localize`Failed to load share link bundles.`)
this.toastService.showError(
$localize`Error retrieving share link bundles.`,
error
)
return of(null)
})
)
return this.shareLinkBundleService
.list(
this.page(),
this.pageSize,
this.sortField(),
this.sortReverse()
)
.pipe(
catchError((error) => {
if (!silent) {
this.loading.set(false)
}
this.error.set($localize`Failed to load share link bundles.`)
this.toastService.showError(
$localize`Error retrieving share link bundles.`,
error
)
return of(null)
})
)
}),
takeUntil(this.unsubscribeNotifier)
)
.subscribe((results) => {
if (results) {
this.bundles.set(results)
this.bundles.set(results.results)
this.total.set(results.count)
this.copiedSlug.set(null)
}
this.loading.set(false)
@@ -98,6 +145,18 @@ export class ShareLinkBundleManageDialogComponent
}`
}
setPage(page: number): void {
this.page.set(page)
this.triggerRefresh(false)
}
onSort(event: SortEvent): void {
this.sortField.set(event.column || 'created')
this.sortReverse.set(event.column ? event.reverse : true)
this.page.set(1)
this.triggerRefresh(false)
}
copy(bundle: ShareLinkBundleSummary): void {
if (bundle.status !== ShareLinkBundleStatus.Ready) {
return
@@ -108,7 +167,6 @@ export class ShareLinkBundleManageDialogComponent
setTimeout(() => {
this.copiedSlug.set(null)
}, 3000)
this.toastService.showInfo($localize`Share link copied to clipboard.`)
}
}
@@ -117,6 +175,9 @@ export class ShareLinkBundleManageDialogComponent
this.loading.set(true)
this.shareLinkBundleService.delete(bundle).subscribe({
next: () => {
if (this.bundles().length === 1 && this.page() > 1) {
this.page.update((page) => page - 1)
}
this.toastService.showInfo($localize`Share link bundle deleted.`)
this.triggerRefresh(false)
},
@@ -153,8 +214,8 @@ export class ShareLinkBundleManageDialogComponent
return SHARE_LINK_BUNDLE_FILE_VERSION_LABELS[version] ?? version
}
close(): void {
this.activeModal.close()
isExpired(expiration?: string): boolean {
return !!expiration && Date.parse(expiration) <= Date.now()
}
private replaceBundle(updated: ShareLinkBundleSummary): void {
@@ -0,0 +1,110 @@
<div class="border border-top-0 rounded-bottom p-3">
@if (!loading() && error()) {
<div class="alert alert-danger mb-0" role="alert">{{ error() }}</div>
}
@if (!loading() && !error() && links().length === 0) {
<p class="mb-0 text-muted fst-italic" i18n>
No document share links currently exist.
</p>
}
@if (!loading() && !error() && links().length > 0) {
<div class="table-responsive">
<table class="table table-sm align-middle mb-0 bg-body">
<thead>
<tr>
<th scope="col" class="fw-normal" pngxSortable="document__title" [currentSortField]="sortField()" [currentSortReverse]="sortReverse()" (sort)="onSort($event)" i18n>Document</th>
<th scope="col" class="fw-normal" pngxSortable="created" [currentSortField]="sortField()" [currentSortReverse]="sortReverse()" (sort)="onSort($event)" i18n>Created</th>
<th scope="col" class="fw-normal" pngxSortable="expiration" [currentSortField]="sortField()" [currentSortReverse]="sortReverse()" (sort)="onSort($event)" i18n>Expires</th>
<th scope="col" i18n>File version</th>
<th scope="col" class="text-end" i18n>Actions</th>
</tr>
</thead>
<tbody>
@for (link of links(); track link.id) {
<tr>
<td>
<a routerLink="/documents/{{ link.document }}">{{ link.document_title | documentTitle }}</a>
<span class="badge bg-primary text-primary-text-contrast ms-3 small fs-normal cursor-pointer" (click)="copyDocumentID(link.document)">
@if (copiedDocumentID() === link.document) {
<i-bs width="1em" height="1em" name="clipboard-check" class="me-1"></i-bs><ng-container i18n>Copied!</ng-container>
} @else {
ID: {{link.document}}
}
</span>
</td>
<td>{{ link.created | date: 'short' }}</td>
<td>
@if (link.expiration) {
{{ link.expiration | date: 'short' }}
@if (isExpired(link.expiration)) {
<span class="badge text-bg-danger ms-2" i18n>Expired</span>
}
} @else {
<span i18n>Never</span>
}
</td>
<td>{{ fileVersionLabel(link.file_version) }}</td>
<td class="text-end">
<div class="d-inline-block position-relative">
<span
class="badge bg-primary small fade position-absolute top-50 end-100 translate-middle-y me-2 pe-none z-3 text-nowrap"
[class.show]="copiedID() === link.id"
i18n
>Copied!</span>
<div class="btn-group btn-group-sm">
<button
type="button"
class="btn btn-outline-primary"
(click)="copy(link)"
title="Copy share link"
i18n-title
>
@if (copiedID() === link.id) {
<i-bs name="clipboard-check"></i-bs>
} @else {
<i-bs name="clipboard"></i-bs>
}
<span class="visually-hidden" i18n>Copy share link</span>
</button>
<pngx-confirm-button
*pngxIfPermissions="{ action: PermissionAction.Delete, type: PermissionType.ShareLink }"
buttonClasses="btn btn-sm btn-outline-danger"
(confirm)="delete(link)"
iconName="trash"
>
<span class="visually-hidden" i18n>Delete share link</span>
</pngx-confirm-button>
</div>
</div>
</td>
</tr>
}
</tbody>
</table>
</div>
<div class="d-flex flex-wrap justify-content-end align-items-center gap-3 mt-3 ms-auto">
<div class="d-flex align-items-center">
<label class="small text-muted me-2" for="shareLinkPageSize" i18n>Show:</label>
<select id="shareLinkPageSize" class="form-select form-select-sm w-auto" [(ngModel)]="pageSize">
<option [ngValue]="25">25</option>
<option [ngValue]="50">50</option>
<option [ngValue]="100">100</option>
</select>
<span class="small text-muted ms-2 d-none d-md-inline" i18n>per page</span>
</div>
@if (total() > pageSize) {
<ngb-pagination
class="mb-0"
[pageSize]="pageSize"
[collectionSize]="total()"
[page]="page()"
[maxSize]="5"
(pageChange)="setPage($event)"
size="sm"
aria-label="Share links pagination"
i18n-aria-label
></ngb-pagination>
}
</div>
}
</div>
@@ -0,0 +1,155 @@
import { Clipboard } from '@angular/cdk/clipboard'
import { ComponentFixture, TestBed } from '@angular/core/testing'
import { RouterTestingModule } from '@angular/router/testing'
import { NgxBootstrapIconsModule, allIcons } from 'ngx-bootstrap-icons'
import { of, throwError } from 'rxjs'
import { FileVersion, ShareLink } from 'src/app/data/share-link'
import { SETTINGS_KEYS } from 'src/app/data/ui-settings'
import { ShareLinkService } from 'src/app/services/rest/share-link.service'
import { SettingsService } from 'src/app/services/settings.service'
import { ToastService } from 'src/app/services/toast.service'
import { ShareLinkListComponent } from './share-link-list.component'
describe('ShareLinkListComponent', () => {
let component: ShareLinkListComponent
let fixture: ComponentFixture<ShareLinkListComponent>
let service: jest.Mocked<Pick<ShareLinkService, 'list' | 'delete'>>
let clipboard: Clipboard
let toastService: jest.Mocked<Pick<ToastService, 'showInfo' | 'showError'>>
const link = {
id: 1,
document: 42,
document_title: 'Test document',
slug: 'share-slug',
created: new Date().toISOString(),
expiration: null,
file_version: FileVersion.Archive,
} as ShareLink
beforeEach(() => {
service = {
list: jest.fn().mockReturnValue(of({ count: 1, results: [link] })),
delete: jest.fn().mockReturnValue(of(true)),
}
toastService = {
showInfo: jest.fn(),
showError: jest.fn(),
}
TestBed.configureTestingModule({
imports: [
ShareLinkListComponent,
NgxBootstrapIconsModule.pick(allIcons),
RouterTestingModule,
],
providers: [
{ provide: ShareLinkService, useValue: service },
{ provide: ToastService, useValue: toastService },
],
})
fixture = TestBed.createComponent(ShareLinkListComponent)
component = fixture.componentInstance
clipboard = TestBed.inject(Clipboard)
})
afterEach(() => {
jest.clearAllTimers()
jest.useRealTimers()
})
it('loads and renders document share links', () => {
fixture.detectChanges()
expect(service.list).toHaveBeenCalledWith(1, 25, 'created', true)
expect(component.links()).toEqual([link])
expect(fixture.nativeElement.textContent).toContain('Test document')
expect(fixture.nativeElement.textContent).toContain('ID: 42')
})
it('loads another page', () => {
fixture.detectChanges()
component.setPage(2)
expect(service.list).toHaveBeenLastCalledWith(2, 25, 'created', true)
})
it('sorts links and returns to the first page', () => {
fixture.detectChanges()
component.page.set(2)
component.onSort({ column: 'expiration', reverse: false })
expect(component.page()).toBe(1)
expect(service.list).toHaveBeenLastCalledWith(1, 25, 'expiration', false)
})
it('marks expired share links', () => {
service.list.mockReturnValue(
of({
count: 1,
results: [
{
...link,
expiration: '2000-01-01T00:00:00.000Z',
},
],
})
)
fixture.detectChanges()
expect(fixture.nativeElement.textContent).toContain('Expired')
})
it('stores a changed page size and reloads from the first page', () => {
const settingsService = TestBed.inject(SettingsService)
jest.spyOn(settingsService, 'get').mockReturnValueOnce({ share_links: 25 })
const setSpy = jest.spyOn(settingsService, 'set')
jest.spyOn(settingsService, 'storeSettings').mockReturnValue(of({}))
const reloadSpy = jest.spyOn(component, 'reload')
component.page.set(2)
component.pageSize = 50
expect(setSpy).toHaveBeenCalledWith(SETTINGS_KEYS.OBJECT_LIST_SIZES, {
share_links: 50,
})
expect(component.page()).toBe(1)
expect(reloadSpy).toHaveBeenCalled()
})
it('shows local copy feedback without a toast', () => {
jest.useFakeTimers()
jest.spyOn(clipboard, 'copy').mockReturnValue(true)
fixture.detectChanges()
component.copy(link)
fixture.detectChanges()
expect(component.copiedID()).toBe(link.id)
expect(fixture.nativeElement.querySelector('.badge.show')).not.toBeNull()
expect(toastService.showInfo).not.toHaveBeenCalled()
jest.advanceTimersByTime(3000)
expect(component.copiedID()).toBeNull()
})
it('deletes a link and reloads the list', () => {
fixture.detectChanges()
component.delete(link)
expect(service.delete).toHaveBeenCalledWith(link)
expect(service.list).toHaveBeenCalledTimes(2)
expect(toastService.showInfo).toHaveBeenCalled()
})
it('shows an error when loading fails', () => {
service.list.mockReturnValue(throwError(() => new Error('load failed')))
fixture.detectChanges()
expect(component.error()).toContain('Failed to load share links.')
expect(toastService.showError).toHaveBeenCalled()
})
})
@@ -0,0 +1,172 @@
import { Clipboard } from '@angular/cdk/clipboard'
import { CommonModule } from '@angular/common'
import { Component, OnInit, inject, signal } from '@angular/core'
import { FormsModule } from '@angular/forms'
import { RouterModule } from '@angular/router'
import { NgbPaginationModule } from '@ng-bootstrap/ng-bootstrap'
import { NgxBootstrapIconsModule } from 'ngx-bootstrap-icons'
import { takeUntil } from 'rxjs'
import { ConfirmButtonComponent } from 'src/app/components/common/confirm-button/confirm-button.component'
import { LoadingComponentWithPermissions } from 'src/app/components/loading-component/loading.component'
import { FileVersion, ShareLink } from 'src/app/data/share-link'
import { SHARE_LINK_BUNDLE_FILE_VERSION_LABELS } from 'src/app/data/share-link-bundle'
import { SETTINGS_KEYS } from 'src/app/data/ui-settings'
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
import {
SortEvent,
SortableDirective,
} from 'src/app/directives/sortable.directive'
import { DocumentTitlePipe } from 'src/app/pipes/document-title.pipe'
import {
PermissionAction,
PermissionType,
} from 'src/app/services/permissions.service'
import { ShareLinkService } from 'src/app/services/rest/share-link.service'
import { SettingsService } from 'src/app/services/settings.service'
import { ToastService } from 'src/app/services/toast.service'
import { environment } from 'src/environments/environment'
@Component({
selector: 'pngx-share-link-list',
templateUrl: './share-link-list.component.html',
imports: [
CommonModule,
ConfirmButtonComponent,
DocumentTitlePipe,
FormsModule,
IfPermissionsDirective,
NgbPaginationModule,
NgxBootstrapIconsModule,
RouterModule,
SortableDirective,
],
})
export class ShareLinkListComponent
extends LoadingComponentWithPermissions
implements OnInit
{
private readonly clipboard = inject(Clipboard)
private readonly shareLinkService = inject(ShareLinkService)
private readonly settingsService = inject(SettingsService)
private readonly toastService = inject(ToastService)
readonly links = signal<ShareLink[]>([])
readonly total = signal(0)
readonly page = signal(1)
readonly sortField = signal('created')
readonly sortReverse = signal(true)
readonly copiedID = signal<number | null>(null)
readonly copiedDocumentID = signal<number | null>(null)
readonly error = signal<string | null>(null)
readonly PermissionAction = PermissionAction
readonly PermissionType = PermissionType
get pageSize(): number {
return (
this.settingsService.get(SETTINGS_KEYS.OBJECT_LIST_SIZES)?.share_links ||
25
)
}
set pageSize(pageSize: number) {
this.settingsService.set(SETTINGS_KEYS.OBJECT_LIST_SIZES, {
...this.settingsService.get(SETTINGS_KEYS.OBJECT_LIST_SIZES),
share_links: pageSize,
})
this.settingsService.storeSettings().subscribe({
next: () => {
this.page.set(1)
this.reload()
},
error: (error) => {
this.toastService.showError($localize`Error saving settings`, error)
},
})
}
ngOnInit(): void {
this.reload()
}
reload(): void {
this.loading.set(true)
this.error.set(null)
this.shareLinkService
.list(this.page(), this.pageSize, this.sortField(), this.sortReverse())
.pipe(takeUntil(this.unsubscribeNotifier))
.subscribe({
next: (results) => {
this.links.set(results.results)
this.total.set(results.count)
this.loading.set(false)
},
error: (error) => {
this.loading.set(false)
this.error.set($localize`Failed to load share links.`)
this.toastService.showError(
$localize`Error retrieving share links.`,
error
)
},
})
}
setPage(page: number): void {
this.page.set(page)
this.reload()
}
onSort(event: SortEvent): void {
this.sortField.set(event.column || 'created')
this.sortReverse.set(event.column ? event.reverse : true)
this.page.set(1)
this.reload()
}
getShareUrl(link: ShareLink): string {
const apiURL = new URL(environment.apiBaseUrl)
return `${apiURL.origin}${apiURL.pathname.replace(/\/api\/$/, '/share/')}${
link.slug
}`
}
fileVersionLabel(version: FileVersion): string {
return SHARE_LINK_BUNDLE_FILE_VERSION_LABELS[version] ?? version
}
isExpired(expiration?: string): boolean {
return !!expiration && Date.parse(expiration) <= Date.now()
}
copy(link: ShareLink): void {
if (this.clipboard.copy(this.getShareUrl(link))) {
this.copiedID.set(link.id)
setTimeout(() => this.copiedID.set(null), 3000)
}
}
delete(link: ShareLink): void {
this.shareLinkService.delete(link).subscribe({
next: () => {
if (this.links().length === 1 && this.page() > 1) {
this.page.update((page) => page - 1)
}
this.toastService.showInfo($localize`Share link deleted.`)
this.reload()
},
error: (error) => {
this.toastService.showError(
$localize`Error deleting share link.`,
error
)
},
})
}
copyDocumentID(documentID: number): void {
if (this.clipboard.copy(documentID.toString())) {
this.copiedDocumentID.set(documentID)
setTimeout(() => this.copiedDocumentID.set(null), 3000)
}
}
}
@@ -0,0 +1,34 @@
<pngx-page-header
title="Share links"
i18n-title
info="Manage public links to individual documents and document bundles."
i18n-info
[loading]="loading()"
></pngx-page-header>
<ul
ngbNav
#nav="ngbNav"
class="nav-tabs"
[activeId]="activeNavID()"
(activeIdChange)="selectTab($event)"
>
@if (canViewDocumentLinks) {
<li [ngbNavItem]="ShareLinksNavIDs.DocumentLinks">
<button ngbNavLink i18n>Document links</button>
<ng-template ngbNavContent>
<pngx-share-link-list></pngx-share-link-list>
</ng-template>
</li>
}
@if (canViewBundles) {
<li [ngbNavItem]="ShareLinksNavIDs.Bundles">
<button ngbNavLink i18n>Bundles</button>
<ng-template ngbNavContent>
<pngx-share-link-bundle-list></pngx-share-link-bundle-list>
</ng-template>
</li>
}
</ul>
<div class="bg-body" [ngbNavOutlet]="nav"></div>
@@ -0,0 +1,102 @@
import { ComponentFixture, TestBed } from '@angular/core/testing'
import { ActivatedRoute, convertToParamMap, Router } from '@angular/router'
import { NgbNavModule } from '@ng-bootstrap/ng-bootstrap'
import { NgxBootstrapIconsModule, allIcons } from 'ngx-bootstrap-icons'
import { of } from 'rxjs'
import {
PermissionAction,
PermissionsService,
PermissionType,
} from 'src/app/services/permissions.service'
import { ShareLinkBundleService } from 'src/app/services/rest/share-link-bundle.service'
import { ShareLinkService } from 'src/app/services/rest/share-link.service'
import { ToastService } from 'src/app/services/toast.service'
import { PageHeaderComponent } from '../../common/page-header/page-header.component'
import { ShareLinksComponent, ShareLinksNavIDs } from './share-links.component'
describe('ShareLinksComponent', () => {
let fixture: ComponentFixture<ShareLinksComponent>
let permissionsService: PermissionsService
let router: Router
const configure = async (type: string = null) => {
await TestBed.configureTestingModule({
imports: [
ShareLinksComponent,
NgbNavModule,
NgxBootstrapIconsModule.pick(allIcons),
PageHeaderComponent,
],
providers: [
PermissionsService,
{
provide: ActivatedRoute,
useValue: {
snapshot: { queryParamMap: convertToParamMap({ type }) },
},
},
{
provide: Router,
useValue: { navigate: jest.fn().mockResolvedValue(true) },
},
{
provide: ShareLinkBundleService,
useValue: {
list: jest.fn().mockReturnValue(of({ count: 0, results: [] })),
rebuildBundle: jest.fn(),
delete: jest.fn(),
},
},
{
provide: ShareLinkService,
useValue: {
list: jest.fn().mockReturnValue(of({ count: 0, results: [] })),
delete: jest.fn(),
},
},
{
provide: ToastService,
useValue: { showInfo: jest.fn(), showError: jest.fn() },
},
],
}).compileComponents()
permissionsService = TestBed.inject(PermissionsService)
router = TestBed.inject(Router)
}
afterEach(() => TestBed.resetTestingModule())
it('uses the requested bundles tab when permitted', async () => {
await configure(ShareLinksNavIDs.Bundles)
jest
.spyOn(permissionsService, 'currentUserCan')
.mockImplementation(
(action, type) =>
action === PermissionAction.View &&
type === PermissionType.ShareLinkBundle
)
fixture = TestBed.createComponent(ShareLinksComponent)
fixture.detectChanges()
expect(fixture.componentInstance.activeNavID()).toBe(
ShareLinksNavIDs.Bundles
)
expect(fixture.nativeElement.textContent).not.toContain('Document links')
})
it('updates the URL when a tab is selected', async () => {
await configure()
jest.spyOn(permissionsService, 'currentUserCan').mockReturnValue(true)
fixture = TestBed.createComponent(ShareLinksComponent)
fixture.componentInstance.selectTab(ShareLinksNavIDs.Bundles)
expect(router.navigate).toHaveBeenCalledWith([], {
relativeTo: TestBed.inject(ActivatedRoute),
queryParams: { type: ShareLinksNavIDs.Bundles },
queryParamsHandling: 'merge',
})
})
})
@@ -0,0 +1,78 @@
import { Component, computed, inject, signal, viewChild } from '@angular/core'
import { ActivatedRoute, Router } from '@angular/router'
import { NgbNavModule } from '@ng-bootstrap/ng-bootstrap'
import {
PermissionAction,
PermissionsService,
PermissionType,
} from 'src/app/services/permissions.service'
import { PageHeaderComponent } from '../../common/page-header/page-header.component'
import { ShareLinkBundleListComponent } from './share-link-bundle-list/share-link-bundle-list.component'
import { ShareLinkListComponent } from './share-link-list/share-link-list.component'
export enum ShareLinksNavIDs {
DocumentLinks = 'documents',
Bundles = 'bundles',
}
@Component({
selector: 'pngx-share-links',
templateUrl: './share-links.component.html',
imports: [
NgbNavModule,
PageHeaderComponent,
ShareLinkBundleListComponent,
ShareLinkListComponent,
],
})
export class ShareLinksComponent {
private readonly route = inject(ActivatedRoute)
private readonly router = inject(Router)
private readonly permissionsService = inject(PermissionsService)
readonly ShareLinksNavIDs = ShareLinksNavIDs
readonly activeNavID = signal(this.getInitialNavID())
private readonly documentLinks = viewChild(ShareLinkListComponent)
private readonly bundles = viewChild(ShareLinkBundleListComponent)
readonly loading = computed(() => {
const activeList =
this.activeNavID() === ShareLinksNavIDs.DocumentLinks
? this.documentLinks()
: this.bundles()
return activeList?.loading() ?? true
})
get canViewDocumentLinks(): boolean {
return this.permissionsService.currentUserCan(
PermissionAction.View,
PermissionType.ShareLink
)
}
get canViewBundles(): boolean {
return this.permissionsService.currentUserCan(
PermissionAction.View,
PermissionType.ShareLinkBundle
)
}
selectTab(tab: ShareLinksNavIDs): void {
this.activeNavID.set(tab)
void this.router.navigate([], {
relativeTo: this.route,
queryParams: { type: tab },
queryParamsHandling: 'merge',
})
}
private getInitialNavID(): ShareLinksNavIDs {
const requestedTab = this.route.snapshot.queryParamMap.get('type')
if (requestedTab === ShareLinksNavIDs.Bundles && this.canViewBundles) {
return ShareLinksNavIDs.Bundles
}
if (this.canViewDocumentLinks) {
return ShareLinksNavIDs.DocumentLinks
}
return ShareLinksNavIDs.Bundles
}
}
-8
View File
@@ -49,7 +49,6 @@ export const FILTER_MODIFIED_AFTER = 16
export const FILTER_TITLE_CONTENT = 19 // Deprecated in favor of Tantivy-backed `text` filtervar. Keep for now for existing saved views
export const FILTER_SIMPLE_TITLE = 48
export const FILTER_SIMPLE_TEXT = 49
export const FILTER_HAS_DUPLICATES = 50
export const FILTER_FULLTEXT_QUERY = 20
export const FILTER_FULLTEXT_MORELIKE = 21
@@ -383,13 +382,6 @@ export const FILTER_RULE_TYPES: FilterRuleType[] = [
datatype: 'string',
multi: false,
},
{
id: FILTER_HAS_DUPLICATES,
filtervar: 'has_duplicates',
datatype: 'boolean',
multi: false,
default: true,
},
]
export interface FilterRuleType {
-1
View File
@@ -422,7 +422,6 @@ export const PaperlessConfigOptions: ConfigOption[] = [
]
export interface PaperlessConfig extends ObjectWithId {
externally_configured_variables: string[]
output_type: OutputTypeConfig
pages: number
language: string
+2
View File
@@ -26,5 +26,7 @@ export interface ShareLink extends ObjectWithPermissions {
document: number // Document
document_title?: string
file_version: string
}
+2 -7
View File
@@ -84,8 +84,6 @@ export const SETTINGS_KEYS = {
'general-settings:document-editing:remove-inbox-tags',
DOCUMENT_EDITING_OVERLAY_THUMBNAIL:
'general-settings:document-editing:overlay-thumbnail',
DOCUMENT_EDITING_AUTO_SUGGEST:
'general-settings:document-editing:auto-suggest',
DOCUMENT_DETAILS_HIDDEN_FIELDS:
'general-settings:document-details:hidden-fields',
SEARCH_DB_ONLY: 'general-settings:search:db-only',
@@ -230,6 +228,8 @@ export const SETTINGS: UiSetting[] = [
document_types: 25,
tags: 25,
storage_paths: 25,
share_links: 25,
share_link_bundles: 25,
},
},
{
@@ -302,11 +302,6 @@ export const SETTINGS: UiSetting[] = [
type: 'boolean',
default: true,
},
{
key: SETTINGS_KEYS.DOCUMENT_EDITING_AUTO_SUGGEST,
type: 'boolean',
default: true,
},
{
key: SETTINGS_KEYS.DOCUMENT_DETAILS_HIDDEN_FIELDS,
type: 'array',
@@ -48,13 +48,4 @@ describe('ShareLinkBundleService', () => {
expect(req.request.body).toEqual({})
req.flush({})
})
it('lists bundles with expected parameters', () => {
subscription = service.listAllBundles().subscribe()
const req = httpTestingController.expectOne(
`${environment.apiBaseUrl}${endpoint}/?page=1&page_size=1000&ordering=-created`
)
expect(req.request.method).toBe('GET')
req.flush({ results: [] })
})
})
@@ -1,6 +1,5 @@
import { Injectable } from '@angular/core'
import { Observable } from 'rxjs'
import { map } from 'rxjs/operators'
import {
ShareLinkBundleCreatePayload,
ShareLinkBundleSummary,
@@ -32,10 +31,4 @@ export class ShareLinkBundleService extends AbstractNameFilterService<ShareLinkB
{}
)
}
listAllBundles(): Observable<ShareLinkBundleSummary[]> {
return this.list(1, 1000, 'created', true).pipe(
map((response) => response.results)
)
}
}
-23
View File
@@ -7,7 +7,6 @@ import {
FILTER_HAS_ANY_TAG,
FILTER_HAS_CUSTOM_FIELDS_ALL,
FILTER_HAS_CUSTOM_FIELDS_ANY,
FILTER_HAS_DUPLICATES,
FILTER_HAS_TAGS_ALL,
FILTER_SIMPLE_TEXT,
FILTER_SIMPLE_TITLE,
@@ -133,16 +132,6 @@ describe('QueryParams Utils', () => {
is_tagged: 0,
})
params = queryParamsFromFilterRules([
{
rule_type: FILTER_HAS_DUPLICATES,
value: 'false',
},
])
expect(params).toEqual({
has_duplicates: 0,
})
params = queryParamsFromFilterRules([
{
rule_type: FILTER_TITLE_CONTENT,
@@ -258,18 +247,6 @@ describe('QueryParams Utils', () => {
},
])
rules = filterRulesFromQueryParams(
convertToParamMap({
has_duplicates: 'true',
})
)
expect(rules).toEqual([
{
rule_type: FILTER_HAS_DUPLICATES,
value: 'true',
},
])
rules = filterRulesFromQueryParams(
convertToParamMap({
correspondent__isnull: '1',
Binary file not shown.

Before

Width:  |  Height:  |  Size: 7.6 KiB

After

Width:  |  Height:  |  Size: 6.1 KiB

+3
View File
@@ -0,0 +1,3 @@
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 198.4 238.9" style="enable-background:new 0 0 198.4 238.9" xml:space="preserve">
<path d="M194.7 0C164.211 70.943 17.64 79.733 64.55 194.06c.59 1.468-10.848 17-18.47 29.897-1.758-6.453-3.816-13.486-3.516-14.075 38.109-45.141-27.26-70.643-30.776-107.583-16.423 29.318-22.286 80.623 27.25 110.23.29 0 2.637 11.138 3.816 16.712-1.169 2.348-2.348 4.695-2.927 6.454-1.168 2.926 7.622 2.637 7.622 3.226.879-.29 21.697-36.94 22.276-37.23C187.667 174.711 208.485 68.596 194.699 0zm-60.096 74.749c-55.11 49.246-64.49 85.897-62.732 103.777-18.47-43.682 35.772-91.76 62.732-103.777zM28.2 145.102c10.548 9.67 28.14 39.278 13.196 56.58 3.506-7.912 4.684-25.793-13.196-56.58z"/>
</svg>

After

Width:  |  Height:  |  Size: 727 B

+4
View File
@@ -0,0 +1,4 @@
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 2897.4 896.6" style="enable-background:new 0 0 2897.4 896.6" xml:space="preserve">
<path d="M1022.3 428.7c-17.8-19.9-42.7-29.8-74.7-29.8-22.3 0-42.4 5.7-60.5 17.3-18.1 11.6-32.3 27.5-42.5 47.8s-15.3 42.9-15.3 67.8 5.1 47.5 15.3 67.8c10.3 20.3 24.4 36.2 42.5 47.8 18.1 11.5 38.3 17.3 60.5 17.3 32 0 56.9-9.9 74.7-29.8V655.5h84.5V408.3h-84.5v20.4zM1010.5 575c-10.2 11.7-23.6 17.6-40.2 17.6s-29.9-5.9-40-17.6-15.1-26.1-15.1-43.3c0-17.1 5-31.6 15.1-43.3s23.4-17.6 40-17.6 30 5.9 40.2 17.6 15.3 26.1 15.3 43.3-5.1 31.6-15.3 43.3zM1381 416.1c-18.1-11.5-38.3-17.3-60.5-17.4-32 0-56.9 9.9-74.7 29.8v-20.4h-84.5v390.7h84.5v-164c17.8 19.9 42.7 29.8 74.7 29.8 22.3 0 42.4-5.7 60.5-17.3s32.3-27.5 42.5-47.8c10.2-20.3 15.3-42.9 15.3-67.8s-5.1-47.5-15.3-67.8c-10.3-20.3-24.4-36.2-42.5-47.8zM1337.9 575c-10.1 11.7-23.4 17.6-40 17.6s-29.9-5.9-40-17.6-15.1-26.1-15.1-43.3c0-17.1 5-31.6 15.1-43.3s23.4-17.6 40-17.6 29.9 5.9 40 17.6 15.1 26.1 15.1 43.3-5.1 31.6-15.1 43.3zM1672.2 416.8c-20.5-12-43-18-67.6-18-24.9 0-47.6 5.9-68 17.6-20.4 11.7-36.5 27.7-48.2 48s-17.6 42.7-17.6 67.3c.3 25.2 6.2 47.8 17.8 68 11.5 20.2 28 36 49.3 47.6 21.3 11.5 45.9 17.3 73.8 17.3 48.6 0 86.8-14.7 114.7-44l-52.5-48.9c-8.6 8.3-17.6 14.6-26.7 19-9.3 4.3-21.1 6.4-35.3 6.4-11.6 0-22.5-3.6-32.7-10.9-10.3-7.3-17.1-16.5-20.7-27.8h180l.4-11.6c0-29.6-6-55.7-18-78.2s-28.3-39.8-48.7-51.8zm-113.9 86.4c2.1-12.1 7.5-21.8 16.2-29.1s18.7-10.9 30-10.9 21.2 3.6 29.8 10.9c8.6 7.2 13.9 16.9 16 29.1h-92zM1895.3 411.7c-11 5.6-20.3 13.7-28 24.4h-.1v-28h-84.5v247.3h84.5V536.3c0-22.6 4.7-38.1 14.2-46.5 9.5-8.5 22.7-12.7 39.6-12.7 6.2 0 13.5 1 21.8 3.1l10.7-72c-5.9-3.3-14.5-4.9-25.8-4.9-10.6 0-21.4 2.8-32.4 8.4zM1985 277.4h84.5v377.8H1985zM2313.2 416.8c-20.5-12-43-18-67.6-18-24.9 0-47.6 5.9-68 17.6s-36.5 27.7-48.2 48c-11.7 20.3-17.6 42.7-17.6 67.3.3 25.2 6.2 47.8 17.8 68 11.5 20.2 28 36 49.3 47.6 21.3 11.5 45.9 17.3 73.8 17.3 48.6 0 86.8-14.7 114.7-44l-52.5-48.9c-8.6 8.3-17.6 14.6-26.7 19-9.3 4.3-21.1 6.4-35.3 6.4-11.6 0-22.5-3.6-32.7-10.9-10.3-7.3-17.1-16.5-20.7-27.8h180l.4-11.6c0-29.6-6-55.7-18-78.2s-28.3-39.8-48.7-51.8zm-113.9 86.4c2.1-12.1 7.5-21.8 16.2-29.1s18.7-10.9 30-10.9 21.2 3.6 29.8 10.9c8.6 7.2 13.9 16.9 16 29.1h-92zM2583.6 507.7c-13.8-4.4-30.6-8.1-50.5-11.1-15.1-2.7-26.1-5.2-32.9-7.6-6.8-2.4-10.2-6.1-10.2-11.1s2.3-8.7 6.7-10.9c4.4-2.2 11.5-3.3 21.3-3.3 11.6 0 24.3 2.4 38.1 7.2 13.9 4.8 26.2 11 36.9 18.4l32.4-58.2c-11.3-7.4-26.2-14.7-44.9-21.8-18.7-7.1-39.6-10.7-62.7-10.7-33.7 0-60.2 7.6-79.3 22.7-19.1 15.1-28.7 36.1-28.7 63.1 0 19 4.8 33.9 14.4 44.7 9.6 10.8 21 18.5 34 22.9 13.1 4.5 28.9 8.3 47.6 11.6 14.6 2.7 25.1 5.3 31.6 7.8s9.8 6.5 9.8 11.8c0 10.4-9.7 15.6-29.3 15.6-13.7 0-28.5-2.3-44.7-6.9-16.1-4.6-29.2-11.3-39.3-20.2l-33.3 60c9.2 7.4 24.6 14.7 46.2 22 21.7 7.3 45.2 10.9 70.7 10.9 34.7 0 62.9-7.4 84.5-22.4 21.7-15 32.5-37.3 32.5-66.9 0-19.3-5-34.2-15.1-44.9s-22-18.3-35.8-22.7zM2883.4 575.3c0-19.3-5-34.2-15.1-44.9s-22-18.3-35.8-22.7c-13.8-4.4-30.6-8.1-50.5-11.1-15.1-2.7-26.1-5.2-32.9-7.6-6.8-2.4-10.2-6.1-10.2-11.1s2.3-8.7 6.7-10.9c4.4-2.2 11.5-3.3 21.3-3.3 11.6 0 24.3 2.4 38.1 7.2 13.9 4.8 26.2 11 36.9 18.4l32.4-58.2c-11.3-7.4-26.2-14.7-44.9-21.8-18.7-7.1-39.6-10.7-62.7-10.7-33.7 0-60.2 7.6-79.3 22.7-19.1 15.1-28.7 36.1-28.7 63.1 0 19 4.8 33.9 14.4 44.7 9.6 10.8 21 18.5 34 22.9 13.1 4.5 28.9 8.3 47.6 11.6 14.6 2.7 25.1 5.3 31.6 7.8s9.8 6.5 9.8 11.8c0 10.4-9.7 15.6-29.3 15.6-13.7 0-28.5-2.3-44.7-6.9-16.1-4.6-29.2-11.3-39.3-20.2l-33.3 60c9.2 7.4 24.6 14.7 46.2 22 21.7 7.3 45.2 10.9 70.7 10.9 34.7 0 62.9-7.4 84.5-22.4 21.7-15 32.5-37.3 32.5-66.9zM2460.7 738.7h59.6v17.2h-59.6zM2596.5 706.4c-5.7 0-11 1-15.8 3s-9 5-12.5 8.9v-9.4h-19.4v93.6h19.4v-52c0-8.6 2.1-15.3 6.3-20 4.2-4.7 9.5-7.1 15.9-7.1 7.8 0 13.4 2.3 16.8 6.7 3.4 4.5 5.1 11.3 5.1 20.5v52h19.4v-56.8c0-12.8-3.2-22.6-9.5-29.3-6.4-6.7-14.9-10.1-25.7-10.1zM2733.8 717.7c-3.6-3.4-7.9-6.1-13.1-8.2s-10.6-3.1-16.2-3.1c-8.7 0-16.5 2.1-23.5 6.3s-12.5 10-16.5 17.3c-4 7.3-6 15.4-6 24.4 0 8.9 2 17.1 6 24.3 4 7.3 9.5 13 16.5 17.2s14.9 6.3 23.5 6.3c5.6 0 11-1 16.2-3.1 5.1-2.1 9.5-4.8 13.1-8.2v24.4c0 8.5-2.5 14.8-7.6 18.7-5 3.9-11 5.9-18 5.9-6.7 0-12.4-1.6-17.3-4.7-4.8-3.1-7.6-7.7-8.3-13.8h-19.4c.6 7.7 2.9 14.2 7.1 19.5s9.6 9.3 16.2 12c6.6 2.7 13.8 4 21.7 4 12.8 0 23.5-3.4 32-10.1 8.6-6.7 12.8-17.1 12.8-31.1V708.9h-19.2v8.8zm-1.6 52.4c-2.5 4.7-6 8.3-10.4 11.2-4.4 2.7-9.4 4-14.9 4-5.7 0-10.8-1.4-15.2-4.3s-7.8-6.7-10.2-11.4c-2.3-4.8-3.5-9.8-3.5-15.2 0-5.5 1.1-10.6 3.5-15.3s5.8-8.5 10.2-11.3 9.5-4.2 15.2-4.2c5.5 0 10.5 1.4 14.9 4s7.9 6.3 10.4 11 3.8 10 3.8 15.8-1.3 11-3.8 15.7zM2867.9 708.9h-21.4l-25.6 33-25.4-33h-22.4l36 46.1-37.6 47.5h21.4l27.2-34.6 27.1 34.7h22.4l-37.6-48.2zM757.6 293.7c-20-10.8-42.6-16.2-67.8-16.2H600c-8.5 39.2-21.1 76.4-37.6 111.3-9.9 20.8-21.1 40.6-33.6 59.4v207.2h88.9V521.5h72c25.2 0 47.8-5.4 67.8-16.2s35.7-25.6 47.1-44.2c11.4-18.7 17.1-39.1 17.1-61.3.1-22.7-5.6-43.3-17-61.9-11.4-18.7-27.1-33.4-47.1-44.2zm-41 140.6c-9.3 8.9-21.6 13.3-36.7 13.3l-62.2.4v-92.5l62.2-.4c15.1 0 27.3 4.4 36.7 13.3 9.4 8.9 14 19.9 14 32.9 0 13.2-4.6 24.1-14 33z"/>
<path d="M140 713.7c-3.4-16.4-10.3-49.1-11.2-49.1C-16.9 577.5.4 426.6 48.6 340.4 59 449 251.2 524 139.1 656.8c-.9 1.7 5.2 22.4 10.3 41.4 22.4-37.9 56-83.6 54.3-87.9C65.9 273.9 496.9 248.1 586.6 39.4c40.5 201.8-20.7 513.9-367.2 593.2-1.7.9-62.9 108.6-65.5 109.5 0-1.7-25.9-.9-22.4-9.5 1.6-5.2 5.1-12 8.5-18.9zm-4.3-81.1c44-50.9-7.8-137.9-38.8-166.4 52.6 90.5 49.1 143.1 38.8 166.4z" style="fill:#17541f"/>
</svg>

After

Width:  |  Height:  |  Size: 5.4 KiB

+3
View File
@@ -0,0 +1,3 @@
<svg xmlns="http://www.w3.org/2000/svg" width="264.567" height="318.552" viewBox="0 0 70 84.284">
<path style="fill:#17541f;stroke-width:1.10017" d="M752.438 82.365C638.02 348.605 87.938 381.61 263.964 810.674c2.2 5.5-40.706 63.81-69.31 112.217-6.602-24.204-14.304-50.607-13.204-52.807C324.473 700.658 79.136 604.944 65.934 466.322 4.324 576.34-17.678 768.868 168.25 879.984c1.1 0 9.902 41.808 14.303 62.711-4.4 8.802-8.802 17.602-11.002 24.203-4.4 11.002 28.603 9.902 28.603 12.102 3.3-1.1 81.413-138.62 83.614-139.72 442.267-101.216 520.377-499.476 468.67-756.915ZM526.904 362.906c-206.831 184.828-242.036 322.35-235.435 389.46-69.31-163.926 134.22-344.353 235.435-389.46ZM127.543 626.947c39.606 36.306 105.616 147.422 49.508 212.332 13.202-29.704 17.602-96.814-49.508-212.332z" transform="matrix(.094 0 0 .094 -2.042 -7.742)" fill="#17541F"/>
</svg>

After

Width:  |  Height:  |  Size: 855 B

+3
View File
@@ -0,0 +1,3 @@
<svg xmlns="http://www.w3.org/2000/svg" width="264.567" height="318.552" viewBox="0 0 70 84.284">
<path style="fill:#fff;stroke-width:1.10017" d="M752.438 82.365C638.02 348.605 87.938 381.61 263.964 810.674c2.2 5.5-40.706 63.81-69.31 112.217-6.602-24.204-14.304-50.607-13.204-52.807C324.473 700.658 79.136 604.944 65.934 466.322 4.324 576.34-17.678 768.868 168.25 879.984c1.1 0 9.902 41.808 14.303 62.711-4.4 8.802-8.802 17.602-11.002 24.203-4.4 11.002 28.603 9.902 28.603 12.102 3.3-1.1 81.413-138.62 83.614-139.72 442.267-101.216 520.377-499.476 468.67-756.915ZM526.904 362.906c-206.831 184.828-242.036 322.35-235.435 389.46-69.31-163.926 134.22-344.353 235.435-389.46ZM127.543 626.947c39.606 36.306 105.616 147.422 49.508 212.332 13.202-29.704 17.602-96.814-49.508-212.332z" transform="matrix(.094 0 0 .094 -2.042 -7.742)" fill="#fff"/>
</svg>

After

Width:  |  Height:  |  Size: 849 B

+4
View File
@@ -0,0 +1,4 @@
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 2897.4 896.6" style="enable-background:new 0 0 2897.4 896.6" xml:space="preserve">
<path d="M1022.3 428.7c-17.8-19.9-42.7-29.8-74.7-29.8-22.3 0-42.4 5.7-60.5 17.3-18.1 11.6-32.3 27.5-42.5 47.8s-15.3 42.9-15.3 67.8 5.1 47.5 15.3 67.8c10.3 20.3 24.4 36.2 42.5 47.8 18.1 11.5 38.3 17.3 60.5 17.3 32 0 56.9-9.9 74.7-29.8V655.5h84.5V408.3h-84.5v20.4zM1010.5 575c-10.2 11.7-23.6 17.6-40.2 17.6s-29.9-5.9-40-17.6-15.1-26.1-15.1-43.3c0-17.1 5-31.6 15.1-43.3s23.4-17.6 40-17.6 30 5.9 40.2 17.6 15.3 26.1 15.3 43.3-5.1 31.6-15.3 43.3zM1381 416.1c-18.1-11.5-38.3-17.3-60.5-17.4-32 0-56.9 9.9-74.7 29.8v-20.4h-84.5v390.7h84.5v-164c17.8 19.9 42.7 29.8 74.7 29.8 22.3 0 42.4-5.7 60.5-17.3s32.3-27.5 42.5-47.8c10.2-20.3 15.3-42.9 15.3-67.8s-5.1-47.5-15.3-67.8c-10.3-20.3-24.4-36.2-42.5-47.8zM1337.9 575c-10.1 11.7-23.4 17.6-40 17.6s-29.9-5.9-40-17.6-15.1-26.1-15.1-43.3c0-17.1 5-31.6 15.1-43.3s23.4-17.6 40-17.6 29.9 5.9 40 17.6 15.1 26.1 15.1 43.3-5.1 31.6-15.1 43.3zM1672.2 416.8c-20.5-12-43-18-67.6-18-24.9 0-47.6 5.9-68 17.6-20.4 11.7-36.5 27.7-48.2 48s-17.6 42.7-17.6 67.3c.3 25.2 6.2 47.8 17.8 68 11.5 20.2 28 36 49.3 47.6 21.3 11.5 45.9 17.3 73.8 17.3 48.6 0 86.8-14.7 114.7-44l-52.5-48.9c-8.6 8.3-17.6 14.6-26.7 19-9.3 4.3-21.1 6.4-35.3 6.4-11.6 0-22.5-3.6-32.7-10.9-10.3-7.3-17.1-16.5-20.7-27.8h180l.4-11.6c0-29.6-6-55.7-18-78.2s-28.3-39.8-48.7-51.8zm-113.9 86.4c2.1-12.1 7.5-21.8 16.2-29.1s18.7-10.9 30-10.9 21.2 3.6 29.8 10.9c8.6 7.2 13.9 16.9 16 29.1h-92zM1895.3 411.7c-11 5.6-20.3 13.7-28 24.4h-.1v-28h-84.5v247.3h84.5V536.3c0-22.6 4.7-38.1 14.2-46.5 9.5-8.5 22.7-12.7 39.6-12.7 6.2 0 13.5 1 21.8 3.1l10.7-72c-5.9-3.3-14.5-4.9-25.8-4.9-10.6 0-21.4 2.8-32.4 8.4zM1985 277.4h84.5v377.8H1985zM2313.2 416.8c-20.5-12-43-18-67.6-18-24.9 0-47.6 5.9-68 17.6s-36.5 27.7-48.2 48c-11.7 20.3-17.6 42.7-17.6 67.3.3 25.2 6.2 47.8 17.8 68 11.5 20.2 28 36 49.3 47.6 21.3 11.5 45.9 17.3 73.8 17.3 48.6 0 86.8-14.7 114.7-44l-52.5-48.9c-8.6 8.3-17.6 14.6-26.7 19-9.3 4.3-21.1 6.4-35.3 6.4-11.6 0-22.5-3.6-32.7-10.9-10.3-7.3-17.1-16.5-20.7-27.8h180l.4-11.6c0-29.6-6-55.7-18-78.2s-28.3-39.8-48.7-51.8zm-113.9 86.4c2.1-12.1 7.5-21.8 16.2-29.1s18.7-10.9 30-10.9 21.2 3.6 29.8 10.9c8.6 7.2 13.9 16.9 16 29.1h-92zM2583.6 507.7c-13.8-4.4-30.6-8.1-50.5-11.1-15.1-2.7-26.1-5.2-32.9-7.6-6.8-2.4-10.2-6.1-10.2-11.1s2.3-8.7 6.7-10.9c4.4-2.2 11.5-3.3 21.3-3.3 11.6 0 24.3 2.4 38.1 7.2 13.9 4.8 26.2 11 36.9 18.4l32.4-58.2c-11.3-7.4-26.2-14.7-44.9-21.8-18.7-7.1-39.6-10.7-62.7-10.7-33.7 0-60.2 7.6-79.3 22.7-19.1 15.1-28.7 36.1-28.7 63.1 0 19 4.8 33.9 14.4 44.7 9.6 10.8 21 18.5 34 22.9 13.1 4.5 28.9 8.3 47.6 11.6 14.6 2.7 25.1 5.3 31.6 7.8s9.8 6.5 9.8 11.8c0 10.4-9.7 15.6-29.3 15.6-13.7 0-28.5-2.3-44.7-6.9-16.1-4.6-29.2-11.3-39.3-20.2l-33.3 60c9.2 7.4 24.6 14.7 46.2 22 21.7 7.3 45.2 10.9 70.7 10.9 34.7 0 62.9-7.4 84.5-22.4 21.7-15 32.5-37.3 32.5-66.9 0-19.3-5-34.2-15.1-44.9s-22-18.3-35.8-22.7zM2883.4 575.3c0-19.3-5-34.2-15.1-44.9s-22-18.3-35.8-22.7c-13.8-4.4-30.6-8.1-50.5-11.1-15.1-2.7-26.1-5.2-32.9-7.6-6.8-2.4-10.2-6.1-10.2-11.1s2.3-8.7 6.7-10.9c4.4-2.2 11.5-3.3 21.3-3.3 11.6 0 24.3 2.4 38.1 7.2 13.9 4.8 26.2 11 36.9 18.4l32.4-58.2c-11.3-7.4-26.2-14.7-44.9-21.8-18.7-7.1-39.6-10.7-62.7-10.7-33.7 0-60.2 7.6-79.3 22.7-19.1 15.1-28.7 36.1-28.7 63.1 0 19 4.8 33.9 14.4 44.7 9.6 10.8 21 18.5 34 22.9 13.1 4.5 28.9 8.3 47.6 11.6 14.6 2.7 25.1 5.3 31.6 7.8s9.8 6.5 9.8 11.8c0 10.4-9.7 15.6-29.3 15.6-13.7 0-28.5-2.3-44.7-6.9-16.1-4.6-29.2-11.3-39.3-20.2l-33.3 60c9.2 7.4 24.6 14.7 46.2 22 21.7 7.3 45.2 10.9 70.7 10.9 34.7 0 62.9-7.4 84.5-22.4 21.7-15 32.5-37.3 32.5-66.9zM2460.7 738.7h59.6v17.2h-59.6zM2596.5 706.4c-5.7 0-11 1-15.8 3s-9 5-12.5 8.9v-9.4h-19.4v93.6h19.4v-52c0-8.6 2.1-15.3 6.3-20 4.2-4.7 9.5-7.1 15.9-7.1 7.8 0 13.4 2.3 16.8 6.7 3.4 4.5 5.1 11.3 5.1 20.5v52h19.4v-56.8c0-12.8-3.2-22.6-9.5-29.3-6.4-6.7-14.9-10.1-25.7-10.1zM2733.8 717.7c-3.6-3.4-7.9-6.1-13.1-8.2s-10.6-3.1-16.2-3.1c-8.7 0-16.5 2.1-23.5 6.3s-12.5 10-16.5 17.3c-4 7.3-6 15.4-6 24.4 0 8.9 2 17.1 6 24.3 4 7.3 9.5 13 16.5 17.2s14.9 6.3 23.5 6.3c5.6 0 11-1 16.2-3.1 5.1-2.1 9.5-4.8 13.1-8.2v24.4c0 8.5-2.5 14.8-7.6 18.7-5 3.9-11 5.9-18 5.9-6.7 0-12.4-1.6-17.3-4.7-4.8-3.1-7.6-7.7-8.3-13.8h-19.4c.6 7.7 2.9 14.2 7.1 19.5s9.6 9.3 16.2 12c6.6 2.7 13.8 4 21.7 4 12.8 0 23.5-3.4 32-10.1 8.6-6.7 12.8-17.1 12.8-31.1V708.9h-19.2v8.8zm-1.6 52.4c-2.5 4.7-6 8.3-10.4 11.2-4.4 2.7-9.4 4-14.9 4-5.7 0-10.8-1.4-15.2-4.3s-7.8-6.7-10.2-11.4c-2.3-4.8-3.5-9.8-3.5-15.2 0-5.5 1.1-10.6 3.5-15.3s5.8-8.5 10.2-11.3 9.5-4.2 15.2-4.2c5.5 0 10.5 1.4 14.9 4s7.9 6.3 10.4 11 3.8 10 3.8 15.8-1.3 11-3.8 15.7zM2867.9 708.9h-21.4l-25.6 33-25.4-33h-22.4l36 46.1-37.6 47.5h21.4l27.2-34.6 27.1 34.7h22.4l-37.6-48.2zM757.6 293.7c-20-10.8-42.6-16.2-67.8-16.2H600c-8.5 39.2-21.1 76.4-37.6 111.3-9.9 20.8-21.1 40.6-33.6 59.4v207.2h88.9V521.5h72c25.2 0 47.8-5.4 67.8-16.2s35.7-25.6 47.1-44.2c11.4-18.7 17.1-39.1 17.1-61.3.1-22.7-5.6-43.3-17-61.9-11.4-18.7-27.1-33.4-47.1-44.2zm-41 140.6c-9.3 8.9-21.6 13.3-36.7 13.3l-62.2.4v-92.5l62.2-.4c15.1 0 27.3 4.4 36.7 13.3 9.4 8.9 14 19.9 14 32.9 0 13.2-4.6 24.1-14 33z"/>
<path d="M140 713.7c-3.4-16.4-10.3-49.1-11.2-49.1C-16.9 577.5.4 426.6 48.6 340.4 59 449 251.2 524 139.1 656.8c-.9 1.7 5.2 22.4 10.3 41.4 22.4-37.9 56-83.6 54.3-87.9C65.9 273.9 496.9 248.1 586.6 39.4c40.5 201.8-20.7 513.9-367.2 593.2-1.7.9-62.9 108.6-65.5 109.5 0-1.7-25.9-.9-22.4-9.5 1.6-5.2 5.1-12 8.5-18.9zm-4.3-81.1c44-50.9-7.8-137.9-38.8-166.4 52.6 90.5 49.1 143.1 38.8 166.4z" style="fill:#17541f"/>
</svg>

After

Width:  |  Height:  |  Size: 5.4 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 108 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 7.6 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 22 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 8.2 KiB

+1
View File
@@ -8,6 +8,7 @@
<meta name="color-scheme" content="dark light">
<meta name="theme-color" content="#17541f" />
<link rel="manifest" href="manifest.webmanifest">
<link rel="icon" type="image/x-icon" href="favicon.ico">
<link rel="apple-touch-icon" href="apple-touch-icon.png">
</head>
<body>
+4 -12
View File
@@ -4,20 +4,12 @@
"display": "standalone",
"icons": [
{
"src": "icon-192.png",
"sizes": "192x192",
"type": "image/png"
"src": "favicon.ico",
"sizes": "256x256"
},
{
"src": "icon-512.png",
"sizes": "512x512",
"type": "image/png"
},
{
"src": "icon-512-maskable.png",
"sizes": "512x512",
"type": "image/png",
"purpose": "maskable"
"src": "assets/logo-notext.svg",
"sizes": "any"
}
],
"name": "Paperless-ngx",
+4
View File
@@ -626,6 +626,10 @@ ul.pagination {
table.table {
--bs-table-color: var(--bs-body-color);
--bs-table-bg: var(--bs-light-rgb);
&.bg-body {
--bs-table-bg: var(--bs-body-bg);
}
}
.close {
-40
View File
@@ -25,7 +25,6 @@ from django.db.models import Sum
from django.db.models import Value
from django.db.models import When
from django.db.models.functions import Cast
from django.db.models.functions import NullIf
from django.utils.translation import gettext_lazy as _
from django_filters import DateFilter
from django_filters.rest_framework import BooleanFilter
@@ -51,7 +50,6 @@ from documents.models import ShareLink
from documents.models import ShareLinkBundle
from documents.models import StoragePath
from documents.models import Tag
from documents.permissions import permitted_document_ids
from documents.permissions import permitted_object_ids
if TYPE_CHECKING:
@@ -795,12 +793,6 @@ class CustomFieldQueryFilter(Filter):
class DocumentFilterSet(FilterSet):
has_duplicates = BooleanFilter(method="filter_has_duplicates")
def __init__(self, *args: Any, user: Any = None, **kwargs: Any) -> None:
super().__init__(*args, **kwargs)
self._user = user
is_tagged = BooleanFilter(
label="Is tagged",
field_name="tags",
@@ -860,38 +852,6 @@ class DocumentFilterSet(FilterSet):
mime_type = MimeTypeFilter()
def filter_has_duplicates(self, queryset, name, value):
if value is None:
return queryset
user = (
self._user
if self._user is not None
else getattr(self.request, "user", None)
)
queryset = queryset.alias(
nonempty_archive_checksum=NullIf("archive_checksum", Value("")),
)
visible_root_documents = Document.global_objects.filter(
root_document__isnull=True,
pk__in=permitted_document_ids(
user,
include_deleted=True,
),
).exclude(pk=OuterRef("pk"))
# see serialisers._get_viewable_duplicates().
matching_duplicates = visible_root_documents.filter(
Q(checksum=OuterRef("checksum"))
| Q(checksum=OuterRef("nonempty_archive_checksum"))
| Q(archive_checksum=OuterRef("checksum"))
| Q(archive_checksum=OuterRef("nonempty_archive_checksum")),
)
return queryset.alias(
has_visible_duplicates=Exists(matching_duplicates),
).filter(has_visible_duplicates=value)
# Backwards compatibility
created__date__gt = DateFilter(field_name="created", lookup_expr="gt")
created__date__gte = DateFilter(field_name="created", lookup_expr="gte")
@@ -1,86 +0,0 @@
# Generated by Django 5.2.16 on 2026-09-05 16:29
from django.db import migrations
from django.db import models
class Migration(migrations.Migration):
dependencies = [
("documents", "0025_workflowaction_apply_ai_suggestions"),
]
operations = [
migrations.AlterField(
model_name="document",
name="archive_checksum",
field=models.CharField(
blank=True,
db_index=True,
editable=False,
help_text="The checksum of the archived document.",
max_length=64,
null=True,
verbose_name="archive checksum",
),
),
migrations.AlterField(
model_name="savedviewfilterrule",
name="rule_type",
field=models.PositiveSmallIntegerField(
choices=[
(0, "title contains"),
(1, "content contains"),
(2, "ASN is"),
(3, "correspondent is"),
(4, "document type is"),
(5, "is in inbox"),
(6, "has tag"),
(7, "has any tag"),
(8, "created before"),
(9, "created after"),
(10, "created year is"),
(11, "created month is"),
(12, "created day is"),
(13, "added before"),
(14, "added after"),
(15, "modified before"),
(16, "modified after"),
(17, "does not have tag"),
(18, "does not have ASN"),
(19, "title or content contains"),
(20, "fulltext query"),
(21, "more like this"),
(22, "has tags in"),
(23, "ASN greater than"),
(24, "ASN less than"),
(25, "storage path is"),
(26, "has correspondent in"),
(27, "does not have correspondent in"),
(28, "has document type in"),
(29, "does not have document type in"),
(30, "has storage path in"),
(31, "does not have storage path in"),
(32, "owner is"),
(33, "has owner in"),
(34, "does not have owner"),
(35, "does not have owner in"),
(36, "has custom field value"),
(37, "is shared by me"),
(38, "has custom fields"),
(39, "has custom field in"),
(40, "does not have custom field in"),
(41, "does not have custom field"),
(42, "custom fields query"),
(43, "created to"),
(44, "created from"),
(45, "added to"),
(46, "added from"),
(47, "mime type is"),
(48, "simple title search"),
(49, "simple text search"),
(50, "has duplicates"),
],
verbose_name="rule type",
),
),
]
-2
View File
@@ -227,7 +227,6 @@ class Document(SoftDeleteModel, ModelWithOwner): # type: ignore[django-manager-
editable=False,
blank=True,
null=True,
db_index=True,
help_text=_("The checksum of the archived document."),
)
@@ -707,7 +706,6 @@ class SavedViewFilterRule(models.Model):
(47, _("mime type is")),
(48, _("simple title search")),
(49, _("simple text search")),
(50, _("has duplicates")),
]
saved_view = models.ForeignKey(
+2 -10
View File
@@ -6,20 +6,13 @@ from documents.search._backend import TantivyRelevanceList
from documents.search._backend import WriteBatch
from documents.search._backend import get_backend
from documents.search._backend import reset_backend
from documents.search._errors import InvalidDateQuery
from documents.search._errors import InvalidNumberQuery
from documents.search._errors import MultipleSearchQueryErrors
from documents.search._errors import QueryTooLongError
from documents.search._errors import SearchQueryError
from documents.search._errors import search_query_error_messages
from documents.search._schema import needs_rebuild
from documents.search._schema import wipe_index
from documents.search._translate import InvalidDateQuery
from documents.search._translate import SearchQueryError
__all__ = [
"InvalidDateQuery",
"InvalidNumberQuery",
"MultipleSearchQueryErrors",
"QueryTooLongError",
"SearchHit",
"SearchIndexLockError",
"SearchMode",
@@ -30,6 +23,5 @@ __all__ = [
"get_backend",
"needs_rebuild",
"reset_backend",
"search_query_error_messages",
"wipe_index",
]
+9 -99
View File
@@ -22,6 +22,7 @@ import tantivy
from django.conf import settings
from django.utils.timezone import get_current_timezone
from documents.search._query import build_permission_filter
from documents.search._query import extract_cjk_text
from documents.search._query import parse_simple_text_highlight_query
from documents.search._query import parse_simple_text_query
@@ -39,7 +40,6 @@ from documents.utils import QuerySetStream
from documents.utils import identity
if TYPE_CHECKING:
from collections.abc import Iterable
from collections.abc import Iterator
from collections.abc import Sequence
from pathlib import Path
@@ -284,87 +284,6 @@ class WriteBatch:
tantivy.Query.term_query(self._backend._schema, "id", doc_id),
)
def add_or_update_ids(self, ids: Sequence[int]) -> None:
"""
Add or update multiple documents in the batch by primary key.
Unlike calling ``add_or_update()`` once per document, this resolves
viewer permissions and effective (versioned) content in bulk against
the ids as a whole, instead of once per document -- see
``_DocumentViewerStream`` and ``annotate_effective_content``. Use
this whenever more than one document is being written in the same
batch.
An id with no matching document (e.g. deleted between the caller
collecting ids and the batch running) is silently skipped, matching
``add_or_update()``'s existing single-document deferred-task behavior
rather than erroring or leaving a stale index entry.
Args:
ids: Primary keys of Document instances to index
"""
from documents.models import Document
from documents.versioning import annotate_effective_content
ids = list(ids)
if not ids:
return
queryset = annotate_effective_content(
Document.objects.filter(pk__in=ids)
.select_related("correspondent", "document_type", "storage_path", "owner")
.prefetch_related("tags", "notes__user", "custom_fields__field"),
)
for document, grant in _DocumentViewerStream(queryset, chunk_size=1000):
self.remove(document.pk)
doc = self._backend._build_tantivy_doc(
document,
viewer_ids=grant.viewer_ids,
viewer_group_ids=grant.viewer_group_ids,
)
self._writer.add_document(doc)
def build_permission_filter(
schema: tantivy.Schema,
user: AbstractUser,
viewer_group_ids: Iterable[int] = (),
) -> tantivy.Query:
"""
Build a query filter for user document permissions.
Creates a query that matches only documents visible to the specified user
according to paperless-ngx permission rules:
- Public documents (no owner) are visible to all users
- Private documents are visible to their owner
- Documents explicitly shared with the user are visible
- Documents shared with one of the user's current groups are visible
Args:
schema: Tantivy schema for field validation
user: User to check permissions for
viewer_group_ids: Current group memberships for the user
Returns:
Tantivy query that filters results to visible documents
"""
owner_any = tantivy.Query.exists_query("owner_id")
no_owner = tantivy.Query.boolean_query(
[
(tantivy.Occur.Must, tantivy.Query.all_query()),
(tantivy.Occur.MustNot, owner_any),
],
)
owned = tantivy.Query.term_query(schema, "owner_id", user.pk)
shared = tantivy.Query.term_query(schema, "viewer_id", user.pk)
group_shared = [
tantivy.Query.term_query(schema, "viewer_group_id", group_id)
for group_id in viewer_group_ids
]
return tantivy.Query.disjunction_max_query(
[no_owner, owned, shared, *group_shared],
)
class TantivyBackend:
"""
@@ -539,6 +458,7 @@ class TantivyBackend:
doc.add_text("correspondent_sort", document.correspondent.name)
if cjk_corr := extract_cjk_text(document.correspondent.name):
doc.add_text("bigram_correspondent", cjk_corr)
doc.add_unsigned("correspondent_id", document.correspondent_id)
# Document type
if document.document_type:
@@ -546,10 +466,12 @@ class TantivyBackend:
doc.add_text("type_sort", document.document_type.name)
if cjk_type := extract_cjk_text(document.document_type.name):
doc.add_text("bigram_document_type", cjk_type)
doc.add_unsigned("document_type_id", document.document_type_id)
# Storage path
if document.storage_path:
doc.add_text("storage_path", document.storage_path.name)
doc.add_unsigned("storage_path_id", document.storage_path_id)
# Tags — collect names for autocomplete in the same pass
tag_names: list[str] = []
@@ -557,13 +479,12 @@ class TantivyBackend:
doc.add_text("tag", tag.name)
if cjk_tag := extract_cjk_text(tag.name):
doc.add_text("bigram_tag", cjk_tag)
doc.add_unsigned("tag_id", tag.pk)
tag_names.append(tag.name)
# Notes — JSON for structured queries (notes.user:alice, notes.note:text).
# notes_text is a plain-text companion for snippet/highlight generation;
# tantivy's SnippetGenerator does not support JSON fields. It is not in
# _DEFAULT_SEARCH_FIELDS, so an unqualified query never searches it: a
# note matches through the JSON field or not at all.
# tantivy's SnippetGenerator does not support JSON fields.
num_notes = 0
note_texts: list[str] = []
for note in document.notes.all():
@@ -579,9 +500,8 @@ class TantivyBackend:
if note_texts:
doc.add_text("notes_text", " ".join(note_texts))
# Custom fields: JSON for structured queries (custom_fields.name:x,
# custom_fields.value:y). There is no companion text field here, unlike
# notes: custom field values are reachable only through the JSON field.
# Custom fields JSON for structured queries (custom_fields.name:x, custom_fields.value:y),
# companion text field for default full-text search.
for cfi in document.custom_fields.all():
search_value = cfi.value_for_search
# Skip fields where there is no value yet
@@ -748,17 +668,7 @@ class TantivyBackend:
user_query = self._parse_query(query, search_mode)
highlight_query = user_query
if search_mode is SearchMode.TEXT:
try:
highlight_query = parse_simple_text_highlight_query(
self._index,
query,
)
except ValueError:
logger.debug(
"Skipping simple text highlight query: token string is not "
"valid tantivy query syntax: %r",
query,
)
highlight_query = parse_simple_text_highlight_query(self._index, query)
# For notes_text snippet generation, we need a query that targets the
# notes_text field directly. user_query may contain JSON-field terms
+171
View File
@@ -0,0 +1,171 @@
from __future__ import annotations
from datetime import UTC
from datetime import date
from datetime import datetime
from datetime import timedelta
from typing import TYPE_CHECKING
from typing import Final
from dateutil.relativedelta import relativedelta
if TYPE_CHECKING:
from datetime import tzinfo
_DATE_ONLY_FIELDS = frozenset({"created"})
_TODAY: Final[str] = "today"
_YESTERDAY: Final[str] = "yesterday"
_PREVIOUS_WEEK: Final[str] = "previous week"
_THIS_MONTH: Final[str] = "this month"
_PREVIOUS_MONTH: Final[str] = "previous month"
_THIS_YEAR: Final[str] = "this year"
_PREVIOUS_YEAR: Final[str] = "previous year"
_PREVIOUS_QUARTER: Final[str] = "previous quarter"
_DATE_KEYWORDS = frozenset(
{
_TODAY,
_YESTERDAY,
_PREVIOUS_WEEK,
_THIS_MONTH,
_PREVIOUS_MONTH,
_THIS_YEAR,
_PREVIOUS_YEAR,
_PREVIOUS_QUARTER,
},
)
def _fmt(dt: datetime) -> str:
"""Format a datetime as an ISO 8601 UTC string for use in Tantivy range queries."""
return dt.astimezone(UTC).strftime("%Y-%m-%dT%H:%M:%SZ")
def _iso_range(lo: datetime, hi: datetime) -> str:
"""
Format a half-open ``[lo TO hi)`` range in ISO 8601 for Tantivy query syntax.
``hi`` is always the exclusive ceiling of a computed period (the start of
the *next* day/week/month/quarter/year), so the closing bracket must be
the Tantivy exclusive-range brace ``}`` rather than ``]`` otherwise the
first instant of the following period (e.g. the 1st of next month) is
incorrectly included in the match.
"""
return f"[{_fmt(lo)} TO {_fmt(hi)}}}"
def _quarter_start(d: date) -> date:
"""Return the first day of the calendar quarter containing ``d``."""
return date(d.year, ((d.month - 1) // 3) * 3 + 1, 1)
def _midnight(d: date, tz: tzinfo) -> datetime:
"""Convert a calendar date at local-timezone midnight to a UTC datetime."""
return datetime(d.year, d.month, d.day, tzinfo=tz).astimezone(UTC)
def _keyword_bounds(keyword: str, tz: tzinfo) -> tuple[date, date]:
"""
Map a relative date keyword to ``(start, exclusive_end)`` calendar dates.
``tz`` only determines what "today" is; the caller decides how the returned
dates become UTC datetime boundaries (date-only vs. local-midnight offset).
"""
today = datetime.now(tz).date()
if keyword == _TODAY:
return today, today + timedelta(days=1)
if keyword == _YESTERDAY:
return today - timedelta(days=1), today
if keyword == _PREVIOUS_WEEK:
this_monday = today - timedelta(days=today.weekday())
return this_monday - timedelta(weeks=1), this_monday
if keyword == _THIS_MONTH:
first = today.replace(day=1)
return first, first + relativedelta(months=1)
if keyword == _PREVIOUS_MONTH:
this_first = today.replace(day=1)
return this_first - relativedelta(months=1), this_first
if keyword == _THIS_YEAR:
return date(today.year, 1, 1), date(today.year + 1, 1, 1)
if keyword == _PREVIOUS_YEAR:
return date(today.year - 1, 1, 1), date(today.year, 1, 1)
if keyword == _PREVIOUS_QUARTER:
this_quarter = _quarter_start(today)
return this_quarter - relativedelta(months=3), this_quarter
raise ValueError(f"Unknown keyword: {keyword}")
def _date_only_range(keyword: str, tz: tzinfo) -> str:
"""
For `created` (DateField): use the local calendar date, converted to
midnight UTC boundaries. No offset arithmetic date only.
"""
start, end = _keyword_bounds(keyword, tz)
lo = datetime(start.year, start.month, start.day, tzinfo=UTC)
hi = datetime(end.year, end.month, end.day, tzinfo=UTC)
return _iso_range(lo, hi)
def _datetime_range(keyword: str, tz: tzinfo) -> str:
"""
For `added` / `modified` (DateTimeField, stored as UTC): convert local day
boundaries to UTC full offset arithmetic required.
"""
start, end = _keyword_bounds(keyword, tz)
return _iso_range(_midnight(start, tz), _midnight(end, tz))
def _precision_bounds(digits: str) -> tuple[date, date] | None:
"""
Map a 4/6/8-digit date token to (start, exclusive_end) calendar dates.
YYYY -> whole year, YYYYMM -> whole month, YYYYMMDD -> single day.
Returns None for any unparsable or out-of-range value (e.g. month 23),
so callers can emit a no-match clause instead of erroring (Whoosh parity).
"""
try:
if len(digits) == 4:
year = int(digits)
return date(year, 1, 1), date(year + 1, 1, 1)
if len(digits) == 6:
year, month = int(digits[:4]), int(digits[4:6])
start = date(year, month, 1)
end = date(year + 1, 1, 1) if month == 12 else date(year, month + 1, 1)
return start, end
if len(digits) == 8:
start = date(int(digits[:4]), int(digits[4:6]), int(digits[6:8]))
return start, start + timedelta(days=1)
except ValueError:
return None
return None
def _utc_bounds_for_field(
field: str,
start: date,
end: date,
tz: tzinfo,
) -> tuple[datetime, datetime]:
"""
Convert calendar-date bounds to UTC datetimes per the field's storage type.
For DateField (``created``) the bounds are UTC midnight (no offset). For
DateTimeField (``added``/``modified``) the bounds are local-tz midnight
converted to UTC, matching how each field is indexed.
"""
if field in _DATE_ONLY_FIELDS:
return (
datetime(start.year, start.month, start.day, tzinfo=UTC),
datetime(end.year, end.month, end.day, tzinfo=UTC),
)
return (
datetime(start.year, start.month, start.day, tzinfo=tz).astimezone(UTC),
datetime(end.year, end.month, end.day, tzinfo=tz).astimezone(UTC),
)
def _field_range_from_dates(field: str, start: date, end: date, tz: tzinfo) -> str:
"""Build a Tantivy ``field:[lo TO hi]`` ISO range from calendar-date bounds."""
lo, hi = _utc_bounds_for_field(field, start, end, tz)
return f"{field}:{_iso_range(lo, hi)}"
-71
View File
@@ -1,71 +0,0 @@
from __future__ import annotations
from typing import TYPE_CHECKING
if TYPE_CHECKING:
from collections.abc import Sequence
class SearchQueryError(ValueError):
"""
Base for user-fixable search query errors.
Carries a message safe to surface to the user (no internal details). The
view layer catches this and returns an HTTP 400, so any future subclass
gets the same treatment.
"""
class InvalidDateQuery(SearchQueryError):
"""Raised when a date field value or range bound cannot be parsed."""
def __init__(self, field: str | None, value: str | None) -> None:
self.field = field
self.value = value
super().__init__(f"Invalid date value {value!r} for field {field!r}.")
class InvalidNumberQuery(SearchQueryError):
"""Raised when a numeric field value or range bound cannot be parsed."""
def __init__(self, field: str | None, value: str | None) -> None:
self.field = field
self.value = value
super().__init__(f"Invalid numeric value {value!r} for field {field!r}.")
class QueryTooLongError(SearchQueryError):
"""Raised when a query string exceeds the maximum allowed length.
whoosh-compat's fieldname tagger is O(n^2) in plain word characters, so an
unbounded query is a CPU-exhaustion vector against a single request
handler. This is a hard boundary, not a validation nicety.
"""
def __init__(self, length: int, limit: int) -> None:
self.length = length
self.limit = limit
super().__init__(
f"The search query is too long ({length} characters). "
f"The maximum allowed length is {limit} characters.",
)
class MultipleSearchQueryErrors(SearchQueryError):
"""Aggregates every user-fixable error from one parse, not just the first."""
def __init__(self, errors: Sequence[SearchQueryError]) -> None:
self.errors = tuple(errors)
super().__init__("; ".join(str(e) for e in self.errors))
def search_query_error_messages(e: SearchQueryError) -> list[str]:
"""The user-facing message list for a SearchQueryError.
Every offending value's message, not just the first, so the user can
fix them all in one round-trip. Shared by every view that maps
SearchQueryError to an HTTP 400.
"""
if isinstance(e, MultipleSearchQueryErrors):
return [str(sub) for sub in e.errors]
return [str(e)]
-42
View File
@@ -1,42 +0,0 @@
from __future__ import annotations
from whoosh_compat import FieldKind
from whoosh_compat import FieldSpec
from whoosh_compat import SubpathSpec
# Internal-only schema fields with no query-syntax meaning of their own
# (sort shadow fields, bigram CJK fields, simple_title/simple_content,
# autocomplete_word, notes_text) are NOT represented here, they are
# declared in _schema.py's field_descriptors().
#
# analyzer/pattern_normalizer are deliberately left at FieldSpec's default
# (None): they're language-specific and only meaningful to whoosh-compat's
# parser, so _registry.py attaches them per-language via dataclasses.replace()
# rather than PUBLIC_FIELDS declaring them itself. _schema.py only reads
# name/kind/fast and never sees the analyzer at all.
PUBLIC_FIELDS: tuple[FieldSpec, ...] = (
FieldSpec("title", FieldKind.TEXT),
FieldSpec("content", FieldKind.TEXT),
FieldSpec("correspondent", FieldKind.TEXT),
FieldSpec("document_type", FieldKind.TEXT, aliases=("type",)),
FieldSpec("storage_path", FieldKind.TEXT, aliases=("path",)),
FieldSpec("original_filename", FieldKind.TEXT),
FieldSpec("tag", FieldKind.TEXT, comma_values=True),
FieldSpec("checksum", FieldKind.KEYWORD),
FieldSpec("asn", FieldKind.U64, fast=True),
FieldSpec("page_count", FieldKind.U64, fast=True),
FieldSpec("num_notes", FieldKind.U64, fast=True),
FieldSpec("created", FieldKind.DATE, date_only=True, fast=True),
FieldSpec("modified", FieldKind.DATETIME, fast=True),
FieldSpec("added", FieldKind.DATETIME, fast=True),
FieldSpec(
"notes",
FieldKind.JSON,
subpaths={"user": SubpathSpec(), "note": SubpathSpec(default=True)},
),
FieldSpec(
"custom_fields",
FieldKind.JSON,
subpaths={"name": SubpathSpec(), "value": SubpathSpec(default=True)},
),
)
+145 -450
View File
@@ -6,30 +6,22 @@ from typing import Final
import regex
import tantivy
import whoosh_compat as wc
from django.conf import settings
from whoosh_compat.emitters.tantivy_ import emit as tantivy_emit
from whoosh_compat.errors import Cause
from whoosh_compat.errors import Diagnostic
from whoosh_compat.errors import DiagnosticKind
from whoosh_compat.errors import QueryError
from documents.search._errors import InvalidDateQuery
from documents.search._errors import InvalidNumberQuery
from documents.search._errors import MultipleSearchQueryErrors
from documents.search._errors import SearchQueryError
from documents.search._registry import get_field_registry
from documents.search._tokenizer import simple_search_tokens
from documents.search._translate import SearchQueryError
from documents.search._translate import translate_query
if TYPE_CHECKING:
from collections.abc import Iterable
from datetime import tzinfo
from django.contrib.auth.base_user import AbstractBaseUser
logger = logging.getLogger("paperless.search")
# Maximum seconds any single regex substitution over user-supplied query text
# may run. The one remaining use is a character class, which cannot backtrack,
# so the bound is an upper limit on that substitution's cost, not the ReDoS
# guard it was originally written as.
# Maximum seconds any single regex substitution may run.
# Prevents ReDoS on adversarial user-supplied query strings.
_REGEX_TIMEOUT: Final[float] = 1.0
# Matches CJK/Hangul characters so queries can be routed to bigram fields.
@@ -37,64 +29,6 @@ _REGEX_TIMEOUT: Final[float] = 1.0
_CJK_RE: Final = regex.compile(r"[\p{Han}\p{Hiragana}\p{Katakana}\p{Hangul}]+")
def _user_facing_emit_message(d: Diagnostic) -> str:
"""A user-safe message for an emit-time QueryError's Diagnostic.
Built from the Diagnostic's structured fields (kind, field), never from
d.message: whoosh-compat documents that as developer/log output with no
stability guarantee, and PATTERN_TOO_COMPLEX embeds the raw backend
error text in it.
"""
field = str(d.field) if d.field is not None else None
if d.kind is DiagnosticKind.EXISTS_REQUIRES_FAST:
return f"Existence searches (field:*) are not supported for field {field!r}."
if d.kind is DiagnosticKind.TEXT_RANGE:
return f"Range searches are not supported for field {field!r}."
if d.kind is DiagnosticKind.PATTERN_TOO_COMPLEX:
return f"The wildcard pattern for field {field!r} is too complex."
if d.kind is DiagnosticKind.SCHEMA_FIELD_MISSING:
return f"Field {field!r} is not available in the search index."
logger.warning("Unmapped emit diagnostic %s: %s", d.kind, d.message)
return "The search query could not be executed."
def _map_emit_error(e: QueryError) -> SearchQueryError:
"""Route an emit-time QueryError by its Diagnostic's Cause.
INVALID_INPUT/UNSUPPORTED are user-input errors, exactly like a parse
diagnostic, and map to a 400. INTERNAL means a defect in whoosh-compat
or in our own AST handling, never the user's query, so the QueryError is
re-raised rather than converted, reaching the generic 500 handler instead
of blaming the query. MISCONFIGURED is deliberately both: the registry and
the index schema disagree, which only an operator can fix, so it is logged
as an error, but a request is still waiting and the query cannot run
either way, so it also returns a 400.
EXISTS_REQUIRES_FAST is the one MISCONFIGURED kind that is not a
disagreement. whoosh-compat derives it from the registry's own FieldSpec
(kind plus fast) without ever consulting the index schema, so it fires
whenever a non-fast field of a kind that cannot answer "exists" is asked
to: for us that is only the JSON fields, which field_descriptors() builds
non-fast on purpose. "notes:*" and the five other spellings of it are
ordinary user error that no operator action can clear, so they get the
400 without the alert.
"""
d = e.diagnostic
if d.cause is Cause.INTERNAL:
raise e
if (
d.cause is Cause.MISCONFIGURED
and d.kind is not DiagnosticKind.EXISTS_REQUIRES_FAST
):
logger.error(
"Search index misconfiguration for field %s (%s): %s",
d.field,
d.kind.name,
d.message,
)
return SearchQueryError(_user_facing_emit_message(d))
def _has_cjk(text: str) -> bool:
"""Return True if text contains any CJK characters."""
return bool(_CJK_RE.search(text))
@@ -103,36 +37,14 @@ def _has_cjk(text: str) -> bool:
def extract_cjk_text(text: str) -> str:
"""Join the CJK runs in ``text`` for indexing into bigram (char-ngram) fields.
Mirrors the query side, which extracts the CJK runs of whatever it is
about to search for (the raw string in simple modes, the parsed query's
free-text tokens in query mode): only CJK runs are ever searched against
the bigram fields, so only CJK runs are worth indexing there. Latin text
fed to a character-bigram field is never matched and only bloats the
Mirrors the query side (``_build_cjk_query``): only CJK runs are ever searched
against the bigram fields, so only CJK runs are worth indexing there. Latin
text fed to a character-bigram field is never matched and only bloats the
index and slows indexing/merge. Returns "" when there is no CJK text.
"""
return " ".join(_CJK_RE.findall(text))
def _parse_cjk_text(
index: tantivy.Index,
cjk_text: str,
fields: list[str],
) -> tantivy.Query | None:
"""Parse a plain CJK run string against ``fields``, or None if it won't parse."""
try:
return index.parse_query(cjk_text, fields)
except Exception:
# Broad on purpose, unlike _try_parse_fuzzy_query's narrower
# ValueError: cjk_text isn't filtered to a guaranteed-safe token
# set the way the fuzzy blend's word string is, so the exact
# failure mode tantivy could raise here isn't pinned down.
logger.debug(
"Skipping CJK search clause: could not parse CJK text: %r",
cjk_text,
)
return None
def _build_cjk_query(
index: tantivy.Index,
raw_query: str,
@@ -140,259 +52,91 @@ def _build_cjk_query(
) -> tantivy.Query | None:
"""Build a bigram-field query from the CJK runs in ``raw_query``.
For the simple (TEXT/TITLE) modes, whose input is plain text and carries
no query grammar to respect. Only the CJK character runs are extracted, so
a stray ``field:`` prefix or ``-``/``+`` in the input can neither leak
field semantics nor fail the parse, and no Latin token reaches the
character-bigram matcher (where it would produce spurious matches against
unrelated Latin text). Returns None when there is no CJK text or the parse
fails.
Only the CJK character runs are extracted and parsed; ASCII field prefixes,
boolean operators and date keywords are discarded. This keeps the CJK clause
plain-text and consistent across query/simple modes (no leaked ``field:``
semantics, no parse failures from spaced ``-``/``+``), and avoids feeding
Latin tokens into the character-bigram matcher (which would produce spurious
matches against unrelated Latin text). Returns None when there is no CJK
text or the parse fails.
"""
cjk_text = extract_cjk_text(raw_query)
cjk_text = " ".join(_CJK_RE.findall(raw_query))
if not cjk_text:
return None
return _parse_cjk_text(index, cjk_text, fields)
def _build_ast_cjk_query(
index: tantivy.Index,
ast: wc.ast.Node,
registry: wc.FieldRegistry,
) -> tantivy.Query | None:
"""Build the bigram clause of a QUERY-mode search from the parsed AST.
Same discipline as the fuzzy clause (see _try_parse_fuzzy_query): the CJK
runs come from whoosh_compat's ``free_text_tokens`` over the parsed tree,
never from the raw query string, so a term the user negated or restricted
to a field outside the default search fields contributes nothing, instead
of resurfacing as a top-level clause matching every bigram field.
``free_text_tokens`` reports no field of its own, so the tokens are
collected one default field at a time: a bare term, which the parser has
already copied onto every default field, is therefore searched across
every bigram field, while ``title:東京`` reaches ``bigram_title`` alone.
Fields whose CJK text is identical (the bare-term case) share a single
parse over all of their bigram fields at once.
Raw (``analyzed=False``) tokens are used because the bigram fields have
their own character-ngram analyzer: the default fields' word analyzers
have no useful say over a CJK run, and running them first would only
risk dropping it (remove_long) before the run is ever extracted.
Returns None when the query has no CJK free text.
"""
fields_by_text: dict[str, list[str]] = {}
for field, bigram_field in _CJK_BIGRAM_FIELDS.items():
tokens = wc.free_text_tokens(
ast,
registry=registry,
fields=[field],
analyzed=False,
)
cjk_text = extract_cjk_text(" ".join(tokens))
if cjk_text:
fields_by_text.setdefault(cjk_text, []).append(bigram_field)
clauses: list[tuple[tantivy.Occur, tantivy.Query]] = [
(tantivy.Occur.Should, query)
for cjk_text, bigram_fields in fields_by_text.items()
if (query := _parse_cjk_text(index, cjk_text, bigram_fields)) is not None
]
return _any_of(clauses) if clauses else None
# A joined fuzzy word string must stay plain words: it goes back through
# tantivy's own query parser, and the raw query text the clause collects
# routinely carries characters that parser reads as grammar (a colon, a
# bracket, a quote, a leading -). Each token is cut into its word runs and
# only those are kept, so no field syntax, pattern, range or grouping can
# reach the parser. Cutting rather than dropping the whole token is what
# keeps ordinary hyphenated, dotted and quoted input ("COVID-19",
# "hello@example.com", "tax reports") contributing to the clause at all.
_WORD_RUN_RE = regex.compile(r"\w+")
# The one piece of tantivy grammar that survives the cut: its boolean
# keywords are themselves word runs. Only these exact spellings are
# grammar there ("And"/"and" are ordinary terms), so lowercasing exactly
# these turns them back into the ordinary terms the field analyzer used to
# make of them, before the clause switched to raw text. Left alone, a
# quoted phrase would silently restructure the clause ("tax AND reports"
# becoming a conjunction) or fail to parse and drop it entirely
# ("tax AND", or "IN" anywhere).
#
# Only these words are touched: tantivy lowercases query terms with the
# field's own analyzer, and doing it ourselves first is not always the
# same operation (Python folds a final sigma to a different letter than
# tantivy does, and turns Turkish 'İ' into a sequence tantivy then splits
# in two), which would search for terms the index does not contain.
_TANTIVY_KEYWORDS: Final[frozenset[str]] = frozenset({"AND", "OR", "NOT", "IN"})
def _try_parse_fuzzy_query(
index: tantivy.Index,
ast: wc.ast.Node,
registry: wc.FieldRegistry,
) -> tantivy.Query | None:
"""Build the fuzzy blend clause from the parsed query's free-text
words, or None if it has none.
The clause is built by handing tantivy's own query parser a plain
word string (there's no clean AST-level fuzzy equivalent to
whoosh-compat's parse tree, and fuzzy matching was always an
approximate, secondary, 0.1-boosted clause). The words come from
whoosh_compat's ``free_text_tokens`` over the already-parsed AST,
never from the raw query string: raw whoosh grammar (date keywords,
``[2005 to 2009]`` ranges, bracket-class wildcards) is not tantivy
syntax, and feeding it here used to knock the fuzzy clause out for
the whole query the moment any such construct appeared alongside a
typo'd word. The helper also keeps excluded terms out: a ``NOT``'d
word must not resurface through the fuzzy clause.
Chosen trade-off: a term explicitly fielded on one of the default
search fields (``correspondent:acme``) contributes its text to the
word string UNFIELDED, so the fuzzy clause searches it across all
default fields rather than just the one the user named. That is
recall-only widening on a secondary 0.1-boosted clause the score
threshold already disciplines, accepted in exchange for never feeding
field syntax to tantivy's parser. What the word string guarantees is
exactly that: no field prefix, pattern, range, grouping or quoting
survives, and the boolean keywords that do survive (they are word
runs) are lowercased into ordinary terms; see _TANTIVY_KEYWORDS.
The words are the query's RAW text, not the analyzer's output
(``analyzed=False``), because ``index.parse_query`` analyzes whatever
it is given and analysis is not idempotent: ``universities`` stems to
``univers``, and handing that back stems it again to ``univ``, a term
the index does not contain. ``prefix=True`` hid this as over-broad
matching (``univ`` also prefixes ``unicycle``) rather than as no
matches at all. Raw text is untokenized, which is why it is cut into
word runs above rather than taken whole.
The ValueError guard stays as insurance (the word string is plain
tokens, so tantivy accepting it is expected, not assumed): on a parse
failure the fuzzy clause is skipped and the exact/CJK clauses stand,
rather than the whole query failing.
"""
tokens = wc.free_text_tokens(
ast,
registry=registry,
fields=_DEFAULT_SEARCH_FIELDS,
analyzed=False,
)
words = list(
dict.fromkeys(
word.lower() if word in _TANTIVY_KEYWORDS else word
for token in tokens
for word in _WORD_RUN_RE.findall(token)
),
)
if not words:
return None
fuzzy_text = " ".join(words)
try:
return index.parse_query(
fuzzy_text,
_DEFAULT_SEARCH_FIELDS,
field_boosts=_FIELD_BOOSTS,
fuzzy_fields={f: (True, 1, True) for f in _DEFAULT_SEARCH_FIELDS},
)
except ValueError:
logger.debug(
"Skipping fuzzy search clause: token string is not valid "
"tantivy query syntax: %r",
fuzzy_text,
)
return index.parse_query(cjk_text, fields)
except Exception:
return None
_DEFAULT_SEARCH_FIELDS: Final[list[str]] = [
def build_permission_filter(
schema: tantivy.Schema,
user: AbstractBaseUser,
viewer_group_ids: Iterable[int] = (),
) -> tantivy.Query:
"""
Build a query filter for user document permissions.
Creates a query that matches only documents visible to the specified user
according to paperless-ngx permission rules:
- Public documents (no owner) are visible to all users
- Private documents are visible to their owner
- Documents explicitly shared with the user are visible
- Documents shared with one of the user's current groups are visible
Args:
schema: Tantivy schema for field validation
user: User to check permissions for
viewer_group_ids: Current group memberships for the user
Returns:
Tantivy query that filters results to visible documents
"""
owner_any = tantivy.Query.exists_query("owner_id")
no_owner = tantivy.Query.boolean_query(
[
(tantivy.Occur.Must, tantivy.Query.all_query()),
(tantivy.Occur.MustNot, owner_any),
],
)
owned = tantivy.Query.term_query(schema, "owner_id", user.pk)
shared = tantivy.Query.term_query(schema, "viewer_id", user.pk)
group_shared = [
tantivy.Query.term_query(schema, "viewer_group_id", group_id)
for group_id in viewer_group_ids
]
return tantivy.Query.disjunction_max_query(
[no_owner, owned, shared, *group_shared],
)
DEFAULT_SEARCH_FIELDS = [
"title",
"content",
"correspondent",
"document_type",
"tag",
]
_SIMPLE_SEARCH_FIELDS: Final[list[str]] = ["simple_title", "simple_content"]
_TITLE_SEARCH_FIELDS: Final[list[str]] = ["simple_title"]
# The bigram (character-ngram) companion of each default search field.
_CJK_BIGRAM_FIELDS: Final[dict[str, str]] = {
field: f"bigram_{field}" for field in _DEFAULT_SEARCH_FIELDS
}
SIMPLE_SEARCH_FIELDS = ["simple_title", "simple_content"]
TITLE_SEARCH_FIELDS = ["simple_title"]
_CJK_ALL_FIELDS: Final[list[str]] = [
"bigram_content",
"bigram_title",
"bigram_correspondent",
"bigram_document_type",
"bigram_tag",
]
_CJK_CONTENT_FIELDS: Final[list[str]] = ["bigram_content"]
_CJK_TITLE_FIELDS: Final[list[str]] = ["bigram_title"]
_FIELD_BOOSTS = {"title": 2.0}
_SIMPLE_FIELD_BOOSTS = {"simple_title": 2.0}
class _ConjunctiveNegations(wc.ast.Visitor[tuple["wc.ast.Node", ...]]):
"""Collect the subtrees an AST excludes from every document it matches.
A negation reached through ``And``/``AndNot``/``Require`` (and through
the required half of an ``AndMaybe``) constrains the whole query, so it
can be re-stated above the blend. ``Or`` is deliberately not descended
into: in ``invoice OR NOT secret`` the negation is one branch's own
condition, and hoisting it would throw away documents the other branch
matches. Nor is a collected subtree descended into, since a negation
inside a negation is not an exclusion.
Node types with no negation to contribute (every leaf, ``Or``) fall
through to ``generic_visit``.
"""
def generic_visit(self, node: wc.ast.Node) -> tuple[wc.ast.Node, ...]:
return ()
def visit_not(self, node: wc.ast.Not) -> tuple[wc.ast.Node, ...]:
return (node.child,)
def visit_andnot(self, node: wc.ast.AndNot) -> tuple[wc.ast.Node, ...]:
return (*self.visit(node.positive), node.negative)
def visit_and(self, node: wc.ast.And) -> tuple[wc.ast.Node, ...]:
return tuple(
negation for child in node.children for negation in self.visit(child)
)
def visit_boosted(self, node: wc.ast.Boosted) -> tuple[wc.ast.Node, ...]:
return self.visit(node.child)
def visit_andmaybe(self, node: wc.ast.AndMaybe) -> tuple[wc.ast.Node, ...]:
return self.visit(node.required)
def visit_require(self, node: wc.ast.Require) -> tuple[wc.ast.Node, ...]:
return (*self.visit(node.scored), *self.visit(node.filter_only))
def _negation_clauses(
index: tantivy.Index,
ast: wc.ast.Node,
registry: wc.FieldRegistry,
) -> list[tuple[tantivy.Occur, tantivy.Query]]:
"""MustNot clauses for everything ``ast`` excludes conjunctively.
Each excluded subtree is emitted as its own positive query and attached
with ``MustNot``, rather than emitting a negative query and hoping
tantivy accepts a bare one.
"""
try:
return [
(
tantivy.Occur.MustNot,
tantivy_emit(negation, index=index, registry=registry),
)
for negation in _ConjunctiveNegations().visit(ast)
]
except QueryError as e:
raise _map_emit_error(e) from e
def _any_of(clauses: list[tuple[tantivy.Occur, tantivy.Query]]) -> tantivy.Query:
"""Collapse a clause list: none -> empty, one -> itself (no wasted
single-clause boolean_query wrapping), many -> boolean_query(clauses)."""
if not clauses:
return tantivy.Query.empty_query()
if len(clauses) == 1:
return clauses[0][1]
return tantivy.Query.boolean_query(clauses)
def _simple_query_tokens(raw_query: str) -> list[str]:
# Tokenize and fold via the same analyzer used to index simple_title /
# simple_content, so query terms fold identically to the indexed terms
# (single source of truth for ASCII folding).
return simple_search_tokens(raw_query)
def _build_simple_token_query(
@@ -424,7 +168,9 @@ def _build_simple_token_query(
query = tantivy.Query.boost_query(query, boost)
field_queries.append((tantivy.Occur.Should, query))
return _any_of(field_queries)
if len(field_queries) == 1:
return field_queries[0][1]
return tantivy.Query.boolean_query(field_queries)
def parse_user_query(
@@ -433,53 +179,52 @@ def parse_user_query(
tz: tzinfo,
) -> tantivy.Query:
"""
Parse user query through whoosh-compat, then blend in fuzzy/CJK clauses.
Parse user query through the complete preprocessing pipeline.
1. wc.parse() against the shared FieldRegistry (whoosh grammar -> AST).
Bare notes:/custom_fields: prefixes resolve to their default subpath
(notes.note:/custom_fields.value:) directly in the registry, via
each JSON field's SubpathSpec(default=True).
2. Any diagnostics (bad dates/numbers) map to SearchQueryError subclasses
and raise, the view returns HTTP 400 with every offending field
listed, not just the first.
3. emit() turns the AST into a tantivy.Query directly (no string
round-trip). A QueryError is routed by its Diagnostic's Cause
(_map_emit_error): a construct that parses but can't execute against
tantivy (e.g. a text-field range) is a 400, a registry/schema
mismatch is logged and a 400, and an INTERNAL defect is re-raised.
4. Optional fuzzy blend (ADVANCED_FUZZY_SEARCH_THRESHOLD) builds a
plain word string from the parsed AST's free-text tokens
(whoosh_compat.free_text_tokens) and feeds THAT to
index.parse_query, never raw_query, whose whoosh grammar (date
keywords, bracket-class wildcards, etc.) tantivy's parser rejects,
which used to silently knock the fuzzy clause out of any mixed
query (see _try_parse_fuzzy_query).
5. Optional CJK bigram clause, built from the same parsed AST for the
same reason (see _build_ast_cjk_query): a CJK term the query negated
or fielded must not resurface through it.
6. When any optional clause was added, the query's conjunctive
exclusions are restated as MustNot above the blend
(_negation_clauses): a clause built from positive terms cannot
express them, and as a bare Should it would undo them.
Transforms the raw user query through multiple stages:
1. Date keyword rewriting (today ISO 8601 ranges)
2. Query normalization (comma expansion, whitespace cleanup)
3. Tantivy parsing with field boosts
4. Optional fuzzy query blending (if ADVANCED_FUZZY_SEARCH_THRESHOLD set)
Args:
index: Tantivy index with registered tokenizers
raw_query: Original user query string
tz: Timezone for date boundary calculations
Returns:
Parsed Tantivy query ready for execution
Note:
When ADVANCED_FUZZY_SEARCH_THRESHOLD is configured, adds a low-priority
fuzzy query as a Should clause (0.1 boost) to catch approximate matches
while keeping exact matches ranked higher. The threshold value is applied
as a post-search score filter, not during query construction.
"""
registry = get_field_registry(settings.SEARCH_LANGUAGE)
result = wc.parse(
raw_query,
registry=registry,
default_fields=_DEFAULT_SEARCH_FIELDS,
field_boosts=_FIELD_BOOSTS,
tz=tz,
)
if result.diagnostics:
raise _diagnostics_to_error(result.diagnostics)
try:
exact = tantivy_emit(result.ast, index=index, registry=registry)
except QueryError as e:
raise _map_emit_error(e) from e
query_str = translate_query(raw_query, tz)
except SearchQueryError:
# Intentional, user-fixable error (e.g. an unparsable date). Propagate so
# the view can return a 400 with a helpful message rather than falling
# back to the raw (still-invalid) query.
raise
except Exception: # pragma: no cover - defensive
logger.warning("Query translation failed; using raw query", exc_info=True)
query_str = raw_query
exact = index.parse_query(
query_str,
DEFAULT_SEARCH_FIELDS,
field_boosts=_FIELD_BOOSTS,
)
# The standard analyzer keeps a whitespace-free CJK run as a single token,
# so substring queries can't match content/title (and long runs are dropped
# by remove_long). Route CJK queries to the bigram fields, whose ngram
# tokenizer indexes overlapping 2-grams for substring matching.
cjk_query = (
_build_ast_cjk_query(index, result.ast, registry)
_build_cjk_query(index, raw_query, _CJK_ALL_FIELDS)
if _has_cjk(raw_query)
else None
)
@@ -490,73 +235,22 @@ def parse_user_query(
threshold = settings.ADVANCED_FUZZY_SEARCH_THRESHOLD
if threshold is not None:
fuzzy = _try_parse_fuzzy_query(index, result.ast, registry)
if fuzzy is not None:
clauses.append(
(tantivy.Occur.Should, tantivy.Query.boost_query(fuzzy, 0.1)),
)
fuzzy = index.parse_query(
query_str,
DEFAULT_SEARCH_FIELDS,
field_boosts=_FIELD_BOOSTS,
# (prefix=True, distance=1, transposition_cost_one=True) — edit-distance fuzziness
fuzzy_fields={f: (True, 1, True) for f in DEFAULT_SEARCH_FIELDS},
)
# 0.1 boost keeps fuzzy hits ranked below exact matches (intentional)
clauses.append((tantivy.Occur.Should, tantivy.Query.boost_query(fuzzy, 0.1)))
if cjk_query is not None:
clauses.append((tantivy.Occur.Should, cjk_query))
if len(clauses) == 1:
return exact
# The fuzzy and CJK clauses are built from positive terms only, so as
# plain Shoulds beside the exact clause they re-admit exactly the
# documents the query excluded. Restate the exclusions once, above the
# whole blend. Redundant against the exact clause, which already
# carries them, but idempotently so.
negations = _negation_clauses(index, result.ast, registry)
if not negations:
return _any_of(clauses)
return tantivy.Query.boolean_query(
[(tantivy.Occur.Must, _any_of(clauses)), *negations],
)
# The three whoosh-compat kinds for a wildcard on a field that cannot
# carry one. d.field_kind supplies the discriminator, so naming the field's
# type needs no second trip through the registry.
_PATTERN_ON_KINDS: Final = frozenset(
{
DiagnosticKind.PATTERN_ON_NUMERIC,
DiagnosticKind.PATTERN_ON_BOOLEAN_EXISTS,
DiagnosticKind.PATTERN_ON_SUBPATH,
},
)
def _diagnostics_to_error(diagnostics: tuple[Diagnostic, ...]) -> SearchQueryError:
errors = [_single_diagnostic_to_error(d) for d in diagnostics]
return errors[0] if len(errors) == 1 else MultipleSearchQueryErrors(errors)
def _single_diagnostic_to_error(d: Diagnostic) -> SearchQueryError:
# d.field is a FieldRef, not a str: str(d.field) gives the canonical
# dotted name (an aliased query, e.g. type:, reports document_type).
field_name = str(d.field) if d.field is not None else None
if d.kind is DiagnosticKind.BAD_DATE:
return InvalidDateQuery(field_name, d.raw_value)
if d.kind is DiagnosticKind.BAD_NUMBER:
return InvalidNumberQuery(field_name, d.raw_value)
if d.kind is DiagnosticKind.TOO_DEEP:
return SearchQueryError("The search query is nested too deeply.")
if d.kind in _PATTERN_ON_KINDS:
kind_label = f" ({d.field_kind.name.lower()})" if d.field_kind else ""
return SearchQueryError(
f"Wildcard patterns are not supported for field "
f"{field_name!r}{kind_label}.",
)
if d.kind is DiagnosticKind.SINGLE_CHAR_BRACKET_RANGE:
field_label = f" for field {field_name!r}" if field_name else ""
return SearchQueryError(
f"{d.raw_value!r} looks like a bracket range{field_label}, but "
"'[' is not a wildcard character on its own. Combine it with a "
"wildcard, e.g. a trailing '*', or double-quote the value to "
"search it as literal text.",
)
logger.warning("Unmapped parse diagnostic %s: %s", d.kind, d.message)
return SearchQueryError("The search query could not be executed.")
return tantivy.Query.boolean_query(clauses)
def parse_simple_query(
@@ -574,7 +268,7 @@ def parse_simple_query(
CJK substrings the simple analyzer can't (long whitespace-free runs are
dropped by remove_long).
"""
tokens = simple_search_tokens(raw_query)
tokens = _simple_query_tokens(raw_query)
clauses: list[tuple[tantivy.Occur, tantivy.Query]] = []
if tokens:
@@ -597,14 +291,23 @@ def parse_simple_query(
)
for token in tokens
]
clauses.append((tantivy.Occur.Should, _any_of(token_queries)))
simple_query = (
token_queries[0][1]
if len(token_queries) == 1
else tantivy.Query.boolean_query(token_queries)
)
clauses.append((tantivy.Occur.Should, simple_query))
if cjk_fields and _has_cjk(raw_query):
cjk_q = _build_cjk_query(index, raw_query, cjk_fields)
if cjk_q is not None:
clauses.append((tantivy.Occur.Should, cjk_q))
return _any_of(clauses)
if not clauses:
return tantivy.Query.empty_query()
if len(clauses) == 1:
return clauses[0][1]
return tantivy.Query.boolean_query(clauses)
def parse_simple_text_highlight_query(
@@ -619,21 +322,13 @@ def parse_simple_text_highlight_query(
# Strip Tantivy operator chars before tokenizing: this is a plain-text
# highlight query, not a structured boolean query, so +/- are separators.
tokens = simple_search_tokens(
tokens = _simple_query_tokens(
regex.sub(r"[-+]", " ", raw_query, timeout=_REGEX_TIMEOUT),
)
if not tokens:
return tantivy.Query.empty_query()
# Quote each token as its own phrase, escaping backslashes and embedded
# quotes. simple search tokens can carry arbitrary Tantivy syntax
# characters (`"`, `:`, `(`, `[`, `/`, ...) that the query-string parser
# would otherwise interpret as query grammar rather than literal text.
quoted_tokens = [
'"' + token.replace("\\", "\\\\").replace('"', '\\"') + '"' for token in tokens
]
return index.parse_query(" ".join(quoted_tokens), ["content"])
return index.parse_query(" ".join(tokens), ["content"])
def parse_simple_text_query(
@@ -647,7 +342,7 @@ def parse_simple_text_query(
return parse_simple_query(
index,
raw_query,
_SIMPLE_SEARCH_FIELDS,
SIMPLE_SEARCH_FIELDS,
cjk_fields=_CJK_CONTENT_FIELDS,
)
@@ -663,6 +358,6 @@ def parse_simple_title_query(
return parse_simple_query(
index,
raw_query,
_TITLE_SEARCH_FIELDS,
TITLE_SEARCH_FIELDS,
cjk_fields=_CJK_TITLE_FIELDS,
)
-91
View File
@@ -1,91 +0,0 @@
from __future__ import annotations
import dataclasses
from typing import TYPE_CHECKING
from whoosh_compat import FieldKind
from whoosh_compat import FieldRegistry
from documents.search._fields import PUBLIC_FIELDS
from documents.search._tokenizer import ascii_fold
from documents.search._tokenizer import paperless_text_analyzer
from documents.search._tokenizer import stem_pattern_text
if TYPE_CHECKING:
from whoosh_compat import PatternNormalizer
_registry_cache: dict[str | None, FieldRegistry] = {}
def _identity_analyzer(text: str) -> list[str]:
"""Analyzer for KEYWORD fields indexed with the raw tokenizer (no splitting)."""
return [text]
def _fold_normalizer(text: str) -> str:
"""Wildcard/regex literal-run normalizer for fields indexed without stemming."""
return ascii_fold(text.lower())
def _make_pattern_normalizer(language: str | None) -> PatternNormalizer:
"""Build the wildcard/regex literal-run normalizer for a search language."""
def _pattern_normalizer(text: str) -> tuple[str, ...]:
"""Normalize a literal run into the forms a term may match.
TEXT index terms go through lowercase -> ascii_fold -> stem, so a
pattern that skips stemming can never match one: "invoice*" would look
for a term starting with "invoice" while the index holds "invoic". The
run is therefore offered stemmed as well. KEYWORD fields are indexed
raw and get _fold_normalizer instead, so their patterns stay literal.
Both forms are returned, as alternatives, because neither is a prefix
of the other in general: English stemming substitutes as well as
truncates ("copy" -> "copi"), so the stem alone loses the compounds
the typed run reaches ("copyright") while the typed run alone loses
the inflections the stem reaches ("copies"). whoosh-compat ORs the
alternatives per literal run and deduplicates them, so a run the
stemmer leaves alone costs exactly the one branch it did before.
Inside a bracket class the emitter calls this once per character and
uses the answer only if it is a single one-character form; two forms
there leave the character as typed. A stemmer does not change a lone
character, so the two forms deduplicate to one and the class body is
folded as before.
"""
folded = ascii_fold(text.lower())
stemmed = stem_pattern_text(folded, language)
return (folded, stemmed)
return _pattern_normalizer
def get_field_registry(language: str | None) -> FieldRegistry:
"""Build (or return the cached) FieldRegistry for the given search language.
Cached keyed by language, rebuilt on the same trigger register_tokenizers()
uses (settings.SEARCH_LANGUAGE change). A fresh call with a new language
builds and caches a new registry rather than mutating the old one.
"""
if language in _registry_cache:
return _registry_cache[language]
text_analyzer = paperless_text_analyzer(language).analyze
pattern_normalizer = _make_pattern_normalizer(language)
specs = [
dataclasses.replace(
field,
analyzer=_identity_analyzer
if field.kind is FieldKind.KEYWORD
else text_analyzer,
pattern_normalizer=_fold_normalizer
if field.kind is FieldKind.KEYWORD
else pattern_normalizer,
)
for field in PUBLIC_FIELDS
]
registry = FieldRegistry(specs)
_registry_cache[language] = registry
return registry
+83 -222
View File
@@ -1,19 +1,14 @@
from __future__ import annotations
import hashlib
import json
import logging
import shutil
from typing import TYPE_CHECKING
from typing import Final
from typing import NamedTuple
from typing import cast
import tantivy
from django.conf import settings
from whoosh_compat import FieldKind
from documents.search._fields import PUBLIC_FIELDS
if TYPE_CHECKING:
from pathlib import Path
@@ -21,185 +16,7 @@ if TYPE_CHECKING:
logger = logging.getLogger("paperless.search")
# v1 - Initial tantivy schema format
# v2 - build_schema() derived from PUBLIC_FIELDS, changing the field declaration
# order, and the write-only correspondent/document_type/storage_path/tag id
# columns dropped. tantivy compares schemas by ordered field list, so an
# index built by v1 rejects every write against the v2 schema.
SCHEMA_VERSION: Final[int] = 2
class FieldDescriptor(NamedTuple):
"""One tantivy field, in declaration order.
The descriptor vocabulary is paperless', not tantivy-py's: it is both the
input to the SchemaBuilder and the input to schema_fingerprint(), so the
persisted fingerprint cannot move under a tantivy-py upgrade.
"""
name: str
kind: str
stored: bool
indexed: bool
fast: bool
tokenizer: str | None
# (schema kind, tokenizer) for the FieldKind -> FieldDescriptor mapping that
# doesn't need special-casing. JSON is handled separately below since it can
# emit a second, synthetic descriptor.
_KIND_TABLE: Final[dict[FieldKind, tuple[str, str | None]]] = {
FieldKind.TEXT: ("text", "paperless_text"),
FieldKind.KEYWORD: ("text", "raw"),
FieldKind.U64: ("u64", None),
FieldKind.DATE: ("date", None),
FieldKind.DATETIME: ("date", None),
}
# Kinds whose fast-field flag follows FieldSpec.fast rather than always False.
_FAST_FROM_FIELD: Final[frozenset[FieldKind]] = frozenset(
{FieldKind.U64, FieldKind.DATE, FieldKind.DATETIME},
)
def _public_field_descriptors() -> list[FieldDescriptor]:
"""Descriptors for the query-visible fields declared in PUBLIC_FIELDS."""
descriptors: list[FieldDescriptor] = []
for field in PUBLIC_FIELDS:
if field.kind is FieldKind.JSON:
descriptors.append(
FieldDescriptor(
field.name,
"json",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
)
if field.name == "notes":
# Plain-text companion for snippet generation: tantivy's
# SnippetGenerator does not support JSON fields. Schema-only,
# no query-syntax meaning, not in PUBLIC_FIELDS.
descriptors.append(
FieldDescriptor(
"notes_text",
"text",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
)
continue
schema_kind, tokenizer = _KIND_TABLE[field.kind]
descriptors.append(
FieldDescriptor(
field.name,
schema_kind,
stored=True,
indexed=True,
fast=field.fast if field.kind in _FAST_FROM_FIELD else False,
tokenizer=tokenizer,
),
)
return descriptors
def field_descriptors() -> list[FieldDescriptor]:
"""Every field of the document index, in the order tantivy declares them.
tantivy compares schemas by *ordered* field list, so the order here is
part of the on-disk contract: schema_fingerprint() hashes it and
needs_rebuild() acts on the result.
"""
return [
FieldDescriptor(
"id",
"u64",
stored=True,
indexed=True,
fast=True,
tokenizer=None,
),
*_public_field_descriptors(),
# Shadow sort fields - fast, not stored
*(
FieldDescriptor(
name,
"text",
stored=False,
indexed=True,
fast=True,
tokenizer="simple_analyzer",
)
for name in ("title_sort", "correspondent_sort", "type_sort")
),
# CJK support - not stored, indexed only
*(
FieldDescriptor(
name,
"text",
stored=False,
indexed=True,
fast=False,
tokenizer="bigram_analyzer",
)
for name in (
"bigram_content",
"bigram_title",
"bigram_correspondent",
"bigram_document_type",
"bigram_tag",
)
),
# Simple substring search support for title/content - not stored,
# indexed only
*(
FieldDescriptor(
name,
"text",
stored=False,
indexed=True,
fast=False,
tokenizer="simple_search_analyzer",
)
for name in ("simple_title", "simple_content")
),
# Autocomplete prefix scan via terms_with_prefix, which walks the
# field's term dictionary - so the field must be indexed (term dict),
# not stored. The stored value is never read back, so storing it only
# wastes space.
FieldDescriptor(
"autocomplete_word",
"text",
stored=False,
indexed=True,
fast=False,
tokenizer="raw",
),
# Permission filter columns, read by build_permission_filter.
*(
FieldDescriptor(
name,
"u64",
stored=False,
indexed=True,
fast=True,
tokenizer=None,
)
for name in ("owner_id", "viewer_id", "viewer_group_id")
),
]
def schema_fingerprint() -> str:
"""Hash of the field descriptors, stamped into .index_settings.json.
Changes whenever a field is added, removed, retyped, re-optioned or
reordered, so an index built from a different schema shape is detected
even when SCHEMA_VERSION was not bumped.
"""
payload = json.dumps([list(descriptor) for descriptor in field_descriptors()])
return hashlib.blake2b(payload.encode()).hexdigest()
SCHEMA_VERSION: Final[int] = 1
def build_schema() -> tantivy.Schema:
@@ -215,37 +32,85 @@ def build_schema() -> tantivy.Schema:
"""
sb = tantivy.SchemaBuilder()
for descriptor in field_descriptors():
if descriptor.kind == "text":
sb.add_text_field(
descriptor.name,
stored=descriptor.stored,
fast=descriptor.fast,
tokenizer_name=cast("str", descriptor.tokenizer),
)
elif descriptor.kind == "json":
sb.add_json_field(
descriptor.name,
stored=descriptor.stored,
fast=descriptor.fast,
tokenizer_name=cast("str", descriptor.tokenizer),
)
elif descriptor.kind == "u64":
sb.add_unsigned_field(
descriptor.name,
stored=descriptor.stored,
indexed=descriptor.indexed,
fast=descriptor.fast,
)
elif descriptor.kind == "date":
sb.add_date_field(
descriptor.name,
stored=descriptor.stored,
indexed=descriptor.indexed,
fast=descriptor.fast,
)
else:
raise ValueError(f"Unknown schema field kind: {descriptor.kind}")
sb.add_unsigned_field("id", stored=True, indexed=True, fast=True)
sb.add_text_field("checksum", stored=True, tokenizer_name="raw")
for field in (
"title",
"correspondent",
"document_type",
"storage_path",
"original_filename",
"content",
):
sb.add_text_field(field, stored=True, tokenizer_name="paperless_text")
# Shadow sort fields - fast, not stored/indexed
for field in ("title_sort", "correspondent_sort", "type_sort"):
sb.add_text_field(
field,
stored=False,
tokenizer_name="simple_analyzer",
fast=True,
)
# CJK support - not stored, indexed only
sb.add_text_field("bigram_content", stored=False, tokenizer_name="bigram_analyzer")
sb.add_text_field("bigram_title", stored=False, tokenizer_name="bigram_analyzer")
sb.add_text_field(
"bigram_correspondent",
stored=False,
tokenizer_name="bigram_analyzer",
)
sb.add_text_field(
"bigram_document_type",
stored=False,
tokenizer_name="bigram_analyzer",
)
sb.add_text_field("bigram_tag", stored=False, tokenizer_name="bigram_analyzer")
# Simple substring search support for title/content - not stored, indexed only
sb.add_text_field(
"simple_title",
stored=False,
tokenizer_name="simple_search_analyzer",
)
sb.add_text_field(
"simple_content",
stored=False,
tokenizer_name="simple_search_analyzer",
)
# Autocomplete prefix scan via terms_with_prefix, which walks the field's
# term dictionary - so the field must be indexed (term dict), not stored.
# The stored value is never read back, so storing it only wastes space.
sb.add_text_field("autocomplete_word", stored=False, tokenizer_name="raw")
sb.add_text_field("tag", stored=True, tokenizer_name="paperless_text")
# JSON fields — structured queries: notes.user:alice, custom_fields.name:invoice
sb.add_json_field("notes", stored=True, tokenizer_name="paperless_text")
# Plain-text companion for notes — tantivy's SnippetGenerator does not support
# JSON fields, so highlights require a text field with the same content.
sb.add_text_field("notes_text", stored=True, tokenizer_name="paperless_text")
sb.add_json_field("custom_fields", stored=True, tokenizer_name="paperless_text")
for field in (
"correspondent_id",
"document_type_id",
"storage_path_id",
"tag_id",
"owner_id",
"viewer_id",
"viewer_group_id",
):
sb.add_unsigned_field(field, stored=False, indexed=True, fast=True)
for field in ("created", "modified", "added"):
sb.add_date_field(field, stored=True, indexed=True, fast=True)
for field in ("asn", "page_count", "num_notes"):
sb.add_unsigned_field(field, stored=True, indexed=True, fast=True)
return sb.build()
@@ -254,9 +119,9 @@ def needs_rebuild(index_dir: Path) -> bool:
"""
Check if the search index needs rebuilding.
Reads .index_settings.json to compare the stored schema version, search
language and schema fingerprint against the current configuration. Returns
True if the file is missing, unparsable, or any value mismatches.
Reads .index_settings.json to compare the stored schema version and
search language against the current configuration. Returns True if the
file is missing, unparsable, or either value mismatches.
Args:
index_dir: Path to the search index directory
@@ -275,9 +140,6 @@ def needs_rebuild(index_dir: Path) -> bool:
if "language" not in data or data["language"] != settings.SEARCH_LANGUAGE:
logger.info("Search index language changed - rebuilding.")
return True
if data.get("schema_fingerprint") != schema_fingerprint():
logger.info("Search index schema fingerprint mismatch - rebuilding.")
return True
except ValueError:
return True
return False
@@ -308,7 +170,6 @@ def _write_sentinels(index_dir: Path) -> None:
{
"schema_version": SCHEMA_VERSION,
"language": settings.SEARCH_LANGUAGE,
"schema_fingerprint": schema_fingerprint(),
},
),
)
+2 -51
View File
@@ -1,7 +1,6 @@
from __future__ import annotations
import logging
from functools import cache
from typing import Final
import tantivy
@@ -72,7 +71,7 @@ def register_tokenizers(index: tantivy.Index, language: str | None) -> None:
use fast=True and Tantivy requires fast-field tokenizers to exist
even for documents that omit those fields.
"""
index.register_tokenizer("paperless_text", paperless_text_analyzer(language))
index.register_tokenizer("paperless_text", _paperless_text(language))
index.register_tokenizer("simple_analyzer", _simple_analyzer())
index.register_tokenizer("bigram_analyzer", _bigram_analyzer())
index.register_tokenizer("simple_search_analyzer", _simple_search_analyzer())
@@ -80,7 +79,7 @@ def register_tokenizers(index: tantivy.Index, language: str | None) -> None:
index.register_fast_field_tokenizer("simple_analyzer", _simple_analyzer())
def paperless_text_analyzer(language: str | None) -> tantivy.TextAnalyzer:
def _paperless_text(language: str | None) -> tantivy.TextAnalyzer:
"""Main full-text tokenizer for content, title, etc: simple -> remove_long(129) -> lowercase -> ascii_fold [-> stemmer]"""
builder = (
tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.simple())
@@ -101,54 +100,6 @@ def paperless_text_analyzer(language: str | None) -> tantivy.TextAnalyzer:
return builder.build()
@cache
def _pattern_stemmer(language: str | None) -> tantivy.TextAnalyzer | None:
"""The stemming tail of paperless_text_analyzer, over a whole literal run.
Same language gate and same Snowball stemmer paperless_text_analyzer
applies at index time, so query patterns follow SEARCH_LANGUAGE. Returns
None when that gate disables stemming; paperless_text_analyzer already
warns about an unsupported language, so this stays quiet.
The raw tokenizer keeps the run whole (a wildcard literal is a fragment,
not necessarily a word), and remove_long is kept so an over-long run is
treated the same way the index treats it.
"""
if not language:
return None
tantivy_lang = _LANGUAGE_MAP.get(language.lower())
if tantivy_lang is None:
return None
return (
tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.raw())
.filter(tantivy.Filter.remove_long(_TOKEN_REMOVE_LONG_LIMIT))
.filter(tantivy.Filter.stemmer(tantivy_lang))
.build()
)
def stem_pattern_text(text: str, language: str | None) -> str:
"""Stem an already lowercased/ascii-folded run the way index terms are.
Returns text unchanged when stemming is disabled for language, and also
when the stem step does not yield exactly one token: remove_long drops a run
past the length limit, leaving no stem to substitute. Falling back to the
text as typed is the safe direction for a pattern prefix, since it can only
be as narrow as it was before stemming was considered.
The raw tokenizer emits one token whatever the input and the stemmer is
1-to-1, so only the zero-token case can fire today; the guard covers both
counts so a tokenizer change cannot turn this into an IndexError.
"""
analyzer = _pattern_stemmer(language)
if analyzer is None:
return text
tokens = analyzer.analyze(text)
if len(tokens) != 1:
return text
return tokens[0]
def _simple_analyzer() -> tantivy.TextAnalyzer:
"""Tokenizer for shadow sort fields (title_sort, correspondent_sort, type_sort): simple -> lowercase -> ascii_fold."""
return (
+610
View File
@@ -0,0 +1,610 @@
from __future__ import annotations
from dataclasses import dataclass
from datetime import UTC
from datetime import datetime
from datetime import timedelta
from typing import TYPE_CHECKING
from typing import TypeAlias
import regex
from dateutil.relativedelta import relativedelta
from documents.search._dates import _DATE_KEYWORDS
from documents.search._dates import _DATE_ONLY_FIELDS
from documents.search._dates import _date_only_range
from documents.search._dates import _datetime_range
from documents.search._dates import _field_range_from_dates
from documents.search._dates import _fmt
from documents.search._dates import _precision_bounds
from documents.search._dates import _utc_bounds_for_field
# Compiled regex that matches any known multi-word (or single-word) date keyword
# at the start of a match position, longest alternatives first so "previous week"
# wins over a hypothetical shorter "previous".
_KEYWORD_VALUE_RE = regex.compile(
"|".join(sorted((regex.escape(k) for k in _DATE_KEYWORDS), key=len, reverse=True)),
regex.IGNORECASE,
)
if TYPE_CHECKING:
from datetime import tzinfo
# TODO: this module translates date queries into Tantivy *string* syntax, which
# forces a workaround for something Tantivy's string parser cannot express on
# date fields: open-ended ranges use far-past/far-future string sentinels
# (OPEN_LO/OPEN_HI). These can be replaced with a real tantivy.Query object
# (Query.range_query(..., None) for open bounds) once tantivy-py accepts Python
# datetimes in range_query/term_query on Date fields. That support exists on
# tantivy-py master (PRs #655 + #666) but postdates the pinned 0.26.0 wheel, so
# it is blocked only on a published release > 0.26.0 and a dependency bump.
# (Unparsable dates now raise InvalidDateQuery -> HTTP 400 rather than using a
# no-match string sentinel.)
# Fields that store exact, non-analyzed comma-joined tokens in the index and so
# need explicit comma->AND expansion (Whoosh KEYWORD(commas=True) set).
MULTI_VALUE_FIELDS = frozenset({"tag", "tag_id", "viewer_id"})
# Date fields whose values/ranges get rewritten to RFC3339 Tantivy ranges.
DATE_FIELDS = frozenset({"created", "modified", "added"})
# Field aliases: Whoosh (v2) field names that were renamed in the Tantivy schema.
# Preserved here so v2 queries using the old names continue to work without 400
# errors instead of silently failing. Applied by _render to non-date field tokens.
FIELD_ALIASES: dict[str, str] = {
"type": "document_type",
"type_id": "document_type_id",
"path": "storage_path",
"path_id": "storage_path_id",
}
# Known schema fields: a comma immediately followed by ``<known>:`` is a clause
# separator. Restricting to known fields prevents URL-like ``http:`` misfires.
KNOWN_FIELDS = frozenset(
{
"title",
"content",
"correspondent",
"document_type",
"type", # v2 alias -> document_type
"storage_path",
"path", # v2 alias -> storage_path
"tag",
"tag_id",
"correspondent_id",
"document_type_id",
"type_id", # v2 alias -> document_type_id
"storage_path_id",
"path_id", # v2 alias -> storage_path_id
"owner_id",
"viewer_id",
"asn",
"page_count",
"num_notes",
"created",
"modified",
"added",
"original_filename",
"checksum",
"notes",
"custom_fields",
},
)
_FIELD_RE = regex.compile(r"(?P<field>\w+):")
# Matches the TO separator inside a range bracket. Handles three forms:
# middle: "lo TO hi" (either lo or hi may be empty)
# trailing: "lo TO" (open upper bound)
# leading: "TO hi" (open lower bound)
# Bounds MAY contain internal spaces (e.g. "-7 days"), so we use .*? / .+?
# and split on the whitespace-delimited " TO " / " to " separator.
_RANGE_RE = regex.compile(
r"^\s*(?P<lo>.*?)\s+[Tt][Oo]\s+(?P<hi>.+?)\s*$"
r"|"
r"^\s*(?P<lo2>.+?)\s+[Tt][Oo]\s*$"
r"|"
r"^\s*[Tt][Oo]\s+(?P<hi2>.+?)\s*$",
)
@dataclass(frozen=True, slots=True)
class FieldValue:
field: str
value: str
# Produced by the comma-resolution pass (not by scan()).
@dataclass(frozen=True, slots=True)
class FieldValueList:
field: str
values: tuple[str, ...]
@dataclass(frozen=True, slots=True)
class FieldRange:
field: str
open: str
lo: str
hi: str
close: str
# Produced by the comma-resolution pass (not by scan()).
@dataclass(frozen=True, slots=True)
class Comma:
pass
@dataclass(frozen=True, slots=True)
class Passthrough:
raw: str
Token: TypeAlias = FieldValue | FieldValueList | FieldRange | Comma | Passthrough
_CLOSE: dict[str, str] = {"[": "]", "{": "}"}
def scan(query: str) -> list[Token]:
"""
Tokenize a raw query into date/comma-aware tokens, leaving everything else
as verbatim ``Passthrough`` runs. Non-recursive: finds the first matching
close bracket/quote. Nested brackets are not valid Tantivy range syntax and
pass through verbatim on mismatch.
"""
tokens: list[Token] = []
buf: list[str] = [] # accumulates passthrough chars
i, n = 0, len(query)
while i < n:
matched = _match_field_token(query, i)
if matched is None:
buf.append(query[i])
i += 1
continue
token, i = matched
if buf and buf[-1] == ",":
buf.pop()
_flush(buf, tokens)
tokens.append(Comma())
else:
_flush(buf, tokens)
tokens.append(token)
i = _maybe_comma(query, i, tokens)
_flush(buf, tokens)
return tokens
def _flush(buf: list[str], tokens: list[Token]) -> None:
"""Emit any accumulated passthrough characters as a single token."""
if buf:
tokens.append(Passthrough("".join(buf)))
buf.clear()
def _at_word_boundary(query: str, i: int) -> bool:
"""A field token may begin only at the start or after a non-word character."""
return i == 0 or not (query[i - 1].isalnum() or query[i - 1] == "_")
def _match_field_token(query: str, i: int) -> tuple[Token, int] | None:
"""
If a known ``field:`` token starts at ``i``, consume it and return
``(token, end_index)``; otherwise return None so the caller treats the
character as passthrough. Handles both ``field:[range]`` and ``field:value``,
and returns None when the range/value cannot be consumed.
"""
m = _FIELD_RE.match(query, i)
if m is None or m.group("field") not in KNOWN_FIELDS:
return None
if not _at_word_boundary(query, i):
return None
field = m.group("field")
j = m.end()
if j < len(query) and query[j] in "[{":
return _consume_range(query, j, field)
consumed = _consume_field_value(query, field, j)
if consumed is None:
return None
value, end = consumed
return FieldValue(field, value), end
def _consume_field_value(query: str, field: str, start: int) -> tuple[str, int] | None:
"""
Consume a field value starting at ``start``: a multi-word date keyword phrase
(date fields only), or a bare/quoted value, then absorb any comma-joined
continuation that is not a clause separator. ``resolve_commas`` later splits a
multi-value field's joined value into a ``FieldValueList``; for other fields
the comma stays literal.
"""
n = len(query)
consumed = None
if field in DATE_FIELDS:
km = _KEYWORD_VALUE_RE.match(query, start)
if km is not None and (km.end() >= n or query[km.end()] in " \t),"):
consumed = (km.group(0), km.end())
if consumed is None:
consumed = _consume_value(query, start)
if consumed is None:
return None
value, k = consumed
while k < n and query[k] == ",":
if _looks_like_known_field(query, k + 1):
break # clause separator: left for _maybe_comma to emit a Comma()
more = _consume_value(query, k + 1)
if more is None:
break
value = f"{value},{more[0]}"
k = more[1]
return value, k
def _consume_range(
query: str,
start: int,
field: str,
) -> tuple[FieldRange, int] | None:
"""Consume ``[lo TO hi]`` / ``{lo TO hi}`` from ``start`` (the bracket)."""
open_br = query[start]
close_br = _CLOSE[open_br]
end = query.find(close_br, start + 1)
if end == -1:
return None
inner = query[start + 1 : end]
m = _RANGE_RE.match(inner)
if m is not None:
if m.group("lo") is not None or m.group("hi") is not None:
# Middle form: "lo TO hi" (either may be empty string)
lo = (m.group("lo") or "").strip()
hi = (m.group("hi") or "").strip()
elif m.group("lo2") is not None:
# Trailing form: "lo TO"
lo = m.group("lo2").strip()
hi = ""
else:
# Leading form: "TO hi"
lo = ""
hi = (m.group("hi2") or "").strip()
else:
lo, hi = inner.strip(), ""
return FieldRange(field, open_br, lo, hi, close_br), end + 1
def _consume_value(query: str, start: int) -> tuple[str, int] | None:
"""Consume a bare or quoted field value from ``start``, stopping at comma."""
n = len(query)
if start >= n or query[start] in " \t":
return None
if query[start] in "\"'":
quote = query[start]
end = query.find(quote, start + 1)
if end == -1:
return None
return query[start : end + 1], end + 1
j = start
while j < n and query[j] not in " \t),":
j += 1
return query[start:j], j
def _looks_like_known_field(query: str, pos: int) -> bool:
"""True if a known ``field:`` token starts at ``pos``."""
m = _FIELD_RE.match(query, pos)
return bool(m and m.group("field") in KNOWN_FIELDS)
def _maybe_comma(query: str, i: int, tokens: list) -> int:
"""If a clause-separator comma follows at ``i``, emit ``Comma()`` and advance."""
if i < len(query) and query[i] == "," and _looks_like_known_field(query, i + 1):
tokens.append(Comma())
return i + 1
return i
def resolve_commas(tokens: list) -> list:
"""
Collapse value-list commas into ``FieldValueList`` and keep clause-separator
commas as ``Comma``. (Clause-sep commas are already emitted by ``scan`` via
the value-stop logic; this pass folds value-lists.)
"""
out: list = []
for tok in tokens:
if (
isinstance(tok, FieldValue)
and tok.field in MULTI_VALUE_FIELDS
and "," in tok.value
):
values = tuple(v for v in tok.value.split(",") if v)
out.append(FieldValueList(tok.field, values))
else:
out.append(tok)
return out
class SearchQueryError(ValueError):
"""
Base for user-fixable search query errors.
Carries a message safe to surface to the user (no internal details). The view
layer catches this and returns an HTTP 400, so any future subclass (unknown
field, malformed range, wrapped parser errors) gets the same treatment.
"""
class InvalidDateQuery(SearchQueryError):
"""Raised when a date field value or range bound cannot be parsed."""
def __init__(self, field: str, value: str) -> None:
self.field = field
self.value = value
super().__init__(f"Invalid date value {value!r} for field {field!r}.")
_DIGITS_RE = regex.compile(r"^\d{4}(?:\d{2}){0,2}$")
_ISO_RE = regex.compile(r"^\d{4}(?:-\d{2}(?:-\d{2})?)?$")
def translate_scalar(field: str, value: str, tz: tzinfo) -> str:
"""Translate a bare date-field value to a Tantivy range string."""
bare = value.strip("\"'").lower()
if bare in _DATE_KEYWORDS:
if field in _DATE_ONLY_FIELDS:
return f"{field}:{_date_only_range(bare, tz)}"
return f"{field}:{_datetime_range(bare, tz)}"
digits = value.replace("-", "")
if _DIGITS_RE.match(value) or _ISO_RE.match(value):
bounds = _precision_bounds(digits)
if bounds is None:
raise InvalidDateQuery(field, value)
return _field_range_from_dates(field, bounds[0], bounds[1], tz)
if regex.fullmatch(r"\d{14}", value):
try:
dt = datetime(
int(value[0:4]),
int(value[4:6]),
int(value[6:8]),
int(value[8:10]),
int(value[10:12]),
int(value[12:14]),
tzinfo=UTC,
)
except ValueError:
raise InvalidDateQuery(field, value) from None
iso = _fmt(dt)
return f"{field}:[{iso} TO {iso}]"
# Unrecognized shape -> tell the user their date is malformed rather than
# silently matching nothing or emitting invalid Tantivy syntax.
raise InvalidDateQuery(field, value)
# Open-bound sentinels for date ranges. These far-past/far-future strings allow
# open-ended ranges to be expressed as Tantivy string queries until tantivy-py
# exposes Query.range_query(..., None) on Date fields (see module TODO).
OPEN_LO = "0001-01-01T00:00:00Z"
OPEN_HI = "9999-12-31T23:59:59Z"
# Matches compact now-offset tokens like now-7d, now+1h, now-30m.
_NOW_COMPACT_RE = regex.compile(
r"^now(?P<sign>[+-])(?P<n>\d+)(?P<unit>[dhm])$",
regex.IGNORECASE,
)
# Matches "±N <unit>" Whoosh-style offsets (e.g. -7 days, -1 week, +3 hours).
# Whoosh's own date parser (qparser.dateparse.PlusMinus) additionally accepted
# abbreviated unit spellings (e.g. "yrs", "yr", "y", "mos", "wks", "hrs", "mins",
# "secs"); saved views/searches created under the old Whoosh backend can still
# contain those tokens (e.g. "-999yrs"), so they are accepted here too and
# normalized to a canonical unit via _UNIT_ALIASES below.
_NOW_SPACED_RE = regex.compile(
r"^(?P<sign>[+-])(?P<n>\d+)\s*"
r"(?P<unit>years|year|yrs|yr|ys|y"
r"|months|month|mons|mon|mos|mo"
r"|weeks|week|wks|wk|ws|w"
r"|days|day|dys|dy|ds|d"
r"|hours|hour|hrs|hr|hs|h"
r"|minutes|minute|mins|min|ms|m"
r"|seconds|second|secs|sec|s)$",
regex.IGNORECASE,
)
# Maps every accepted unit spelling (including Whoosh-era abbreviations) to the
# canonical unit name used as a key into the delta map in _resolve_relative_bound.
_UNIT_ALIASES: dict[str, str] = {
alias: canonical
for canonical, aliases in {
"year": ("years", "year", "yrs", "yr", "ys", "y"),
"month": ("months", "month", "mons", "mon", "mos", "mo"),
"week": ("weeks", "week", "wks", "wk", "ws", "w"),
"day": ("days", "day", "dys", "dy", "ds", "d"),
"hour": ("hours", "hour", "hrs", "hr", "hs", "h"),
"minute": ("minutes", "minute", "mins", "min", "ms", "m"),
"second": ("seconds", "second", "secs", "sec", "s"),
}.items()
for alias in aliases
}
def _resolve_relative_bound(token: str) -> datetime | None:
"""
Resolve a relative bound token to an exact UTC instant, or return None.
Supported forms:
- ``now`` -> current UTC instant
- ``now+/-<n>d/h/m`` -> now +/- timedelta (d=days, h=hours, m=minutes)
- ``±N <unit>`` -> now +/- delta; month/year use relativedelta;
unit also accepts Whoosh-era abbreviations
(e.g. "yrs", "mos", "wks", "hrs", "mins", "secs")
"""
stripped = token.strip()
low = stripped.lower()
now = datetime.now(UTC)
if low == "now":
return now
m = _NOW_COMPACT_RE.match(stripped)
if m:
sign = 1 if m.group("sign") == "+" else -1
n = int(m.group("n"))
unit = m.group("unit").lower()
delta = (
sign
* {
"d": timedelta(days=n),
"h": timedelta(hours=n),
"m": timedelta(minutes=n),
}[unit]
)
return now + delta
m = _NOW_SPACED_RE.match(stripped)
if m:
sign = 1 if m.group("sign") == "+" else -1
n = int(m.group("n"))
unit = _UNIT_ALIASES[m.group("unit").lower()]
delta_map: dict[str, timedelta | relativedelta] = {
"second": timedelta(seconds=n),
"minute": timedelta(minutes=n),
"hour": timedelta(hours=n),
"day": timedelta(days=n),
"week": timedelta(weeks=n),
"month": relativedelta(months=n),
"year": relativedelta(years=n),
}
return now - delta_map[unit] if sign == -1 else now + delta_map[unit]
return None
def _bound_datetimes(
field: str,
token: str,
tz: tzinfo,
) -> tuple[datetime, datetime] | None:
"""
Return (floor_dt, ceil_dt) UTC datetimes for a single range bound token, or
None if the token is unparsable. ``now`` and relative offsets resolve to the
current instant (floor == ceil == that instant; no day-flooring).
"""
token = token.strip()
# Try relative/now forms first (before stripping hyphens which would mangle them).
rel = _resolve_relative_bound(token)
if rel is not None:
return rel, rel
# Full ISO datetime token (contains "T"): parse directly and return an exact
# instant (floor == ceil). Python 3.11+ datetime.fromisoformat accepts trailing Z.
if "T" in token:
try:
dt = datetime.fromisoformat(token)
# Ensure timezone-aware UTC result.
dt = dt.replace(tzinfo=UTC) if dt.tzinfo is None else dt.astimezone(UTC)
return dt, dt
except ValueError:
return None
digits = token.replace("-", "")
bounds = _precision_bounds(digits)
if bounds is None:
return None
start, end = bounds
return _utc_bounds_for_field(field, start, end, tz)
def _render(tok: Token, tz: tzinfo) -> str:
"""Render a single token back to a Tantivy query string fragment."""
if isinstance(tok, Passthrough):
return tok.raw
if isinstance(tok, Comma):
return " AND "
if isinstance(tok, FieldValueList):
field = FIELD_ALIASES.get(tok.field, tok.field)
return " AND ".join(f"{field}:{v}" for v in tok.values)
if isinstance(tok, FieldValue):
field = FIELD_ALIASES.get(tok.field, tok.field)
if field in DATE_FIELDS:
return translate_scalar(field, tok.value, tz)
return f"{field}:{tok.value}"
if isinstance(tok, FieldRange):
field = FIELD_ALIASES.get(tok.field, tok.field)
if field in DATE_FIELDS:
return translate_range(field, tok.lo, tok.hi, tz)
return f"{field}:{tok.open}{tok.lo} TO {tok.hi}{tok.close}"
return "" # pragma: no cover
# Post-render operator normalization patterns: collapse repeated whitespace and
# strip spaced/trailing Tantivy boolean operators that would otherwise be invalid.
_MULTI_SPACE_RE = regex.compile(r" {2,}")
_TRAILING_OP_RE = regex.compile(r"\s+[-+]+\s*$")
_SPACED_OP_RE = regex.compile(r"\s+[-+]\s+")
def _normalize_operators(text: str) -> str:
"""
Collapse multiple spaces, strip trailing dangling operators, and replace
spaced operators (`` - `` / `` + ``) with a single space.
Applied only to Passthrough fragments (the rendered output is scanned for
operator artifacts outside bracketed ranges) via a post-render pass on the
full rendered string. This preserves date ranges (``[... TO ...]``) verbatim
while cleaning natural-language separators in the surrounding text.
"""
text = _MULTI_SPACE_RE.sub(" ", text)
text = _TRAILING_OP_RE.sub("", text).strip()
text = _SPACED_OP_RE.sub(" ", text).strip()
return text
def translate_query(raw: str, tz: tzinfo) -> str:
"""Translate a raw Whoosh-style query into Tantivy-compatible syntax."""
tokens = resolve_commas(scan(raw))
rendered = "".join(_render(t, tz) for t in tokens)
return _normalize_operators(rendered)
def translate_range(field: str, lo: str, hi: str, tz: tzinfo) -> str:
"""Translate a date-field ``[lo TO hi]`` range to a Tantivy ISO range string.
Handles partial-date bounds (YYYY, YYYYMM, YYYYMMDD, ISO dash variants),
open bounds (empty string -> OPEN_LO/OPEN_HI), ``now``, and reversed ranges
(swaps tokens before computing floor/ceil so the span is always correct).
"""
lo_s = lo.strip()
hi_s = hi.strip()
# Parse both bounds to (floor, ceil) pairs when present.
lo_pair: tuple[datetime, datetime] | None = None
hi_pair: tuple[datetime, datetime] | None = None
if lo_s:
lo_pair = _bound_datetimes(field, lo_s, tz)
if lo_pair is None:
raise InvalidDateQuery(field, lo_s)
if hi_s:
hi_pair = _bound_datetimes(field, hi_s, tz)
if hi_pair is None:
raise InvalidDateQuery(field, hi_s)
# Detect a reversed range: only swap when BOTH bounds are present.
if lo_pair is not None and hi_pair is not None and lo_pair[0] > hi_pair[0]:
lo_pair, hi_pair = hi_pair, lo_pair
lo_iso = _fmt(lo_pair[0]) if lo_pair is not None else OPEN_LO
# A bound resolves to (floor, ceil) where floor == ceil for an exact instant
# (a full ISO datetime, "now", or a "+/-N unit" offset) and floor != ceil for
# a coarser period token (year/month/day precision). Only the latter needs a
# half-open close: its ceil is the start of the *next* period and must be
# excluded, or that instant (e.g. the 1st of next month) wrongly matches.
if hi_pair is not None:
hi_iso = _fmt(hi_pair[1])
hi_close = "]" if hi_pair[0] == hi_pair[1] else "}"
else:
hi_iso = OPEN_HI
hi_close = "]"
return f"{field}:[{lo_iso} TO {hi_iso}{hi_close}"
+6
View File
@@ -2812,6 +2812,11 @@ class AcknowledgeTasksViewSerializer(serializers.Serializer[dict[str, Any]]):
class ShareLinkSerializer(OwnedObjectSerializer):
document_title = serializers.CharField(
source="document.title",
read_only=True,
)
class Meta:
model = ShareLink
fields = (
@@ -2820,6 +2825,7 @@ class ShareLinkSerializer(OwnedObjectSerializer):
"expiration",
"slug",
"document",
"document_title",
"file_version",
)
+3 -5
View File
@@ -312,10 +312,7 @@ def bulk_update_documents(document_ids) -> None:
from documents.search import get_backend
document_ids = list(document_ids)
# Annotated so the signal handlers below (e.g. matching) don't query the
# versions of each document. Indexing re-queries and re-annotates its own
# copy via add_or_update_ids() below, after these signals (and any
# workflow they trigger) have had a chance to mutate the documents.
# Annotated so indexing below doesn't query the versions of each document
documents = annotate_effective_content(
Document.objects.filter(id__in=document_ids),
)
@@ -331,7 +328,8 @@ def bulk_update_documents(document_ids) -> None:
post_save.send(Document, instance=doc, created=False)
with get_backend().batch_update() as batch:
batch.add_or_update_ids(document_ids)
for doc in documents:
batch.add_or_update(doc)
ai_config = AIConfig()
if ai_config.llm_index_enabled:
+3 -3
View File
@@ -10,7 +10,7 @@ import pytest
from django.contrib.auth import get_user_model
from django.contrib.contenttypes.models import ContentType
from guardian.shortcuts import clear_ct_cache
from pytest_django.fixtures import Settings
from pytest_django.fixtures import SettingsWrapper
from rest_framework.test import APIClient
from documents.tests.factories import DocumentFactory
@@ -100,7 +100,7 @@ def sample_doc(
@pytest.fixture()
def _search_index(
tmp_path: Path,
settings: Settings,
settings: SettingsWrapper,
) -> Generator[None, None, None]:
"""Create a temp index directory and point INDEX_DIR at it.
@@ -118,7 +118,7 @@ def _search_index(
@pytest.fixture()
def settings_timezone(settings: Settings) -> zoneinfo.ZoneInfo:
def settings_timezone(settings: SettingsWrapper) -> zoneinfo.ZoneInfo:
return zoneinfo.ZoneInfo(settings.TIME_ZONE)
+1 -1
View File
@@ -70,7 +70,7 @@ def clear_lru_cache() -> Generator[None, None, None]:
@pytest.fixture
def mock_date_parser_settings(settings: pytest_django.fixtures.Settings) -> Any:
def mock_date_parser_settings(settings: pytest_django.fixtures.SettingsWrapper) -> Any:
"""
Override Django settings for the duration of date parser tests.
"""
+3 -3
View File
@@ -6,7 +6,7 @@ from pathlib import Path
import pytest
import pytest_mock
from pytest_django.fixtures import Settings
from pytest_django.fixtures import SettingsWrapper
from documents.export.sinks import DirectoryExportSink
from documents.export.sinks import ExportSink
@@ -242,7 +242,7 @@ class TestZipExportSink:
self,
tmp_path: Path,
source_file: Path,
settings: Settings,
settings: SettingsWrapper,
) -> None:
scratch_dir = tmp_path / "scratch"
settings.SCRATCH_DIR = scratch_dir
@@ -261,7 +261,7 @@ class TestZipExportSink:
def test_abort_after_manifest_written_cleans_up_pending_tmp(
self,
tmp_path: Path,
settings: Settings,
settings: SettingsWrapper,
) -> None:
scratch_dir = tmp_path / "scratch"
settings.SCRATCH_DIR = scratch_dir
+14 -2
View File
@@ -1,21 +1,25 @@
from __future__ import annotations
import tempfile
from typing import TYPE_CHECKING
import pytest
import tantivy
from documents.search._backend import TantivyBackend
from documents.search._backend import reset_backend
from documents.search._schema import build_schema
from documents.search._tokenizer import register_tokenizers
if TYPE_CHECKING:
from collections.abc import Generator
from pathlib import Path
from pytest_django.fixtures import Settings
from pytest_django.fixtures import SettingsWrapper
@pytest.fixture
def index_dir(tmp_path: Path, settings: Settings) -> Path:
def index_dir(tmp_path: Path, settings: SettingsWrapper) -> Path:
path = tmp_path / "index"
path.mkdir()
settings.INDEX_DIR = path
@@ -31,3 +35,11 @@ def backend() -> Generator[TantivyBackend, None, None]:
finally:
b.close()
reset_backend()
@pytest.fixture(scope="module")
def index() -> tantivy.Index:
"""A real Tantivy index for parse-acceptance tests (module scope for speed)."""
idx = tantivy.Index(build_schema(), path=tempfile.mkdtemp())
register_tokenizers(idx, "english")
return idx
@@ -1,411 +0,0 @@
"""Result-level acceptance corpus: real documents indexed via build_schema(),
real queries run through parse_user_query(), matched-document-ID sets
asserted, not intermediate ASTs or query strings. This is paperless-ngx's
analogue of whoosh-compat's own tests/emitter/test_acceptance_e2e.py.
Supersedes test_query.py's TestParseUserQuery result-level cases.
"""
from __future__ import annotations
from datetime import UTC
from datetime import datetime
from typing import TYPE_CHECKING
import pytest
import time_machine
from django.contrib.auth.models import User
from documents.models import CustomField
from documents.models import CustomFieldInstance
from documents.models import Document
from documents.models import DocumentType
from documents.models import Note
from documents.models import StoragePath
from documents.search._query import parse_user_query
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
FROZEN_NOW = datetime(2026, 6, 15, 12, 0, tzinfo=UTC)
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
"""Create a Document and index it in one step, for the common case
where nothing needs to happen between the two (no related Note/
CustomFieldInstance to attach first)."""
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
@pytest.fixture
def indexed_documents(backend: TantivyBackend) -> dict[str, int]:
"""Index a small fixture set, return {label: doc_id} for corpus queries."""
docs = {
"invoice_2020": _index(
backend,
title="Invoice 2020",
content="invoice total due",
checksum="acc-invoice-2020",
archive_serial_number=100,
),
"invoice_2021": _index(
backend,
title="Invoice 2021",
content="invoice total due",
checksum="acc-invoice-2021",
archive_serial_number=101,
),
"invoice_2023": _index(
backend,
title="Invoice 2023",
content="invoice total due",
checksum="acc-invoice-2023",
archive_serial_number=102,
),
"receipt_2022": _index(
backend,
title="Receipt 2022",
content="receipt total due",
checksum="acc-receipt-2022",
archive_serial_number=103,
),
}
return {label: doc.pk for label, doc in docs.items()}
class TestIssue13568BracketWildcard:
"""paperless-ngx#13568: title:202[0-3]* must keep its character class,
not fold to a prefix query that silently drops it."""
def test_bracket_class_wildcard_matches_only_in_range_years(
self,
backend: TantivyBackend,
indexed_documents: dict[str, int],
) -> None:
# [0-1] (not [0-3]) is deliberate: the fixture's four years are
# 2020/2021/2022/2023, i.e. their trailing digit is 0/1/2/3
# respectively - a [0-3] class would match all four and the test
# would pass even if the character class were silently dropped and
# folded to an unconstrained "202*" prefix. [0-1] partitions the
# fixture into a genuine in-range/out-of-range split.
matched = _matched_ids(backend, "title:202[0-1]*")
expected = {
indexed_documents["invoice_2020"],
indexed_documents["invoice_2021"],
}
assert matched == expected, (
"title:202[0-1]* must match 2020/2021 titles and exclude 2022/2023 "
"- if this matches everything, the wildcard's character class was "
"silently dropped (issue #13568's original bug)"
)
class TestFieldBoosts:
def test_title_boost_ranks_title_match_above_content_only_match(
self,
backend: TantivyBackend,
) -> None:
title_match = _index(
backend,
title="urgent",
content="nothing else relevant",
checksum="acc-boost-title",
)
_index(
backend,
title="nothing",
content="urgent matter here",
checksum="acc-boost-content",
)
query = parse_user_query(backend._index, "urgent", UTC)
searcher = backend._index.searcher()
results = searcher.search(query, limit=10)
ranked_ids = [
searcher.doc(addr).to_dict()["id"][0] for _score, addr in results.hits
]
assert ranked_ids[0] == title_match.pk
class TestJsonSubpaths:
def test_notes_user_matches_document_with_that_note_author(
self,
backend: TantivyBackend,
) -> None:
alice = User.objects.create_user(username="alice")
doc_with_note = Document.objects.create(
title="Has note",
content="x",
checksum="acc-note-with",
)
Note.objects.create(document=doc_with_note, user=alice, note="reminder")
backend.add_or_update(doc_with_note)
_index(backend, title="No note", content="x", checksum="acc-note-without")
matched = _matched_ids(backend, "notes.user:alice")
assert matched == {doc_with_note.pk}
def test_custom_fields_name_and_value_combine(
self,
backend: TantivyBackend,
) -> None:
field = CustomField.objects.create(
name="Contract Number",
data_type=CustomField.FieldDataType.STRING,
)
other_field = CustomField.objects.create(
name="Other Field",
data_type=CustomField.FieldDataType.STRING,
)
matching = Document.objects.create(
title="Matching",
content="x",
checksum="acc-cf-matching",
)
CustomFieldInstance.objects.create(
document=matching,
field=field,
value_text="policy",
)
backend.add_or_update(matching)
non_matching = Document.objects.create(
title="Non-matching",
content="x",
checksum="acc-cf-nonmatching",
)
CustomFieldInstance.objects.create(
document=non_matching,
field=other_field,
value_text="policy",
)
backend.add_or_update(non_matching)
matched = _matched_ids(
backend,
'custom_fields.name:"Contract Number" custom_fields.value:policy',
)
assert matched == {matching.pk}
class TestUnregisteredIdFieldFoldsToLiteralText:
"""tag_id, owner_id, etc. are intentionally excluded from the
FieldRegistry - always internal index columns, never meant to be
query-addressable. Prove an unregistered field folds to a literal
text search that matches nothing, rather than erroring."""
def test_tag_id_query_matches_nothing(
self,
backend: TantivyBackend,
indexed_documents: dict[str, int],
) -> None:
matched = _matched_ids(backend, "tag_id:5")
assert matched == set()
class TestFuzzyBlendSurvivesWhooshGrammar:
"""A query mixing whoosh-only grammar (a date keyword) with a typo'd
free-text word must still fuzzy-match the intended document when
ADVANCED_FUZZY_SEARCH_THRESHOLD is enabled. The fuzzy clause is built
from the parsed query's free-text tokens (whoosh_compat's
free_text_tokens), never from the raw query string, so whoosh grammar
that tantivy's own parser rejects cannot knock the fuzzy clause out."""
def test_typo_fuzzy_matches_alongside_date_keyword(
self,
backend: TantivyBackend,
settings,
) -> None:
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.5
with time_machine.travel(FROZEN_NOW, tick=False):
doc = _index(
backend,
title="Receipt March",
content="receipt total due",
checksum="fuzzy-blend-1",
archive_serial_number=900,
)
# Sanity: the exact spelling matches through the exact clause.
assert doc.pk in _matched_ids(backend, "added:today receipt")
# The regression: the misspelling (one transposition) only
# matches via the fuzzy clause, and "added:today" is
# whoosh-only grammar tantivy's parser rejects, so raw-string
# fuzzy parsing skips the clause entirely and this returns
# nothing. The typo is deliberate; keep codespell away from it.
typo_query = "added:today reciept" # codespell:ignore reciept
assert doc.pk in _matched_ids(backend, typo_query)
def test_negated_words_do_not_fuzzy_match(
self,
backend: TantivyBackend,
settings,
) -> None:
# A term the user excluded must not resurface through the fuzzy
# clause. The shape is chosen so this genuinely discriminates: the
# indexed document contains the NOT'd word but NOT the positive
# word, so nothing matches the exact clause, and a fuzzy string
# naively built from ALL words (including the NOT'd one) would
# make this document the sole hit, normalize its score to 1.0,
# and survive any threshold. (A shape with an exact-matching
# sibling document does NOT discriminate: normalization ranks the
# resurfaced doc far below the exact match and the threshold cuts
# it even for a naive implementation.)
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.5
with time_machine.travel(FROZEN_NOW, tick=False):
_index(
backend,
title="Receipt Archive",
content="receipt archived stack",
checksum="fuzzy-blend-2",
archive_serial_number=901,
)
assert _matched_ids(backend, "added:today total NOT receipt") == set()
class TestUnquotedDateKeywordPhrases:
"""The unquoted spelling (added:previous month) is honored natively by
whoosh-compat's own grammar for this closed phrase vocabulary, no
app-level rewrite is involved. Pins that the historically supported
spelling keeps working now that paperless no longer pre-quotes it."""
@pytest.fixture
def period_documents(self, backend: TantivyBackend) -> dict[str, int]:
with time_machine.travel(FROZEN_NOW, tick=False):
in_may = _index(
backend,
title="May Doc",
content="statement",
checksum="kw-may",
archive_serial_number=910,
added=datetime(2026, 5, 20, 12, 0, tzinfo=UTC),
)
in_june = _index(
backend,
title="June Doc",
content="statement",
checksum="kw-june",
archive_serial_number=911,
added=datetime(2026, 6, 10, 12, 0, tzinfo=UTC),
)
return {"in_may": in_may.pk, "in_june": in_june.pk}
@pytest.mark.parametrize(
"query",
[
pytest.param("added:previous month", id="unquoted"),
pytest.param('added:"previous month"', id="quoted"),
pytest.param("added:Previous Month", id="unquoted-mixed-case"),
],
)
def test_unquoted_matches_the_same_documents_as_quoted(
self,
backend: TantivyBackend,
period_documents: dict[str, int],
query: str,
) -> None:
with time_machine.travel(FROZEN_NOW, tick=False):
assert _matched_ids(backend, query) == {period_documents["in_may"]}
@pytest.mark.parametrize(
"query",
[
pytest.param("added:this month", id="this-month"),
pytest.param("added:this year", id="this-year"),
pytest.param("added:previous week", id="previous-week"),
pytest.param("added:previous quarter", id="previous-quarter"),
pytest.param("added:previous year", id="previous-year"),
pytest.param("created:previous month", id="created-field"),
pytest.param("modified:previous month", id="modified-field"),
],
)
def test_every_phrase_and_date_field_parses_without_error(
self,
backend: TantivyBackend,
period_documents: dict[str, int],
query: str,
) -> None:
# The whole vocabulary times every date field must at least parse
# and search cleanly (no SearchQueryError -> no HTTP 400); exact
# window semantics are whoosh-compat's, pinned in its own suite.
with time_machine.travel(FROZEN_NOW, tick=False):
_matched_ids(backend, query)
def test_text_field_keyword_words_are_ordinary_text(
self,
backend: TantivyBackend,
period_documents: dict[str, int],
) -> None:
# "previous month" after a TEXT field (or unfielded) is ordinary
# text, not a date phrase: a title actually containing the words
# matches, and the date-window documents do not.
with time_machine.travel(FROZEN_NOW, tick=False):
wordy = _index(
backend,
title="Notes from the previous month",
content="meeting notes",
checksum="kw-text",
archive_serial_number=912,
)
assert _matched_ids(backend, "title:previous month") == {wordy.pk}
class TestFieldAliases:
"""type:/path: are registry aliases for document_type:/storage_path:.
The only other alias coverage is parse-shape; these prove resolution
end-to-end against a real index."""
def test_type_alias_and_canonical_name_match_the_same_document(
self,
backend: TantivyBackend,
) -> None:
invoice_type = DocumentType.objects.create(name="invoice")
# Discriminating shape: document_type is itself a default search
# field, so if alias resolution ever broke and "type:invoice"
# demoted to unfielded text, the token would STILL match the typed
# document through the field value. The decoy carries the query
# word in content, so a demoted search matches BOTH documents and
# the exact-set assertions fail. (The title avoids stemming to
# "type": english stems Typed -> type.)
typed = _index(
backend,
title="First",
content="quarterly statement",
checksum="alias-type-1",
document_type=invoice_type,
)
_index(
backend,
title="Second",
content="invoice mentioned in body",
checksum="alias-type-2",
)
assert _matched_ids(backend, "type:invoice") == {typed.pk}
assert _matched_ids(backend, "document_type:invoice") == {typed.pk}
def test_path_alias_and_canonical_name_match_the_same_document(
self,
backend: TantivyBackend,
) -> None:
archive = StoragePath.objects.create(name="archive", path="archive/{title}")
stored = _index(
backend,
title="Stored",
content="quarterly statement",
checksum="alias-path-1",
storage_path=archive,
)
# storage_path is NOT a default search field today, so a demoted
# "path:archive" already matches nothing; the content decoy keeps
# this test discriminating even if it ever joins the defaults.
_index(
backend,
title="Loose",
content="archive mentioned in body",
checksum="alias-path-2",
)
assert _matched_ids(backend, "path:archive") == {stored.pk}
assert _matched_ids(backend, "storage_path:archive") == {stored.pk}
-187
View File
@@ -4,8 +4,6 @@ from pathlib import Path
import pytest
from django.contrib.auth.models import Group
from django.contrib.auth.models import User
from django.db import connection
from django.test.utils import CaptureQueriesContext
from guardian.shortcuts import assign_perm
from pytest_mock import MockerFixture
@@ -104,191 +102,6 @@ class TestWriteBatch:
assert len(backend.search_ids("indexable", user=None)) == 1
class TestAddOrUpdateIds:
"""Test WriteBatch.add_or_update_ids(), the bulk id-based upsert path.
Unlike add_or_update() called once per document, this resolves viewer
permissions and effective (versioned) content in bulk against the ids as
a whole, so it must produce identical indexed output to the per-document
path while issuing a constant number of queries regardless of batch size.
"""
def test_missing_id_is_skipped_not_errored(
self,
backend: TantivyBackend,
) -> None:
doc = Document.objects.create(
title="doc",
content="present",
checksum="EXIST1",
pk=1,
)
missing_pk = 999
with backend.batch_update() as batch:
batch.add_or_update_ids([doc.pk, missing_pk])
assert backend.search_ids("present", user=None) == [doc.pk]
def test_query_count_does_not_scale_with_batch_size(
self,
backend: TantivyBackend,
) -> None:
"""Each query count must stay far below N, not merely match between
two runs -- an exact-equality assertion between two measurements is
at the mercy of incidental process-level caches (e.g. Django's
ContentType.objects.get_for_model) warming on whichever run happens
first, which makes counts differ by a query for reasons unrelated to
batch size. A generous fixed bound sidesteps that: the old
per-document path issued roughly 8 queries per document, so 50
documents under a bound this low proves the fix regardless of cache
state.
"""
max_queries_for_any_batch_size = 15
small_docs = [
Document.objects.create(
title="doc",
content=f"unique{i}",
checksum=f"SMALL{i}",
pk=i,
)
for i in range(1, 3)
]
with CaptureQueriesContext(connection) as ctx_small:
with backend.batch_update() as batch:
batch.add_or_update_ids([d.pk for d in small_docs])
assert len(ctx_small.captured_queries) <= max_queries_for_any_batch_size
large_docs = [
Document.objects.create(
title="doc",
content=f"unique{i}",
checksum=f"LARGE{i}",
pk=i,
)
for i in range(100, 150)
]
with CaptureQueriesContext(connection) as ctx_large:
with backend.batch_update() as batch:
batch.add_or_update_ids([d.pk for d in large_docs])
assert len(ctx_large.captured_queries) <= max_queries_for_any_batch_size
for doc in large_docs:
assert backend.search_ids(f"unique{doc.pk}", user=None) == [doc.pk]
def test_resolves_direct_user_grant_in_bulk(
self,
backend: TantivyBackend,
) -> None:
owner = UserFactory()
user = UserFactory()
doc = Document.objects.create(
title="doc",
checksum="PERM1",
pk=1,
owner=owner,
)
assign_perm("view_document", user, doc)
with backend.batch_update() as batch:
batch.add_or_update_ids([doc.pk])
assert backend.search_ids("doc", user=user) == [doc.pk]
other = UserFactory()
assert backend.search_ids("doc", user=other) == []
def test_resolves_group_grant_in_bulk(self, backend: TantivyBackend) -> None:
owner = UserFactory()
group = Group.objects.create(name="reviewers")
user = UserFactory()
user.groups.add(group)
doc = Document.objects.create(
title="doc",
checksum="GPERM1",
pk=1,
owner=owner,
)
assign_perm("view_document", group, doc)
with backend.batch_update() as batch:
batch.add_or_update_ids([doc.pk])
assert backend.search_ids("doc", user=user) == [doc.pk]
other = UserFactory()
assert backend.search_ids("doc", user=other) == []
def test_indexes_notes_and_custom_fields(self, backend: TantivyBackend) -> None:
note_author = UserFactory(username="noter")
field = CustomField.objects.create(
name="Invoice Number",
data_type=CustomField.FieldDataType.STRING,
)
doc = Document.objects.create(title="doc", checksum="RICH1", pk=1)
Note.objects.create(document=doc, note="Reviewed", user=note_author)
CustomFieldInstance.objects.create(
document=doc,
field=field,
value_text="INV-42",
)
with backend.batch_update() as batch:
batch.add_or_update_ids([doc.pk])
assert backend.search_ids("notes.user:noter", user=None) == [doc.pk]
assert backend.search_ids("custom_fields.value:INV-42", user=None) == [
doc.pk,
]
def test_uses_effective_content_for_versioned_documents(
self,
backend: TantivyBackend,
) -> None:
root = Document.objects.create(
title="Statement",
content="stale text",
checksum="ROOT1",
pk=1,
)
Document.objects.create(
title="Statement",
content="latest version text",
checksum="VER1",
pk=2,
root_document=root,
version_index=1,
)
with backend.batch_update() as batch:
batch.add_or_update_ids([root.pk])
assert backend.search_ids("latest", user=None) == [root.pk]
assert backend.search_ids("stale", user=None) == []
def test_reindexes_documents_already_in_the_index(
self,
backend: TantivyBackend,
) -> None:
"""add_or_update_ids must upsert, matching add_or_update's behaviour."""
doc = Document.objects.create(
title="doc",
content="original",
checksum="UP1",
pk=1,
)
backend.add_or_update(doc)
assert backend.search_ids("original", user=None) == [doc.pk]
doc.content = "updated"
doc.save()
with backend.batch_update() as batch:
batch.add_or_update_ids([doc.pk])
assert backend.search_ids("original", user=None) == []
assert backend.search_ids("updated", user=None) == [doc.pk]
class TestSearch:
"""Test search query parsing and matching via search_ids."""
@@ -1,57 +0,0 @@
"""``checksum`` wildcard patterns stay literal end to end, once user queries
route through whoosh-compat.
The registry-level fact (the pattern normalizer folds a KEYWORD pattern
rather than stemming it) is pinned on its own in
``test_keyword_pattern_literal.py``. This proves it actually reaches a real
query: ``checksum:ceded*`` must match only the document whose checksum
starts with "ceded", not the one whose checksum stems to the same run.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.models import Document
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
CEDEF00D = "cedef00ddeadbeef0123456789abcdef01234567"
CEDEDEAD = "cededeadbeef567801234567" + "89abcdef01234567"
class TestChecksumPrefixQueries:
@pytest.fixture
def indexed(self, backend: TantivyBackend) -> None:
for i, checksum in enumerate((CEDEF00D, CEDEDEAD)):
doc = Document.objects.create(
title=f"Checksum doc {i}",
content="invoices for the quarter",
checksum=checksum,
archive_serial_number=940 + i,
)
backend.add_or_update(doc)
def _ids(self, backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def test_prefix_matches_only_the_document_that_starts_with_it(
self,
backend: TantivyBackend,
indexed: None,
) -> None:
matched = self._ids(backend, "checksum:ceded*")
expected = Document.objects.get(checksum=CEDEDEAD).pk
assert matched == {expected}
def test_text_prefix_still_reaches_the_stemmed_index(
self,
backend: TantivyBackend,
indexed: None,
) -> None:
assert len(self._ids(backend, "invoice*")) == 2
@@ -1,133 +0,0 @@
"""The CJK bigram clause blended into QUERY-mode searches.
The clause exists so CJK runs are matchable at all (the default analyzers
keep a whitespace-free CJK run as one indivisible token), but it must not
widen the query beyond what the user asked for: a CJK term the query
excludes, or restricts to one field, must not come back through it.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.models import Document
if TYPE_CHECKING:
from pytest_django.fixtures import SettingsWrapper
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
class TestCjkClauseFollowsTheParsedQuery:
def test_negated_cjk_term_is_excluded(self, backend: TantivyBackend) -> None:
"""'invoice NOT 漢字' must not return the document containing 漢字."""
with_cjk = _index(
backend,
title="Invoice A",
content="invoice total 漢字",
checksum="cjk-neg-1",
)
without_cjk = _index(
backend,
title="Invoice B",
content="invoice total only",
checksum="cjk-neg-2",
)
assert _matched_ids(backend, "invoice") == {with_cjk.pk, without_cjk.pk}
assert _matched_ids(backend, "invoice NOT 漢字") == {without_cjk.pk}
@pytest.mark.parametrize(
("threshold", "expected"),
[
pytest.param(None, {"titled"}, id="fuzzy_off"),
pytest.param(0.0, {"titled", "content_only"}, id="fuzzy_on"),
],
)
def test_fielded_cjk_term_searches_only_that_field(
self,
backend: TantivyBackend,
settings: SettingsWrapper,
threshold: float | None,
expected: set[str],
) -> None:
"""'title:東京' must not match a document whose 東京 is in the content.
The CJK clause honours the field. The fuzzy clause, when enabled,
does not: it contributes every free-text term UNFIELDED by design
(see _try_parse_fuzzy_query), so it brings the content-only
document back on its own 0.1-boosted terms. That is the documented
trade-off, pinned here so it stays deliberate.
"""
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = threshold
content_only = _index(
backend,
title="Tokyo report",
content="東京都の人口は約1400万人です",
checksum="cjk-field-1",
)
titled = _index(
backend,
title="東京都の報告書",
content="an english summary",
checksum="cjk-field-2",
)
pks = {"titled": titled.pk, "content_only": content_only.pk}
assert _matched_ids(backend, "東京") == set(pks.values())
assert _matched_ids(backend, "title:東京") == {pks[label] for label in expected}
def test_cjk_on_a_non_default_field_builds_no_clause(
self,
backend: TantivyBackend,
) -> None:
"""A CJK term restricted to a field outside the default search fields
has nothing to contribute to the bigram clause: 'notes:東京' must not
fall back to matching 東京 in the content."""
_index(
backend,
title="Tokyo report",
content="東京都の人口は約1400万人です",
checksum="cjk-notes-1",
)
assert _matched_ids(backend, "notes:東京") == set()
def test_bare_cjk_term_still_matches_every_default_field(
self,
backend: TantivyBackend,
) -> None:
"""The clause's reason for existing: an unfielded CJK run matches
wherever it is indexed, and does so alongside a latin term."""
in_content = _index(
backend,
title="report",
content="本文に重要な情報",
checksum="cjk-bare-1",
)
in_title = _index(
backend,
title="重要な報告書",
content="english only",
checksum="cjk-bare-2",
)
assert _matched_ids(backend, "重要") == {in_content.pk, in_title.pk}
assert _matched_ids(backend, "重要 OR report") == {
in_content.pk,
in_title.pk,
}
@@ -1,75 +0,0 @@
"""Whoosh's compact, separator-free date spelling, resolved end to end.
whoosh-compat owns both widths of this spelling and asserts both of each
form's bounds directly: ``test_compact_numeric_datetime`` pins the 8-digit
form as a whole calendar day (lower bound, upper bound and exclusivity), and
``test_compact_numeric_datetime_full_width_is_a_single_second_instant`` pins
the 14-digit form as one instant. The 14-digit form is kept here as the single
representative because it is the one that exercises paperless's ``added``
DATETIME fast field at full precision: the corpus separates a document at
the named instant from one on the same calendar day at another hour and one
on the next day at the same hour, so a query that degrades into a whole-day
window, or drops the time of day, matches the wrong set rather than passing
on a corpus that could not tell the difference.
"""
from __future__ import annotations
from datetime import UTC
from datetime import datetime
from typing import TYPE_CHECKING
import pytest
from documents.models import Document
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
@pytest.fixture
def docs(backend: TantivyBackend) -> dict[str, int]:
return {
"instant": _index(
backend,
title="On the instant",
content="x",
checksum="compact-date-instant",
added=datetime(2005, 3, 4, 15, 30, tzinfo=UTC),
).pk,
"same_day": _index(
backend,
title="Same day, other hour",
content="x",
checksum="compact-date-same-day",
added=datetime(2005, 3, 4, 9, 0, tzinfo=UTC),
).pk,
"next_day": _index(
backend,
title="Next day, same hour",
content="x",
checksum="compact-date-next-day",
added=datetime(2005, 3, 5, 15, 30, tzinfo=UTC),
).pk,
}
def test_fourteen_digits_is_a_single_instant(
backend: TantivyBackend,
docs: dict[str, int],
) -> None:
# same_day is what tells this apart from the 8-digit day-window form,
# next_day from a form that ignored the time altogether.
assert _matched_ids(backend, "added:20050304153000") == {docs["instant"]}
@@ -1,72 +0,0 @@
"""Pins the correctness gained by deleting the pre-parse
_quote_date_keyword_phrases rewrite.
That rewrite matched date-keyword phrases (e.g. "previous month" after a
date field) anywhere in the raw query string, including inside an
unrelated quoted string, and inserted quotes mid-phrase there too. Its
own docstring gave ``title:"see added:previous month notes"`` as the
example of what it corrupted. whoosh-compat's grammar accepts the same
phrase vocabulary unquoted natively (see TestUnquotedDateKeywordPhrases
in test_acceptance.py), so the rewrite was redundant everywhere it was
safe and actively wrong everywhere it was not. This is the one case that
tells the two apart: a literal title phrase that happens to contain
"added:previous month" as running text.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.models import Document
if TYPE_CHECKING:
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
class TestQuotedStringContainingDateKeywordText:
"""A quoted title phrase containing the literal text
"added:previous month" as running words must match on that literal
text alone, never spill into an unfielded search for "previous" and
"month" across the default search fields the way the deleted rewrite
would have decomposed it into."""
def test_matches_only_the_literal_phrase(
self,
backend: TantivyBackend,
) -> None:
literal = _index(
backend,
title="see added:previous month notes",
content="quarterly filing",
checksum="dkp-literal",
archive_serial_number=920,
)
# Under the deleted rewrite, this decoy would incorrectly match:
# its title contains the "see added:" and " notes" fragments the
# corrupted parse required as title phrases, and its content
# supplies "previous" and "month" as the decomposed word-match
# clauses the rewrite turned the middle of the phrase into.
decoy = _index(
backend,
title="see added: quarterly report notes",
content="we reviewed the previous statement about month end",
checksum="dkp-decoy",
archive_serial_number=921,
)
query = 'title:"see added:previous month notes"'
assert _matched_ids(backend, query) == {literal.pk}
assert decoy.pk not in _matched_ids(backend, query)
@@ -1,83 +0,0 @@
"""Date keyword phrases (``today``, etc.) resolved in a non-UTC timezone,
end to end.
paperless's own ``tz=get_current_timezone()`` plumbing
(``TantivyBackend._parse_query``) is exercised elsewhere only for
relative *ranges* (``added:[-1 week to now]``, in
documents/tests/test_api_search.py). This covers a date *keyword*
(``today``), whose day boundary depends on the active timezone the same
way but goes through whoosh-compat's DateParserPlugin resolution instead
of an explicit range.
Discriminating shape: frozen at 2026-06-15T02:00 UTC, which is
2026-06-14T22:00 in America/New_York -- still "today" (06-14) there, but
already "today" (06-15) in UTC. Two documents pin both directions of the
mistake a hardcoded-UTC bug would make:
- ``in_ny_today`` (added 2026-06-14T20:00 UTC = 2026-06-14T16:00 NY) is
inside New York's "today" window and outside a naive UTC-calendar-day
window. A ``tz``-ignoring bug would miss it.
- ``in_utc_calendar_day_only`` (added 2026-06-15T10:00 UTC =
2026-06-15T06:00 NY) is inside a naive UTC-calendar-day window but
outside New York's actual "today" window. A ``tz``-ignoring bug would
wrongly match it.
"""
from __future__ import annotations
from datetime import UTC
from datetime import datetime
from typing import TYPE_CHECKING
import pytest
import time_machine
from documents.models import Document
if TYPE_CHECKING:
from pytest_django.fixtures import SettingsWrapper
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
FROZEN_NOW = datetime(2026, 6, 15, 2, 0, tzinfo=UTC)
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
class TestDateKeywordUsesTheActiveTimezone:
def test_today_matches_the_new_york_calendar_day_not_the_utc_one(
self,
backend: TantivyBackend,
settings: SettingsWrapper,
) -> None:
settings.TIME_ZONE = "America/New_York"
with time_machine.travel(FROZEN_NOW, tick=False):
in_ny_today = _index(
backend,
title="NY today",
content="x",
checksum="tz-keyword-ny-today",
added=datetime(2026, 6, 14, 20, 0, tzinfo=UTC),
)
# Not captured: the exact-set assertion below already proves
# this document (inside a naive UTC-calendar-day window, but
# outside New York's actual "today") does not match.
_index(
backend,
title="UTC calendar day only",
content="x",
checksum="tz-keyword-utc-calendar-day-only",
added=datetime(2026, 6, 15, 10, 0, tzinfo=UTC),
)
assert _matched_ids(backend, "added:today") == {in_ny_today.pk}
@@ -1,20 +0,0 @@
"""``_DEFAULT_SEARCH_FIELDS`` must stay a subset of the registered public
field names.
Nothing enforced this before: a rename in PUBLIC_FIELDS not mirrored in
``_DEFAULT_SEARCH_FIELDS`` (documents/search/_query.py) would 400 every
unfielded search at request time, since ``index.parse_query`` and the
fuzzy/CJK clause builders are handed a field name the schema no longer
has.
"""
from __future__ import annotations
from documents.search._fields import PUBLIC_FIELDS
from documents.search._query import _DEFAULT_SEARCH_FIELDS
class TestDefaultSearchFieldsAreRegistered:
def test_every_default_search_field_is_a_public_field(self) -> None:
public_field_names = {f.name for f in PUBLIC_FIELDS}
assert set(_DEFAULT_SEARCH_FIELDS) <= public_field_names
@@ -1,355 +0,0 @@
"""Pins the search syntax that ``docs/usage.md`` promises users.
Every query here appears verbatim, or as a direct paraphrase, in the
"Document searches" section of ``docs/usage.md``. Each case indexes real
documents and asserts on matched document IDs rather than on the parsed
query, because a query that parses cleanly is not necessarily a query that
means what the documentation says it means: ``added:now`` parses without a
single diagnostic and then matches nothing, because it resolves to an
instant rather than to a span.
The negative cases matter as much as the positive ones. They pin the
behaviours the docs explicitly warn about, so that if any of them ever
starts working the warning can be removed deliberately rather than being
left standing as a lie.
"""
from __future__ import annotations
from datetime import UTC
from datetime import datetime
from typing import TYPE_CHECKING
import pytest
import time_machine
from documents.models import Document
from documents.models import Note
from documents.models import Tag
from documents.search._errors import InvalidDateQuery
if TYPE_CHECKING:
from collections.abc import Generator
from django.contrib.auth.models import User
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
# A Monday, so that "next monday"/"last monday" land a clean week either side.
FROZEN_NOW = datetime(2026, 6, 15, 12, 0, tzinfo=UTC)
# The checksum used in the docs' `checksum:` example.
DOC_CHECKSUM = "9f86d081884c7d659a2feaa0c55ad015a3bf4f1b2b0b822cd15d6c15b0f00a08"
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
class TestLogicalExpressions:
@pytest.fixture
def docs(self, backend: TantivyBackend) -> dict[str, int]:
return {
"secret": _index(
backend,
title="Invoice one",
content="invoice secret contents",
checksum="doc-syntax-secret",
).pk,
"plain": _index(
backend,
title="Invoice two",
content="invoice ordinary contents",
checksum="doc-syntax-plain",
).pk,
}
def test_not_excludes_a_term(
self,
backend: TantivyBackend,
docs: dict[str, int],
) -> None:
assert _matched_ids(backend, "invoice NOT secret") == {docs["plain"]}
def test_leading_hyphen_requires_the_term_instead_of_excluding_it(
self,
backend: TantivyBackend,
docs: dict[str, int],
) -> None:
# The docs warn about exactly this: separators are stripped at index
# time, so "-secret" is the term "secret" and the query is an AND.
assert _matched_ids(backend, "invoice -secret") == {docs["secret"]}
def test_or_inside_parentheses_matches_either_branch(
self,
backend: TantivyBackend,
docs: dict[str, int],
) -> None:
matched = _matched_ids(backend, "invoice AND (secret OR ordinary)")
assert matched == {docs["secret"], docs["plain"]}
class TestPhraseSearch:
def test_quoted_phrase_requires_the_words_in_order(
self,
backend: TantivyBackend,
) -> None:
doc = _index(
backend,
title="Phrase",
content="the quick brown fox jumps",
checksum="doc-syntax-phrase",
)
assert _matched_ids(backend, '"quick brown fox"') == {doc.pk}
assert _matched_ids(backend, '"brown quick fox"') == set()
class TestTagCommaList:
"""``tag:bills,unpaid`` is published syntax (docs/usage.md), so this checks
that the documented spelling still returns what the docs promise: only the
document carrying every listed tag.
It is deliberately not proof of paperless's field configuration, and must
not be read as such. Removing ``comma_values`` from the ``tag`` FieldSpec
leaves this test passing, because paperless's analyzer splits the literal
value "bills,unpaid" into the same two tokens the value-list reading
produces, so the two readings select the same documents. The registry fact
-- that ``tag`` opts in and no other field does -- is observable only at
the registry, and is owned by test_registry.py's
``test_tag_is_comma_values``/``test_correspondent_is_not_comma_values``.
"""
def test_comma_list_requires_every_listed_tag(
self,
backend: TantivyBackend,
) -> None:
bills = Tag.objects.create(name="bills")
unpaid = Tag.objects.create(name="unpaid")
archived = Tag.objects.create(name="archived")
both = Document.objects.create(
title="Both tags",
content="body",
checksum="doc-syntax-tag-both",
)
both.tags.add(bills, unpaid)
backend.add_or_update(both)
one = Document.objects.create(
title="One tag",
content="body",
checksum="doc-syntax-tag-one",
)
one.tags.add(bills, archived)
backend.add_or_update(one)
assert _matched_ids(backend, "tag:bills,unpaid") == {both.pk}
assert _matched_ids(backend, "tag:bills") == {both.pk, one.pk}
class TestArchiveMetadataFields:
@pytest.fixture
def doc(self, backend: TantivyBackend, admin_user: User) -> Document:
doc = Document.objects.create(
title="Metadata",
content="body",
checksum=DOC_CHECKSUM,
archive_serial_number=100,
page_count=12,
original_filename="invoice.pdf",
)
Note.objects.create(document=doc, user=admin_user, note="a note")
backend.add_or_update(doc)
return doc
@pytest.mark.parametrize(
"query",
[
"asn:100",
"asn:[50 to 150]",
"page_count:12",
"page_count:[10 to 20]",
"num_notes:1",
"num_notes:[1 to 5]",
"original_filename:invoice.pdf",
f"checksum:{DOC_CHECKSUM}",
"checksum:9f86d081*",
# A checksum term is stored verbatim, but a checksum *pattern* is
# lowercased before it is matched, which the docs now say outright
# next to the "only a complete, lowercase checksum matches" rule
# that the uppercase term in the negative list below pins.
"checksum:9F86D081*",
],
)
def test_documented_metadata_query_matches(
self,
backend: TantivyBackend,
doc: Document,
query: str,
) -> None:
assert _matched_ids(backend, query) == {doc.pk}
@pytest.mark.parametrize(
"query",
[
# The docs say only a complete, lowercase checksum matches.
"checksum:9f86d081",
f"checksum:{DOC_CHECKSUM.upper()}",
],
)
def test_partial_or_uppercase_checksum_matches_nothing(
self,
backend: TantivyBackend,
doc: Document,
query: str,
) -> None:
assert _matched_ids(backend, query) == set()
class TestDocumentedDateForms:
@pytest.fixture(autouse=True)
def frozen_now(self) -> Generator[None, None, None]:
with time_machine.travel(FROZEN_NOW, tick=False):
yield
@pytest.fixture
def dated(self, backend: TantivyBackend) -> dict[str, int]:
stamps = {
"today": datetime(2026, 6, 15, 9, 0, tzinfo=UTC),
"yesterday": datetime(2026, 6, 14, 9, 0, tzinfo=UTC),
"tomorrow": datetime(2026, 6, 16, 9, 0, tzinfo=UTC),
"next_monday": datetime(2026, 6, 22, 10, 0, tzinfo=UTC),
"last_monday": datetime(2026, 6, 8, 10, 0, tzinfo=UTC),
"january": datetime(2026, 1, 10, 10, 0, tzinfo=UTC),
"old": datetime(2005, 3, 4, 15, 30, tzinfo=UTC),
}
return {
label: _index(
backend,
title=label,
content="dated body",
checksum=f"doc-syntax-date-{label}",
added=stamp,
).pk
for label, stamp in stamps.items()
}
@pytest.mark.parametrize(
("query", "label"),
[
("added:today", "today"),
("added:yesterday", "yesterday"),
("added:tomorrow", "tomorrow"),
('added:"next monday"', "next_monday"),
('added:"last monday"', "last_monday"),
("added:january", "january"),
("added:2005-03-04", "old"),
("added:2005-03", "old"),
("added:[2005-01-01 to 2005-12-31]", "old"),
("added:[2005 to 2009]", "old"),
# A full timestamp works, but only quoted when it stands alone,
# and only unquoted when it is a range bound. The bare standalone
# spelling is pinned as a non-match below.
('added:"2005-03-04T15:30:00Z"', "old"),
("added:[2005-03-04T09:00:00Z to 2005-03-04T17:00:00Z]", "old"),
# A quoted range bound works when the quotes are single ones; the
# double-quoted spelling is pinned as an error below.
("added:['2005-03-04' to 2005-03-05]", "old"),
],
)
def test_documented_date_form_matches_its_day_or_month(
self,
backend: TantivyBackend,
dated: dict[str, int],
query: str,
label: str,
) -> None:
assert _matched_ids(backend, query) == {dated[label]}
@pytest.mark.parametrize(
"query",
[
# Zero-width: these resolve to a single instant, not a span, so
# nothing in a realistic corpus lands on them. The docs warn
# about them rather than presenting them as usable.
"added:now",
"added:noon",
"added:midnight",
# Quoting is what rescues the other multi-word date expressions,
# so pin that it does not rescue these: the problem is the width
# of the resulting range, not the way the value is delimited.
# One quoted spelling is enough for that; which keyword sits
# inside the quotes is grammar whoosh-compat owns.
'added:"now"',
# A relative offset, which the warning in the docs names by this
# exact spelling. Standing alone it is an instant like the rest of
# this list; the same offset used as a range bound is a real
# window, pinned by the test below.
'added:"-1 week"',
],
)
def test_forms_the_docs_warn_about_match_nothing(
self,
backend: TantivyBackend,
dated: dict[str, int],
query: str,
) -> None:
assert _matched_ids(backend, query) == set()
def test_bare_timestamp_is_rejected_rather_than_matching_nothing(
self,
backend: TantivyBackend,
dated: dict[str, int],
) -> None:
"""The bare, unquoted spelling of a full timestamp. The quoted and
range-bound spellings pinned above do work and match this fixture's
document; this one is a user-fixable error rather than an empty
result set, so the docs tell the user to quote it.
The reported value is the whole contiguous fragment the user typed,
not just the prefix the date grammar's tokenizer first split on.
"""
with pytest.raises(InvalidDateQuery) as exc_info:
_matched_ids(backend, "added:2005-03-04T15:30:00Z")
assert exc_info.value.field == "added"
assert exc_info.value.value == "2005-03-04T15:30:00Z"
def test_relative_offset_as_a_range_bound_is_a_real_window(
self,
backend: TantivyBackend,
dated: dict[str, int],
) -> None:
"""The same offset that matches nothing on its own spans the last
seven days as a lower bound. The docs say so, next to the warning
about the standalone form, so both readings are pinned together.
"last_monday" is indexed at 2026-06-08T10:00, two hours before the
window opens, so its exclusion is what shows the bound is the offset
and not a whole-day rounding of it.
"""
assert _matched_ids(backend, "added:['-1 week' to now]") == {
dated["today"],
dated["yesterday"],
}
def test_double_quoted_range_bound_is_rejected(
self,
backend: TantivyBackend,
dated: dict[str, int],
) -> None:
"""Quoting a range bound is allowed, but only with single quotes: the
double-quoted spelling reaches the date grammar with its quotes still
attached and is not a recognizable date. The docs say so, so pin which
of the two quote characters is the one that fails.
"""
with pytest.raises(InvalidDateQuery) as exc_info:
_matched_ids(backend, 'added:["2005-03-04" to 2005-03-05]')
assert exc_info.value.value == '"2005-03-04"'
@@ -1,247 +0,0 @@
"""Diagnostics route by Cause, and user-facing messages are host-owned.
whoosh-compat documents ``Diagnostic.message`` as developer output with no
stability guarantee, so it must never reach an HTTP response body.
"""
from __future__ import annotations
import logging
from datetime import UTC
import pytest
import tantivy
from whoosh_compat.errors import Diagnostic
from whoosh_compat.errors import DiagnosticKind
from whoosh_compat.errors import QueryError
from whoosh_compat.errors import cause_for
from whoosh_compat.fields import FieldKind
from whoosh_compat.fields import FieldRef
from documents.search._errors import SearchQueryError
from documents.search._query import _map_emit_error
from documents.search._query import _single_diagnostic_to_error
from documents.search._query import parse_user_query
from documents.search._schema import build_schema
from documents.search._tokenizer import register_tokenizers
pytestmark = pytest.mark.search
_LIBRARY_PROSE = "INTERNAL LIBRARY WORDING WITH raw tantivy detail"
@pytest.fixture(scope="module")
def query_index() -> tantivy.Index:
"""An in-memory, unstemmed index; these tests only parse, never index."""
idx = tantivy.Index(build_schema(), path=None)
register_tokenizers(idx, "")
return idx
def _diagnostic(
kind: DiagnosticKind,
*,
field: FieldRef | None = FieldRef("title"),
field_kind: FieldKind | None = FieldKind.TEXT,
) -> Diagnostic:
"""A Diagnostic shaped like the emitter's, with the library's own
kind -> cause mapping rather than a hand-picked cause."""
return Diagnostic(
kind=kind,
cause=cause_for(kind),
message=_LIBRARY_PROSE,
field=field,
field_kind=field_kind,
)
class TestEmitErrorRouting:
"""Every Cause gets a distinguishable treatment, not just "a 400"."""
@pytest.mark.parametrize(
"kind",
[
DiagnosticKind.BACKEND_REJECTED,
DiagnosticKind.AST_INVALID_SHAPE,
DiagnosticKind.AST_UNKNOWN_FIELD,
],
)
def test_internal_cause_is_not_converted(self, kind: DiagnosticKind) -> None:
"""A library defect must surface as a 500 monitoring can see, not a
400 blaming the user."""
error = QueryError(_diagnostic(kind))
with pytest.raises(QueryError) as excinfo:
_map_emit_error(error)
assert excinfo.value is error
def test_misconfigured_cause_is_logged_and_becomes_a_400(
self,
caplog: pytest.LogCaptureFixture,
) -> None:
kind = DiagnosticKind.SCHEMA_FIELD_MISSING
with caplog.at_level(logging.ERROR, logger="paperless.search"):
error = _map_emit_error(
QueryError(_diagnostic(kind, field=FieldRef("asn"))),
)
assert isinstance(error, SearchQueryError)
errors = [r for r in caplog.records if r.levelno == logging.ERROR]
assert len(errors) == 1
assert "asn" in errors[0].getMessage()
assert kind.name in errors[0].getMessage()
@pytest.mark.parametrize(
"kind",
[
DiagnosticKind.TEXT_RANGE,
DiagnosticKind.PATTERN_TOO_COMPLEX,
DiagnosticKind.EXISTS_REQUIRES_FAST,
],
)
def test_unsupported_cause_is_a_400_with_no_operator_log(
self,
kind: DiagnosticKind,
caplog: pytest.LogCaptureFixture,
) -> None:
"""A query tantivy cannot run is the user's to fix; it must not page
an operator the way a registry/schema mismatch does.
EXISTS_REQUIRES_FAST is nominally MISCONFIGURED but belongs here: it
is decided from the registry's own FieldSpec, so it never reports a
disagreement anyone could resolve."""
with caplog.at_level(logging.WARNING, logger="paperless.search"):
error = _map_emit_error(QueryError(_diagnostic(kind)))
assert isinstance(error, SearchQueryError)
assert caplog.records == []
@pytest.mark.parametrize(
"kind",
[
DiagnosticKind.TEXT_RANGE,
DiagnosticKind.PATTERN_TOO_COMPLEX,
DiagnosticKind.EXISTS_REQUIRES_FAST,
DiagnosticKind.SCHEMA_FIELD_MISSING,
],
)
def test_user_facing_message_never_echoes_library_prose(
self,
kind: DiagnosticKind,
) -> None:
error = _map_emit_error(QueryError(_diagnostic(kind)))
assert _LIBRARY_PROSE not in str(error)
@pytest.mark.parametrize(
"kind",
[
DiagnosticKind.TEXT_RANGE,
DiagnosticKind.PATTERN_TOO_COMPLEX,
DiagnosticKind.EXISTS_REQUIRES_FAST,
DiagnosticKind.SCHEMA_FIELD_MISSING,
],
)
def test_user_facing_message_names_the_field(
self,
kind: DiagnosticKind,
) -> None:
"""FieldRef.__str__ yields the canonical dotted name, including a
JSON subpath, so every user-reachable emit kind can name it."""
diagnostic = _diagnostic(
kind,
field=FieldRef("custom_fields", "value"),
field_kind=FieldKind.JSON,
)
error = _map_emit_error(QueryError(diagnostic))
assert "custom_fields.value" in str(error)
class TestParseDiagnosticMessages:
"""Parse-time diagnostics are host-worded too, off field_kind."""
def test_too_deep_is_a_400_without_library_prose(self) -> None:
error = _single_diagnostic_to_error(
_diagnostic(DiagnosticKind.TOO_DEEP, field=None, field_kind=None),
)
assert isinstance(error, SearchQueryError)
assert _LIBRARY_PROSE not in str(error)
@pytest.mark.parametrize(
("kind", "field_kind"),
[
(DiagnosticKind.PATTERN_ON_NUMERIC, FieldKind.U64),
(DiagnosticKind.PATTERN_ON_BOOLEAN_EXISTS, FieldKind.BOOLEAN_EXISTS),
(DiagnosticKind.PATTERN_ON_SUBPATH, FieldKind.JSON),
],
)
def test_pattern_on_kinds_name_the_field_and_its_kind(
self,
kind: DiagnosticKind,
field_kind: FieldKind,
) -> None:
error = _single_diagnostic_to_error(
_diagnostic(kind, field=FieldRef("asn"), field_kind=field_kind),
)
message = str(error)
assert _LIBRARY_PROSE not in message
assert "asn" in message
assert field_kind.name.lower() in message
def test_single_char_bracket_range_names_the_field_and_the_value(self) -> None:
diagnostic = Diagnostic(
kind=DiagnosticKind.SINGLE_CHAR_BRACKET_RANGE,
cause=cause_for(DiagnosticKind.SINGLE_CHAR_BRACKET_RANGE),
message=_LIBRARY_PROSE,
field=FieldRef("title"),
field_kind=FieldKind.TEXT,
raw_value="200[1-9]",
)
error = _single_diagnostic_to_error(diagnostic)
message = str(error)
assert isinstance(error, SearchQueryError)
assert _LIBRARY_PROSE not in message
assert "title" in message
assert "200[1-9]" in message
class TestRealQueriesRouteCorrectly:
"""The routing table against diagnostics emit() really produces."""
def test_text_range_is_a_400_naming_the_field(
self,
query_index: tantivy.Index,
) -> None:
with pytest.raises(SearchQueryError) as excinfo:
parse_user_query(query_index, "title:[a to b]", UTC)
assert "title" in str(excinfo.value)
def test_wildcard_on_a_numeric_field_is_a_400_naming_the_field(
self,
query_index: tantivy.Index,
) -> None:
with pytest.raises(SearchQueryError) as excinfo:
parse_user_query(query_index, "asn:12*", UTC)
assert "asn" in str(excinfo.value)
def test_single_char_bracket_range_is_a_400_naming_field_and_value(
self,
query_index: tantivy.Index,
) -> None:
with pytest.raises(SearchQueryError) as excinfo:
parse_user_query(query_index, "title:200[1-9]", UTC)
message = str(excinfo.value)
assert "title" in message
assert "200[1-9]" in message
def test_internal_diagnostic_escapes_as_a_query_error(
self,
query_index: tantivy.Index,
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""The one case with no query text that reaches it: emit() reporting
a defect in itself must not be converted to a user-facing 400."""
import documents.search._query as query_mod
def raise_internal(*args: object, **kwargs: object) -> None:
raise QueryError(_diagnostic(DiagnosticKind.BACKEND_REJECTED))
monkeypatch.setattr(query_mod, "tantivy_emit", raise_internal)
with pytest.raises(QueryError):
parse_user_query(query_index, "invoice", UTC)
@@ -1,92 +0,0 @@
"""``field:*`` on a JSON field is user error, not an operator alert.
whoosh-compat classifies EXISTS_REQUIRES_FAST as MISCONFIGURED, and
_map_emit_error used to route every MISCONFIGURED diagnostic to an ERROR log.
But the kind is decided from the registry's own FieldSpec (kind plus fast)
without consulting the index schema, and field_descriptors() builds the JSON
fields non-fast deliberately, so nothing is misconfigured and no operator
action can clear the condition. Any authenticated user could otherwise emit
ERROR lines in a loop by repeating ``notes:*``.
SCHEMA_FIELD_MISSING, the other MISCONFIGURED kind, does compare the registry
against the live schema, so it stays an ERROR.
"""
from __future__ import annotations
import logging
from datetime import UTC
import pytest
import tantivy
from whoosh_compat.errors import Diagnostic
from whoosh_compat.errors import DiagnosticKind
from whoosh_compat.errors import QueryError
from whoosh_compat.errors import cause_for
from whoosh_compat.fields import FieldKind
from whoosh_compat.fields import FieldRef
from documents.search._errors import SearchQueryError
from documents.search._query import _map_emit_error
from documents.search._query import parse_user_query
from documents.search._schema import build_schema
from documents.search._tokenizer import register_tokenizers
pytestmark = pytest.mark.search
# Every spelling of "does this JSON field have a value" a user can type.
EXISTS_QUERIES = [
"notes:*",
"notes.note:*",
"notes.user:*",
"custom_fields:*",
"custom_fields.name:*",
"custom_fields.value:*",
]
@pytest.fixture(scope="module")
def query_index() -> tantivy.Index:
idx = tantivy.Index(build_schema(), path=None)
register_tokenizers(idx, "")
return idx
class TestJsonExistsIsUserError:
@pytest.mark.parametrize("query", EXISTS_QUERIES)
def test_query_is_a_400_that_emits_no_error_log(
self,
query_index: tantivy.Index,
caplog: pytest.LogCaptureFixture,
query: str,
) -> None:
with caplog.at_level(logging.WARNING, logger="paperless.search"):
with pytest.raises(SearchQueryError) as excinfo:
parse_user_query(query_index, query, UTC)
assert query.split(":", maxsplit=1)[0] in str(excinfo.value)
assert [r for r in caplog.records if r.levelno >= logging.ERROR] == []
class TestGenuineMisconfigurationStillLogs:
def test_schema_field_missing_is_an_error_log(
self,
caplog: pytest.LogCaptureFixture,
) -> None:
"""The registry naming a field the index schema does not have is a
real mismatch an operator can fix, so it keeps the alert."""
kind = DiagnosticKind.SCHEMA_FIELD_MISSING
error = QueryError(
Diagnostic(
kind=kind,
cause=cause_for(kind),
message="field 'asn' is not defined in the index schema",
field=FieldRef("asn"),
field_kind=FieldKind.U64,
),
)
with caplog.at_level(logging.ERROR, logger="paperless.search"):
mapped = _map_emit_error(error)
assert isinstance(mapped, SearchQueryError)
records = [r for r in caplog.records if r.levelno == logging.ERROR]
assert len(records) == 1
assert kind.name in records[0].getMessage()
-10
View File
@@ -1,10 +0,0 @@
from whoosh_compat import FieldKind
from documents.search._fields import PUBLIC_FIELDS
class TestPublicFields:
def test_json_fields_have_subpaths(self) -> None:
for field in PUBLIC_FIELDS:
if field.kind is FieldKind.JSON:
assert field.subpaths, f"{field.name} is JSON but has no subpaths"
@@ -1,174 +0,0 @@
"""The words the fuzzy blend clause hands back to tantivy's parser.
The clause re-parses a word string through tantivy, which analyzes it
again, so the words must be the query's raw text rather than the analyzed
text (analysis is not idempotent), and must still be split into plain
words so that hyphenated, dotted and quoted terms keep contributing.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.models import Document
if TYPE_CHECKING:
from pytest_django.fixtures import SettingsWrapper
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
return set(backend.search_ids(query, user=None))
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
doc = Document.objects.create(**kwargs)
backend.add_or_update(doc)
return doc
@pytest.fixture(autouse=True)
def fuzzy_enabled(settings: SettingsWrapper) -> None:
"""Enable the fuzzy blend clause. The threshold doubles as a minimum
score filter, so it is set to 0.0: every hit passes and the test sees
the clause's matching behaviour, not the filter's."""
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.0
class TestFuzzyClauseWords:
def test_a_stemmed_word_is_not_stemmed_a_second_time(
self,
backend: TantivyBackend,
) -> None:
"""'universities' stems to 'univers'; feeding that back to tantivy
stems it again to 'univ', whose fuzzy prefix reaches unrelated
words. The clause must stay wide enough for a typo and no wider."""
wanted = _index(
backend,
title="A",
content="universities of europe",
checksum="fuzz-stem-1",
)
typo = _index(
backend,
title="B",
content="universties of europe",
checksum="fuzz-stem-2",
)
_index(
backend,
title="C",
content="univalent chemical bonds",
checksum="fuzz-stem-3",
)
_index(
backend,
title="D",
content="unicycle repair manual",
checksum="fuzz-stem-4",
)
assert _matched_ids(backend, "universities") == {wanted.pk, typo.pk}
def test_a_hyphenated_term_still_reaches_the_clause(
self,
backend: TantivyBackend,
) -> None:
"""'COVID-19' is one raw token: unless it is split into words, it
carries characters the re-parse would read as grammar, is dropped,
and the whole query loses its fuzzy clause."""
misspelled = _index(
backend,
title="A",
content="covidx testing results",
checksum="fuzz-hyphen-1",
)
assert _matched_ids(backend, "COVID-19") == {misspelled.pk}
def test_a_phrase_still_reaches_the_clause(
self,
backend: TantivyBackend,
) -> None:
"""A phrase is one raw token carrying a space, and is the whole
query's only free text here."""
near_miss = _index(
backend,
title="A",
content="taxation reportage weekly",
checksum="fuzz-phrase-1",
)
assert _matched_ids(backend, '"tax reports"') == {near_miss.pk}
class TestBooleanKeywordsInRawText:
"""Tantivy's boolean keywords are word runs, so they survive the cut
into words and its own parser reads them as grammar. Raw query text
reaches that parser with its case intact, so a quoted phrase can carry
them in."""
@pytest.fixture
def corpus(self, backend: TantivyBackend) -> dict[str, int]:
both = _index(
backend,
title="A",
content="taxation reportage weekly",
checksum="fuzz-kw-1",
)
tax_only = _index(
backend,
title="B",
content="taxation only here",
checksum="fuzz-kw-2",
)
report_only = _index(
backend,
title="C",
content="reportage only here",
checksum="fuzz-kw-3",
)
return {
"both": both.pk,
"tax_only": tax_only.pk,
"report_only": report_only.pk,
}
@pytest.mark.parametrize(
"query",
[
pytest.param('"tax AND reports"', id="and"),
pytest.param('"tax OR reports"', id="or"),
pytest.param('"tax NOT reports"', id="not"),
pytest.param('"tax IN reports"', id="in"),
],
)
def test_a_keyword_inside_a_phrase_stays_an_ordinary_word(
self,
backend: TantivyBackend,
corpus: dict[str, int],
query: str,
) -> None:
"""The phrase asks for three words, so the clause must stay the
disjunction it is for '"tax reports"': AND must not turn it into a
conjunction, NOT must not give it its own exclusion, IN must not
fail the parse."""
assert _matched_ids(backend, '"tax reports"') == set(corpus.values())
assert _matched_ids(backend, query) == set(corpus.values())
def test_a_trailing_keyword_does_not_drop_the_clause(
self,
backend: TantivyBackend,
corpus: dict[str, int],
) -> None:
"""'tax AND' is a syntax error to tantivy's parser, which would
cost the whole query its fuzzy clause."""
assert _matched_ids(backend, '"tax AND"') == {
corpus["both"],
corpus["tax_only"],
}
@@ -1,192 +0,0 @@
"""Regression coverage for the unguarded TEXT-mode highlight query.
parse_simple_text_highlight_query re-parses simple-search tokens through
Tantivy's query-string parser to build a SnippetGenerator-compatible query.
Simple-search tokens keep arbitrary punctuation (quotes, colons, brackets,
slashes), so any token carrying Tantivy query grammar raised an unguarded
ValueError. The search itself had already succeeded by the time this ran:
only the highlight step failed, and with the DocumentViewSet.list
exception handler narrowed elsewhere on this branch, that ValueError now
reaches the client as a bare 500 rather than a 400.
Covers three angles:
- the query builder itself: quoting each token as its own escaped phrase
should let it parse instead of raising, for every failure mode a plain-
text query can trigger (syntax error, unknown field, unsupported regex).
- highlight_hits: even when a token still can't be expressed as a
highlight query, the guard must fall back to a query that still
produces usable highlight HTML, not silently empty ones.
- the real API endpoint: pinning the previously-500 status to 200.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
import tantivy
from rest_framework import status
from documents.search._backend import SearchMode
from documents.search._query import parse_simple_text_highlight_query
from documents.search._schema import build_schema
from documents.search._tokenizer import register_tokenizers
from documents.tests.factories import DocumentFactory
if TYPE_CHECKING:
from rest_framework.test import APIClient
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
# Each spelling below trips a different Tantivy parser failure mode:
# 'a"b' -> Syntax Error (unterminated quote)
# foo:bar -> unknown field
# (a -> Syntax Error (unbalanced group)
# [a -> Syntax Error (unbalanced range)
# /a/ -> Unsupported query (regex queries disallowed)
_MALFORMED_QUERIES = [
pytest.param('a"b', id="unterminated_quote"),
pytest.param("foo:bar", id="unknown_field"),
pytest.param("(a", id="unbalanced_group"),
pytest.param("[a", id="unbalanced_range"),
pytest.param("/a/", id="unsupported_regex"),
]
@pytest.fixture(scope="module")
def query_index() -> tantivy.Index:
"""An in-memory, unstemmed index for parse-only tests."""
schema = build_schema()
idx = tantivy.Index(schema, path=None)
register_tokenizers(idx, "")
return idx
class TestParseSimpleTextHighlightQueryDoesNotRaise:
"""The query builder itself must tolerate Tantivy syntax in its tokens."""
@pytest.mark.parametrize("raw_query", _MALFORMED_QUERIES)
def test_malformed_token_does_not_raise(
self,
query_index: tantivy.Index,
raw_query: str,
) -> None:
assert isinstance(
parse_simple_text_highlight_query(query_index, raw_query),
tantivy.Query,
)
class TestHighlightHitsProducesUsableHighlights:
"""highlight_hits must keep producing real <b>-wrapped snippet HTML for
these queries, not merely avoid raising."""
@pytest.mark.parametrize(
"raw_query",
[*_MALFORMED_QUERIES, pytest.param("plain text", id="plain_text_sanity")],
)
def test_highlight_still_contains_matched_text(
self,
backend: TantivyBackend,
raw_query: str,
) -> None:
doc = DocumentFactory.create(
title="probe",
content=f"needle content containing {raw_query} literally here",
)
backend.add_or_update(doc)
hits = backend.highlight_hits(
raw_query,
[doc.pk],
search_mode=SearchMode.TEXT,
)
assert len(hits) == 1
highlights = hits[0]["highlights"]
assert "content" in highlights, (
f"Expected a content highlight for {raw_query!r}, got: {highlights!r}"
)
assert "<b>" in highlights["content"], (
f"Highlight for {raw_query!r} carries no matched-term markup: "
f"{highlights['content']!r}"
)
class TestHighlightGuardDiscriminatesOnValueError:
"""The guard added to highlight_hits must catch exactly ValueError, the
same shape as the sibling notes_text guard, and let anything else
through -- so a real library defect is never mistaken for a harmless
syntax error."""
def test_non_value_error_is_not_swallowed(
self,
backend: TantivyBackend,
monkeypatch: pytest.MonkeyPatch,
) -> None:
import documents.search._backend as backend_mod
def raise_runtime_error(*args: object, **kwargs: object) -> object:
raise RuntimeError("synthetic bug, unrelated to query syntax")
monkeypatch.setattr(
backend_mod,
"parse_simple_text_highlight_query",
raise_runtime_error,
)
doc = DocumentFactory.create(title="probe", content="anything here")
backend.add_or_update(doc)
with pytest.raises(RuntimeError):
backend.highlight_hits(
"anything",
[doc.pk],
search_mode=SearchMode.TEXT,
)
@pytest.mark.usefixtures("_search_index")
class TestApiNoLongerReturns500:
"""Pins the actual regression: a matching TEXT-mode search whose query
string carries Tantivy syntax must return results, not a server error."""
@pytest.mark.parametrize("raw_query", _MALFORMED_QUERIES)
def test_malformed_text_query_returns_200(
self,
admin_client: APIClient,
raw_query: str,
) -> None:
from documents.search import get_backend
doc = DocumentFactory.create(
title="probe",
content=f"needle content containing {raw_query} literally here",
)
get_backend().add_or_update(doc)
response = admin_client.get(f"/api/documents/?text={raw_query}")
assert response.status_code == status.HTTP_200_OK
assert response.data["count"] == 1
def test_plain_text_query_still_returns_200(
self,
admin_client: APIClient,
) -> None:
"""Sanity check: the guard must not mask a total failure of the
ordinary highlight path."""
from documents.search import get_backend
doc = DocumentFactory.create(
title="probe",
content="needle content containing plain text literally here",
)
get_backend().add_or_update(doc)
response = admin_client.get("/api/documents/?text=plain text")
assert response.status_code == status.HTTP_200_OK
assert response.data["count"] == 1

Some files were not shown because too many files have changed in this diff Show More