mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-08-26 12:43:19 +00:00
Compare commits
25
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a98d0669e4 | ||
|
|
01c40aa0bd | ||
|
|
5f70b6eee7 | ||
|
|
e3e4b26944 | ||
|
|
a0908f6b4a | ||
|
|
bab9129ff8 | ||
|
|
1e13f86174 | ||
|
|
22c57fb392 | ||
|
|
08a7f6ccc0 | ||
|
|
3f4d327f64 | ||
|
|
bd326540fa | ||
|
|
33f9adb05a | ||
|
|
d24606a03b | ||
|
|
294328f174 | ||
|
|
0458bad5f2 | ||
|
|
a510d03c77 | ||
|
|
d7b3612a41 | ||
|
|
7e4a644714 | ||
|
|
bbcd6af2fe | ||
|
|
0431939f18 | ||
|
|
bed95ea301 | ||
|
|
42034c3c77 | ||
|
|
705220fb5a | ||
|
|
2d084983c8 | ||
|
|
e2c284f64e |
@@ -68,7 +68,7 @@ services:
|
||||
- "--chromium-disable-javascript=true"
|
||||
- "--chromium-allow-list=file:///tmp/.*"
|
||||
tika:
|
||||
image: docker.io/apache/tika:latest
|
||||
image: docker.io/apache/tika:3.3.1.0
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
data:
|
||||
|
||||
@@ -115,6 +115,3 @@ celerybeat-schedule*
|
||||
|
||||
# Git worktree local folder
|
||||
.worktrees
|
||||
|
||||
# Agent workflow scratch (ledgers, briefs, review packages)
|
||||
.superpowers/
|
||||
|
||||
@@ -81,7 +81,7 @@ services:
|
||||
- "--chromium-disable-javascript=true"
|
||||
- "--chromium-allow-list=file:///tmp/.*"
|
||||
tika:
|
||||
image: docker.io/apache/tika:latest
|
||||
image: docker.io/apache/tika:3.3.1.0
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
data:
|
||||
|
||||
@@ -76,7 +76,7 @@ services:
|
||||
- "--chromium-disable-javascript=true"
|
||||
- "--chromium-allow-list=file:///tmp/.*"
|
||||
tika:
|
||||
image: docker.io/apache/tika:latest
|
||||
image: docker.io/apache/tika:3.3.1.0
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
data:
|
||||
|
||||
@@ -65,7 +65,7 @@ services:
|
||||
- "--chromium-disable-javascript=true"
|
||||
- "--chromium-allow-list=file:///tmp/.*"
|
||||
tika:
|
||||
image: docker.io/apache/tika:latest
|
||||
image: docker.io/apache/tika:3.3.1.0
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
data:
|
||||
|
||||
@@ -521,8 +521,7 @@ Pass `--recreate` to wipe the existing index before rebuilding. Use this when th
|
||||
index is corrupted or you want a fully clean rebuild.
|
||||
|
||||
Pass `--if-needed` to skip the rebuild if the index is already up to date (schema
|
||||
version, schema fingerprint and search language all match). Safe to run on every
|
||||
startup or upgrade.
|
||||
version and search language match). Safe to run on every startup or upgrade.
|
||||
|
||||
Specify `optimize` to optimize the index. This command is regularly invoked by the
|
||||
task scheduler.
|
||||
|
||||
@@ -129,6 +129,10 @@ At a minimum you need to enable AI and choose an LLM backend:
|
||||
and/or [`PAPERLESS_AI_LLM_ENDPOINT`](configuration.md#PAPERLESS_AI_LLM_ENDPOINT). Ollama
|
||||
requires `PAPERLESS_AI_LLM_ENDPOINT` pointing at your Ollama server.
|
||||
|
||||
See the community-maintained wiki page on
|
||||
[choosing AI models](https://github.com/paperless-ngx/paperless-ngx/wiki/AI-Model-Recommendations)
|
||||
for suggested generation and embedding models.
|
||||
|
||||
### AI-assisted suggestions
|
||||
|
||||
With AI enabled, Paperless-ngx can suggest a title, tags, correspondent, document type,
|
||||
@@ -808,7 +812,8 @@ Third-party parser plugins extend Paperless-ngx to support additional file
|
||||
formats. A plugin is a Python package that advertises itself under the
|
||||
`paperless_ngx.parsers` entry point group. Refer to the
|
||||
[developer documentation](development.md#making-custom-parsers) for how to
|
||||
create one.
|
||||
create one, or see the wiki for a community-maintained list of
|
||||
[parser plugins](https://github.com/paperless-ngx/paperless-ngx/wiki/Related-Projects#parser-plugins).
|
||||
|
||||
!!! warning "Third-party plugins are not officially supported"
|
||||
|
||||
|
||||
@@ -1215,7 +1215,7 @@ should be a valid crontab(5) expression describing when to run.
|
||||
|
||||
: If set to the string "disable", no emails will be fetched automatically.
|
||||
|
||||
Defaults to `*/10 * * * *` or every ten minutes.
|
||||
Defaults to every ten minutes, with an installation-specific minute offset.
|
||||
|
||||
#### [`PAPERLESS_TRAIN_TASK_CRON=<cron expression>`](#PAPERLESS_TRAIN_TASK_CRON) {#PAPERLESS_TRAIN_TASK_CRON}
|
||||
|
||||
@@ -2088,6 +2088,8 @@ suggestions. This setting is required to be set to true in order to use the AI f
|
||||
models supported by the current embedding backend. If not supplied, defaults to
|
||||
"text-embedding-3-small" for the OpenAI-compatible backend,
|
||||
"sentence-transformers/all-MiniLM-L6-v2" for Huggingface, and "embeddinggemma" for Ollama.
|
||||
See [choosing AI models](https://github.com/paperless-ngx/paperless-ngx/wiki/AI-Model-Recommendations)
|
||||
for language and resource considerations.
|
||||
|
||||
Defaults to None.
|
||||
|
||||
@@ -2144,6 +2146,8 @@ setting is required to be set to use the AI features.
|
||||
: The model to use for the AI backend, i.e. "gpt-3.5-turbo", "gpt-4" or any of the models supported
|
||||
by the current backend. If not supplied, defaults to "gpt-3.5-turbo" for the OpenAI-compatible
|
||||
backend and "llama3.1" for Ollama.
|
||||
See [choosing AI models](https://github.com/paperless-ngx/paperless-ngx/wiki/AI-Model-Recommendations)
|
||||
for local versus remote and model-size considerations.
|
||||
|
||||
Defaults to None.
|
||||
|
||||
|
||||
@@ -156,7 +156,7 @@ The new settings are independent:
|
||||
|
||||
### Database configuration
|
||||
|
||||
If you changed OCR settings via the admin UI (ApplicationConfiguration), the database values are **migrated automatically** during the upgrade. `mode` values (`skip` / `skip_noarchive`) are mapped to their new equivalents and `skip_archive_file` values are converted to the new `archive_file_generation` field. After upgrading, review the OCR settings in the admin UI to confirm the migrated values match your intent.
|
||||
If you changed OCR settings via the admin UI (ApplicationConfiguration), the database values are **migrated automatically** during the upgrade. `mode` values (`skip` / `skip_noarchive`) are mapped to their new equivalents and explicit `skip_archive_file` values are converted to the new `archive_file_generation` field. Users who relied on the old defaults must set `archive_file_generation` to `always` to preserve the v2 behaviour of always creating an archive. After upgrading, review the OCR settings in the admin UI to confirm the migrated values match your intent.
|
||||
|
||||
### Action Required
|
||||
|
||||
@@ -165,8 +165,9 @@ Remove any `PAPERLESS_OCR_SKIP_ARCHIVE_FILE` variable from your environment. If
|
||||
```bash
|
||||
# v2: skip OCR when text present, always archive
|
||||
PAPERLESS_OCR_MODE=skip
|
||||
# v3: equivalent (auto is the new default)
|
||||
# No change needed - auto is the default
|
||||
# v3: equivalent
|
||||
PAPERLESS_OCR_MODE=auto
|
||||
PAPERLESS_ARCHIVE_FILE_GENERATION=always
|
||||
|
||||
# v2: skip OCR when text present, skip archive too
|
||||
PAPERLESS_OCR_MODE=skip_noarchive
|
||||
|
||||
+4
-90
@@ -886,19 +886,6 @@ Matching documents with logical expressions:
|
||||
|
||||
```
|
||||
shopname AND (product1 OR product2)
|
||||
invoice NOT draft
|
||||
```
|
||||
|
||||
`AND`, `OR` and `NOT` must be written in capitals, and parentheses group sub-expressions. Terms written next to each other with no operator between them are combined with `AND`.
|
||||
|
||||
!!! warning
|
||||
|
||||
A leading `-` does **not** exclude a term. Separators are stripped during indexing, so `invoice -secret` searches for `invoice` and `secret`, which is the opposite of what you probably intended. Use `NOT` to exclude a term: `invoice NOT secret`.
|
||||
|
||||
Matching an exact phrase, in order, by quoting it:
|
||||
|
||||
```
|
||||
"quick brown fox"
|
||||
```
|
||||
|
||||
Matching specific tags, correspondents or types:
|
||||
@@ -906,12 +893,8 @@ Matching specific tags, correspondents or types:
|
||||
```
|
||||
type:invoice tag:unpaid
|
||||
correspondent:university certificate
|
||||
tag:bills,unpaid
|
||||
```
|
||||
|
||||
- `document_type` may be abbreviated to `type`, and `storage_path` to `path`.
|
||||
- A comma-separated list after `tag:` requires **all** of the listed tags, so `tag:bills,unpaid` matches only documents tagged both `bills` and `unpaid`.
|
||||
|
||||
Matching dates:
|
||||
|
||||
```
|
||||
@@ -920,58 +903,14 @@ added:yesterday
|
||||
modified:today
|
||||
```
|
||||
|
||||
Matching by archive metadata:
|
||||
|
||||
```
|
||||
asn:100
|
||||
page_count:12
|
||||
num_notes:0
|
||||
checksum:9f86d081884c7d659a2feaa0c55ad015a3bf4f1b2b0b822cd15d6c15b0f00a08
|
||||
original_filename:invoice.pdf
|
||||
```
|
||||
|
||||
- `asn` matches a document's Archive Serial Number.
|
||||
- `page_count` matches a document's page count.
|
||||
- `num_notes` matches how many notes a document has.
|
||||
- `checksum` matches the checksum of the original document file (not the archived/processed version). Unlike the text fields, this one is stored verbatim rather than tokenized, so only a complete, lowercase checksum matches. To search by the first few characters instead, use a wildcard: `checksum:9f86d081*`. Wildcard patterns on the text fields are also tried stemmed, to line up with the stemmed index, but `checksum` is indexed without stemming, so its patterns are not stemmed either: a wildcard prefix is matched literally, apart from being lowercased first. `checksum:9F86D081*` therefore does find the document, even though the plain uppercase term does not.
|
||||
- `original_filename` matches the filename of the document as originally consumed.
|
||||
|
||||
`asn`, `page_count` and `num_notes` are numeric and also accept ranges, for example `asn:[50 to 150]`.
|
||||
|
||||
Matching inexact words:
|
||||
|
||||
```
|
||||
invoice*
|
||||
title:Invoice*
|
||||
produ*name
|
||||
```
|
||||
|
||||
Wildcards are matched against the _stemmed_ terms stored in the index, not
|
||||
against the words as they appear in the document. Each literal part of a
|
||||
pattern is tried both as you typed it and in its stemmed form, so a trailing
|
||||
`*` matches a word and its inflections (`invoice*` finds "invoice", "invoices"
|
||||
and "invoiced") as well as longer words whose stored term still begins with
|
||||
what you typed (`copy*` finds "copyright" alongside "copy" and "copies").
|
||||
|
||||
It is still not a plain prefix search over the original text. A trailing `*`
|
||||
matches a stored term when either the run you typed or its stemmed form is a
|
||||
prefix of that term, so a fragment that stops part-way between the two matches
|
||||
neither: `universities*` finds "university" and "universities", which are both
|
||||
stored as `univers`, while the shorter `universit*` finds nothing at all. For
|
||||
the same reason `happine*` does not find "happiness", which is stored as
|
||||
`happi`. And a pattern that requires letters after the wildcard which stemming
|
||||
has removed cannot match either: `productname` is stored as `productnam`, so
|
||||
`produ*name` finds nothing.
|
||||
|
||||
Matching natural date keywords:
|
||||
|
||||
The multi-word date keywords listed below work quoted or unquoted after a
|
||||
date field (`added:"previous month"` and `added:previous month` are
|
||||
equivalent); elsewhere in a query the same words are treated as ordinary
|
||||
search text. Other date expressions the parser accepts (relative offsets
|
||||
like `-1 week`, or specific dates like `12 december 2019`) must be quoted when
|
||||
they stand alone as a value; inside a range's brackets they work unquoted, as
|
||||
in `added:[-1 week to now]`.
|
||||
|
||||
```
|
||||
added:today
|
||||
modified:yesterday
|
||||
@@ -984,30 +923,6 @@ Supported date keywords: `today`, `yesterday`, `previous week`,
|
||||
`this month`, `previous month`, `this year`, `previous year`,
|
||||
`previous quarter`.
|
||||
|
||||
These other date forms also work after a date field:
|
||||
|
||||
```
|
||||
added:tomorrow
|
||||
created:2005-03-04
|
||||
added:january
|
||||
modified:"next monday"
|
||||
added:"last monday"
|
||||
added:"2005-01-01T00:00:00Z"
|
||||
created:[2005-01-01 to 2005-01-31]
|
||||
added:[2005-06-15T09:00:00Z to 2005-06-15T17:00:00Z]
|
||||
```
|
||||
|
||||
- `tomorrow`, like `today` and `yesterday`, covers that whole day.
|
||||
- An ISO date such as `2005-03-04` covers that whole day, and `2005-01` covers that whole month.
|
||||
- A month name such as `january` covers that whole month in the current year.
|
||||
- `next <weekday>` and `last <weekday>` each cover that whole day and must be quoted. A bare weekday name such as `monday` is not accepted.
|
||||
- A full timestamp such as `2005-01-01T00:00:00Z` matches that exact instant. Like the other expressions above, it has to be quoted when it stands on its own: `added:"2005-01-01T00:00:00Z"`. The unquoted spelling is rejected with an error rather than searched, because only part of it can be read as a date.
|
||||
- A range takes two of the above as its bounds, for example `created:[2005 to 2009]` or `added:[2005-01-01 to 2005-01-31]`. Bounds may carry a time of day. A bound is normally written without quotes; if you do quote one, use single quotes (`added:['-1 week' to now]`), because a double-quoted bound is rejected with an error.
|
||||
|
||||
!!! warning
|
||||
|
||||
As a value on its own, `now`, `noon`, `midnight` and relative offsets such as `"-3 days"` or `"-1 week"` are accepted by the parser but resolve to a single instant rather than to a span of time, so they match only a document whose timestamp is exactly that instant, which in practice means no documents at all. Quoting does not change this. As a *range bound* they are the opposite of a trap and are what you want: `added:['-1 week' to now]` covers the whole of the last seven days. Spellings like `now-3days` and `"3 days ago"` are rejected outright wherever they appear.
|
||||
|
||||
#### Searching custom fields
|
||||
|
||||
Custom field names and values are included in the full-text index, but they
|
||||
@@ -1023,7 +938,6 @@ custom_fields.name:Insurance custom_fields.value:policy
|
||||
- `custom_fields.value` matches against the value of any custom field.
|
||||
- `custom_fields.name` matches the name of the field (use quotes for multi-word names).
|
||||
- Combine both to find documents where a specific named field contains a specific value.
|
||||
- The bare `custom_fields:` prefix is shorthand for `custom_fields.value:`.
|
||||
|
||||
Because separators are stripped during indexing, individual parts of formatted
|
||||
codes are searchable on their own. A value stored as `A-1312/99.50` produces the
|
||||
@@ -1051,9 +965,9 @@ notes.note:reminder
|
||||
notes.user:alice notes.note:insurance
|
||||
```
|
||||
|
||||
The bare `notes:` prefix is shorthand for `notes.note:`.
|
||||
|
||||
All of these constructs can be combined as you see fit. What is described above is the whole of the query language paperless supports. It resembles other search query languages without being identical to any of them, so a construct that is not documented here is most likely treated as ordinary search text rather than as syntax, and an unrecognized field name is searched as text too.
|
||||
All of these constructs can be combined as you see fit. If you want to
|
||||
learn more about the query language used by paperless, see the
|
||||
[Tantivy query language documentation](https://docs.rs/tantivy/latest/tantivy/query/struct.QueryParser.html).
|
||||
|
||||
!!! note
|
||||
|
||||
|
||||
@@ -77,7 +77,6 @@ dependencies = [
|
||||
"torch~=2.13.0",
|
||||
"watchfiles>=1.2",
|
||||
"whitenoise~=6.11",
|
||||
"whoosh-compat[tantivy]",
|
||||
"zxing-cpp~=3.1.0",
|
||||
]
|
||||
[project.optional-dependencies]
|
||||
@@ -169,9 +168,6 @@ psycopg-c = [
|
||||
torch = [
|
||||
{ index = "pytorch-cpu" },
|
||||
]
|
||||
# TODO: switch to a pinned PyPI version once whoosh-compat releases; fall back
|
||||
# to a pinned git commit SHA if that release slips.
|
||||
whoosh-compat = { path = "../whoosh-compat" }
|
||||
|
||||
[tool.ruff]
|
||||
target-version = "py311"
|
||||
|
||||
@@ -33,7 +33,7 @@ test('should warn on unsaved changes', async ({ page }) => {
|
||||
await page.getByRole('button', { name: 'Close', exact: true }).click()
|
||||
await expect(page.getByRole('dialog')).toHaveText(/unsaved changes/)
|
||||
await page.getByRole('button', { name: 'Cancel' }).click()
|
||||
await page.getByRole('link', { name: 'Close all' }).click()
|
||||
await page.getByRole('button', { name: 'Close all' }).click()
|
||||
await expect(page.getByRole('dialog')).toHaveText(/unsaved changes/)
|
||||
})
|
||||
|
||||
|
||||
+431
-121
File diff suppressed because it is too large
Load Diff
@@ -89,7 +89,7 @@
|
||||
</div>
|
||||
</div>
|
||||
@if (filterText?.length) {
|
||||
<button class="btn btn-link btn-sm px-2 position-absolute top-0 end-0 z-10" (click)="resetFilter()">
|
||||
<button class="btn btn-link btn-sm px-2 position-absolute top-0 end-0 z-10" (click)="resetFilter()" aria-label="Clear search" i18n-aria-label>
|
||||
<i-bs width="1em" height="1em" name="x"></i-bs>
|
||||
</button>
|
||||
}
|
||||
@@ -208,7 +208,7 @@
|
||||
</td>
|
||||
}
|
||||
<td class="d-lg-none">
|
||||
<button class="btn btn-link" (click)="expandTask(task); $event.stopPropagation();">
|
||||
<button class="btn btn-link" (click)="expandTask(task); $event.stopPropagation();" aria-label="View task details" i18n-aria-label>
|
||||
<i-bs width="1.2em" height="1.2em" name="info-circle"></i-bs>
|
||||
</button>
|
||||
</td>
|
||||
|
||||
@@ -18,9 +18,11 @@
|
||||
</button>
|
||||
</pngx-page-header>
|
||||
|
||||
<div class="row mb-3">
|
||||
<ngb-pagination class="col-auto" [pageSize]="25" [collectionSize]="totalDocuments()" [page]="page()" [maxSize]="5" (pageChange)="page.set($event); reload()" size="sm" aria-label="Pagination"></ngb-pagination>
|
||||
</div>
|
||||
@if (totalDocuments() > 25) {
|
||||
<div class="row mb-3">
|
||||
<ngb-pagination class="col-auto" [pageSize]="25" [collectionSize]="totalDocuments()" [page]="page()" [maxSize]="5" (pageChange)="page.set($event); reload()" size="sm" aria-label="Pagination"></ngb-pagination>
|
||||
</div>
|
||||
}
|
||||
|
||||
<div class="card border table-responsive mb-3">
|
||||
<table class="table table-striped align-middle shadow-sm mb-0">
|
||||
@@ -64,7 +66,7 @@
|
||||
<td scope="row">
|
||||
<div class="btn-group d-block d-sm-none">
|
||||
<div ngbDropdown container="body" class="d-inline-block">
|
||||
<button type="button" class="btn btn-link" id="actionsMenuMobile" (click)="$event.stopPropagation()" ngbDropdownToggle>
|
||||
<button type="button" class="btn btn-link" id="actionsMenuMobile" (click)="$event.stopPropagation()" ngbDropdownToggle aria-label="Actions" i18n-aria-label>
|
||||
<i-bs name="three-dots-vertical"></i-bs>
|
||||
</button>
|
||||
<div ngbDropdownMenu aria-labelledby="actionsMenuMobile">
|
||||
|
||||
@@ -4,27 +4,34 @@
|
||||
(click)="closeMobileSearch(); toggleMenuCollapsed()">
|
||||
<span class="navbar-toggler-icon"></span>
|
||||
</button>
|
||||
<a class="navbar-brand d-flex align-items-center me-0 px-3 py-3 order-sm-0"
|
||||
[ngClass]="{ 'slim': slimSidebarEnabled, 'col-auto col-md-3 col-lg-2 col-xxxl-1' : !slimSidebarEnabled, 'py-3' : !customAppTitle?.length || slimSidebarEnabled, 'py-2': customAppTitle?.length }"
|
||||
<a class="navbar-brand d-flex align-items-center me-0 ps-md-3 py-0 order-sm-0"
|
||||
[ngClass]="{ 'slim': slimSidebarEnabled, '' : !slimSidebarEnabled }"
|
||||
routerLink="/dashboard"
|
||||
tourAnchor="tour.intro">
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 1000 1000" width="1.5em" height="1.5em" fill="currentColor">
|
||||
<path d="M341,949.1c-6.9-20.3-20.7-61.2-21.9-61-199.6-88.9-182.5-229.8-134.3-347.5,30,137.2,268.8,148.9,146.2,336-.9,2.2,10,27.8,19.5,51.3,22.7-51.9,58.6-115.5,55.8-120.8C178,398.7,724.9,299,807.1,18.5c83,251.5,53.1,659.8-377.4,814.9-2,1.4-63.5,148.6-66.9,150.2-.2-2.1-33.2,2.9-30.1-8.7,1.6-7,4.8-16.2,8.2-25.6h0v-.2h.1ZM323.1,846.2c48.3-71.9-12.7-120.8-56.9-152.2,81.2,107.4,66.4,120.8,56.9,152.2h0Z"/>
|
||||
</svg>
|
||||
<div class="ms-2 ms-md-3 d-inline-block" [class.d-md-none]="slimSidebarEnabled">
|
||||
@if (customAppTitle?.length) {
|
||||
<div class="d-flex flex-column align-items-start custom-title">
|
||||
<span class="title">{{customAppTitle}}</span>
|
||||
<span class="byline text-uppercase font-monospace" i18n>by Paperless-ngx</span>
|
||||
</div>
|
||||
@if (!hasCustomBranding) {
|
||||
<pngx-logo extra_classes="navbar-official-logo px-1" height="2.4rem"></pngx-logo>
|
||||
<svg class="brand-mark brand-mark-slim d-none" xmlns="http://www.w3.org/2000/svg" viewBox="0 0 1000 1000" width="1.5em" height="1.5em" fill="currentColor">
|
||||
<path d="M341,949.1c-6.9-20.3-20.7-61.2-21.9-61-199.6-88.9-182.5-229.8-134.3-347.5,30,137.2,268.8,148.9,146.2,336-.9,2.2,10,27.8,19.5,51.3,22.7-51.9,58.6-115.5,55.8-120.8C178,398.7,724.9,299,807.1,18.5c83,251.5,53.1,659.8-377.4,814.9-2,1.4-63.5,148.6-66.9,150.2-.2-2.1-33.2,2.9-30.1-8.7,1.6-7,4.8-16.2,8.2-25.6h0v-.2h.1ZM323.1,846.2c48.3-71.9-12.7-120.8-56.9-152.2,81.2,107.4,66.4,120.8,56.9,152.2h0Z"/>
|
||||
</svg>
|
||||
} @else {
|
||||
@if (customAppLogo) {
|
||||
<img class="brand-logo" [src]="customAppLogo" alt="" />
|
||||
} @else {
|
||||
Paperless-ngx
|
||||
<svg class="brand-mark" xmlns="http://www.w3.org/2000/svg" viewBox="0 0 1000 1000" width="1.5em" height="1.5em" fill="currentColor">
|
||||
<path d="M341,949.1c-6.9-20.3-20.7-61.2-21.9-61-199.6-88.9-182.5-229.8-134.3-347.5,30,137.2,268.8,148.9,146.2,336-.9,2.2,10,27.8,19.5,51.3,22.7-51.9,58.6-115.5,55.8-120.8C178,398.7,724.9,299,807.1,18.5c83,251.5,53.1,659.8-377.4,814.9-2,1.4-63.5,148.6-66.9,150.2-.2-2.1-33.2,2.9-30.1-8.7,1.6-7,4.8-16.2,8.2-25.6h0v-.2h.1ZM323.1,846.2c48.3-71.9-12.7-120.8-56.9-152.2,81.2,107.4,66.4,120.8,56.9,152.2h0Z"/>
|
||||
</svg>
|
||||
}
|
||||
</div>
|
||||
<div class="brand-copy ms-2 text-truncate" [class.d-md-none]="slimSidebarEnabled">
|
||||
<span class="brand-title text-truncate">{{ appTitle }}</span>
|
||||
@if (customAppTitle) {
|
||||
<span class="byline text-uppercase font-monospace" i18n>by Paperless-ngx</span>
|
||||
}
|
||||
</div>
|
||||
}
|
||||
</a>
|
||||
<div class="search-container flex-grow-1 py-2 pb-3 pb-sm-2 px-3 ps-md-4 me-sm-auto order-3 order-sm-1"
|
||||
<div class="search-container flex-grow-1 py-2 pb-3 pb-sm-2 px-3 ps-md-3 me-sm-auto order-3 order-sm-1"
|
||||
[class.mobile-hidden]="mobileSearchHidden()">
|
||||
<div class="col-12 col-md-7">
|
||||
<div class="col-12 header-search">
|
||||
<pngx-global-search></pngx-global-search>
|
||||
</div>
|
||||
</div>
|
||||
@@ -34,7 +41,7 @@
|
||||
}
|
||||
<pngx-toasts-dropdown></pngx-toasts-dropdown>
|
||||
<li ngbDropdown class="nav-item dropdown">
|
||||
<button class="btn ps-1 border-0" id="userDropdown" ngbDropdownToggle>
|
||||
<button class="btn navbar-action border-0" id="userDropdown" ngbDropdownToggle aria-label="User menu" i18n-aria-label>
|
||||
<i-bs width="1.3em" height="1.3em" name="person-circle"></i-bs>
|
||||
<span class="small ms-2 d-none d-sm-inline">
|
||||
{{this.settingsService.displayName}}
|
||||
@@ -71,7 +78,7 @@
|
||||
[ngClass]="slimSidebarEnabled ? 'slim' : 'col-md-3 col-lg-2 col-xxxl-1'" [class.animating]="slimSidebarAnimating()"
|
||||
[ngbCollapse]="isMenuCollapsed()">
|
||||
@if (canSaveSettings) {
|
||||
<button class="btn btn-sm btn-dark sidebar-slim-toggler" (click)="toggleSlimSidebar()">
|
||||
<button class="btn btn-sm btn-dark sidebar-slim-toggler" (click)="toggleSlimSidebar()" [aria-label]="slimSidebarEnabled ? 'Expand sidebar' : 'Collapse sidebar'" i18n-aria-label>
|
||||
@if (slimSidebarEnabled) {
|
||||
<i-bs width="0.9em" height="0.9em" name="chevron-double-right"></i-bs>
|
||||
} @else {
|
||||
@@ -79,7 +86,7 @@
|
||||
}
|
||||
</button>
|
||||
}
|
||||
<div class="sidebar-sticky pt-3 d-flex flex-column justify-space-around">
|
||||
<div class="sidebar-sticky pt-3 pb-1 d-flex flex-column justify-space-around">
|
||||
<ul class="nav flex-column">
|
||||
<li class="nav-item app-link">
|
||||
<a class="nav-link" routerLink="dashboard" routerLinkActive="active" (click)="closeMenu()"
|
||||
@@ -89,7 +96,9 @@
|
||||
</a>
|
||||
</li>
|
||||
<li class="nav-item app-link" *pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.Document }">
|
||||
<a class="nav-link" routerLink="documents" routerLinkActive="active" (click)="closeMenu()"
|
||||
<a class="nav-link" routerLink="documents" routerLinkActive="active"
|
||||
[routerLinkActiveOptions]="{ paths: 'exact', queryParams: 'ignored', matrixParams: 'ignored', fragment: 'ignored' }"
|
||||
(click)="closeMenu()"
|
||||
ngbPopover="Documents" i18n-ngbPopover [disablePopover]="!slimSidebarEnabled" placement="end"
|
||||
container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
||||
<i-bs class="me-2" name="files"></i-bs><span><ng-container i18n>Documents</ng-container></span>
|
||||
@@ -159,11 +168,12 @@
|
||||
}
|
||||
@if (openDocuments.length >= 1) {
|
||||
<li class="nav-item w-100 app-link">
|
||||
<a class="nav-link app-link" [class.text-truncate]="!slimSidebarEnabled" [routerLink]="[]" (click)="closeAll()"
|
||||
<button type="button" class="nav-link nav-link-action app-link w-100 text-start"
|
||||
[class.text-truncate]="!slimSidebarEnabled" (click)="closeAll()"
|
||||
ngbPopover="Close all" i18n-ngbPopover [disablePopover]="!slimSidebarEnabled" placement="end"
|
||||
container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
||||
<i-bs class="me-2" name="x"></i-bs><span><ng-container i18n>Close all</ng-container></span>
|
||||
</a>
|
||||
</button>
|
||||
</li>
|
||||
}
|
||||
</ul>
|
||||
@@ -177,10 +187,11 @@
|
||||
@if (canManageAttributes) {
|
||||
<li class="nav-item app-link" tourAnchor="tour.tags">
|
||||
<div class="d-flex align-items-center attributes-row">
|
||||
<a class="nav-link flex-fill" routerLink="attributes" routerLinkActive="active" (click)="closeMenu()"
|
||||
<a class="nav-link flex-fill" routerLink="attributes" routerLinkActive="active"
|
||||
[routerLinkActiveOptions]="{ exact: !(slimSidebarEnabled || attributesSectionsCollapsed) }" (click)="closeMenu()"
|
||||
ngbPopover="Attributes" i18n-ngbPopover [disablePopover]="!slimSidebarEnabled" placement="end"
|
||||
container="body" triggers="mouseenter:mouseleave" popoverClass="popover-slim">
|
||||
<i-bs class="me-2" name="stack"></i-bs><span><ng-container i18n>Attributes</ng-container></span>
|
||||
<i-bs name="stack"></i-bs><span class="ms-2"><ng-container i18n>Attributes</ng-container></span>
|
||||
</a>
|
||||
@if (!slimSidebarEnabled && canSaveSettings) {
|
||||
<button
|
||||
@@ -363,7 +374,7 @@
|
||||
</a>
|
||||
}
|
||||
} @else {
|
||||
<a *pngxIfPermissions="{ action: PermissionAction.Change, type: PermissionType.UISettings }" class="small text-decoration-none" routerLink="/settings" fragment="update-checking"
|
||||
<a *pngxIfPermissions="{ action: PermissionAction.Change, type: PermissionType.UISettings }" class="small text-decoration-none" routerLink="/settings" fragment="update-checking" aria-label="Configure update checking" i18n-aria-label
|
||||
[ngbPopover]="updateCheckingNotEnabledPopContent" popoverClass="shadow" triggers="mouseenter"
|
||||
container="body">
|
||||
<i-bs width="1.2em" height="1.2em" name="info-circle"></i-bs>
|
||||
|
||||
@@ -7,8 +7,8 @@
|
||||
bottom: 0;
|
||||
left: 0;
|
||||
z-index: 995; /* Behind the navbar */
|
||||
padding: 50px 0 0; /* Height of navbar */
|
||||
box-shadow: inset -1px 0 0 rgba(0, 0, 0, .1);
|
||||
padding: 64px 0 0; /* Height of navbar */
|
||||
border-right: 1px solid color-mix(in srgb, var(--bs-border-color) 65%, transparent);
|
||||
overflow-y: auto;
|
||||
--pngx-sidebar-width: 100%;
|
||||
max-width: var(--pngx-sidebar-width);
|
||||
@@ -25,7 +25,7 @@
|
||||
}
|
||||
|
||||
.view-name {
|
||||
max-width: calc(100% - 50px)
|
||||
max-width: calc(100% - 55px)
|
||||
}
|
||||
|
||||
.nav-group:not(:has(.app-link)) .sidebar-heading {
|
||||
@@ -47,7 +47,7 @@
|
||||
}
|
||||
@media (max-width: 767.98px) {
|
||||
.sidebar {
|
||||
top: 3.5rem;
|
||||
top: 4rem;
|
||||
}
|
||||
|
||||
.search-container {
|
||||
@@ -71,12 +71,14 @@
|
||||
|
||||
main {
|
||||
transition: all .2s ease;
|
||||
padding-top: 110px;
|
||||
padding-top: 118px;
|
||||
background: var(--pngx-bg-darker);
|
||||
min-height: 100vh;
|
||||
}
|
||||
|
||||
@media (min-width: 768px) {
|
||||
main {
|
||||
padding-top: 56px;
|
||||
padding-top: 64px;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -91,7 +93,7 @@ main {
|
||||
|
||||
@media(min-width: 768px) {
|
||||
.sidebar.slim {
|
||||
max-width: 50px;
|
||||
max-width: 55px;
|
||||
|
||||
li.nav-item span.badge {
|
||||
display: inline-block;
|
||||
@@ -150,7 +152,7 @@ main {
|
||||
display: block;
|
||||
position: fixed;
|
||||
left: calc(var(--pngx-sidebar-width) - 12px);
|
||||
top: 60px;
|
||||
top: 72px;
|
||||
z-index: 996;
|
||||
--bs-btn-padding-x: 0.35rem;
|
||||
--bs-btn-padding-y: 0.125rem;
|
||||
@@ -158,7 +160,7 @@ main {
|
||||
}
|
||||
|
||||
.sidebar.slim .sidebar-slim-toggler {
|
||||
--pngx-sidebar-width: 50px !important;
|
||||
--pngx-sidebar-width: 56px !important;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -171,7 +173,7 @@ main {
|
||||
position: relative;
|
||||
top: 0;
|
||||
height: 100%;
|
||||
padding-top: 0.5rem;
|
||||
padding: .75rem .5rem 1rem;
|
||||
overflow-x: hidden;
|
||||
overflow-y: auto; /* Scrollable contents if viewport is shorter than content. */
|
||||
min-height: min-content;
|
||||
@@ -186,11 +188,14 @@ main {
|
||||
.sidebar .nav-link {
|
||||
font-weight: 500;
|
||||
white-space: nowrap;
|
||||
border-left: 2px solid transparent;
|
||||
transition: color .15s ease-in-out;
|
||||
border-radius: .55rem;
|
||||
margin: 1px 0;
|
||||
padding: .55rem .7rem;
|
||||
transition: color .15s ease-in-out, background-color .15s ease-in-out;
|
||||
|
||||
&:hover, &.active, &:focus {
|
||||
&:hover, &:focus {
|
||||
color: var(--bs-primary);
|
||||
background-color: color-mix(in srgb, var(--bs-primary) 8%, transparent);
|
||||
}
|
||||
|
||||
&:focus-visible {
|
||||
@@ -199,8 +204,21 @@ main {
|
||||
}
|
||||
|
||||
&.active {
|
||||
font-weight: bold;
|
||||
border-left-color: var(--bs-primary);
|
||||
font-weight: 600;
|
||||
color: var(--bs-primary);
|
||||
background-color: color-mix(in srgb, var(--bs-primary) 13%, transparent);
|
||||
}
|
||||
|
||||
&.nav-link-action {
|
||||
color: var(--bs-secondary-color);
|
||||
font-weight: 400;
|
||||
background-color: transparent;
|
||||
|
||||
&:hover,
|
||||
&:focus {
|
||||
color: var(--bs-primary);
|
||||
background-color: transparent;
|
||||
}
|
||||
}
|
||||
|
||||
i-bs {
|
||||
@@ -209,20 +227,45 @@ main {
|
||||
}
|
||||
}
|
||||
|
||||
// sub-page gets marker only
|
||||
.nav-item:has(.attributes-submenu.show .nav-link.active) > .attributes-row > .nav-link.active {
|
||||
border-left-color: transparent;
|
||||
}
|
||||
.attributes-row {
|
||||
border-radius: .55rem;
|
||||
margin: .1rem 0;
|
||||
transition: color .15s ease-in-out, background-color .15s ease-in-out;
|
||||
|
||||
// bring sub-menu markers back out to L edge
|
||||
.attributes-submenu .nav-link {
|
||||
margin-left: -1rem;
|
||||
padding-left: calc(var(--bs-nav-link-padding-x) + 1rem);
|
||||
> .nav-link {
|
||||
margin: 0;
|
||||
|
||||
&:hover,
|
||||
&:focus,
|
||||
&.active {
|
||||
background-color: transparent;
|
||||
}
|
||||
}
|
||||
|
||||
&:hover,
|
||||
&:has(> .nav-link:focus-visible) {
|
||||
background-color: color-mix(in srgb, var(--bs-primary) 8%, transparent);
|
||||
}
|
||||
|
||||
&:has(> .nav-link.active) {
|
||||
background-color: color-mix(in srgb, var(--bs-primary) 13%, transparent);
|
||||
}
|
||||
}
|
||||
|
||||
.attributes-row .attributes-expand-btn {
|
||||
opacity: 0.2;
|
||||
width: 1.75rem;
|
||||
height: 1.75rem;
|
||||
margin-right: .35rem !important;
|
||||
border-radius: 50%;
|
||||
box-shadow: none !important;
|
||||
transition: opacity 0.15s ease-in-out;
|
||||
|
||||
&:focus-visible {
|
||||
outline: 2px solid color-mix(in srgb, var(--bs-primary) 55%, transparent);
|
||||
outline-offset: 1px;
|
||||
opacity: 1;
|
||||
}
|
||||
}
|
||||
|
||||
.attributes-row:hover .attributes-expand-btn {
|
||||
@@ -230,8 +273,10 @@ main {
|
||||
}
|
||||
|
||||
.sidebar-heading {
|
||||
font-size: 0.75rem;
|
||||
font-size: 0.68rem;
|
||||
text-transform: uppercase;
|
||||
font-weight: 700;
|
||||
opacity: .75;
|
||||
}
|
||||
|
||||
.nav {
|
||||
@@ -288,16 +333,118 @@ main {
|
||||
*/
|
||||
|
||||
.navbar-brand {
|
||||
font-size: 1rem;
|
||||
--pngx-navbar-brand-shadow-rgb: 0, 0, 0;
|
||||
font-size: 1.0625rem;
|
||||
min-height: 64px;
|
||||
letter-spacing: -0.015em;
|
||||
|
||||
.flex-column {
|
||||
padding: 0.15rem 0;
|
||||
&:hover,
|
||||
&:focus-visible {
|
||||
::ng-deep .navbar-official-logo,
|
||||
.brand-mark,
|
||||
.brand-logo {
|
||||
filter: drop-shadow(0 2px 3px rgba(var(--pngx-navbar-brand-shadow-rgb), .5));
|
||||
}
|
||||
}
|
||||
|
||||
::ng-deep .navbar-official-logo {
|
||||
filter: drop-shadow(0 1px 2px rgba(var(--pngx-navbar-brand-shadow-rgb), .3));
|
||||
transition: filter .15s ease-in-out;
|
||||
|
||||
@media screen and (max-width: 575.98px) {
|
||||
max-height: 2rem;
|
||||
}
|
||||
}
|
||||
|
||||
.brand-mark {
|
||||
width: 1.65rem;
|
||||
height: 1.65rem;
|
||||
flex: 0 0 auto;
|
||||
transition: filter .15s ease-in-out;
|
||||
}
|
||||
|
||||
.brand-copy {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
align-items: flex-start;
|
||||
min-width: 0;
|
||||
line-height: 1.1;
|
||||
transition: transform .15s ease-in-out;
|
||||
}
|
||||
|
||||
.brand-title {
|
||||
font-weight: 600;
|
||||
max-width: 100%;
|
||||
min-width: 0;
|
||||
}
|
||||
|
||||
.byline {
|
||||
font-size: 0.5rem;
|
||||
letter-spacing: 0.1rem;
|
||||
margin-top: .15rem;
|
||||
font-size: .5rem;
|
||||
font-weight: 500;
|
||||
letter-spacing: .1rem;
|
||||
opacity: .8;
|
||||
}
|
||||
|
||||
.brand-logo {
|
||||
width: auto;
|
||||
height: 2.75rem;
|
||||
max-width: 5rem;
|
||||
flex: 0 0 auto;
|
||||
object-fit: contain;
|
||||
transition: filter .15s ease-in-out, transform .15s ease-in-out;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
:host-context(.primary-light) .navbar-brand {
|
||||
--pngx-navbar-brand-shadow-rgb: 255, 255, 255; // Light app color, use white shadow for dark text
|
||||
}
|
||||
|
||||
:host ::ng-deep .navbar-official-logo {
|
||||
.leaf {
|
||||
fill: color-mix(in srgb, var(--pngx-primary-text-contrast) 70%, var(--bs-primary)) !important;
|
||||
}
|
||||
|
||||
.text {
|
||||
fill: var(--pngx-primary-text-contrast) !important;
|
||||
}
|
||||
}
|
||||
|
||||
.navbar {
|
||||
min-height: 64px;
|
||||
box-shadow: 0 1px 0 rgba(0, 0, 0, .12), 0 4px 18px rgba(0, 0, 0, .08) !important;
|
||||
}
|
||||
|
||||
.navbar > ul {
|
||||
align-items: center;
|
||||
gap: .125rem;
|
||||
padding-right: .5rem;
|
||||
}
|
||||
|
||||
:host ::ng-deep .navbar-action {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
min-width: 2.5rem;
|
||||
min-height: 2.5rem;
|
||||
padding: .45rem .55rem;
|
||||
border-radius: .6rem;
|
||||
transition: background-color .15s ease-in-out, opacity .15s ease-in-out;
|
||||
|
||||
&:hover,
|
||||
&:focus-visible {
|
||||
background-color: rgba(0, 0, 0, .14);
|
||||
}
|
||||
}
|
||||
|
||||
#userDropdown {
|
||||
padding-left: .75rem;
|
||||
}
|
||||
|
||||
.header-search {
|
||||
width: 100%;
|
||||
max-width: 44rem;
|
||||
}
|
||||
|
||||
@media screen and (max-width: 575.98px) {
|
||||
@@ -346,18 +493,44 @@ main {
|
||||
grid-area: actions;
|
||||
justify-self: end;
|
||||
flex-wrap: nowrap;
|
||||
gap: 0;
|
||||
padding-right: .25rem;
|
||||
}
|
||||
|
||||
:host ::ng-deep .navbar-action {
|
||||
min-width: 2.25rem;
|
||||
padding-right: .4rem;
|
||||
padding-left: .4rem;
|
||||
}
|
||||
|
||||
#userDropdown {
|
||||
padding-right: .35rem;
|
||||
padding-left: .45rem;
|
||||
}
|
||||
}
|
||||
|
||||
@media screen and (min-width: 768px) {
|
||||
.navbar-brand.slim {
|
||||
max-width: 50px;
|
||||
max-width: 55px;
|
||||
|
||||
.brand-logo {
|
||||
width: 1.65rem;
|
||||
max-width: 1.65rem;
|
||||
}
|
||||
|
||||
.brand-mark-slim {
|
||||
display: block !important;
|
||||
}
|
||||
}
|
||||
|
||||
:host ::ng-deep .navbar-brand.slim .navbar-official-logo {
|
||||
display: none;
|
||||
}
|
||||
}
|
||||
|
||||
:host ::ng-deep .dropdown.show .dropdown-toggle,
|
||||
:host ::ng-deep .dropdown-toggle:hover {
|
||||
opacity: 0.7;
|
||||
opacity: 1;
|
||||
}
|
||||
|
||||
.dropdown-toggle::after {
|
||||
|
||||
@@ -45,6 +45,7 @@ import { TasksService } from 'src/app/services/tasks.service'
|
||||
import { ToastService } from 'src/app/services/toast.service'
|
||||
import { environment } from 'src/environments/environment'
|
||||
import { ChatComponent } from '../chat/chat/chat.component'
|
||||
import { LogoComponent } from '../common/logo/logo.component'
|
||||
import { ProfileEditDialogComponent } from '../common/profile-edit-dialog/profile-edit-dialog.component'
|
||||
import { DocumentDetailComponent } from '../document-detail/document-detail.component'
|
||||
import { ComponentWithPermissions } from '../with-permissions/with-permissions.component'
|
||||
@@ -59,6 +60,7 @@ const SCROLL_THRESHOLD = 16
|
||||
styleUrls: ['./app-frame.component.scss'],
|
||||
imports: [
|
||||
GlobalSearchComponent,
|
||||
LogoComponent,
|
||||
DocumentTitlePipe,
|
||||
IfPermissionsDirective,
|
||||
ToastsDropdownComponent,
|
||||
@@ -190,11 +192,34 @@ export class AppFrameComponent
|
||||
return `${environment.appTitle} v${this.settingsService.get(SETTINGS_KEYS.VERSION)}${environment.tag === 'prod' ? '' : ` #${environment.tag}`}`
|
||||
}
|
||||
|
||||
get appTitle(): string {
|
||||
this.settingsService.trackChanges()
|
||||
return (
|
||||
this.settingsService.get(SETTINGS_KEYS.APP_TITLE) || environment.appTitle
|
||||
)
|
||||
}
|
||||
|
||||
get customAppTitle(): string {
|
||||
this.settingsService.trackChanges()
|
||||
return this.settingsService.get(SETTINGS_KEYS.APP_TITLE)
|
||||
}
|
||||
|
||||
get hasCustomBranding(): boolean {
|
||||
this.settingsService.trackChanges()
|
||||
return !!(
|
||||
this.settingsService.get(SETTINGS_KEYS.APP_TITLE)?.length ||
|
||||
this.settingsService.get(SETTINGS_KEYS.APP_LOGO)?.length
|
||||
)
|
||||
}
|
||||
|
||||
get customAppLogo(): string {
|
||||
this.settingsService.trackChanges()
|
||||
const logo = this.settingsService.get(SETTINGS_KEYS.APP_LOGO)
|
||||
return logo?.length
|
||||
? environment.apiBaseUrl.replace(/\/api\/$/, logo)
|
||||
: null
|
||||
}
|
||||
|
||||
get canSaveSettings(): boolean {
|
||||
return (
|
||||
this.permissionsService.currentUserCan(
|
||||
|
||||
@@ -4,12 +4,12 @@ form {
|
||||
> i-bs[name="search"] {
|
||||
position: absolute;
|
||||
left: 0.6rem;
|
||||
top: .35rem;
|
||||
top: .25rem;
|
||||
color: rgba(255, 255, 255, 0.6);
|
||||
|
||||
@media screen and (min-width: 768px) {
|
||||
// adjust for smaller font size on non-mobile
|
||||
top: 0.25rem;
|
||||
top: .15rem;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -37,8 +37,9 @@ form {
|
||||
}
|
||||
|
||||
.form-control {
|
||||
color: rgba(255, 255, 255, 0.3);
|
||||
background-color: rgba(0, 0, 0, 0.15);
|
||||
min-height: 2.25rem;
|
||||
color: rgba(255, 255, 255, 0.55);
|
||||
background-color: rgba(0, 0, 0, 0.16);
|
||||
padding-left: 1.8rem;
|
||||
border-color: rgba(255, 255, 255, 0.2);
|
||||
transition: all .3s ease, padding-left 0s ease, background-color 0s ease; // Safari requires all
|
||||
@@ -52,7 +53,7 @@ form {
|
||||
}
|
||||
|
||||
&:focus-within {
|
||||
background-color: rgba(0, 0, 0, 0.3);
|
||||
background-color: rgba(0, 0, 0, 0.26);
|
||||
color: var(--bs-light);
|
||||
flex-grow: 1;
|
||||
padding-left: 0.5rem;
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
|
||||
<li ngbDropdown class="nav-item mx-1" (openChange)="onOpenChange($event)">
|
||||
<li ngbDropdown class="nav-item position-relative" (openChange)="onOpenChange($event)">
|
||||
@if (toasts().length) {
|
||||
<span class="badge rounded-pill z-3 pe-none bg-secondary me-2 position-absolute top-0 left-0">{{ toasts().length }}</span>
|
||||
<span class="notification-count badge rounded-pill z-3 pe-none bg-secondary position-absolute">{{ toasts().length }}</span>
|
||||
}
|
||||
<button class="btn border-0" id="notificationsDropdown" ngbDropdownToggle>
|
||||
<button class="btn navbar-action border-0" id="notificationsDropdown" ngbDropdownToggle aria-label="Notifications" i18n-aria-label>
|
||||
<i-bs width="1.3em" height="1.3em" name="bell"></i-bs>
|
||||
</button>
|
||||
<div ngbDropdownMenu class="dropdown-menu-end shadow p-3" aria-labelledby="notificationsDropdown">
|
||||
|
||||
@@ -11,6 +11,16 @@
|
||||
display: none;
|
||||
}
|
||||
|
||||
.notification-count {
|
||||
top: -.2rem;
|
||||
right: -.2rem;
|
||||
min-width: 1.15rem;
|
||||
height: 1.15rem;
|
||||
padding: .18rem .32rem;
|
||||
font-size: .68rem;
|
||||
line-height: 1;
|
||||
}
|
||||
|
||||
.dropdown-item {
|
||||
white-space: initial;
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
|
||||
<li ngbDropdown class="nav-item me-n2" (openChange)="onOpenChange($event)">
|
||||
<button class="btn border-0" id="chatDropdown" ngbDropdownToggle>
|
||||
<li ngbDropdown class="nav-item" (openChange)="onOpenChange($event)">
|
||||
<button class="btn navbar-action border-0" id="chatDropdown" ngbDropdownToggle aria-label="Chat" i18n-aria-label>
|
||||
<i-bs width="1.3em" height="1.3em" name="chatSquareDots"></i-bs>
|
||||
</button>
|
||||
<div ngbDropdownMenu class="dropdown-menu-end shadow p-3" aria-labelledby="chatDropdown">
|
||||
|
||||
+2
-2
@@ -6,7 +6,7 @@
|
||||
<div class="modal-body">
|
||||
<div class="row">
|
||||
<div class="col-2 d-flex justify-content-end">
|
||||
<button class="btn btn-secondary mt-auto" (click)="rotate(false)">
|
||||
<button class="btn btn-secondary mt-auto" (click)="rotate(false)" aria-label="Rotate counterclockwise" i18n-aria-label>
|
||||
<i-bs name="arrow-counterclockwise"></i-bs>
|
||||
</button>
|
||||
</div>
|
||||
@@ -16,7 +16,7 @@
|
||||
}
|
||||
</div>
|
||||
<div class="col-2 d-flex">
|
||||
<button class="btn btn-secondary mt-auto" (click)="rotate()">
|
||||
<button class="btn btn-secondary mt-auto" (click)="rotate()" aria-label="Rotate clockwise" i18n-aria-label>
|
||||
<i-bs name="arrow-clockwise"></i-bs>
|
||||
</button>
|
||||
</div>
|
||||
|
||||
+1
-1
@@ -1,5 +1,5 @@
|
||||
<div ngbDropdown #fieldDropdown="ngbDropdown" (openChange)="onOpenClose($event)" [popperOptions]="popperOptions">
|
||||
<button type="button" class="btn btn-sm btn-outline-primary" id="customFieldsDropdown" [disabled]="disabled" ngbDropdownToggle>
|
||||
<button type="button" class="btn btn-sm btn-outline-primary" id="customFieldsDropdown" [disabled]="disabled" ngbDropdownToggle aria-label="Custom Fields" i18n-aria-label>
|
||||
<i-bs name="ui-radios"></i-bs><div class="d-none d-lg-inline ms-1"><ng-container i18n>Custom Fields</ng-container></div>
|
||||
</button>
|
||||
<div ngbDropdownMenu aria-labelledby="customFieldsDropdown" class="shadow custom-fields-dropdown">
|
||||
|
||||
+4
-4
@@ -1,6 +1,6 @@
|
||||
@if (useDropdown) {
|
||||
<div class="btn-group w-100" role="group" ngbDropdown #dropdown="ngbDropdown" (openChange)="onOpenChange($event)" [popperOptions]="popperOptions">
|
||||
<button class="btn btn-sm btn-outline-primary" id="dropdown_toggle" ngbDropdownToggle [disabled]="disabled">
|
||||
<button class="btn btn-sm btn-outline-primary" id="dropdown_toggle" ngbDropdownToggle [disabled]="disabled" [aria-label]="title">
|
||||
<i-bs name="{{icon}}"></i-bs><div class="d-none d-sm-inline ms-1">{{title}}</div>
|
||||
@if (isActive) {
|
||||
<pngx-clearable-badge [selected]="isActive" (cleared)="reset()"></pngx-clearable-badge>
|
||||
@@ -38,7 +38,7 @@
|
||||
ngbDatepicker
|
||||
#d="ngbDatepicker"
|
||||
[footerTemplate]="datePickerFooterTemplate" />
|
||||
<button class="btn btn-sm btn-outline-secondary rounded-end" (click)="d.toggle()" type="button">
|
||||
<button class="btn btn-sm btn-outline-secondary rounded-end" (click)="d.toggle()" type="button" aria-label="Open date picker" i18n-aria-label>
|
||||
<i-bs name="calendar-event"></i-bs>
|
||||
</button>
|
||||
<ng-template #datePickerFooterTemplate>
|
||||
@@ -143,7 +143,7 @@
|
||||
<input class="w-25 form-control rounded-end" type="text" [(ngModel)]="atom.value" [disabled]="disabled">
|
||||
}
|
||||
}
|
||||
<button class="btn btn-link btn-sm text-danger pe-0" type="button" (click)="removeElement(atom)" [disabled]="disabled">
|
||||
<button class="btn btn-link btn-sm text-danger pe-0" type="button" (click)="removeElement(atom)" [disabled]="disabled" aria-label="Remove query" i18n-aria-label>
|
||||
<i-bs name="x-circle"></i-bs>
|
||||
</button>
|
||||
</div>
|
||||
@@ -185,7 +185,7 @@
|
||||
<i-bs name="braces"></i-bs>
|
||||
</button>
|
||||
@if (expression.depth > 0) {
|
||||
<button type="button" class="btn btn-sm btn-outline-secondary text-danger" (click)="removeElement(expression)" [disabled]="disabled">
|
||||
<button type="button" class="btn btn-sm btn-outline-secondary text-danger" (click)="removeElement(expression)" [disabled]="disabled" aria-label="Remove expression" i18n-aria-label>
|
||||
<i-bs name="x-circle"></i-bs>
|
||||
</button>
|
||||
}
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
<div class="btn-group w-100" ngbDropdown role="group" [popperOptions]="popperOptions" [placement]="placement">
|
||||
<button class="btn btn-sm" id="dropdown{{title}}" ngbDropdownToggle [ngClass]="createdDateTo || createdDateFrom ? 'btn-primary' : 'btn-outline-primary'" [disabled]="disabled">
|
||||
<button class="btn btn-sm" id="dropdown{{title}}" ngbDropdownToggle [ngClass]="createdDateTo || createdDateFrom ? 'btn-primary' : 'btn-outline-primary'" [disabled]="disabled" [aria-label]="title">
|
||||
<i-bs width="1em" height="1em" name="calendar-event-fill"></i-bs><div class="d-none d-sm-inline ms-1">{{title}}</div>
|
||||
<pngx-clearable-badge [selected]="isActive" (cleared)="reset()"></pngx-clearable-badge><span class="visually-hidden">selected</span>
|
||||
</button>
|
||||
@@ -9,7 +9,7 @@
|
||||
<div class="list-group-item d-flex p-2 select-item" role="menuitem">
|
||||
<div class="selected-icon">
|
||||
@if (createdRelativeDate) {
|
||||
<a class="text-light focus-variants" href="javascript:void(0)" (click)="clearCreatedRelativeDate()">
|
||||
<a class="text-light focus-variants" href="javascript:void(0)" (click)="clearCreatedRelativeDate()" aria-label="Clear created relative date" i18n-aria-label>
|
||||
<i-bs width="1em" height="1em" name="check" class="variant-unfocused text-dark"></i-bs>
|
||||
<i-bs width="1em" height="1em" name="x" class="variant-focused text-primary"></i-bs>
|
||||
</a>
|
||||
@@ -33,7 +33,7 @@
|
||||
<div class="list-group-item d-flex p-2" role="menuitem">
|
||||
<div class="selected-icon">
|
||||
@if (createdDateFrom) {
|
||||
<a class="text-light focus-variants" href="javascript:void(0)" (click)="clearCreatedFrom()">
|
||||
<a class="text-light focus-variants" href="javascript:void(0)" (click)="clearCreatedFrom()" aria-label="Clear created from date" i18n-aria-label>
|
||||
<i-bs width="1em" height="1em" name="check" class="variant-unfocused"></i-bs>
|
||||
<i-bs width="1em" height="1em" name="x" class="variant-focused text-primary"></i-bs>
|
||||
</a>
|
||||
@@ -43,7 +43,7 @@
|
||||
<span class="input-group-text w-25 small text-muted" i18n>From</span>
|
||||
<input class="form-control small" [placeholder]="datePlaceHolder" (dateSelect)="onChangeDebounce()" (change)="onChangeDebounce()" (keypress)="onKeyPress($event)"
|
||||
maxlength="10" [(ngModel)]="createdDateFrom" ngbDatepicker #createdDateFromPicker="ngbDatepicker" [footerTemplate]="createdFromFooterTemplate">
|
||||
<button class="btn btn-outline-secondary" (click)="createdDateFromPicker.toggle()" type="button">
|
||||
<button class="btn btn-outline-secondary" (click)="createdDateFromPicker.toggle()" type="button" aria-label="Open created from date picker" i18n-aria-label>
|
||||
<i-bs width="1em" height="1em" name="calendar"></i-bs>
|
||||
</button>
|
||||
<ng-template #createdFromFooterTemplate>
|
||||
@@ -57,7 +57,7 @@
|
||||
<div class="list-group-item d-flex p-2" role="menuitem">
|
||||
<div class="selected-icon">
|
||||
@if (createdDateTo) {
|
||||
<a class="text-light focus-variants" href="javascript:void(0)" (click)="clearCreatedTo()">
|
||||
<a class="text-light focus-variants" href="javascript:void(0)" (click)="clearCreatedTo()" aria-label="Clear created to date" i18n-aria-label>
|
||||
<i-bs width="1em" height="1em" name="check" class="variant-unfocused"></i-bs>
|
||||
<i-bs width="1em" height="1em" name="x" class="variant-focused text-primary"></i-bs>
|
||||
</a>
|
||||
@@ -67,7 +67,7 @@
|
||||
<span class="input-group-text w-25 small text-muted" i18n>To</span>
|
||||
<input class="form-control small" [placeholder]="datePlaceHolder" (dateSelect)="onChangeDebounce()" (change)="onChangeDebounce()" (keypress)="onKeyPress($event)"
|
||||
maxlength="10" [(ngModel)]="createdDateTo" ngbDatepicker #createdDateToPicker="ngbDatepicker" [footerTemplate]="createdToFooterTemplate">
|
||||
<button class="btn btn-outline-secondary" (click)="createdDateToPicker.toggle()" type="button">
|
||||
<button class="btn btn-outline-secondary" (click)="createdDateToPicker.toggle()" type="button" aria-label="Open created to date picker" i18n-aria-label>
|
||||
<i-bs width="1em" height="1em" name="calendar"></i-bs>
|
||||
</button>
|
||||
<ng-template #createdToFooterTemplate>
|
||||
@@ -85,7 +85,7 @@
|
||||
<div class="list-group-item d-flex p-2 select-item" role="menuitem">
|
||||
<div class="selected-icon">
|
||||
@if (addedRelativeDate) {
|
||||
<a class="text-light focus-variants" href="javascript:void(0)" (click)="clearAddedRelativeDate()">
|
||||
<a class="text-light focus-variants" href="javascript:void(0)" (click)="clearAddedRelativeDate()" aria-label="Clear added relative date" i18n-aria-label>
|
||||
<i-bs width="1em" height="1em" name="check" class="variant-unfocused text-dark"></i-bs>
|
||||
<i-bs width="1em" height="1em" name="x" class="variant-focused text-primary"></i-bs>
|
||||
</a>
|
||||
@@ -109,7 +109,7 @@
|
||||
<div class="list-group-item d-flex p-2" role="menuitem">
|
||||
<div class="selected-icon">
|
||||
@if (addedDateFrom) {
|
||||
<a class="text-light focus-variants" href="javascript:void(0)" (click)="clearAddedFrom()">
|
||||
<a class="text-light focus-variants" href="javascript:void(0)" (click)="clearAddedFrom()" aria-label="Clear added from date" i18n-aria-label>
|
||||
<i-bs width="1em" height="1em" name="check" class="variant-unfocused"></i-bs>
|
||||
<i-bs width="1em" height="1em" name="x" class="variant-focused text-primary"></i-bs>
|
||||
</a>
|
||||
@@ -119,7 +119,7 @@
|
||||
<span class="input-group-text w-25 small text-muted" i18n>From</span>
|
||||
<input class="form-control small" [placeholder]="datePlaceHolder" (dateSelect)="onChangeDebounce()" (change)="onChangeDebounce()" (keypress)="onKeyPress($event)"
|
||||
maxlength="10" [(ngModel)]="addedDateFrom" ngbDatepicker #addedDateFromPicker="ngbDatepicker" [footerTemplate]="addedFromFooterTemplate">
|
||||
<button class="btn btn-outline-secondary" (click)="addedDateFromPicker.toggle()" type="button">
|
||||
<button class="btn btn-outline-secondary" (click)="addedDateFromPicker.toggle()" type="button" aria-label="Open added from date picker" i18n-aria-label>
|
||||
<i-bs width="1em" height="1em" name="calendar"></i-bs>
|
||||
</button>
|
||||
<ng-template #addedFromFooterTemplate>
|
||||
@@ -133,7 +133,7 @@
|
||||
<div class="list-group-item d-flex p-2" role="menuitem">
|
||||
<div class="selected-icon">
|
||||
@if (addedDateTo) {
|
||||
<a class="text-light focus-variants" href="javascript:void(0)" (click)="clearAddedTo()">
|
||||
<a class="text-light focus-variants" href="javascript:void(0)" (click)="clearAddedTo()" aria-label="Clear added to date" i18n-aria-label>
|
||||
<i-bs width="1em" height="1em" name="check" class="variant-unfocused"></i-bs>
|
||||
<i-bs width="1em" height="1em" name="x" class="variant-focused text-primary"></i-bs>
|
||||
</a>
|
||||
@@ -143,7 +143,7 @@
|
||||
<span class="input-group-text w-25 small text-muted" i18n>To</span>
|
||||
<input class="form-control small" [placeholder]="datePlaceHolder" (dateSelect)="onChangeDebounce()" (change)="onChangeDebounce()" (keypress)="onKeyPress($event)"
|
||||
maxlength="10" [(ngModel)]="addedDateTo" ngbDatepicker #addedDateToPicker="ngbDatepicker" [footerTemplate]="addedToFooterTemplate">
|
||||
<button class="btn btn-outline-secondary" (click)="addedDateToPicker.toggle()" type="button">
|
||||
<button class="btn btn-outline-secondary" (click)="addedDateToPicker.toggle()" type="button" aria-label="Open added to date picker" i18n-aria-label>
|
||||
<i-bs width="1em" height="1em" name="calendar"></i-bs>
|
||||
</button>
|
||||
<ng-template #addedToFooterTemplate>
|
||||
|
||||
+1
-1
@@ -1,5 +1,5 @@
|
||||
<div class="btn-group w-100" ngbDropdown role="group" (openChange)="dropdownOpenChange($event)" #dropdown="ngbDropdown" (keydown)="listKeyDown($event)" [popperOptions]="popperOptions" [autoClose]="!creating()">
|
||||
<button class="btn btn-sm" id="dropdown_{{name}}" ngbDropdownToggle [ngClass]="!editing && selectionModel.selectionSize() > 0 ? 'btn-primary' : 'btn-outline-primary'" [disabled]="disabled">
|
||||
<button class="btn btn-sm" id="dropdown_{{name}}" ngbDropdownToggle [ngClass]="!editing && selectionModel.selectionSize() > 0 ? 'btn-primary' : 'btn-outline-primary'" [disabled]="disabled" [aria-label]="title">
|
||||
<i-bs name="{{icon}}"></i-bs><div class="d-none d-sm-inline ms-1">{{title}}</div>
|
||||
@if (!editing && selectionModel.totalCount > 0) {
|
||||
<pngx-clearable-badge [number]="selectionModel.totalCount" [selected]="selectionModel.selectionSize() > 0" (cleared)="reset()"></pngx-clearable-badge>
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
}
|
||||
|
||||
<div class="input-group" [class.is-invalid]="error">
|
||||
<button type="button" class="input-group-text" [style.background-color]="value" (click)="colorPicker.toggle()"> </button>
|
||||
<button type="button" class="input-group-text" [style.background-color]="value" (click)="colorPicker.toggle()" aria-label="Open color picker" i18n-aria-label> </button>
|
||||
|
||||
<ng-template #popContent>
|
||||
<div style="min-width: 200px;" class="pb-3">
|
||||
@@ -14,7 +14,7 @@
|
||||
|
||||
<input #inputField class="form-control" [class.is-invalid]="error" [id]="inputId" [(ngModel)]="value" (change)="onChange(value)" [autoClose]="'outside'" [ngbPopover]="popContent" #colorPicker="ngbPopover" placement="bottom" popoverClass="shadow">
|
||||
|
||||
<button class="btn btn-outline-secondary" type="button" (click)="randomize()">
|
||||
<button class="btn btn-outline-secondary" type="button" (click)="randomize()" aria-label="Choose a random color" i18n-aria-label>
|
||||
<i-bs name="dice5"></i-bs>
|
||||
</button>
|
||||
|
||||
|
||||
+1
-1
@@ -74,7 +74,7 @@
|
||||
class="flex-grow-1"></pngx-input-textarea>
|
||||
}
|
||||
}
|
||||
<button type="button" class="btn btn-link text-danger" (click)="removeSelectedField.next(fieldId)">
|
||||
<button type="button" class="btn btn-link text-danger" (click)="removeSelectedField.next(fieldId)" aria-label="Remove custom field" i18n-aria-label>
|
||||
<i-bs name="trash"></i-bs>
|
||||
</button>
|
||||
</div>
|
||||
|
||||
@@ -13,7 +13,7 @@
|
||||
<input #inputField class="form-control" [class.is-invalid]="error" [placeholder]="placeholder" [id]="inputId" maxlength="10"
|
||||
(dateSelect)="onChange(value)" (change)="onChange(value)" (keypress)="onKeyPress($event)" (paste)="onPaste($event)"
|
||||
name="dp" [(ngModel)]="value" ngbDatepicker #datePicker="ngbDatepicker" #datePickerContent="ngModel" [disabled]="disabled" [footerTemplate]="datePickerFooterTemplate">
|
||||
<button class="btn btn-outline-secondary calendar" (click)="datePicker.toggle()" type="button" [disabled]="disabled">
|
||||
<button class="btn btn-outline-secondary calendar" (click)="datePicker.toggle()" type="button" [disabled]="disabled" aria-label="Open date picker" i18n-aria-label>
|
||||
<i-bs width="1.2em" height="1.2em" name="calendar"></i-bs>
|
||||
</button>
|
||||
<ng-template #datePickerFooterTemplate>
|
||||
|
||||
@@ -13,7 +13,7 @@
|
||||
<div class="input-group mb-3">
|
||||
<input type="text" class="form-control" [(ngModel)]="entry[0]" (change)="inputChange()" [disabled]="disabled" autocomplete="off">
|
||||
<input type="text" class="form-control" [(ngModel)]="entry[1]" (change)="inputChange()" [disabled]="disabled" autocomplete="off">
|
||||
<button type="button" class="btn btn-outline-secondary" (click)="removeEntry(i)">
|
||||
<button type="button" class="btn btn-outline-secondary" (click)="removeEntry(i)" aria-label="Remove entry" i18n-aria-label>
|
||||
<i-bs class="text-danger" name="trash"></i-bs>
|
||||
</button>
|
||||
</div>
|
||||
|
||||
@@ -50,7 +50,7 @@
|
||||
</ng-template>
|
||||
</ng-select>
|
||||
@if (allowCreateNew && !hideAddButton) {
|
||||
<button class="btn btn-outline-secondary" type="button" (click)="addItem()" [disabled]="disabled">
|
||||
<button class="btn btn-outline-secondary" type="button" (click)="addItem()" [disabled]="disabled" aria-label="Create new item" i18n-aria-label>
|
||||
<i-bs width="1.2em" height="1.2em" name="plus"></i-bs>
|
||||
</button>
|
||||
}
|
||||
|
||||
@@ -48,7 +48,7 @@
|
||||
</ng-template>
|
||||
</ng-select>
|
||||
@if (allowCreate && !hideAddButton) {
|
||||
<button class="btn btn-outline-secondary" type="button" (click)="createTag(null, true)" [disabled]="disabled">
|
||||
<button class="btn btn-outline-secondary" type="button" (click)="createTag(null, true)" [disabled]="disabled" aria-label="Create new tag" i18n-aria-label>
|
||||
<i-bs width="1.2em" height="1.2em" name="plus"></i-bs>
|
||||
</button>
|
||||
}
|
||||
|
||||
@@ -7,6 +7,15 @@ h3 {
|
||||
}
|
||||
}
|
||||
|
||||
:host {
|
||||
display: block;
|
||||
margin-bottom: .35rem;
|
||||
}
|
||||
|
||||
h3 > .h6 {
|
||||
color: var(--bs-secondary-color);
|
||||
}
|
||||
|
||||
@media (min-width: 1200px) {
|
||||
h3 {
|
||||
min-height: 2.8rem;
|
||||
|
||||
+1
-1
@@ -1,5 +1,5 @@
|
||||
<div class="btn-group w-100" ngbDropdown role="group">
|
||||
<button class="btn btn-sm" id="dropdown{{title}}" ngbDropdownToggle [ngClass]="isActive ? 'btn-primary' : 'btn-outline-primary'" [disabled]="disabled">
|
||||
<button class="btn btn-sm" id="dropdown{{title}}" ngbDropdownToggle [ngClass]="isActive ? 'btn-primary' : 'btn-outline-primary'" [disabled]="disabled" [aria-label]="title">
|
||||
<i-bs name="person-fill-lock"></i-bs><div class="d-none d-sm-inline ms-1">{{title}}</div>
|
||||
<pngx-clearable-badge [selected]="isActive" (cleared)="reset()"></pngx-clearable-badge><span class="visually-hidden">selected</span>
|
||||
</button>
|
||||
|
||||
+19
@@ -107,6 +107,25 @@ describe('PermissionsSelectComponent', () => {
|
||||
expect(component.form.get('Tag').get('Change').disabled).toBeTruthy()
|
||||
})
|
||||
|
||||
it('should update checkboxes when inherited permissions change', () => {
|
||||
component.ngOnInit()
|
||||
component.inheritedPermissions = ['documents.change_document']
|
||||
component.writeValue(['delete_document'])
|
||||
expect(component.form.get('Document').get('Change').value).toBeTruthy()
|
||||
expect(component.form.get('Document').get('Change').disabled).toBeTruthy()
|
||||
|
||||
// swap for a group with a different permission, but the same number of them
|
||||
component.inheritedPermissions = ['documents.view_document']
|
||||
|
||||
// the no-longer-inherited permission is unchecked, the explicit one is kept
|
||||
expect(component.permissions).toEqual(['delete_document'])
|
||||
expect(component.form.get('Document').get('Change').value).toBeFalsy()
|
||||
expect(component.form.get('Document').get('Change').disabled).toBeFalsy()
|
||||
expect(component.form.get('Document').get('Delete').value).toBeTruthy()
|
||||
expect(component.form.get('Document').get('View').value).toBeTruthy()
|
||||
expect(component.form.get('Document').get('View').disabled).toBeTruthy()
|
||||
})
|
||||
|
||||
it('should exclude history permissions if disabled', () => {
|
||||
settingsService.set(SETTINGS_KEYS.AUDITLOG_ENABLED, false)
|
||||
fixture = TestBed.createComponent(PermissionsSelectComponent)
|
||||
|
||||
+61
-35
@@ -74,12 +74,22 @@ export class PermissionsSelectComponent
|
||||
? inherited.map((p) => p.replace(/^\w+\./g, ''))
|
||||
: []
|
||||
|
||||
if (this._inheritedPermissions !== newInheritedPermissions) {
|
||||
this._inheritedPermissions = newInheritedPermissions
|
||||
this.writeValue(this.permissions) // updates visual checks etc.
|
||||
}
|
||||
const changed =
|
||||
newInheritedPermissions.length !== this._inheritedPermissions.length ||
|
||||
newInheritedPermissions.some(
|
||||
(p) => !this._inheritedPermissions.includes(p)
|
||||
)
|
||||
|
||||
this.updateDisabledStates()
|
||||
if (changed) {
|
||||
// skip inherited permissions, these are the explicitly set ones
|
||||
this.permissions = this.getSelectedPermissions(
|
||||
this.form.getRawValue()
|
||||
).filter((p) => !this._inheritedPermissions.includes(p))
|
||||
this._inheritedPermissions = newInheritedPermissions
|
||||
this.applyCheckedState()
|
||||
} else {
|
||||
this.updateDisabledStates()
|
||||
}
|
||||
}
|
||||
|
||||
inheritedWarning: string = $localize`Inherited from group`
|
||||
@@ -106,20 +116,29 @@ export class PermissionsSelectComponent
|
||||
}
|
||||
|
||||
this.permissions = permissions ?? []
|
||||
const allPerms = this._inheritedPermissions.concat(this.permissions)
|
||||
this.applyCheckedState()
|
||||
}
|
||||
|
||||
allPerms.forEach((permissionStr) => {
|
||||
const { actionKey, typeKey } =
|
||||
this.permissionsService.getPermissionKeys(permissionStr)
|
||||
// sets every checkbox from inherited + own perms
|
||||
private applyCheckedState(): void {
|
||||
const allPerms = new Set(
|
||||
this._inheritedPermissions.concat(this.permissions)
|
||||
)
|
||||
|
||||
if (actionKey && typeKey) {
|
||||
this.form
|
||||
.get(typeKey)
|
||||
?.get(actionKey)
|
||||
?.patchValue(true, { emitEvent: false })
|
||||
}
|
||||
})
|
||||
this.allowedTypes.forEach((type) => {
|
||||
const typeGroup = this.form.get(type)
|
||||
for (const action of Object.keys(PermissionAction)) {
|
||||
typeGroup.get(action)?.patchValue(
|
||||
allPerms.has(
|
||||
this.permissionsService.getPermissionCode(
|
||||
PermissionAction[action],
|
||||
PermissionType[type]
|
||||
)
|
||||
),
|
||||
{ emitEvent: false } // don't trigger valueChanges now
|
||||
)
|
||||
}
|
||||
|
||||
if (this.typeHasAllActionsSelected(type)) {
|
||||
this.typesWithAllActions.add(type)
|
||||
} else {
|
||||
@@ -150,26 +169,9 @@ export class PermissionsSelectComponent
|
||||
|
||||
ngOnInit(): void {
|
||||
this.form.valueChanges.subscribe((newValue) => {
|
||||
let permissions = []
|
||||
Object.entries(newValue).forEach(([typeKey, typeValue]) => {
|
||||
const selectedActions = Object.entries(typeValue).filter(
|
||||
([actionKey, actionValue]) =>
|
||||
actionValue &&
|
||||
this.isActionSupported(
|
||||
PermissionType[typeKey],
|
||||
PermissionAction[actionKey]
|
||||
)
|
||||
)
|
||||
|
||||
selectedActions.forEach(([actionKey]) => {
|
||||
permissions.push(
|
||||
(PermissionType[typeKey] as string).replace(
|
||||
'%s',
|
||||
PermissionAction[actionKey]
|
||||
)
|
||||
)
|
||||
})
|
||||
const permissions = this.getSelectedPermissions(newValue)
|
||||
|
||||
Object.keys(newValue).forEach((typeKey) => {
|
||||
if (this.typeHasAllActionsSelected(typeKey)) {
|
||||
this.typesWithAllActions.add(typeKey)
|
||||
} else {
|
||||
@@ -269,6 +271,30 @@ export class PermissionsSelectComponent
|
||||
return true
|
||||
}
|
||||
|
||||
private getSelectedPermissions(formValue: object): string[] {
|
||||
const permissions = []
|
||||
Object.entries(formValue).forEach(([typeKey, typeValue]) => {
|
||||
Object.entries(typeValue)
|
||||
.filter(
|
||||
([actionKey, actionValue]) =>
|
||||
actionValue &&
|
||||
this.isActionSupported(
|
||||
PermissionType[typeKey],
|
||||
PermissionAction[actionKey]
|
||||
)
|
||||
)
|
||||
.forEach(([actionKey]) => {
|
||||
permissions.push(
|
||||
this.permissionsService.getPermissionCode(
|
||||
PermissionAction[actionKey],
|
||||
PermissionType[typeKey]
|
||||
)
|
||||
)
|
||||
})
|
||||
})
|
||||
return permissions
|
||||
}
|
||||
|
||||
private typeHasAllActionsSelected(typeKey: string): boolean {
|
||||
return Object.keys(PermissionAction)
|
||||
.filter((action) =>
|
||||
|
||||
+1
-1
@@ -1,5 +1,5 @@
|
||||
<div class="btn-group">
|
||||
<button type="button" class="btn btn-sm btn-outline-primary" (click)="clickSuggest()" [disabled]="disabled() || loading() || (suggestions() && !aiEnabled())">
|
||||
<button type="button" class="btn btn-sm btn-outline-primary" (click)="clickSuggest()" [disabled]="disabled() || loading() || (suggestions() && !aiEnabled())" [aria-label]="noSuggestions ? 'No suggestions' : 'Suggest'" i18n-aria-label>
|
||||
@if (loading()) {
|
||||
<div class="spinner-border spinner-border-sm" role="status"></div>
|
||||
} @else if (noSuggestions) {
|
||||
|
||||
+1
-1
@@ -23,7 +23,7 @@
|
||||
<dd>
|
||||
{{status().pngx_version}}
|
||||
@if (versionMismatch()) {
|
||||
<button class="btn btn-sm d-inline align-items-center btn-dark text-uppercase small" [ngbPopover]="versionPopover" triggers="click mouseenter:mouseleave">
|
||||
<button class="btn btn-sm d-inline align-items-center btn-dark text-uppercase small" [ngbPopover]="versionPopover" triggers="click mouseenter:mouseleave" aria-label="View version mismatch details" i18n-aria-label>
|
||||
<i-bs name="exclamation-triangle-fill" class="text-danger lh-1"></i-bs>
|
||||
</button>
|
||||
}
|
||||
|
||||
@@ -1,9 +1,7 @@
|
||||
<pngx-page-header title="Dashboard" [subTitle]="subtitle" i18n-title tourAnchor="tour.dashboard">
|
||||
<pngx-logo extra_classes="d-none d-md-block mt-n2" height="3rem"></pngx-logo>
|
||||
</pngx-page-header>
|
||||
<pngx-page-header title="Dashboard" [subTitle]="subtitle" i18n-title tourAnchor="tour.dashboard"></pngx-page-header>
|
||||
|
||||
<div class="row">
|
||||
<div class="col-12 col-lg-8 col-xl-9 mb-4">
|
||||
<div class="row dashboard-grid g-4 pb-0">
|
||||
<div class="col-12 col-lg-8 col-xl-9 mb-4 dashboard-main">
|
||||
<div class="row row-cols-1 g-4"
|
||||
cdkDropList
|
||||
[cdkDropListDisabled]="settingsService.globalDropzoneActive()"
|
||||
@@ -58,7 +56,7 @@
|
||||
</ng-container>
|
||||
</div>
|
||||
</div>
|
||||
<div class="col-12 col-lg-4 col-xl-3 col-sidebar">
|
||||
<div class="col-12 col-lg-4 col-xl-3 col-sidebar dashboard-aside">
|
||||
<div class="row row-cols-1 g-4 mb-4 sticky-lg-top z-0">
|
||||
<pngx-upload-file-widget></pngx-upload-file-widget>
|
||||
<pngx-statistics-widget *pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.UISettings }"></pngx-statistics-widget>
|
||||
|
||||
@@ -1,3 +1,13 @@
|
||||
.col-sidebar .row {
|
||||
top: 3.5rem;
|
||||
top: 4.75rem;
|
||||
}
|
||||
|
||||
.dashboard-grid {
|
||||
padding-bottom: 2rem;
|
||||
}
|
||||
|
||||
@media (min-width: 1200px) {
|
||||
.dashboard-main {
|
||||
padding-right: 1rem;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -15,7 +15,6 @@ import { SavedViewService } from 'src/app/services/rest/saved-view.service'
|
||||
import { SettingsService } from 'src/app/services/settings.service'
|
||||
import { ToastService } from 'src/app/services/toast.service'
|
||||
import { environment } from 'src/environments/environment'
|
||||
import { LogoComponent } from '../common/logo/logo.component'
|
||||
import { PageHeaderComponent } from '../common/page-header/page-header.component'
|
||||
import { ComponentWithPermissions } from '../with-permissions/with-permissions.component'
|
||||
import { SavedViewWidgetComponent } from './widgets/saved-view-widget/saved-view-widget.component'
|
||||
@@ -28,7 +27,6 @@ import { WelcomeWidgetComponent } from './widgets/welcome-widget/welcome-widget.
|
||||
templateUrl: './dashboard.component.html',
|
||||
styleUrls: ['./dashboard.component.scss'],
|
||||
imports: [
|
||||
LogoComponent,
|
||||
PageHeaderComponent,
|
||||
SavedViewWidgetComponent,
|
||||
StatisticsWidgetComponent,
|
||||
|
||||
+11
@@ -8,3 +8,14 @@
|
||||
width: 0.6rem;
|
||||
}
|
||||
}
|
||||
|
||||
.list-group {
|
||||
--bs-list-group-border-color: color-mix(in srgb, var(--bs-border-color) 58%, transparent);
|
||||
--bs-list-group-bg: transparent;
|
||||
border-radius: .6rem;
|
||||
overflow: hidden;
|
||||
}
|
||||
|
||||
.list-group-item {
|
||||
padding: .7rem .8rem;
|
||||
}
|
||||
|
||||
+10
@@ -4,6 +4,16 @@
|
||||
|
||||
.btn-outline-dark {
|
||||
--bs-btn-border-color: var(--bs-border-color-translucent);
|
||||
border-style: dashed;
|
||||
border-width: 1px;
|
||||
min-height: 5.25rem;
|
||||
background: color-mix(in srgb, var(--bs-primary) 4%, var(--bs-light)) !important;
|
||||
|
||||
&:hover,
|
||||
&:focus {
|
||||
border-color: var(--bs-primary);
|
||||
background: color-mix(in srgb, var(--bs-primary) 9%, var(--bs-light)) !important;
|
||||
}
|
||||
}
|
||||
|
||||
.smaller {
|
||||
|
||||
+1
-1
@@ -1,5 +1,5 @@
|
||||
@if (!cardless()) {
|
||||
<div class="card shadow-sm bg-light fade" [class.show]="show()" cdkDrag [cdkDragDisabled]="!draggable()" cdkDragPreviewContainer="parent">
|
||||
<div class="card bg-light fade" [class.show]="show()" cdkDrag [cdkDragDisabled]="!draggable()" cdkDragPreviewContainer="parent">
|
||||
<div class="card-header">
|
||||
<div class="d-flex justify-content-between align-items-center">
|
||||
<div class="d-flex align-items-center">
|
||||
|
||||
@@ -2,6 +2,20 @@ i-bs {
|
||||
cursor: move;
|
||||
}
|
||||
|
||||
.card {
|
||||
overflow: hidden;
|
||||
border-color: color-mix(in srgb, var(--bs-border-color) 68%, transparent);
|
||||
|
||||
.card-header {
|
||||
padding: .9rem 1rem;
|
||||
border-bottom-color: color-mix(in srgb, var(--bs-border-color) 55%, transparent);
|
||||
}
|
||||
|
||||
.card-body {
|
||||
padding: 1rem;
|
||||
}
|
||||
}
|
||||
|
||||
.fade.show {
|
||||
animation: pngx-entry-fade 160ms ease-out;
|
||||
}
|
||||
|
||||
@@ -20,7 +20,7 @@
|
||||
</div>
|
||||
}
|
||||
|
||||
<button type="button" class="btn btn-sm btn-outline-danger me-md-4" (click)="delete()" [disabled]="!userIsOwner" *pngxIfPermissions="{ action: PermissionAction.Delete, type: PermissionType.Document }">
|
||||
<button type="button" class="btn btn-sm btn-outline-danger me-md-4" (click)="delete()" [disabled]="!userIsOwner" *pngxIfPermissions="{ action: PermissionAction.Delete, type: PermissionType.Document }" aria-label="Delete" i18n-aria-label>
|
||||
<i-bs width="1.2em" height="1.2em" name="trash"></i-bs><span class="d-none d-lg-inline ps-1" i18n>Delete</span>
|
||||
</button>
|
||||
|
||||
@@ -35,7 +35,7 @@
|
||||
/>
|
||||
|
||||
<div class="btn-group">
|
||||
<button (click)="download()" class="btn btn-sm btn-outline-primary" [disabled]="downloading()">
|
||||
<button (click)="download()" class="btn btn-sm btn-outline-primary" [disabled]="downloading()" aria-label="Download" i18n-aria-label>
|
||||
@if (downloading()) {
|
||||
<div class="spinner-border spinner-border-sm" role="status"></div>
|
||||
} @else {
|
||||
@@ -45,7 +45,7 @@
|
||||
</button>
|
||||
|
||||
<div class="btn-group" ngbDropdown role="group">
|
||||
<button class="btn btn-sm btn-outline-primary dropdown-toggle" [disabled]="downloading()" ngbDropdownToggle></button>
|
||||
<button class="btn btn-sm btn-outline-primary dropdown-toggle" [disabled]="downloading()" ngbDropdownToggle aria-label="Download options" i18n-aria-label></button>
|
||||
<div class="dropdown-menu shadow" ngbDropdownMenu>
|
||||
@if (metadata()?.has_archive_version) {
|
||||
<button ngbDropdownItem (click)="download(true)" [disabled]="downloading()" i18n>Download original</button>
|
||||
@@ -62,7 +62,7 @@
|
||||
</div>
|
||||
|
||||
<div class="ms-auto" ngbDropdown>
|
||||
<button class="btn btn-sm btn-outline-primary" id="actionsDropdown" ngbDropdownToggle>
|
||||
<button class="btn btn-sm btn-outline-primary" id="actionsDropdown" ngbDropdownToggle aria-label="Actions" i18n-aria-label>
|
||||
<i-bs name="three-dots"></i-bs><div class="d-none d-sm-inline ms-1"><ng-container i18n>Actions</ng-container></div>
|
||||
</button>
|
||||
<div ngbDropdownMenu aria-labelledby="actionsDropdown" class="shadow">
|
||||
@@ -91,7 +91,7 @@
|
||||
</div>
|
||||
|
||||
<div class="ms-auto" ngbDropdown>
|
||||
<button class="btn btn-sm btn-outline-primary" id="sendDropdown" ngbDropdownToggle>
|
||||
<button class="btn btn-sm btn-outline-primary" id="sendDropdown" ngbDropdownToggle aria-label="Send" i18n-aria-label>
|
||||
<i-bs name="send"></i-bs><div class="d-none d-sm-inline ms-1"><ng-container i18n>Send</ng-container></div>
|
||||
</button>
|
||||
<div ngbDropdownMenu aria-labelledby="actionsDropdown" class="shadow">
|
||||
|
||||
@@ -2171,6 +2171,11 @@ describe('DocumentDetailComponent', () => {
|
||||
).toBe(10)
|
||||
component.openEmailDocument()
|
||||
expect(modalSpy).toHaveBeenCalled()
|
||||
expect(
|
||||
(
|
||||
modalSpy.mock.results[1].value as NgbModalRef
|
||||
).componentInstance.documentIds()
|
||||
).toEqual([10])
|
||||
})
|
||||
|
||||
it('should set previewText', () => {
|
||||
|
||||
@@ -1973,7 +1973,9 @@ export class DocumentDetailComponent
|
||||
const modal = this.modalService.open(EmailDocumentDialogComponent, {
|
||||
backdrop: 'static',
|
||||
})
|
||||
modal.componentInstance.documentIds.set([this.document().id])
|
||||
modal.componentInstance.documentIds.set([
|
||||
this.selectedVersionId() ?? this.document().id,
|
||||
])
|
||||
modal.componentInstance.hasArchiveVersion.set(
|
||||
this.metadata()?.has_archive_version ??
|
||||
!!this.document()?.archived_file_name
|
||||
|
||||
+4
-1
@@ -1,7 +1,10 @@
|
||||
<div class="btn-group" ngbDropdown autoClose="outside">
|
||||
<button class="btn btn-sm btn-outline-secondary dropdown-toggle" ngbDropdownToggle>
|
||||
<button class="btn btn-sm btn-outline-secondary dropdown-toggle" ngbDropdownToggle aria-label="Versions" i18n-aria-label>
|
||||
<i-bs name="file-earmark-diff"></i-bs>
|
||||
<span class="d-none d-lg-inline ps-1" i18n>Versions</span>
|
||||
@if (versions.length > 1) {
|
||||
<span class="badge text-bg-secondary ms-1">{{ versions.length }}</span>
|
||||
}
|
||||
</button>
|
||||
<div class="dropdown-menu shadow" ngbDropdownMenu>
|
||||
<div class="px-3 py-2 mb-2">
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
<h6>
|
||||
<button type="button" class="btn btn-outline-secondary btn-sm me-2"
|
||||
(click)="expand = !expand">
|
||||
(click)="expand = !expand" aria-label="Toggle document metadata" i18n-aria-label>
|
||||
@if (!expand) {
|
||||
<i-bs width="1.2em" height="1.2em" name="caret-down"></i-bs>
|
||||
}
|
||||
|
||||
@@ -74,7 +74,7 @@
|
||||
</pngx-filterable-dropdown>
|
||||
}
|
||||
<div class="btn-group">
|
||||
<button type="button" class="btn btn-sm btn-outline-primary me-2" (click)="setPermissions()" [disabled]="!userOwnsAll || !userCanEditAll">
|
||||
<button type="button" class="btn btn-sm btn-outline-primary me-2" (click)="setPermissions()" [disabled]="!userOwnsAll || !userCanEditAll" aria-label="Permissions" i18n-aria-label>
|
||||
<i-bs name="person-fill-lock"></i-bs><div class="d-none d-sm-inline ms-1"><ng-container i18n>Permissions</ng-container></div>
|
||||
</button>
|
||||
</div>
|
||||
@@ -82,7 +82,7 @@
|
||||
<div class="d-flex align-items-center gap-2 ms-auto">
|
||||
<div class="btn-toolbar">
|
||||
<div ngbDropdown>
|
||||
<button class="btn btn-sm btn-outline-primary" id="dropdownSelect" [disabled]="!userCanEdit && !userCanAdd" ngbDropdownToggle>
|
||||
<button class="btn btn-sm btn-outline-primary" id="dropdownSelect" [disabled]="!userCanEdit && !userCanAdd" ngbDropdownToggle aria-label="Actions" i18n-aria-label>
|
||||
<i-bs name="three-dots"></i-bs><div class="d-none d-sm-inline ms-1"><ng-container i18n>Actions</ng-container></div>
|
||||
</button>
|
||||
<div ngbDropdownMenu aria-labelledby="dropdownSelect" class="shadow">
|
||||
@@ -107,6 +107,8 @@
|
||||
id="dropdownSend"
|
||||
ngbDropdownToggle
|
||||
[disabled]="disabled || !canSendSelection"
|
||||
aria-label="Send"
|
||||
i18n-aria-label
|
||||
>
|
||||
<i-bs name="send"></i-bs><div class="d-none d-sm-inline ms-1"><ng-container i18n>Send</ng-container>
|
||||
</div>
|
||||
@@ -127,7 +129,7 @@
|
||||
</div>
|
||||
</div>
|
||||
<div class="btn-group btn-group-sm">
|
||||
<button class="btn btn-sm btn-outline-primary" [disabled]="awaitingDownload()" (click)="downloadSelected()">
|
||||
<button class="btn btn-sm btn-outline-primary" [disabled]="awaitingDownload()" (click)="downloadSelected()" aria-label="Download" i18n-aria-label>
|
||||
@if (!awaitingDownload()) {
|
||||
<i-bs name="arrow-down"></i-bs>
|
||||
}
|
||||
@@ -139,7 +141,7 @@
|
||||
<div class="d-none d-sm-inline ms-1"><ng-container i18n>Download</ng-container></div>
|
||||
</button>
|
||||
<div ngbDropdown class="me-2 d-flex btn-group" role="group">
|
||||
<button type="button" class="btn btn-sm btn-outline-primary dropdown-toggle-split rounded-end" ngbDropdownToggle></button>
|
||||
<button type="button" class="btn btn-sm btn-outline-primary dropdown-toggle-split rounded-end" ngbDropdownToggle aria-label="Download options" i18n-aria-label></button>
|
||||
<div ngbDropdownMenu aria-labelledby="dropdownSelect" class="shadow">
|
||||
<form [formGroup]="downloadForm" class="px-3 py-1">
|
||||
<p class="mb-1" i18n>Include:</p>
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
.dropdown-toggle-split {
|
||||
--bs-border-radius: .25rem;
|
||||
--bs-border-radius: .45rem;
|
||||
}
|
||||
|
||||
.dropdown-menu{
|
||||
|
||||
+1
-1
@@ -61,7 +61,7 @@
|
||||
</pngx-input-textarea>
|
||||
}
|
||||
}
|
||||
<button type="button" class="btn btn-outline-danger mb-3" (click)="removeField(field.id)">
|
||||
<button type="button" class="btn btn-outline-danger mb-3" (click)="removeField(field.id)" aria-label="Remove custom field" i18n-aria-label>
|
||||
<i-bs name="x"></i-bs>
|
||||
</button>
|
||||
</div>
|
||||
|
||||
+3
-3
@@ -154,9 +154,9 @@
|
||||
<i-bs name="download"></i-bs>
|
||||
</a>
|
||||
} @else {
|
||||
<button class="btn btn-sm btn-outline-secondary placeholder bg-secondary"></button>
|
||||
<button class="btn btn-sm btn-outline-secondary placeholder bg-secondary"></button>
|
||||
<button class="btn btn-sm btn-outline-secondary placeholder bg-secondary"></button>
|
||||
<span class="btn btn-sm btn-outline-secondary placeholder bg-secondary" aria-hidden="true"></span>
|
||||
<span class="btn btn-sm btn-outline-secondary placeholder bg-secondary" aria-hidden="true"></span>
|
||||
<span class="btn btn-sm btn-outline-secondary placeholder bg-secondary" aria-hidden="true"></span>
|
||||
}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
<pngx-page-header [title]="getTitle()">
|
||||
<div ngbDropdown class="btn-group flex-fill d-sm-none">
|
||||
<button class="btn btn-sm btn-outline-primary" id="dropdownSelectMobile" ngbDropdownToggle>
|
||||
<button class="btn btn-sm btn-outline-primary" id="dropdownSelectMobile" ngbDropdownToggle aria-label="Select" i18n-aria-label>
|
||||
<i-bs name="text-indent-left"></i-bs><div class="d-none d-sm-inline ms-1"><ng-container i18n>Select</ng-container></div>
|
||||
@if (list.hasSelection) {
|
||||
<pngx-clearable-badge [selected]="list.hasSelection" [number]="list.selectedCount" (cleared)="list.selectNone()"></pngx-clearable-badge><span class="visually-hidden">selected</span>
|
||||
@@ -31,7 +31,7 @@
|
||||
</div>
|
||||
</div>
|
||||
<div ngbDropdown class="btn-group flex-fill">
|
||||
<button class="btn btn-sm btn-outline-primary" id="dropdownDisplayFields" ngbDropdownToggle>
|
||||
<button class="btn btn-sm btn-outline-primary" id="dropdownDisplayFields" ngbDropdownToggle aria-label="Show" i18n-aria-label>
|
||||
<i-bs name="card-heading"></i-bs><div class="d-none d-sm-inline ms-1"><ng-container i18n>Show</ng-container></div>
|
||||
</button>
|
||||
<div ngbDropdownMenu aria-labelledby="dropdownDisplayFields" class="shadow">
|
||||
@@ -61,7 +61,7 @@
|
||||
</div>
|
||||
|
||||
<div ngbDropdown class="btn-group flex-fill">
|
||||
<button class="btn btn-outline-primary btn-sm" id="dropdownBasic1" ngbDropdownToggle>
|
||||
<button class="btn btn-outline-primary btn-sm" id="dropdownBasic1" ngbDropdownToggle aria-label="Sort" i18n-aria-label>
|
||||
<i-bs name="arrow-down-up"></i-bs><div class="d-none d-sm-inline ms-1"><ng-container i18n>Sort</ng-container></div>
|
||||
</button>
|
||||
<div ngbDropdownMenu aria-labelledby="dropdownBasic1" class="shadow dropdown-menu-right">
|
||||
@@ -85,8 +85,8 @@
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="btn-group flex-fill" *pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.SavedView }" ngbDropdown role="group">
|
||||
<button class="btn btn-sm btn-outline-primary dropdown-toggle flex-fill" tourAnchor="tour.documents-views" ngbDropdownToggle>
|
||||
<div class="btn-group flex-fill" *pngxIfPermissions="{ action: PermissionAction.View, type: PermissionType.SavedView }" ngbDropdown #viewsDropdown="ngbDropdown" role="group">
|
||||
<button class="btn btn-sm btn-outline-primary dropdown-toggle flex-fill" tourAnchor="tour.documents-views" ngbDropdownToggle aria-label="Views" i18n-aria-label>
|
||||
<i-bs name="window-stack"></i-bs><div class="d-none d-sm-inline ms-1"><ng-container i18n>Views</ng-container></div>
|
||||
@if (savedViewIsModified) {
|
||||
<div class="position-absolute top-0 start-100 p-2 translate-middle badge bg-secondary border border-light rounded-circle">
|
||||
@@ -95,15 +95,15 @@
|
||||
}
|
||||
</button>
|
||||
<div class="dropdown-menu shadow dropdown-menu-right" ngbDropdownMenu>
|
||||
@if (!list.activeSavedViewId) {
|
||||
@for (view of savedViewService.allViews; track view) {
|
||||
<button ngbDropdownItem (click)="loadViewConfig(view.id)">
|
||||
<i-bs class="me-2" [name]="view.icon || 'funnel'"></i-bs>{{view.name}}
|
||||
</button>
|
||||
}
|
||||
@if (savedViewService.allViews.length > 0) {
|
||||
<div class="dropdown-divider"></div>
|
||||
}
|
||||
@if (viewsDropdown.isOpen() && !list.activeSavedViewId && savedViewService.allViews.length > 0) {
|
||||
<div class="views-list overflow-y-auto">
|
||||
@for (view of savedViewService.allViews; track view.id) {
|
||||
<button ngbDropdownItem (click)="loadViewConfig(view.id)">
|
||||
<i-bs class="me-2" [name]="view.icon || 'funnel'"></i-bs>{{view.name}}
|
||||
</button>
|
||||
}
|
||||
</div>
|
||||
<div class="dropdown-divider"></div>
|
||||
}
|
||||
|
||||
@if (list.activeSavedViewId && activeSavedViewCanChange) {
|
||||
@@ -157,7 +157,7 @@
|
||||
</div>
|
||||
</ng-template>
|
||||
|
||||
<div tourAnchor="tour.documents">
|
||||
<div class="mt-3 mb-n2" tourAnchor="tour.documents">
|
||||
<ng-container *ngTemplateOutlet="pagination"></ng-container>
|
||||
</div>
|
||||
|
||||
|
||||
@@ -56,11 +56,11 @@ $paperless-card-breakpoints: (
|
||||
|
||||
.sticky-top {
|
||||
z-index: 990; // below main navbar
|
||||
top: calc(7rem - 2px); // height of navbar + search row (mobile)
|
||||
top: calc(7.5rem - 2px); // height of navbar + search row (mobile)
|
||||
transition: top 0.2s ease;
|
||||
|
||||
@media (min-width: 580px) {
|
||||
top: 3.5rem; // height of navbar
|
||||
top: 4.5em; // height of navbar
|
||||
}
|
||||
}
|
||||
|
||||
@@ -73,7 +73,7 @@ $paperless-card-breakpoints: (
|
||||
|
||||
@media (max-width: 579.98px) {
|
||||
:host-context(main.mobile-search-hidden) .sticky-top {
|
||||
top: calc(3.5rem - 2px); // height of navbar only when search is hidden
|
||||
top: calc(4rem - 2px); // height of navbar only when search is hidden
|
||||
}
|
||||
}
|
||||
|
||||
@@ -93,4 +93,8 @@ a {
|
||||
|
||||
pngx-page-header .dropdown-menu {
|
||||
--bs-dropdown-min-width: 12em;
|
||||
|
||||
.views-list {
|
||||
max-height: min(400px, calc(100vh - 260px)); // leave room for the header above and the actions below
|
||||
}
|
||||
}
|
||||
|
||||
@@ -18,7 +18,7 @@
|
||||
</select>
|
||||
}
|
||||
@if (_textFilter) {
|
||||
<button class="btn btn-link btn-sm px-2 position-absolute top-0 end-0 z-10" (click)="resetTextField()">
|
||||
<button class="btn btn-link btn-sm px-2 position-absolute top-0 end-0 z-10" (click)="resetTextField()" aria-label="Clear search" i18n-aria-label>
|
||||
<i-bs width="1em" height="1em" name="x"></i-bs>
|
||||
</button>
|
||||
}
|
||||
|
||||
+1
-1
@@ -24,7 +24,7 @@
|
||||
<div class="btn-toolbar gap-2">
|
||||
<div class="btn-group d-block d-sm-none">
|
||||
<div ngbDropdown container="body" class="d-inline-block">
|
||||
<button type="button" class="btn btn-link" id="actionsMenuMobile" (click)="$event.stopPropagation()" ngbDropdownToggle>
|
||||
<button type="button" class="btn btn-link" id="actionsMenuMobile" (click)="$event.stopPropagation()" ngbDropdownToggle aria-label="Actions" i18n-aria-label>
|
||||
<i-bs name="three-dots-vertical"></i-bs>
|
||||
</button>
|
||||
<div ngbDropdownMenu aria-labelledby="actionsMenuMobile">
|
||||
|
||||
+1
-1
@@ -7,7 +7,7 @@
|
||||
>
|
||||
@if (activeManagementList) {
|
||||
<div ngbDropdown class="btn-group flex-fill d-sm-none">
|
||||
<button class="btn btn-sm btn-outline-primary" id="dropdownSelectMobile" ngbDropdownToggle>
|
||||
<button class="btn btn-sm btn-outline-primary" id="dropdownSelectMobile" ngbDropdownToggle aria-label="Select" i18n-aria-label>
|
||||
<i-bs name="text-indent-left"></i-bs><div class="d-none d-sm-inline ms-1"><ng-container i18n>Select</ng-container></div>
|
||||
@if (activeManagementList.hasSelection) {
|
||||
<pngx-clearable-badge [selected]="activeManagementList.hasSelection" [number]="activeManagementList.selectedCount" (cleared)="activeManagementList.selectNone()"></pngx-clearable-badge><span class="visually-hidden">selected</span>
|
||||
|
||||
+3
-3
@@ -1,7 +1,7 @@
|
||||
<div class="row mb-3">
|
||||
<div class="col mb-2 mb-xl-0">
|
||||
<div class="form-inline d-flex align-items-center">
|
||||
<label class="text-muted me-2 mb-0" for="managementNameFilter" i18n>Filter by:</label>
|
||||
<label class="me-2 mb-0" for="managementNameFilter" i18n>Filter by:</label>
|
||||
<input id="managementNameFilter" class="form-control form-control-sm flex-fill w-auto" type="text" autofocus [(ngModel)]="nameFilter" (keyup)="onNameFilterKeyUp($event)" placeholder="Name" i18n-placeholder>
|
||||
</div>
|
||||
</div>
|
||||
@@ -9,7 +9,7 @@
|
||||
<div class="col-auto mb-2 mb-xl-0">
|
||||
<div class="form-inline d-flex align-items-center">
|
||||
<div class="input-group input-group-sm w-auto d-none d-md-flex">
|
||||
<label class="input-group-text border-0" for="managementPageSize" i18n>Show:</label>
|
||||
<label class="input-group-text bg-transparent border-0" for="managementPageSize" i18n>Show:</label>
|
||||
</div>
|
||||
<div class="input-group input-group-sm w-auto me-3">
|
||||
<select id="managementPageSize" class="form-select form-select-sm small" [(ngModel)]="pageSize">
|
||||
@@ -113,7 +113,7 @@
|
||||
<div class="btn-toolbar gap-2">
|
||||
<div class="btn-group d-block d-sm-none">
|
||||
<div ngbDropdown container="body" class="d-inline-block">
|
||||
<button type="button" class="btn btn-link" id="actionsMenuMobile" (click)="$event.stopPropagation()" ngbDropdownToggle>
|
||||
<button type="button" class="btn btn-link" id="actionsMenuMobile" (click)="$event.stopPropagation()" ngbDropdownToggle aria-label="Actions" i18n-aria-label>
|
||||
<i-bs name="three-dots-vertical"></i-bs>
|
||||
</button>
|
||||
<div ngbDropdownMenu aria-labelledby="actionsMenuMobile">
|
||||
|
||||
@@ -58,7 +58,7 @@
|
||||
<div class="col">
|
||||
<div class="btn-group d-block d-sm-none">
|
||||
<div ngbDropdown container="body" class="d-inline-block">
|
||||
<button type="button" class="btn btn-link" id="actionsMenuMobile" (click)="$event.stopPropagation()" ngbDropdownToggle>
|
||||
<button type="button" class="btn btn-link" id="actionsMenuMobile" (click)="$event.stopPropagation()" ngbDropdownToggle aria-label="Actions" i18n-aria-label>
|
||||
<i-bs name="three-dots-vertical"></i-bs>
|
||||
</button>
|
||||
<div ngbDropdownMenu aria-labelledby="actionsMenuMobile">
|
||||
@@ -146,7 +146,7 @@
|
||||
<div class="col-3">
|
||||
<div class="btn-group d-block d-sm-none">
|
||||
<div ngbDropdown container="body" class="d-inline-block">
|
||||
<button type="button" class="btn btn-link" id="actionsMenuMobile" (click)="$event.stopPropagation()" ngbDropdownToggle>
|
||||
<button type="button" class="btn btn-link" id="actionsMenuMobile" (click)="$event.stopPropagation()" ngbDropdownToggle aria-label="Actions" i18n-aria-label>
|
||||
<i-bs name="three-dots-vertical"></i-bs>
|
||||
</button>
|
||||
<div ngbDropdownMenu aria-labelledby="actionsMenuMobile">
|
||||
|
||||
@@ -47,7 +47,7 @@
|
||||
|
||||
<div class="btn-group d-block d-sm-none">
|
||||
<div ngbDropdown container="body" class="d-inline-block">
|
||||
<button type="button" class="btn btn-link" id="actionsMenuMobile" (click)="$event.stopPropagation()" ngbDropdownToggle>
|
||||
<button type="button" class="btn btn-link" id="actionsMenuMobile" (click)="$event.stopPropagation()" ngbDropdownToggle aria-label="Actions" i18n-aria-label>
|
||||
<i-bs name="three-dots-vertical"></i-bs>
|
||||
</button>
|
||||
<div ngbDropdownMenu aria-labelledby="actionsMenuMobile">
|
||||
|
||||
@@ -438,6 +438,21 @@ describe('ConsumerStatusService', () => {
|
||||
expect(updated).toBeTruthy()
|
||||
})
|
||||
|
||||
it('should ignore keep-alive heartbeat messages from the server', () => {
|
||||
let updated = false
|
||||
let deleted = false
|
||||
websocketStatusService.onDocumentUpdated().subscribe(() => (updated = true))
|
||||
websocketStatusService.onDocumentDeleted().subscribe(() => (deleted = true))
|
||||
|
||||
websocketStatusService.connect()
|
||||
server.send({ type: WebsocketStatusType.HEARTBEAT })
|
||||
|
||||
expect(updated).toBeFalsy()
|
||||
expect(deleted).toBeFalsy()
|
||||
expect(websocketStatusService.getConsumerStatus()).toHaveLength(0)
|
||||
websocketStatusService.disconnect()
|
||||
})
|
||||
|
||||
it('should ignore document updated events the user cannot view', () => {
|
||||
let updated = false
|
||||
websocketStatusService.onDocumentUpdated().subscribe(() => {
|
||||
|
||||
@@ -11,6 +11,7 @@ export enum WebsocketStatusType {
|
||||
STATUS_UPDATE = 'status_update',
|
||||
DOCUMENTS_DELETED = 'documents_deleted',
|
||||
DOCUMENT_UPDATED = 'document_updated',
|
||||
HEARTBEAT = 'heartbeat',
|
||||
}
|
||||
|
||||
// see ProgressStatusOptions in src/documents/plugins/helpers.py
|
||||
@@ -207,6 +208,10 @@ export class WebsocketStatusService {
|
||||
case WebsocketStatusType.STATUS_UPDATE:
|
||||
this.handleProgressUpdate(messageData as WebsocketProgressMessage)
|
||||
break
|
||||
|
||||
case WebsocketStatusType.HEARTBEAT:
|
||||
// keep-alive from the server, see paperless.consumers.StatusConsumer
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+37
-2
@@ -1,7 +1,7 @@
|
||||
@use "sass:meta";
|
||||
// bs variables
|
||||
$grid-gutter-width: 1.5rem;
|
||||
$border-radius: .375rem;
|
||||
$border-radius: .425rem;
|
||||
$btn-border-width: var(--bs-border-width);
|
||||
|
||||
$form-file-button-bg: var(--bs-body-bg);
|
||||
@@ -68,15 +68,44 @@ body {
|
||||
--pngx-body-font-size: 0.875rem;
|
||||
font-size: var(--pngx-body-font-size);
|
||||
height: 100vh;
|
||||
letter-spacing: -0.005em;
|
||||
}
|
||||
|
||||
* {
|
||||
transition: background-color 0.3s ease, border-color 0.3s ease;
|
||||
}
|
||||
|
||||
.card,
|
||||
.dropdown-menu,
|
||||
.modal-content,
|
||||
.popover {
|
||||
--bs-card-border-color: color-mix(in srgb, var(--bs-border-color) 72%, transparent);
|
||||
border-radius: .55rem;
|
||||
}
|
||||
|
||||
.card {
|
||||
box-shadow: 0 1px 2px rgba(0, 0, 0, .04), 0 8px 24px rgba(0, 0, 0, .035);
|
||||
}
|
||||
|
||||
.btn {
|
||||
--bs-btn-border-radius: .425rem;
|
||||
--bs-border-radius-sm: .425rem;
|
||||
font-weight: 500;
|
||||
}
|
||||
|
||||
.form-control,
|
||||
.form-select,
|
||||
.input-group-text {
|
||||
border-radius: .425rem;
|
||||
}
|
||||
|
||||
.pagination, .input-group {
|
||||
--bs-border-radius-sm: .425rem;
|
||||
}
|
||||
|
||||
@media(min-width: 768px) {
|
||||
.col-slim {
|
||||
padding-left: calc(50px + $grid-gutter-width) !important;
|
||||
padding-left: calc(56px + $grid-gutter-width) !important;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -660,6 +689,10 @@ table.table {
|
||||
--bs-toast-max-width: var(--pngx-toast-max-width);
|
||||
}
|
||||
|
||||
.alert {
|
||||
--bs-border-radius: .425rem;
|
||||
}
|
||||
|
||||
.alert-primary {
|
||||
--bs-alert-color: var(--bs-primary);
|
||||
--bs-alert-bg: var(--pngx-primary-faded);
|
||||
@@ -791,6 +824,8 @@ code {
|
||||
--bs-accordion-bg: var(--bs-light);
|
||||
--bs-accordion-active-color: var(--bs-primary);
|
||||
--bs-accordion-active-bg: var(--pngx-bg-alt);
|
||||
--bs-border-radius: .425rem;
|
||||
--bs-accordion-inner-border-radius: calc(.425rem - 1px);
|
||||
}
|
||||
|
||||
.accordion-button::after {
|
||||
|
||||
@@ -20,7 +20,7 @@
|
||||
--pngx-primary-darken-27: hsl(var(--pngx-primary), calc(var(--pngx-primary-lightness) - 27%));
|
||||
--pngx-success-darken-10: hsl(152, 69%, 11%); // based on success #198754
|
||||
--pngx-bg-alt: #fff;
|
||||
--pngx-bg-darker: var(--bs-gray-100);
|
||||
--pngx-bg-darker: #f4f6f5;
|
||||
--pngx-bg-alt2: var(--bs-gray-200); // #e9ecef
|
||||
--pngx-bg-disabled: #f7f7f7;
|
||||
--pngx-card-hover-border: var(--bs-tertiary-color);
|
||||
@@ -91,7 +91,7 @@ $form-check-radio-checked-bg-image-dark: url("data:image/svg+xml,%3csvg xmlns='h
|
||||
--pngx-body-color-accent: #{$text-color-dark-bg-accent};
|
||||
--pngx-bg-alt: #242529;
|
||||
--pngx-bg-alt2: #232323;
|
||||
--pngx-bg-darker: #101216;
|
||||
--pngx-bg-darker: #121315;
|
||||
--pngx-bg-disabled: var(--pngx-bg-alt);
|
||||
--pngx-card-hover-border: var(--bs-border-color);
|
||||
--pngx-focus-alpha: 0.6;
|
||||
@@ -107,7 +107,7 @@ $form-check-radio-checked-bg-image-dark: url("data:image/svg+xml,%3csvg xmlns='h
|
||||
--bs-light-rgb: 28, 28, 31;
|
||||
--bs-info: var(--pngx-bg-alt);
|
||||
--bs-info-rgb: 36, 36, 39;
|
||||
--bs-border-color: #47494f;
|
||||
--bs-border-color: #34373d;
|
||||
--bs-tertiary-bg: var(--pngx-bg-darker);
|
||||
--bs-dark-border-subtle: var(--pngx-bg-darker);
|
||||
--bs-border-color-translucent: rgba(0, 0, 0, .175); // override bs
|
||||
|
||||
+39
-25
@@ -305,33 +305,49 @@ def modify_custom_fields(
|
||||
else [(field, None) for field in add_custom_fields]
|
||||
)
|
||||
|
||||
custom_fields = CustomField.objects.filter(
|
||||
id__in=[int(field) for field, _ in add_custom_fields],
|
||||
).distinct()
|
||||
custom_fields_by_id: dict[int, CustomField] = {
|
||||
cf.id: cf
|
||||
for cf in CustomField.objects.filter(
|
||||
id__in=[int(field) for field, _ in add_custom_fields],
|
||||
)
|
||||
}
|
||||
# Deferred, not `.only()`: these objects get cached onto the FK
|
||||
# descriptor of newly-created CustomFieldInstance rows below, and
|
||||
# downstream post_save receivers (e.g. the filename-generation signal)
|
||||
# touch other Document fields -- `.only("pk")` would just turn that into
|
||||
# a deferred-field reload per document, trading one N+1 for another.
|
||||
# `content` is the one field guaranteed to be both large (full OCR text)
|
||||
# and unused by anything this function or its receivers touch.
|
||||
docs_by_id: dict[int, Document] = {
|
||||
doc.id: doc
|
||||
for doc in Document.objects.filter(id__in=affected_docs).defer("content")
|
||||
}
|
||||
for field_id, value in add_custom_fields:
|
||||
custom_field = custom_fields_by_id[field_id]
|
||||
value_field = CustomFieldInstance.TYPE_TO_DATA_STORE_NAME_MAP[
|
||||
custom_field.data_type
|
||||
]
|
||||
for doc_id in affected_docs:
|
||||
defaults = {}
|
||||
custom_field = custom_fields.get(id=field_id)
|
||||
if custom_field:
|
||||
value_field = CustomFieldInstance.TYPE_TO_DATA_STORE_NAME_MAP[
|
||||
custom_field.data_type
|
||||
]
|
||||
defaults[value_field] = value
|
||||
if (
|
||||
custom_field.data_type == CustomField.FieldDataType.DOCUMENTLINK
|
||||
and value
|
||||
and doc_id in value
|
||||
):
|
||||
# Prevent self-linking
|
||||
continue
|
||||
defaults = {value_field: value}
|
||||
if (
|
||||
custom_field.data_type == CustomField.FieldDataType.DOCUMENTLINK
|
||||
and value
|
||||
and doc_id in value
|
||||
):
|
||||
# Prevent self-linking
|
||||
continue
|
||||
# Pass the already-resolved objects, not bare ids: this caches
|
||||
# them on the FK descriptor of any newly-created instance, so a
|
||||
# later `.field`/`.document` access (e.g. auditlog's post_save
|
||||
# receiver calling `str(instance)`, which touches `.field.name`)
|
||||
# doesn't trigger its own per-instance re-fetch.
|
||||
CustomFieldInstance.objects.update_or_create(
|
||||
document_id=doc_id,
|
||||
field_id=field_id,
|
||||
document=docs_by_id[doc_id],
|
||||
field=custom_field,
|
||||
defaults=defaults,
|
||||
)
|
||||
if custom_field.data_type == CustomField.FieldDataType.DOCUMENTLINK:
|
||||
doc = Document.objects.get(id=doc_id)
|
||||
reflect_doclinks(doc, custom_field, value)
|
||||
reflect_doclinks(docs_by_id[doc_id], custom_field, value)
|
||||
|
||||
# For doc link fields that are being removed, remove symmetrical links
|
||||
for doclink_being_removed_instance in CustomFieldInstance.objects.filter(
|
||||
@@ -339,12 +355,10 @@ def modify_custom_fields(
|
||||
field__id__in=remove_custom_fields,
|
||||
field__data_type=CustomField.FieldDataType.DOCUMENTLINK,
|
||||
value_document_ids__isnull=False,
|
||||
):
|
||||
).select_related("field"):
|
||||
for target_doc_id in doclink_being_removed_instance.value:
|
||||
remove_doclink(
|
||||
document=Document.objects.get(
|
||||
id=doclink_being_removed_instance.document.id,
|
||||
),
|
||||
document=docs_by_id[doclink_being_removed_instance.document_id],
|
||||
field=doclink_being_removed_instance.field,
|
||||
target_doc_id=target_doc_id,
|
||||
)
|
||||
|
||||
@@ -6,20 +6,13 @@ from documents.search._backend import TantivyRelevanceList
|
||||
from documents.search._backend import WriteBatch
|
||||
from documents.search._backend import get_backend
|
||||
from documents.search._backend import reset_backend
|
||||
from documents.search._errors import InvalidDateQuery
|
||||
from documents.search._errors import InvalidNumberQuery
|
||||
from documents.search._errors import MultipleSearchQueryErrors
|
||||
from documents.search._errors import QueryTooLongError
|
||||
from documents.search._errors import SearchQueryError
|
||||
from documents.search._errors import search_query_error_messages
|
||||
from documents.search._schema import needs_rebuild
|
||||
from documents.search._schema import wipe_index
|
||||
from documents.search._translate import InvalidDateQuery
|
||||
from documents.search._translate import SearchQueryError
|
||||
|
||||
__all__ = [
|
||||
"InvalidDateQuery",
|
||||
"InvalidNumberQuery",
|
||||
"MultipleSearchQueryErrors",
|
||||
"QueryTooLongError",
|
||||
"SearchHit",
|
||||
"SearchIndexLockError",
|
||||
"SearchMode",
|
||||
@@ -30,6 +23,5 @@ __all__ = [
|
||||
"get_backend",
|
||||
"needs_rebuild",
|
||||
"reset_backend",
|
||||
"search_query_error_messages",
|
||||
"wipe_index",
|
||||
]
|
||||
|
||||
@@ -20,11 +20,9 @@ from typing import cast
|
||||
import filelock
|
||||
import tantivy
|
||||
from django.conf import settings
|
||||
from django.contrib.contenttypes.models import ContentType
|
||||
from django.utils.timezone import get_current_timezone
|
||||
from guardian.shortcuts import get_groups_with_perms
|
||||
from guardian.shortcuts import get_users_with_perms
|
||||
|
||||
from documents.search._query import build_permission_filter
|
||||
from documents.search._query import extract_cjk_text
|
||||
from documents.search._query import parse_simple_text_highlight_query
|
||||
from documents.search._query import parse_simple_text_query
|
||||
@@ -42,7 +40,6 @@ from documents.utils import QuerySetStream
|
||||
from documents.utils import identity
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterable
|
||||
from collections.abc import Iterator
|
||||
from collections.abc import Sequence
|
||||
from pathlib import Path
|
||||
@@ -266,11 +263,7 @@ class WriteBatch:
|
||||
if self._lock is not None:
|
||||
self._lock.release()
|
||||
|
||||
def add_or_update(
|
||||
self,
|
||||
document: Document,
|
||||
effective_content: str | None = None,
|
||||
) -> None:
|
||||
def add_or_update(self, document: Document) -> None:
|
||||
"""
|
||||
Add or update a document in the batch.
|
||||
|
||||
@@ -280,11 +273,9 @@ class WriteBatch:
|
||||
|
||||
Args:
|
||||
document: Django Document instance to index
|
||||
effective_content: Override document.content for indexing (used when
|
||||
re-indexing with newer OCR text from document versions)
|
||||
"""
|
||||
self.remove(document.pk)
|
||||
doc = self._backend._build_tantivy_doc(document, effective_content)
|
||||
doc = self._backend._build_tantivy_doc(document)
|
||||
self._writer.add_document(doc)
|
||||
|
||||
def remove(self, doc_id: int) -> None:
|
||||
@@ -294,47 +285,6 @@ class WriteBatch:
|
||||
)
|
||||
|
||||
|
||||
def build_permission_filter(
|
||||
schema: tantivy.Schema,
|
||||
user: AbstractUser,
|
||||
viewer_group_ids: Iterable[int] = (),
|
||||
) -> tantivy.Query:
|
||||
"""
|
||||
Build a query filter for user document permissions.
|
||||
|
||||
Creates a query that matches only documents visible to the specified user
|
||||
according to paperless-ngx permission rules:
|
||||
- Public documents (no owner) are visible to all users
|
||||
- Private documents are visible to their owner
|
||||
- Documents explicitly shared with the user are visible
|
||||
- Documents shared with one of the user's current groups are visible
|
||||
|
||||
Args:
|
||||
schema: Tantivy schema for field validation
|
||||
user: User to check permissions for
|
||||
viewer_group_ids: Current group memberships for the user
|
||||
|
||||
Returns:
|
||||
Tantivy query that filters results to visible documents
|
||||
"""
|
||||
owner_any = tantivy.Query.exists_query("owner_id")
|
||||
no_owner = tantivy.Query.boolean_query(
|
||||
[
|
||||
(tantivy.Occur.Must, tantivy.Query.all_query()),
|
||||
(tantivy.Occur.MustNot, owner_any),
|
||||
],
|
||||
)
|
||||
owned = tantivy.Query.term_query(schema, "owner_id", user.pk)
|
||||
shared = tantivy.Query.term_query(schema, "viewer_id", user.pk)
|
||||
group_shared = [
|
||||
tantivy.Query.term_query(schema, "viewer_group_id", group_id)
|
||||
for group_id in viewer_group_ids
|
||||
]
|
||||
return tantivy.Query.disjunction_max_query(
|
||||
[no_owner, owned, shared, *group_shared],
|
||||
)
|
||||
|
||||
|
||||
class TantivyBackend:
|
||||
"""
|
||||
Tantivy search backend with explicit lifecycle management.
|
||||
@@ -466,18 +416,20 @@ class TantivyBackend:
|
||||
def _build_tantivy_doc(
|
||||
self,
|
||||
document: Document,
|
||||
effective_content: str | None = None,
|
||||
viewer_ids: list[int] | None = None,
|
||||
viewer_group_ids: list[int] | None = None,
|
||||
) -> tantivy.Document:
|
||||
"""Build a tantivy Document from a Django Document instance.
|
||||
|
||||
``effective_content`` overrides ``document.content`` for indexing —
|
||||
used when re-indexing a root document with a newer version's OCR text.
|
||||
A root document is indexed with its effective content, i.e. the newest
|
||||
version's OCR text, so it is never indexed with its own outdated text.
|
||||
Annotate the queryset with ``annotate_effective_content`` when indexing
|
||||
more than a couple of documents, to resolve that without a query each.
|
||||
"""
|
||||
content = (
|
||||
effective_content if effective_content is not None else document.content
|
||||
)
|
||||
from guardian.shortcuts import get_groups_with_perms
|
||||
from guardian.shortcuts import get_users_with_perms
|
||||
|
||||
content = document.get_effective_content() or ""
|
||||
|
||||
doc = tantivy.Document()
|
||||
|
||||
@@ -506,6 +458,7 @@ class TantivyBackend:
|
||||
doc.add_text("correspondent_sort", document.correspondent.name)
|
||||
if cjk_corr := extract_cjk_text(document.correspondent.name):
|
||||
doc.add_text("bigram_correspondent", cjk_corr)
|
||||
doc.add_unsigned("correspondent_id", document.correspondent_id)
|
||||
|
||||
# Document type
|
||||
if document.document_type:
|
||||
@@ -513,10 +466,12 @@ class TantivyBackend:
|
||||
doc.add_text("type_sort", document.document_type.name)
|
||||
if cjk_type := extract_cjk_text(document.document_type.name):
|
||||
doc.add_text("bigram_document_type", cjk_type)
|
||||
doc.add_unsigned("document_type_id", document.document_type_id)
|
||||
|
||||
# Storage path
|
||||
if document.storage_path:
|
||||
doc.add_text("storage_path", document.storage_path.name)
|
||||
doc.add_unsigned("storage_path_id", document.storage_path_id)
|
||||
|
||||
# Tags — collect names for autocomplete in the same pass
|
||||
tag_names: list[str] = []
|
||||
@@ -524,13 +479,12 @@ class TantivyBackend:
|
||||
doc.add_text("tag", tag.name)
|
||||
if cjk_tag := extract_cjk_text(tag.name):
|
||||
doc.add_text("bigram_tag", cjk_tag)
|
||||
doc.add_unsigned("tag_id", tag.pk)
|
||||
tag_names.append(tag.name)
|
||||
|
||||
# Notes — JSON for structured queries (notes.user:alice, notes.note:text).
|
||||
# notes_text is a plain-text companion for snippet/highlight generation;
|
||||
# tantivy's SnippetGenerator does not support JSON fields. It is not in
|
||||
# _DEFAULT_SEARCH_FIELDS, so an unqualified query never searches it: a
|
||||
# note matches through the JSON field or not at all.
|
||||
# tantivy's SnippetGenerator does not support JSON fields.
|
||||
num_notes = 0
|
||||
note_texts: list[str] = []
|
||||
for note in document.notes.all():
|
||||
@@ -546,9 +500,8 @@ class TantivyBackend:
|
||||
if note_texts:
|
||||
doc.add_text("notes_text", " ".join(note_texts))
|
||||
|
||||
# Custom fields — JSON for structured queries (custom_fields.name:x,
|
||||
# custom_fields.value:y). There is no companion text field here, unlike
|
||||
# notes: custom field values are reachable only through the JSON field.
|
||||
# Custom fields — JSON for structured queries (custom_fields.name:x, custom_fields.value:y),
|
||||
# companion text field for default full-text search.
|
||||
for cfi in document.custom_fields.all():
|
||||
search_value = cfi.value_for_search
|
||||
# Skip fields where there is no value yet
|
||||
@@ -624,11 +577,7 @@ class TantivyBackend:
|
||||
|
||||
return doc
|
||||
|
||||
def add_or_update(
|
||||
self,
|
||||
document: Document,
|
||||
effective_content: str | None = None,
|
||||
) -> None:
|
||||
def add_or_update(self, document: Document) -> None:
|
||||
"""
|
||||
Add or update a single document with file locking.
|
||||
|
||||
@@ -641,12 +590,11 @@ class TantivyBackend:
|
||||
|
||||
Args:
|
||||
document: Django Document instance to index
|
||||
effective_content: Override document.content for indexing
|
||||
"""
|
||||
self._ensure_open()
|
||||
try:
|
||||
with self.batch_update(lock_timeout=_LOCK_TIMEOUT_SECONDS) as batch:
|
||||
batch.add_or_update(document, effective_content)
|
||||
batch.add_or_update(document)
|
||||
except SearchIndexLockError:
|
||||
logger.error(
|
||||
"Search index lock exhausted for document %d after %d attempts; "
|
||||
@@ -720,17 +668,7 @@ class TantivyBackend:
|
||||
user_query = self._parse_query(query, search_mode)
|
||||
highlight_query = user_query
|
||||
if search_mode is SearchMode.TEXT:
|
||||
try:
|
||||
highlight_query = parse_simple_text_highlight_query(
|
||||
self._index,
|
||||
query,
|
||||
)
|
||||
except ValueError:
|
||||
logger.debug(
|
||||
"Skipping simple text highlight query: token string is not "
|
||||
"valid tantivy query syntax: %r",
|
||||
query,
|
||||
)
|
||||
highlight_query = parse_simple_text_highlight_query(self._index, query)
|
||||
|
||||
# For notes_text snippet generation, we need a query that targets the
|
||||
# notes_text field directly. user_query may contain JSON-field terms
|
||||
@@ -1077,7 +1015,6 @@ class TantivyBackend:
|
||||
):
|
||||
doc = self._build_tantivy_doc(
|
||||
document,
|
||||
document.get_effective_content(),
|
||||
viewer_ids=viewer_ids,
|
||||
viewer_group_ids=viewer_group_ids,
|
||||
)
|
||||
@@ -1145,6 +1082,7 @@ def _bulk_get_viewer_permissions(
|
||||
"""
|
||||
from collections import defaultdict
|
||||
|
||||
from django.contrib.contenttypes.models import ContentType
|
||||
from guardian.models import GroupObjectPermission
|
||||
from guardian.models import UserObjectPermission
|
||||
|
||||
|
||||
@@ -0,0 +1,171 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC
|
||||
from datetime import date
|
||||
from datetime import datetime
|
||||
from datetime import timedelta
|
||||
from typing import TYPE_CHECKING
|
||||
from typing import Final
|
||||
|
||||
from dateutil.relativedelta import relativedelta
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from datetime import tzinfo
|
||||
|
||||
_DATE_ONLY_FIELDS = frozenset({"created"})
|
||||
|
||||
_TODAY: Final[str] = "today"
|
||||
_YESTERDAY: Final[str] = "yesterday"
|
||||
_PREVIOUS_WEEK: Final[str] = "previous week"
|
||||
_THIS_MONTH: Final[str] = "this month"
|
||||
_PREVIOUS_MONTH: Final[str] = "previous month"
|
||||
_THIS_YEAR: Final[str] = "this year"
|
||||
_PREVIOUS_YEAR: Final[str] = "previous year"
|
||||
_PREVIOUS_QUARTER: Final[str] = "previous quarter"
|
||||
|
||||
_DATE_KEYWORDS = frozenset(
|
||||
{
|
||||
_TODAY,
|
||||
_YESTERDAY,
|
||||
_PREVIOUS_WEEK,
|
||||
_THIS_MONTH,
|
||||
_PREVIOUS_MONTH,
|
||||
_THIS_YEAR,
|
||||
_PREVIOUS_YEAR,
|
||||
_PREVIOUS_QUARTER,
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def _fmt(dt: datetime) -> str:
|
||||
"""Format a datetime as an ISO 8601 UTC string for use in Tantivy range queries."""
|
||||
return dt.astimezone(UTC).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||
|
||||
|
||||
def _iso_range(lo: datetime, hi: datetime) -> str:
|
||||
"""
|
||||
Format a half-open ``[lo TO hi)`` range in ISO 8601 for Tantivy query syntax.
|
||||
|
||||
``hi`` is always the exclusive ceiling of a computed period (the start of
|
||||
the *next* day/week/month/quarter/year), so the closing bracket must be
|
||||
the Tantivy exclusive-range brace ``}`` rather than ``]`` — otherwise the
|
||||
first instant of the following period (e.g. the 1st of next month) is
|
||||
incorrectly included in the match.
|
||||
"""
|
||||
return f"[{_fmt(lo)} TO {_fmt(hi)}}}"
|
||||
|
||||
|
||||
def _quarter_start(d: date) -> date:
|
||||
"""Return the first day of the calendar quarter containing ``d``."""
|
||||
return date(d.year, ((d.month - 1) // 3) * 3 + 1, 1)
|
||||
|
||||
|
||||
def _midnight(d: date, tz: tzinfo) -> datetime:
|
||||
"""Convert a calendar date at local-timezone midnight to a UTC datetime."""
|
||||
return datetime(d.year, d.month, d.day, tzinfo=tz).astimezone(UTC)
|
||||
|
||||
|
||||
def _keyword_bounds(keyword: str, tz: tzinfo) -> tuple[date, date]:
|
||||
"""
|
||||
Map a relative date keyword to ``(start, exclusive_end)`` calendar dates.
|
||||
|
||||
``tz`` only determines what "today" is; the caller decides how the returned
|
||||
dates become UTC datetime boundaries (date-only vs. local-midnight offset).
|
||||
"""
|
||||
today = datetime.now(tz).date()
|
||||
if keyword == _TODAY:
|
||||
return today, today + timedelta(days=1)
|
||||
if keyword == _YESTERDAY:
|
||||
return today - timedelta(days=1), today
|
||||
if keyword == _PREVIOUS_WEEK:
|
||||
this_monday = today - timedelta(days=today.weekday())
|
||||
return this_monday - timedelta(weeks=1), this_monday
|
||||
if keyword == _THIS_MONTH:
|
||||
first = today.replace(day=1)
|
||||
return first, first + relativedelta(months=1)
|
||||
if keyword == _PREVIOUS_MONTH:
|
||||
this_first = today.replace(day=1)
|
||||
return this_first - relativedelta(months=1), this_first
|
||||
if keyword == _THIS_YEAR:
|
||||
return date(today.year, 1, 1), date(today.year + 1, 1, 1)
|
||||
if keyword == _PREVIOUS_YEAR:
|
||||
return date(today.year - 1, 1, 1), date(today.year, 1, 1)
|
||||
if keyword == _PREVIOUS_QUARTER:
|
||||
this_quarter = _quarter_start(today)
|
||||
return this_quarter - relativedelta(months=3), this_quarter
|
||||
raise ValueError(f"Unknown keyword: {keyword}")
|
||||
|
||||
|
||||
def _date_only_range(keyword: str, tz: tzinfo) -> str:
|
||||
"""
|
||||
For `created` (DateField): use the local calendar date, converted to
|
||||
midnight UTC boundaries. No offset arithmetic — date only.
|
||||
"""
|
||||
start, end = _keyword_bounds(keyword, tz)
|
||||
lo = datetime(start.year, start.month, start.day, tzinfo=UTC)
|
||||
hi = datetime(end.year, end.month, end.day, tzinfo=UTC)
|
||||
return _iso_range(lo, hi)
|
||||
|
||||
|
||||
def _datetime_range(keyword: str, tz: tzinfo) -> str:
|
||||
"""
|
||||
For `added` / `modified` (DateTimeField, stored as UTC): convert local day
|
||||
boundaries to UTC — full offset arithmetic required.
|
||||
"""
|
||||
start, end = _keyword_bounds(keyword, tz)
|
||||
return _iso_range(_midnight(start, tz), _midnight(end, tz))
|
||||
|
||||
|
||||
def _precision_bounds(digits: str) -> tuple[date, date] | None:
|
||||
"""
|
||||
Map a 4/6/8-digit date token to (start, exclusive_end) calendar dates.
|
||||
|
||||
YYYY -> whole year, YYYYMM -> whole month, YYYYMMDD -> single day.
|
||||
Returns None for any unparsable or out-of-range value (e.g. month 23),
|
||||
so callers can emit a no-match clause instead of erroring (Whoosh parity).
|
||||
"""
|
||||
try:
|
||||
if len(digits) == 4:
|
||||
year = int(digits)
|
||||
return date(year, 1, 1), date(year + 1, 1, 1)
|
||||
if len(digits) == 6:
|
||||
year, month = int(digits[:4]), int(digits[4:6])
|
||||
start = date(year, month, 1)
|
||||
end = date(year + 1, 1, 1) if month == 12 else date(year, month + 1, 1)
|
||||
return start, end
|
||||
if len(digits) == 8:
|
||||
start = date(int(digits[:4]), int(digits[4:6]), int(digits[6:8]))
|
||||
return start, start + timedelta(days=1)
|
||||
except ValueError:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _utc_bounds_for_field(
|
||||
field: str,
|
||||
start: date,
|
||||
end: date,
|
||||
tz: tzinfo,
|
||||
) -> tuple[datetime, datetime]:
|
||||
"""
|
||||
Convert calendar-date bounds to UTC datetimes per the field's storage type.
|
||||
|
||||
For DateField (``created``) the bounds are UTC midnight (no offset). For
|
||||
DateTimeField (``added``/``modified``) the bounds are local-tz midnight
|
||||
converted to UTC, matching how each field is indexed.
|
||||
"""
|
||||
if field in _DATE_ONLY_FIELDS:
|
||||
return (
|
||||
datetime(start.year, start.month, start.day, tzinfo=UTC),
|
||||
datetime(end.year, end.month, end.day, tzinfo=UTC),
|
||||
)
|
||||
return (
|
||||
datetime(start.year, start.month, start.day, tzinfo=tz).astimezone(UTC),
|
||||
datetime(end.year, end.month, end.day, tzinfo=tz).astimezone(UTC),
|
||||
)
|
||||
|
||||
|
||||
def _field_range_from_dates(field: str, start: date, end: date, tz: tzinfo) -> str:
|
||||
"""Build a Tantivy ``field:[lo TO hi]`` ISO range from calendar-date bounds."""
|
||||
lo, hi = _utc_bounds_for_field(field, start, end, tz)
|
||||
return f"{field}:{_iso_range(lo, hi)}"
|
||||
@@ -1,71 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
|
||||
class SearchQueryError(ValueError):
|
||||
"""
|
||||
Base for user-fixable search query errors.
|
||||
|
||||
Carries a message safe to surface to the user (no internal details). The
|
||||
view layer catches this and returns an HTTP 400, so any future subclass
|
||||
gets the same treatment.
|
||||
"""
|
||||
|
||||
|
||||
class InvalidDateQuery(SearchQueryError):
|
||||
"""Raised when a date field value or range bound cannot be parsed."""
|
||||
|
||||
def __init__(self, field: str | None, value: str | None) -> None:
|
||||
self.field = field
|
||||
self.value = value
|
||||
super().__init__(f"Invalid date value {value!r} for field {field!r}.")
|
||||
|
||||
|
||||
class InvalidNumberQuery(SearchQueryError):
|
||||
"""Raised when a numeric field value or range bound cannot be parsed."""
|
||||
|
||||
def __init__(self, field: str | None, value: str | None) -> None:
|
||||
self.field = field
|
||||
self.value = value
|
||||
super().__init__(f"Invalid numeric value {value!r} for field {field!r}.")
|
||||
|
||||
|
||||
class QueryTooLongError(SearchQueryError):
|
||||
"""Raised when a query string exceeds the maximum allowed length.
|
||||
|
||||
whoosh-compat's fieldname tagger is O(n^2) in plain word characters, so an
|
||||
unbounded query is a CPU-exhaustion vector against a single request
|
||||
handler. This is a hard boundary, not a validation nicety.
|
||||
"""
|
||||
|
||||
def __init__(self, length: int, limit: int) -> None:
|
||||
self.length = length
|
||||
self.limit = limit
|
||||
super().__init__(
|
||||
f"The search query is too long ({length} characters). "
|
||||
f"The maximum allowed length is {limit} characters.",
|
||||
)
|
||||
|
||||
|
||||
class MultipleSearchQueryErrors(SearchQueryError):
|
||||
"""Aggregates every user-fixable error from one parse, not just the first."""
|
||||
|
||||
def __init__(self, errors: Sequence[SearchQueryError]) -> None:
|
||||
self.errors = tuple(errors)
|
||||
super().__init__("; ".join(str(e) for e in self.errors))
|
||||
|
||||
|
||||
def search_query_error_messages(e: SearchQueryError) -> list[str]:
|
||||
"""The user-facing message list for a SearchQueryError.
|
||||
|
||||
Every offending value's message, not just the first, so the user can
|
||||
fix them all in one round-trip. Shared by every view that maps
|
||||
SearchQueryError to an HTTP 400.
|
||||
"""
|
||||
if isinstance(e, MultipleSearchQueryErrors):
|
||||
return [str(sub) for sub in e.errors]
|
||||
return [str(e)]
|
||||
@@ -1,42 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from whoosh_compat import FieldKind
|
||||
from whoosh_compat import FieldSpec
|
||||
from whoosh_compat import SubpathSpec
|
||||
|
||||
# Internal-only schema fields with no query-syntax meaning of their own
|
||||
# (sort shadow fields, bigram CJK fields, simple_title/simple_content,
|
||||
# autocomplete_word, notes_text) are NOT represented here — they are
|
||||
# declared in _schema.py's field_descriptors().
|
||||
#
|
||||
# analyzer/pattern_normalizer are deliberately left at FieldSpec's default
|
||||
# (None): they're language-specific and only meaningful to whoosh-compat's
|
||||
# parser, so _registry.py attaches them per-language via dataclasses.replace()
|
||||
# rather than PUBLIC_FIELDS declaring them itself. _schema.py only reads
|
||||
# name/kind/fast and never sees the analyzer at all.
|
||||
PUBLIC_FIELDS: tuple[FieldSpec, ...] = (
|
||||
FieldSpec("title", FieldKind.TEXT),
|
||||
FieldSpec("content", FieldKind.TEXT),
|
||||
FieldSpec("correspondent", FieldKind.TEXT),
|
||||
FieldSpec("document_type", FieldKind.TEXT, aliases=("type",)),
|
||||
FieldSpec("storage_path", FieldKind.TEXT, aliases=("path",)),
|
||||
FieldSpec("original_filename", FieldKind.TEXT),
|
||||
FieldSpec("tag", FieldKind.TEXT, comma_values=True),
|
||||
FieldSpec("checksum", FieldKind.KEYWORD),
|
||||
FieldSpec("asn", FieldKind.U64, fast=True),
|
||||
FieldSpec("page_count", FieldKind.U64, fast=True),
|
||||
FieldSpec("num_notes", FieldKind.U64, fast=True),
|
||||
FieldSpec("created", FieldKind.DATE, date_only=True, fast=True),
|
||||
FieldSpec("modified", FieldKind.DATETIME, fast=True),
|
||||
FieldSpec("added", FieldKind.DATETIME, fast=True),
|
||||
FieldSpec(
|
||||
"notes",
|
||||
FieldKind.JSON,
|
||||
subpaths={"user": SubpathSpec(), "note": SubpathSpec(default=True)},
|
||||
),
|
||||
FieldSpec(
|
||||
"custom_fields",
|
||||
FieldKind.JSON,
|
||||
subpaths={"name": SubpathSpec(), "value": SubpathSpec(default=True)},
|
||||
),
|
||||
)
|
||||
+145
-442
@@ -6,30 +6,22 @@ from typing import Final
|
||||
|
||||
import regex
|
||||
import tantivy
|
||||
import whoosh_compat as wc
|
||||
from django.conf import settings
|
||||
from whoosh_compat.emitters.tantivy_ import emit as tantivy_emit
|
||||
from whoosh_compat.errors import Cause
|
||||
from whoosh_compat.errors import Diagnostic
|
||||
from whoosh_compat.errors import DiagnosticKind
|
||||
from whoosh_compat.errors import QueryError
|
||||
|
||||
from documents.search._errors import InvalidDateQuery
|
||||
from documents.search._errors import InvalidNumberQuery
|
||||
from documents.search._errors import MultipleSearchQueryErrors
|
||||
from documents.search._errors import SearchQueryError
|
||||
from documents.search._registry import get_field_registry
|
||||
from documents.search._tokenizer import simple_search_tokens
|
||||
from documents.search._translate import SearchQueryError
|
||||
from documents.search._translate import translate_query
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterable
|
||||
from datetime import tzinfo
|
||||
|
||||
from django.contrib.auth.base_user import AbstractBaseUser
|
||||
|
||||
logger = logging.getLogger("paperless.search")
|
||||
|
||||
# Maximum seconds any single regex substitution over user-supplied query text
|
||||
# may run. The one remaining use is a character class, which cannot backtrack,
|
||||
# so the bound is an upper limit on that substitution's cost, not the ReDoS
|
||||
# guard it was originally written as.
|
||||
# Maximum seconds any single regex substitution may run.
|
||||
# Prevents ReDoS on adversarial user-supplied query strings.
|
||||
_REGEX_TIMEOUT: Final[float] = 1.0
|
||||
|
||||
# Matches CJK/Hangul characters so queries can be routed to bigram fields.
|
||||
@@ -37,64 +29,6 @@ _REGEX_TIMEOUT: Final[float] = 1.0
|
||||
_CJK_RE: Final = regex.compile(r"[\p{Han}\p{Hiragana}\p{Katakana}\p{Hangul}]+")
|
||||
|
||||
|
||||
def _user_facing_emit_message(d: Diagnostic) -> str:
|
||||
"""A user-safe message for an emit-time QueryError's Diagnostic.
|
||||
|
||||
Built from the Diagnostic's structured fields (kind, field), never from
|
||||
d.message: whoosh-compat documents that as developer/log output with no
|
||||
stability guarantee, and PATTERN_TOO_COMPLEX embeds the raw backend
|
||||
error text in it.
|
||||
"""
|
||||
field = str(d.field) if d.field is not None else None
|
||||
if d.kind is DiagnosticKind.EXISTS_REQUIRES_FAST:
|
||||
return f"Existence searches (field:*) are not supported for field {field!r}."
|
||||
if d.kind is DiagnosticKind.TEXT_RANGE:
|
||||
return f"Range searches are not supported for field {field!r}."
|
||||
if d.kind is DiagnosticKind.PATTERN_TOO_COMPLEX:
|
||||
return f"The wildcard pattern for field {field!r} is too complex."
|
||||
if d.kind is DiagnosticKind.SCHEMA_FIELD_MISSING:
|
||||
return f"Field {field!r} is not available in the search index."
|
||||
logger.warning("Unmapped emit diagnostic %s: %s", d.kind, d.message)
|
||||
return "The search query could not be executed."
|
||||
|
||||
|
||||
def _map_emit_error(e: QueryError) -> SearchQueryError:
|
||||
"""Route an emit-time QueryError by its Diagnostic's Cause.
|
||||
|
||||
INVALID_INPUT/UNSUPPORTED are user-input errors, exactly like a parse
|
||||
diagnostic, and map to a 400. INTERNAL means a defect in whoosh-compat
|
||||
or in our own AST handling, never the user's query, so the QueryError is
|
||||
re-raised rather than converted, reaching the generic 500 handler instead
|
||||
of blaming the query. MISCONFIGURED is deliberately both: the registry and
|
||||
the index schema disagree, which only an operator can fix, so it is logged
|
||||
as an error, but a request is still waiting and the query cannot run
|
||||
either way, so it also returns a 400.
|
||||
|
||||
EXISTS_REQUIRES_FAST is the one MISCONFIGURED kind that is not a
|
||||
disagreement. whoosh-compat derives it from the registry's own FieldSpec
|
||||
(kind plus fast) without ever consulting the index schema, so it fires
|
||||
whenever a non-fast field of a kind that cannot answer "exists" is asked
|
||||
to: for us that is only the JSON fields, which field_descriptors() builds
|
||||
non-fast on purpose. "notes:*" and the five other spellings of it are
|
||||
ordinary user error that no operator action can clear, so they get the
|
||||
400 without the alert.
|
||||
"""
|
||||
d = e.diagnostic
|
||||
if d.cause is Cause.INTERNAL:
|
||||
raise e
|
||||
if (
|
||||
d.cause is Cause.MISCONFIGURED
|
||||
and d.kind is not DiagnosticKind.EXISTS_REQUIRES_FAST
|
||||
):
|
||||
logger.error(
|
||||
"Search index misconfiguration for field %s (%s): %s",
|
||||
d.field,
|
||||
d.kind.name,
|
||||
d.message,
|
||||
)
|
||||
return SearchQueryError(_user_facing_emit_message(d))
|
||||
|
||||
|
||||
def _has_cjk(text: str) -> bool:
|
||||
"""Return True if text contains any CJK characters."""
|
||||
return bool(_CJK_RE.search(text))
|
||||
@@ -103,36 +37,14 @@ def _has_cjk(text: str) -> bool:
|
||||
def extract_cjk_text(text: str) -> str:
|
||||
"""Join the CJK runs in ``text`` for indexing into bigram (char-ngram) fields.
|
||||
|
||||
Mirrors the query side, which extracts the CJK runs of whatever it is
|
||||
about to search for (the raw string in simple modes, the parsed query's
|
||||
free-text tokens in query mode): only CJK runs are ever searched against
|
||||
the bigram fields, so only CJK runs are worth indexing there. Latin text
|
||||
fed to a character-bigram field is never matched and only bloats the
|
||||
Mirrors the query side (``_build_cjk_query``): only CJK runs are ever searched
|
||||
against the bigram fields, so only CJK runs are worth indexing there. Latin
|
||||
text fed to a character-bigram field is never matched and only bloats the
|
||||
index and slows indexing/merge. Returns "" when there is no CJK text.
|
||||
"""
|
||||
return " ".join(_CJK_RE.findall(text))
|
||||
|
||||
|
||||
def _parse_cjk_text(
|
||||
index: tantivy.Index,
|
||||
cjk_text: str,
|
||||
fields: list[str],
|
||||
) -> tantivy.Query | None:
|
||||
"""Parse a plain CJK run string against ``fields``, or None if it won't parse."""
|
||||
try:
|
||||
return index.parse_query(cjk_text, fields)
|
||||
except Exception:
|
||||
# Broad on purpose, unlike _try_parse_fuzzy_query's narrower
|
||||
# ValueError: cjk_text isn't filtered to a guaranteed-safe token
|
||||
# set the way the fuzzy blend's word string is, so the exact
|
||||
# failure mode tantivy could raise here isn't pinned down.
|
||||
logger.debug(
|
||||
"Skipping CJK search clause: could not parse CJK text: %r",
|
||||
cjk_text,
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _build_cjk_query(
|
||||
index: tantivy.Index,
|
||||
raw_query: str,
|
||||
@@ -140,259 +52,91 @@ def _build_cjk_query(
|
||||
) -> tantivy.Query | None:
|
||||
"""Build a bigram-field query from the CJK runs in ``raw_query``.
|
||||
|
||||
For the simple (TEXT/TITLE) modes, whose input is plain text and carries
|
||||
no query grammar to respect. Only the CJK character runs are extracted, so
|
||||
a stray ``field:`` prefix or ``-``/``+`` in the input can neither leak
|
||||
field semantics nor fail the parse, and no Latin token reaches the
|
||||
character-bigram matcher (where it would produce spurious matches against
|
||||
unrelated Latin text). Returns None when there is no CJK text or the parse
|
||||
fails.
|
||||
Only the CJK character runs are extracted and parsed; ASCII field prefixes,
|
||||
boolean operators and date keywords are discarded. This keeps the CJK clause
|
||||
plain-text and consistent across query/simple modes (no leaked ``field:``
|
||||
semantics, no parse failures from spaced ``-``/``+``), and avoids feeding
|
||||
Latin tokens into the character-bigram matcher (which would produce spurious
|
||||
matches against unrelated Latin text). Returns None when there is no CJK
|
||||
text or the parse fails.
|
||||
"""
|
||||
cjk_text = extract_cjk_text(raw_query)
|
||||
cjk_text = " ".join(_CJK_RE.findall(raw_query))
|
||||
if not cjk_text:
|
||||
return None
|
||||
return _parse_cjk_text(index, cjk_text, fields)
|
||||
|
||||
|
||||
def _build_ast_cjk_query(
|
||||
index: tantivy.Index,
|
||||
ast: wc.ast.Node,
|
||||
registry: wc.FieldRegistry,
|
||||
) -> tantivy.Query | None:
|
||||
"""Build the bigram clause of a QUERY-mode search from the parsed AST.
|
||||
|
||||
Same discipline as the fuzzy clause (see _try_parse_fuzzy_query): the CJK
|
||||
runs come from whoosh_compat's ``free_text_tokens`` over the parsed tree,
|
||||
never from the raw query string, so a term the user negated or restricted
|
||||
to a field outside the default search fields contributes nothing, instead
|
||||
of resurfacing as a top-level clause matching every bigram field.
|
||||
|
||||
``free_text_tokens`` reports no field of its own, so the tokens are
|
||||
collected one default field at a time: a bare term, which the parser has
|
||||
already copied onto every default field, is therefore searched across
|
||||
every bigram field, while ``title:東京`` reaches ``bigram_title`` alone.
|
||||
Fields whose CJK text is identical (the bare-term case) share a single
|
||||
parse over all of their bigram fields at once.
|
||||
|
||||
Raw (``analyzed=False``) tokens are used because the bigram fields have
|
||||
their own character-ngram analyzer: the default fields' word analyzers
|
||||
have no useful say over a CJK run, and running them first would only
|
||||
risk dropping it (remove_long) before the run is ever extracted.
|
||||
Returns None when the query has no CJK free text.
|
||||
"""
|
||||
fields_by_text: dict[str, list[str]] = {}
|
||||
for field, bigram_field in _CJK_BIGRAM_FIELDS.items():
|
||||
tokens = wc.free_text_tokens(
|
||||
ast,
|
||||
registry=registry,
|
||||
fields=[field],
|
||||
analyzed=False,
|
||||
)
|
||||
cjk_text = extract_cjk_text(" ".join(tokens))
|
||||
if cjk_text:
|
||||
fields_by_text.setdefault(cjk_text, []).append(bigram_field)
|
||||
|
||||
clauses: list[tuple[tantivy.Occur, tantivy.Query]] = [
|
||||
(tantivy.Occur.Should, query)
|
||||
for cjk_text, bigram_fields in fields_by_text.items()
|
||||
if (query := _parse_cjk_text(index, cjk_text, bigram_fields)) is not None
|
||||
]
|
||||
return _any_of(clauses) if clauses else None
|
||||
|
||||
|
||||
# A joined fuzzy word string must stay plain words: it goes back through
|
||||
# tantivy's own query parser, and the raw query text the clause collects
|
||||
# routinely carries characters that parser reads as grammar (a colon, a
|
||||
# bracket, a quote, a leading -). Each token is cut into its word runs and
|
||||
# only those are kept, so no field syntax, pattern, range or grouping can
|
||||
# reach the parser. Cutting rather than dropping the whole token is what
|
||||
# keeps ordinary hyphenated, dotted and quoted input ("COVID-19",
|
||||
# "hello@example.com", "tax reports") contributing to the clause at all.
|
||||
_WORD_RUN_RE = regex.compile(r"\w+")
|
||||
|
||||
# The one piece of tantivy grammar that survives the cut: its boolean
|
||||
# keywords are themselves word runs. Only these exact spellings are
|
||||
# grammar there ("And"/"and" are ordinary terms), so lowercasing exactly
|
||||
# these turns them back into the ordinary terms the field analyzer used to
|
||||
# make of them, before the clause switched to raw text. Left alone, a
|
||||
# quoted phrase would silently restructure the clause ("tax AND reports"
|
||||
# becoming a conjunction) or fail to parse and drop it entirely
|
||||
# ("tax AND", or "IN" anywhere).
|
||||
#
|
||||
# Only these words are touched: tantivy lowercases query terms with the
|
||||
# field's own analyzer, and doing it ourselves first is not always the
|
||||
# same operation (Python folds a final sigma to a different letter than
|
||||
# tantivy does, and turns Turkish 'İ' into a sequence tantivy then splits
|
||||
# in two), which would search for terms the index does not contain.
|
||||
_TANTIVY_KEYWORDS: Final[frozenset[str]] = frozenset({"AND", "OR", "NOT", "IN"})
|
||||
|
||||
|
||||
def _try_parse_fuzzy_query(
|
||||
index: tantivy.Index,
|
||||
ast: wc.ast.Node,
|
||||
registry: wc.FieldRegistry,
|
||||
) -> tantivy.Query | None:
|
||||
"""Build the fuzzy blend clause from the parsed query's free-text
|
||||
words, or None if it has none.
|
||||
|
||||
The clause is built by handing tantivy's own query parser a plain
|
||||
word string (there's no clean AST-level fuzzy equivalent to
|
||||
whoosh-compat's parse tree, and fuzzy matching was always an
|
||||
approximate, secondary, 0.1-boosted clause). The words come from
|
||||
whoosh_compat's ``free_text_tokens`` over the already-parsed AST,
|
||||
never from the raw query string: raw whoosh grammar (date keywords,
|
||||
``[2005 to 2009]`` ranges, bracket-class wildcards) is not tantivy
|
||||
syntax, and feeding it here used to knock the fuzzy clause out for
|
||||
the whole query the moment any such construct appeared alongside a
|
||||
typo'd word. The helper also keeps excluded terms out: a ``NOT``'d
|
||||
word must not resurface through the fuzzy clause.
|
||||
|
||||
Chosen trade-off: a term explicitly fielded on one of the default
|
||||
search fields (``correspondent:acme``) contributes its text to the
|
||||
word string UNFIELDED, so the fuzzy clause searches it across all
|
||||
default fields rather than just the one the user named. That is
|
||||
recall-only widening on a secondary 0.1-boosted clause the score
|
||||
threshold already disciplines, accepted in exchange for never feeding
|
||||
field syntax to tantivy's parser. What the word string guarantees is
|
||||
exactly that: no field prefix, pattern, range, grouping or quoting
|
||||
survives, and the boolean keywords that do survive (they are word
|
||||
runs) are lowercased into ordinary terms; see _TANTIVY_KEYWORDS.
|
||||
|
||||
The words are the query's RAW text, not the analyzer's output
|
||||
(``analyzed=False``), because ``index.parse_query`` analyzes whatever
|
||||
it is given and analysis is not idempotent: ``universities`` stems to
|
||||
``univers``, and handing that back stems it again to ``univ``, a term
|
||||
the index does not contain. ``prefix=True`` hid this as over-broad
|
||||
matching (``univ`` also prefixes ``unicycle``) rather than as no
|
||||
matches at all. Raw text is untokenized, which is why it is cut into
|
||||
word runs above rather than taken whole.
|
||||
|
||||
The ValueError guard stays as insurance (the word string is plain
|
||||
tokens, so tantivy accepting it is expected, not assumed): on a parse
|
||||
failure the fuzzy clause is skipped and the exact/CJK clauses stand,
|
||||
rather than the whole query failing.
|
||||
"""
|
||||
tokens = wc.free_text_tokens(
|
||||
ast,
|
||||
registry=registry,
|
||||
fields=_DEFAULT_SEARCH_FIELDS,
|
||||
analyzed=False,
|
||||
)
|
||||
words = list(
|
||||
dict.fromkeys(
|
||||
word.lower() if word in _TANTIVY_KEYWORDS else word
|
||||
for token in tokens
|
||||
for word in _WORD_RUN_RE.findall(token)
|
||||
),
|
||||
)
|
||||
if not words:
|
||||
return None
|
||||
fuzzy_text = " ".join(words)
|
||||
try:
|
||||
return index.parse_query(
|
||||
fuzzy_text,
|
||||
_DEFAULT_SEARCH_FIELDS,
|
||||
field_boosts=_FIELD_BOOSTS,
|
||||
fuzzy_fields={f: (True, 1, True) for f in _DEFAULT_SEARCH_FIELDS},
|
||||
)
|
||||
except ValueError:
|
||||
logger.debug(
|
||||
"Skipping fuzzy search clause: token string is not valid "
|
||||
"tantivy query syntax: %r",
|
||||
fuzzy_text,
|
||||
)
|
||||
return index.parse_query(cjk_text, fields)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
_DEFAULT_SEARCH_FIELDS: Final[list[str]] = [
|
||||
def build_permission_filter(
|
||||
schema: tantivy.Schema,
|
||||
user: AbstractBaseUser,
|
||||
viewer_group_ids: Iterable[int] = (),
|
||||
) -> tantivy.Query:
|
||||
"""
|
||||
Build a query filter for user document permissions.
|
||||
|
||||
Creates a query that matches only documents visible to the specified user
|
||||
according to paperless-ngx permission rules:
|
||||
- Public documents (no owner) are visible to all users
|
||||
- Private documents are visible to their owner
|
||||
- Documents explicitly shared with the user are visible
|
||||
- Documents shared with one of the user's current groups are visible
|
||||
|
||||
Args:
|
||||
schema: Tantivy schema for field validation
|
||||
user: User to check permissions for
|
||||
viewer_group_ids: Current group memberships for the user
|
||||
|
||||
Returns:
|
||||
Tantivy query that filters results to visible documents
|
||||
"""
|
||||
owner_any = tantivy.Query.exists_query("owner_id")
|
||||
no_owner = tantivy.Query.boolean_query(
|
||||
[
|
||||
(tantivy.Occur.Must, tantivy.Query.all_query()),
|
||||
(tantivy.Occur.MustNot, owner_any),
|
||||
],
|
||||
)
|
||||
owned = tantivy.Query.term_query(schema, "owner_id", user.pk)
|
||||
shared = tantivy.Query.term_query(schema, "viewer_id", user.pk)
|
||||
group_shared = [
|
||||
tantivy.Query.term_query(schema, "viewer_group_id", group_id)
|
||||
for group_id in viewer_group_ids
|
||||
]
|
||||
return tantivy.Query.disjunction_max_query(
|
||||
[no_owner, owned, shared, *group_shared],
|
||||
)
|
||||
|
||||
|
||||
DEFAULT_SEARCH_FIELDS = [
|
||||
"title",
|
||||
"content",
|
||||
"correspondent",
|
||||
"document_type",
|
||||
"tag",
|
||||
]
|
||||
_SIMPLE_SEARCH_FIELDS: Final[list[str]] = ["simple_title", "simple_content"]
|
||||
_TITLE_SEARCH_FIELDS: Final[list[str]] = ["simple_title"]
|
||||
# The bigram (character-ngram) companion of each default search field.
|
||||
_CJK_BIGRAM_FIELDS: Final[dict[str, str]] = {
|
||||
field: f"bigram_{field}" for field in _DEFAULT_SEARCH_FIELDS
|
||||
}
|
||||
SIMPLE_SEARCH_FIELDS = ["simple_title", "simple_content"]
|
||||
TITLE_SEARCH_FIELDS = ["simple_title"]
|
||||
_CJK_ALL_FIELDS: Final[list[str]] = [
|
||||
"bigram_content",
|
||||
"bigram_title",
|
||||
"bigram_correspondent",
|
||||
"bigram_document_type",
|
||||
"bigram_tag",
|
||||
]
|
||||
_CJK_CONTENT_FIELDS: Final[list[str]] = ["bigram_content"]
|
||||
_CJK_TITLE_FIELDS: Final[list[str]] = ["bigram_title"]
|
||||
_FIELD_BOOSTS = {"title": 2.0}
|
||||
_SIMPLE_FIELD_BOOSTS = {"simple_title": 2.0}
|
||||
|
||||
|
||||
class _ConjunctiveNegations(wc.ast.Visitor[tuple["wc.ast.Node", ...]]):
|
||||
"""Collect the subtrees an AST excludes from every document it matches.
|
||||
|
||||
A negation reached through ``And``/``AndNot``/``Require`` (and through
|
||||
the required half of an ``AndMaybe``) constrains the whole query, so it
|
||||
can be re-stated above the blend. ``Or`` is deliberately not descended
|
||||
into: in ``invoice OR NOT secret`` the negation is one branch's own
|
||||
condition, and hoisting it would throw away documents the other branch
|
||||
matches. Nor is a collected subtree descended into, since a negation
|
||||
inside a negation is not an exclusion.
|
||||
|
||||
Node types with no negation to contribute (every leaf, ``Or``) fall
|
||||
through to ``generic_visit``.
|
||||
"""
|
||||
|
||||
def generic_visit(self, node: wc.ast.Node) -> tuple[wc.ast.Node, ...]:
|
||||
return ()
|
||||
|
||||
def visit_not(self, node: wc.ast.Not) -> tuple[wc.ast.Node, ...]:
|
||||
return (node.child,)
|
||||
|
||||
def visit_andnot(self, node: wc.ast.AndNot) -> tuple[wc.ast.Node, ...]:
|
||||
return (*self.visit(node.positive), node.negative)
|
||||
|
||||
def visit_and(self, node: wc.ast.And) -> tuple[wc.ast.Node, ...]:
|
||||
return tuple(
|
||||
negation for child in node.children for negation in self.visit(child)
|
||||
)
|
||||
|
||||
def visit_boosted(self, node: wc.ast.Boosted) -> tuple[wc.ast.Node, ...]:
|
||||
return self.visit(node.child)
|
||||
|
||||
def visit_andmaybe(self, node: wc.ast.AndMaybe) -> tuple[wc.ast.Node, ...]:
|
||||
return self.visit(node.required)
|
||||
|
||||
def visit_require(self, node: wc.ast.Require) -> tuple[wc.ast.Node, ...]:
|
||||
return (*self.visit(node.scored), *self.visit(node.filter_only))
|
||||
|
||||
|
||||
def _negation_clauses(
|
||||
index: tantivy.Index,
|
||||
ast: wc.ast.Node,
|
||||
registry: wc.FieldRegistry,
|
||||
) -> list[tuple[tantivy.Occur, tantivy.Query]]:
|
||||
"""MustNot clauses for everything ``ast`` excludes conjunctively.
|
||||
|
||||
Each excluded subtree is emitted as its own positive query and attached
|
||||
with ``MustNot``, rather than emitting a negative query and hoping
|
||||
tantivy accepts a bare one.
|
||||
"""
|
||||
try:
|
||||
return [
|
||||
(
|
||||
tantivy.Occur.MustNot,
|
||||
tantivy_emit(negation, index=index, registry=registry),
|
||||
)
|
||||
for negation in _ConjunctiveNegations().visit(ast)
|
||||
]
|
||||
except QueryError as e:
|
||||
raise _map_emit_error(e) from e
|
||||
|
||||
|
||||
def _any_of(clauses: list[tuple[tantivy.Occur, tantivy.Query]]) -> tantivy.Query:
|
||||
"""Collapse a clause list: none -> empty, one -> itself (no wasted
|
||||
single-clause boolean_query wrapping), many -> boolean_query(clauses)."""
|
||||
if not clauses:
|
||||
return tantivy.Query.empty_query()
|
||||
if len(clauses) == 1:
|
||||
return clauses[0][1]
|
||||
return tantivy.Query.boolean_query(clauses)
|
||||
def _simple_query_tokens(raw_query: str) -> list[str]:
|
||||
# Tokenize and fold via the same analyzer used to index simple_title /
|
||||
# simple_content, so query terms fold identically to the indexed terms
|
||||
# (single source of truth for ASCII folding).
|
||||
return simple_search_tokens(raw_query)
|
||||
|
||||
|
||||
def _build_simple_token_query(
|
||||
@@ -424,7 +168,9 @@ def _build_simple_token_query(
|
||||
query = tantivy.Query.boost_query(query, boost)
|
||||
field_queries.append((tantivy.Occur.Should, query))
|
||||
|
||||
return _any_of(field_queries)
|
||||
if len(field_queries) == 1:
|
||||
return field_queries[0][1]
|
||||
return tantivy.Query.boolean_query(field_queries)
|
||||
|
||||
|
||||
def parse_user_query(
|
||||
@@ -433,53 +179,52 @@ def parse_user_query(
|
||||
tz: tzinfo,
|
||||
) -> tantivy.Query:
|
||||
"""
|
||||
Parse user query through whoosh-compat, then blend in fuzzy/CJK clauses.
|
||||
Parse user query through the complete preprocessing pipeline.
|
||||
|
||||
1. wc.parse() against the shared FieldRegistry (whoosh grammar -> AST).
|
||||
Bare notes:/custom_fields: prefixes resolve to their default subpath
|
||||
(notes.note:/custom_fields.value:) directly in the registry, via
|
||||
each JSON field's SubpathSpec(default=True).
|
||||
2. Any diagnostics (bad dates/numbers) map to SearchQueryError subclasses
|
||||
and raise — the view returns HTTP 400 with every offending field
|
||||
listed, not just the first.
|
||||
3. emit() turns the AST into a tantivy.Query directly (no string
|
||||
round-trip). A QueryError is routed by its Diagnostic's Cause
|
||||
(_map_emit_error): a construct that parses but can't execute against
|
||||
tantivy (e.g. a text-field range) is a 400, a registry/schema
|
||||
mismatch is logged and a 400, and an INTERNAL defect is re-raised.
|
||||
4. Optional fuzzy blend (ADVANCED_FUZZY_SEARCH_THRESHOLD) builds a
|
||||
plain word string from the parsed AST's free-text tokens
|
||||
(whoosh_compat.free_text_tokens) and feeds THAT to
|
||||
index.parse_query — never raw_query, whose whoosh grammar (date
|
||||
keywords, bracket-class wildcards, etc.) tantivy's parser rejects,
|
||||
which used to silently knock the fuzzy clause out of any mixed
|
||||
query (see _try_parse_fuzzy_query).
|
||||
5. Optional CJK bigram clause, built from the same parsed AST for the
|
||||
same reason (see _build_ast_cjk_query): a CJK term the query negated
|
||||
or fielded must not resurface through it.
|
||||
6. When any optional clause was added, the query's conjunctive
|
||||
exclusions are restated as MustNot above the blend
|
||||
(_negation_clauses): a clause built from positive terms cannot
|
||||
express them, and as a bare Should it would undo them.
|
||||
Transforms the raw user query through multiple stages:
|
||||
1. Date keyword rewriting (today → ISO 8601 ranges)
|
||||
2. Query normalization (comma expansion, whitespace cleanup)
|
||||
3. Tantivy parsing with field boosts
|
||||
4. Optional fuzzy query blending (if ADVANCED_FUZZY_SEARCH_THRESHOLD set)
|
||||
|
||||
Args:
|
||||
index: Tantivy index with registered tokenizers
|
||||
raw_query: Original user query string
|
||||
tz: Timezone for date boundary calculations
|
||||
|
||||
Returns:
|
||||
Parsed Tantivy query ready for execution
|
||||
|
||||
Note:
|
||||
When ADVANCED_FUZZY_SEARCH_THRESHOLD is configured, adds a low-priority
|
||||
fuzzy query as a Should clause (0.1 boost) to catch approximate matches
|
||||
while keeping exact matches ranked higher. The threshold value is applied
|
||||
as a post-search score filter, not during query construction.
|
||||
"""
|
||||
registry = get_field_registry(settings.SEARCH_LANGUAGE)
|
||||
result = wc.parse(
|
||||
raw_query,
|
||||
registry=registry,
|
||||
default_fields=_DEFAULT_SEARCH_FIELDS,
|
||||
field_boosts=_FIELD_BOOSTS,
|
||||
tz=tz,
|
||||
)
|
||||
if result.diagnostics:
|
||||
raise _diagnostics_to_error(result.diagnostics)
|
||||
|
||||
try:
|
||||
exact = tantivy_emit(result.ast, index=index, registry=registry)
|
||||
except QueryError as e:
|
||||
raise _map_emit_error(e) from e
|
||||
query_str = translate_query(raw_query, tz)
|
||||
except SearchQueryError:
|
||||
# Intentional, user-fixable error (e.g. an unparsable date). Propagate so
|
||||
# the view can return a 400 with a helpful message rather than falling
|
||||
# back to the raw (still-invalid) query.
|
||||
raise
|
||||
except Exception: # pragma: no cover - defensive
|
||||
logger.warning("Query translation failed; using raw query", exc_info=True)
|
||||
query_str = raw_query
|
||||
|
||||
exact = index.parse_query(
|
||||
query_str,
|
||||
DEFAULT_SEARCH_FIELDS,
|
||||
field_boosts=_FIELD_BOOSTS,
|
||||
)
|
||||
|
||||
# The standard analyzer keeps a whitespace-free CJK run as a single token,
|
||||
# so substring queries can't match content/title (and long runs are dropped
|
||||
# by remove_long). Route CJK queries to the bigram fields, whose ngram
|
||||
# tokenizer indexes overlapping 2-grams for substring matching.
|
||||
cjk_query = (
|
||||
_build_ast_cjk_query(index, result.ast, registry)
|
||||
_build_cjk_query(index, raw_query, _CJK_ALL_FIELDS)
|
||||
if _has_cjk(raw_query)
|
||||
else None
|
||||
)
|
||||
@@ -490,65 +235,22 @@ def parse_user_query(
|
||||
|
||||
threshold = settings.ADVANCED_FUZZY_SEARCH_THRESHOLD
|
||||
if threshold is not None:
|
||||
fuzzy = _try_parse_fuzzy_query(index, result.ast, registry)
|
||||
if fuzzy is not None:
|
||||
clauses.append(
|
||||
(tantivy.Occur.Should, tantivy.Query.boost_query(fuzzy, 0.1)),
|
||||
)
|
||||
fuzzy = index.parse_query(
|
||||
query_str,
|
||||
DEFAULT_SEARCH_FIELDS,
|
||||
field_boosts=_FIELD_BOOSTS,
|
||||
# (prefix=True, distance=1, transposition_cost_one=True) — edit-distance fuzziness
|
||||
fuzzy_fields={f: (True, 1, True) for f in DEFAULT_SEARCH_FIELDS},
|
||||
)
|
||||
# 0.1 boost keeps fuzzy hits ranked below exact matches (intentional)
|
||||
clauses.append((tantivy.Occur.Should, tantivy.Query.boost_query(fuzzy, 0.1)))
|
||||
|
||||
if cjk_query is not None:
|
||||
clauses.append((tantivy.Occur.Should, cjk_query))
|
||||
|
||||
if len(clauses) == 1:
|
||||
return exact
|
||||
# The fuzzy and CJK clauses are built from positive terms only, so as
|
||||
# plain Shoulds beside the exact clause they re-admit exactly the
|
||||
# documents the query excluded. Restate the exclusions once, above the
|
||||
# whole blend. Redundant against the exact clause, which already
|
||||
# carries them, but idempotently so.
|
||||
negations = _negation_clauses(index, result.ast, registry)
|
||||
if not negations:
|
||||
return _any_of(clauses)
|
||||
return tantivy.Query.boolean_query(
|
||||
[(tantivy.Occur.Must, _any_of(clauses)), *negations],
|
||||
)
|
||||
|
||||
|
||||
# The three whoosh-compat kinds for a wildcard on a field that cannot
|
||||
# carry one. d.field_kind supplies the discriminator, so naming the field's
|
||||
# type needs no second trip through the registry.
|
||||
_PATTERN_ON_KINDS: Final = frozenset(
|
||||
{
|
||||
DiagnosticKind.PATTERN_ON_NUMERIC,
|
||||
DiagnosticKind.PATTERN_ON_BOOLEAN_EXISTS,
|
||||
DiagnosticKind.PATTERN_ON_SUBPATH,
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def _diagnostics_to_error(diagnostics: tuple[Diagnostic, ...]) -> SearchQueryError:
|
||||
errors = [_single_diagnostic_to_error(d) for d in diagnostics]
|
||||
return errors[0] if len(errors) == 1 else MultipleSearchQueryErrors(errors)
|
||||
|
||||
|
||||
def _single_diagnostic_to_error(d: Diagnostic) -> SearchQueryError:
|
||||
# d.field is a FieldRef, not a str: str(d.field) gives the canonical
|
||||
# dotted name (an aliased query, e.g. type:, reports document_type).
|
||||
field_name = str(d.field) if d.field is not None else None
|
||||
if d.kind is DiagnosticKind.BAD_DATE:
|
||||
return InvalidDateQuery(field_name, d.raw_value)
|
||||
if d.kind is DiagnosticKind.BAD_NUMBER:
|
||||
return InvalidNumberQuery(field_name, d.raw_value)
|
||||
if d.kind is DiagnosticKind.TOO_DEEP:
|
||||
return SearchQueryError("The search query is nested too deeply.")
|
||||
if d.kind in _PATTERN_ON_KINDS:
|
||||
kind_label = f" ({d.field_kind.name.lower()})" if d.field_kind else ""
|
||||
return SearchQueryError(
|
||||
f"Wildcard patterns are not supported for field "
|
||||
f"{field_name!r}{kind_label}.",
|
||||
)
|
||||
logger.warning("Unmapped parse diagnostic %s: %s", d.kind, d.message)
|
||||
return SearchQueryError("The search query could not be executed.")
|
||||
return tantivy.Query.boolean_query(clauses)
|
||||
|
||||
|
||||
def parse_simple_query(
|
||||
@@ -566,7 +268,7 @@ def parse_simple_query(
|
||||
CJK substrings the simple analyzer can't (long whitespace-free runs are
|
||||
dropped by remove_long).
|
||||
"""
|
||||
tokens = simple_search_tokens(raw_query)
|
||||
tokens = _simple_query_tokens(raw_query)
|
||||
|
||||
clauses: list[tuple[tantivy.Occur, tantivy.Query]] = []
|
||||
if tokens:
|
||||
@@ -589,14 +291,23 @@ def parse_simple_query(
|
||||
)
|
||||
for token in tokens
|
||||
]
|
||||
clauses.append((tantivy.Occur.Should, _any_of(token_queries)))
|
||||
simple_query = (
|
||||
token_queries[0][1]
|
||||
if len(token_queries) == 1
|
||||
else tantivy.Query.boolean_query(token_queries)
|
||||
)
|
||||
clauses.append((tantivy.Occur.Should, simple_query))
|
||||
|
||||
if cjk_fields and _has_cjk(raw_query):
|
||||
cjk_q = _build_cjk_query(index, raw_query, cjk_fields)
|
||||
if cjk_q is not None:
|
||||
clauses.append((tantivy.Occur.Should, cjk_q))
|
||||
|
||||
return _any_of(clauses)
|
||||
if not clauses:
|
||||
return tantivy.Query.empty_query()
|
||||
if len(clauses) == 1:
|
||||
return clauses[0][1]
|
||||
return tantivy.Query.boolean_query(clauses)
|
||||
|
||||
|
||||
def parse_simple_text_highlight_query(
|
||||
@@ -611,21 +322,13 @@ def parse_simple_text_highlight_query(
|
||||
|
||||
# Strip Tantivy operator chars before tokenizing: this is a plain-text
|
||||
# highlight query, not a structured boolean query, so +/- are separators.
|
||||
tokens = simple_search_tokens(
|
||||
tokens = _simple_query_tokens(
|
||||
regex.sub(r"[-+]", " ", raw_query, timeout=_REGEX_TIMEOUT),
|
||||
)
|
||||
if not tokens:
|
||||
return tantivy.Query.empty_query()
|
||||
|
||||
# Quote each token as its own phrase, escaping backslashes and embedded
|
||||
# quotes. simple search tokens can carry arbitrary Tantivy syntax
|
||||
# characters (`"`, `:`, `(`, `[`, `/`, ...) that the query-string parser
|
||||
# would otherwise interpret as query grammar rather than literal text.
|
||||
quoted_tokens = [
|
||||
'"' + token.replace("\\", "\\\\").replace('"', '\\"') + '"' for token in tokens
|
||||
]
|
||||
|
||||
return index.parse_query(" ".join(quoted_tokens), ["content"])
|
||||
return index.parse_query(" ".join(tokens), ["content"])
|
||||
|
||||
|
||||
def parse_simple_text_query(
|
||||
@@ -639,7 +342,7 @@ def parse_simple_text_query(
|
||||
return parse_simple_query(
|
||||
index,
|
||||
raw_query,
|
||||
_SIMPLE_SEARCH_FIELDS,
|
||||
SIMPLE_SEARCH_FIELDS,
|
||||
cjk_fields=_CJK_CONTENT_FIELDS,
|
||||
)
|
||||
|
||||
@@ -655,6 +358,6 @@ def parse_simple_title_query(
|
||||
return parse_simple_query(
|
||||
index,
|
||||
raw_query,
|
||||
_TITLE_SEARCH_FIELDS,
|
||||
TITLE_SEARCH_FIELDS,
|
||||
cjk_fields=_CJK_TITLE_FIELDS,
|
||||
)
|
||||
|
||||
@@ -1,91 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import dataclasses
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from whoosh_compat import FieldKind
|
||||
from whoosh_compat import FieldRegistry
|
||||
|
||||
from documents.search._fields import PUBLIC_FIELDS
|
||||
from documents.search._tokenizer import ascii_fold
|
||||
from documents.search._tokenizer import paperless_text_analyzer
|
||||
from documents.search._tokenizer import stem_pattern_text
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from whoosh_compat import PatternNormalizer
|
||||
|
||||
_registry_cache: dict[str | None, FieldRegistry] = {}
|
||||
|
||||
|
||||
def _identity_analyzer(text: str) -> list[str]:
|
||||
"""Analyzer for KEYWORD fields indexed with the raw tokenizer (no splitting)."""
|
||||
return [text]
|
||||
|
||||
|
||||
def _fold_normalizer(text: str) -> str:
|
||||
"""Wildcard/regex literal-run normalizer for fields indexed without stemming."""
|
||||
return ascii_fold(text.lower())
|
||||
|
||||
|
||||
def _make_pattern_normalizer(language: str | None) -> PatternNormalizer:
|
||||
"""Build the wildcard/regex literal-run normalizer for a search language."""
|
||||
|
||||
def _pattern_normalizer(text: str) -> tuple[str, ...]:
|
||||
"""Normalize a literal run into the forms a term may match.
|
||||
|
||||
TEXT index terms go through lowercase -> ascii_fold -> stem, so a
|
||||
pattern that skips stemming can never match one: "invoice*" would look
|
||||
for a term starting with "invoice" while the index holds "invoic". The
|
||||
run is therefore offered stemmed as well. KEYWORD fields are indexed
|
||||
raw and get _fold_normalizer instead, so their patterns stay literal.
|
||||
|
||||
Both forms are returned, as alternatives, because neither is a prefix
|
||||
of the other in general: English stemming substitutes as well as
|
||||
truncates ("copy" -> "copi"), so the stem alone loses the compounds
|
||||
the typed run reaches ("copyright") while the typed run alone loses
|
||||
the inflections the stem reaches ("copies"). whoosh-compat ORs the
|
||||
alternatives per literal run and deduplicates them, so a run the
|
||||
stemmer leaves alone costs exactly the one branch it did before.
|
||||
|
||||
Inside a bracket class the emitter calls this once per character and
|
||||
uses the answer only if it is a single one-character form; two forms
|
||||
there leave the character as typed. A stemmer does not change a lone
|
||||
character, so the two forms deduplicate to one and the class body is
|
||||
folded as before.
|
||||
"""
|
||||
folded = ascii_fold(text.lower())
|
||||
stemmed = stem_pattern_text(folded, language)
|
||||
return (folded, stemmed)
|
||||
|
||||
return _pattern_normalizer
|
||||
|
||||
|
||||
def get_field_registry(language: str | None) -> FieldRegistry:
|
||||
"""Build (or return the cached) FieldRegistry for the given search language.
|
||||
|
||||
Cached keyed by language, rebuilt on the same trigger register_tokenizers()
|
||||
uses (settings.SEARCH_LANGUAGE change) — a fresh call with a new language
|
||||
builds and caches a new registry rather than mutating the old one.
|
||||
"""
|
||||
if language in _registry_cache:
|
||||
return _registry_cache[language]
|
||||
|
||||
text_analyzer = paperless_text_analyzer(language).analyze
|
||||
pattern_normalizer = _make_pattern_normalizer(language)
|
||||
|
||||
specs = [
|
||||
dataclasses.replace(
|
||||
field,
|
||||
analyzer=_identity_analyzer
|
||||
if field.kind is FieldKind.KEYWORD
|
||||
else text_analyzer,
|
||||
pattern_normalizer=_fold_normalizer
|
||||
if field.kind is FieldKind.KEYWORD
|
||||
else pattern_normalizer,
|
||||
)
|
||||
for field in PUBLIC_FIELDS
|
||||
]
|
||||
|
||||
registry = FieldRegistry(specs)
|
||||
_registry_cache[language] = registry
|
||||
return registry
|
||||
+83
-238
@@ -1,19 +1,14 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import shutil
|
||||
from typing import TYPE_CHECKING
|
||||
from typing import Final
|
||||
from typing import NamedTuple
|
||||
from typing import cast
|
||||
|
||||
import tantivy
|
||||
from django.conf import settings
|
||||
from whoosh_compat import FieldKind
|
||||
|
||||
from documents.search._fields import PUBLIC_FIELDS
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
@@ -21,201 +16,7 @@ if TYPE_CHECKING:
|
||||
logger = logging.getLogger("paperless.search")
|
||||
|
||||
# v1 - Initial tantivy schema format
|
||||
# v2 - build_schema() derived from PUBLIC_FIELDS, changing the field declaration
|
||||
# order, and the write-only correspondent/document_type/storage_path/tag id
|
||||
# columns dropped. tantivy compares schemas by ordered field list, so an
|
||||
# index built by v1 rejects every write against the v2 schema.
|
||||
SCHEMA_VERSION: Final[int] = 2
|
||||
|
||||
|
||||
class FieldDescriptor(NamedTuple):
|
||||
"""One tantivy field, in declaration order.
|
||||
|
||||
The descriptor vocabulary is paperless', not tantivy-py's: it is both the
|
||||
input to the SchemaBuilder and the input to schema_fingerprint(), so the
|
||||
persisted fingerprint cannot move under a tantivy-py upgrade.
|
||||
"""
|
||||
|
||||
name: str
|
||||
kind: str
|
||||
stored: bool
|
||||
indexed: bool
|
||||
fast: bool
|
||||
tokenizer: str | None
|
||||
|
||||
|
||||
def _public_field_descriptors() -> list[FieldDescriptor]:
|
||||
"""Descriptors for the query-visible fields declared in PUBLIC_FIELDS."""
|
||||
descriptors: list[FieldDescriptor] = []
|
||||
for field in PUBLIC_FIELDS:
|
||||
if field.kind is FieldKind.TEXT:
|
||||
descriptors.append(
|
||||
FieldDescriptor(
|
||||
field.name,
|
||||
"text",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
)
|
||||
elif field.kind is FieldKind.KEYWORD:
|
||||
descriptors.append(
|
||||
FieldDescriptor(
|
||||
field.name,
|
||||
"text",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="raw",
|
||||
),
|
||||
)
|
||||
elif field.kind is FieldKind.U64:
|
||||
descriptors.append(
|
||||
FieldDescriptor(
|
||||
field.name,
|
||||
"u64",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=field.fast,
|
||||
tokenizer=None,
|
||||
),
|
||||
)
|
||||
elif field.kind in (FieldKind.DATE, FieldKind.DATETIME):
|
||||
descriptors.append(
|
||||
FieldDescriptor(
|
||||
field.name,
|
||||
"date",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=field.fast,
|
||||
tokenizer=None,
|
||||
),
|
||||
)
|
||||
elif field.kind is FieldKind.JSON:
|
||||
descriptors.append(
|
||||
FieldDescriptor(
|
||||
field.name,
|
||||
"json",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
)
|
||||
if field.name == "notes":
|
||||
# Plain-text companion for snippet generation — tantivy's
|
||||
# SnippetGenerator does not support JSON fields. Schema-only,
|
||||
# no query-syntax meaning, not in PUBLIC_FIELDS.
|
||||
descriptors.append(
|
||||
FieldDescriptor(
|
||||
"notes_text",
|
||||
"text",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="paperless_text",
|
||||
),
|
||||
)
|
||||
return descriptors
|
||||
|
||||
|
||||
def field_descriptors() -> list[FieldDescriptor]:
|
||||
"""Every field of the document index, in the order tantivy declares them.
|
||||
|
||||
tantivy compares schemas by *ordered* field list, so the order here is
|
||||
part of the on-disk contract: schema_fingerprint() hashes it and
|
||||
needs_rebuild() acts on the result.
|
||||
"""
|
||||
return [
|
||||
FieldDescriptor(
|
||||
"id",
|
||||
"u64",
|
||||
stored=True,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer=None,
|
||||
),
|
||||
*_public_field_descriptors(),
|
||||
# Shadow sort fields - fast, not stored
|
||||
*(
|
||||
FieldDescriptor(
|
||||
name,
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer="simple_analyzer",
|
||||
)
|
||||
for name in ("title_sort", "correspondent_sort", "type_sort")
|
||||
),
|
||||
# CJK support - not stored, indexed only
|
||||
*(
|
||||
FieldDescriptor(
|
||||
name,
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="bigram_analyzer",
|
||||
)
|
||||
for name in (
|
||||
"bigram_content",
|
||||
"bigram_title",
|
||||
"bigram_correspondent",
|
||||
"bigram_document_type",
|
||||
"bigram_tag",
|
||||
)
|
||||
),
|
||||
# Simple substring search support for title/content - not stored,
|
||||
# indexed only
|
||||
*(
|
||||
FieldDescriptor(
|
||||
name,
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="simple_search_analyzer",
|
||||
)
|
||||
for name in ("simple_title", "simple_content")
|
||||
),
|
||||
# Autocomplete prefix scan via terms_with_prefix, which walks the
|
||||
# field's term dictionary - so the field must be indexed (term dict),
|
||||
# not stored. The stored value is never read back, so storing it only
|
||||
# wastes space.
|
||||
FieldDescriptor(
|
||||
"autocomplete_word",
|
||||
"text",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=False,
|
||||
tokenizer="raw",
|
||||
),
|
||||
# Permission filter columns, read by build_permission_filter.
|
||||
*(
|
||||
FieldDescriptor(
|
||||
name,
|
||||
"u64",
|
||||
stored=False,
|
||||
indexed=True,
|
||||
fast=True,
|
||||
tokenizer=None,
|
||||
)
|
||||
for name in ("owner_id", "viewer_id", "viewer_group_id")
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
def schema_fingerprint() -> str:
|
||||
"""Hash of the field descriptors, stamped into .index_settings.json.
|
||||
|
||||
Changes whenever a field is added, removed, retyped, re-optioned or
|
||||
reordered, so an index built from a different schema shape is detected
|
||||
even when SCHEMA_VERSION was not bumped.
|
||||
"""
|
||||
payload = json.dumps([list(descriptor) for descriptor in field_descriptors()])
|
||||
return hashlib.blake2b(payload.encode()).hexdigest()
|
||||
SCHEMA_VERSION: Final[int] = 1
|
||||
|
||||
|
||||
def build_schema() -> tantivy.Schema:
|
||||
@@ -231,37 +32,85 @@ def build_schema() -> tantivy.Schema:
|
||||
"""
|
||||
sb = tantivy.SchemaBuilder()
|
||||
|
||||
for descriptor in field_descriptors():
|
||||
if descriptor.kind == "text":
|
||||
sb.add_text_field(
|
||||
descriptor.name,
|
||||
stored=descriptor.stored,
|
||||
fast=descriptor.fast,
|
||||
tokenizer_name=cast("str", descriptor.tokenizer),
|
||||
)
|
||||
elif descriptor.kind == "json":
|
||||
sb.add_json_field(
|
||||
descriptor.name,
|
||||
stored=descriptor.stored,
|
||||
fast=descriptor.fast,
|
||||
tokenizer_name=cast("str", descriptor.tokenizer),
|
||||
)
|
||||
elif descriptor.kind == "u64":
|
||||
sb.add_unsigned_field(
|
||||
descriptor.name,
|
||||
stored=descriptor.stored,
|
||||
indexed=descriptor.indexed,
|
||||
fast=descriptor.fast,
|
||||
)
|
||||
elif descriptor.kind == "date":
|
||||
sb.add_date_field(
|
||||
descriptor.name,
|
||||
stored=descriptor.stored,
|
||||
indexed=descriptor.indexed,
|
||||
fast=descriptor.fast,
|
||||
)
|
||||
else:
|
||||
raise ValueError(f"Unknown schema field kind: {descriptor.kind}")
|
||||
sb.add_unsigned_field("id", stored=True, indexed=True, fast=True)
|
||||
sb.add_text_field("checksum", stored=True, tokenizer_name="raw")
|
||||
|
||||
for field in (
|
||||
"title",
|
||||
"correspondent",
|
||||
"document_type",
|
||||
"storage_path",
|
||||
"original_filename",
|
||||
"content",
|
||||
):
|
||||
sb.add_text_field(field, stored=True, tokenizer_name="paperless_text")
|
||||
|
||||
# Shadow sort fields - fast, not stored/indexed
|
||||
for field in ("title_sort", "correspondent_sort", "type_sort"):
|
||||
sb.add_text_field(
|
||||
field,
|
||||
stored=False,
|
||||
tokenizer_name="simple_analyzer",
|
||||
fast=True,
|
||||
)
|
||||
|
||||
# CJK support - not stored, indexed only
|
||||
sb.add_text_field("bigram_content", stored=False, tokenizer_name="bigram_analyzer")
|
||||
sb.add_text_field("bigram_title", stored=False, tokenizer_name="bigram_analyzer")
|
||||
sb.add_text_field(
|
||||
"bigram_correspondent",
|
||||
stored=False,
|
||||
tokenizer_name="bigram_analyzer",
|
||||
)
|
||||
sb.add_text_field(
|
||||
"bigram_document_type",
|
||||
stored=False,
|
||||
tokenizer_name="bigram_analyzer",
|
||||
)
|
||||
sb.add_text_field("bigram_tag", stored=False, tokenizer_name="bigram_analyzer")
|
||||
|
||||
# Simple substring search support for title/content - not stored, indexed only
|
||||
sb.add_text_field(
|
||||
"simple_title",
|
||||
stored=False,
|
||||
tokenizer_name="simple_search_analyzer",
|
||||
)
|
||||
sb.add_text_field(
|
||||
"simple_content",
|
||||
stored=False,
|
||||
tokenizer_name="simple_search_analyzer",
|
||||
)
|
||||
|
||||
# Autocomplete prefix scan via terms_with_prefix, which walks the field's
|
||||
# term dictionary - so the field must be indexed (term dict), not stored.
|
||||
# The stored value is never read back, so storing it only wastes space.
|
||||
sb.add_text_field("autocomplete_word", stored=False, tokenizer_name="raw")
|
||||
|
||||
sb.add_text_field("tag", stored=True, tokenizer_name="paperless_text")
|
||||
|
||||
# JSON fields — structured queries: notes.user:alice, custom_fields.name:invoice
|
||||
sb.add_json_field("notes", stored=True, tokenizer_name="paperless_text")
|
||||
# Plain-text companion for notes — tantivy's SnippetGenerator does not support
|
||||
# JSON fields, so highlights require a text field with the same content.
|
||||
sb.add_text_field("notes_text", stored=True, tokenizer_name="paperless_text")
|
||||
sb.add_json_field("custom_fields", stored=True, tokenizer_name="paperless_text")
|
||||
|
||||
for field in (
|
||||
"correspondent_id",
|
||||
"document_type_id",
|
||||
"storage_path_id",
|
||||
"tag_id",
|
||||
"owner_id",
|
||||
"viewer_id",
|
||||
"viewer_group_id",
|
||||
):
|
||||
sb.add_unsigned_field(field, stored=False, indexed=True, fast=True)
|
||||
|
||||
for field in ("created", "modified", "added"):
|
||||
sb.add_date_field(field, stored=True, indexed=True, fast=True)
|
||||
|
||||
for field in ("asn", "page_count", "num_notes"):
|
||||
sb.add_unsigned_field(field, stored=True, indexed=True, fast=True)
|
||||
|
||||
return sb.build()
|
||||
|
||||
@@ -270,9 +119,9 @@ def needs_rebuild(index_dir: Path) -> bool:
|
||||
"""
|
||||
Check if the search index needs rebuilding.
|
||||
|
||||
Reads .index_settings.json to compare the stored schema version, search
|
||||
language and schema fingerprint against the current configuration. Returns
|
||||
True if the file is missing, unparsable, or any value mismatches.
|
||||
Reads .index_settings.json to compare the stored schema version and
|
||||
search language against the current configuration. Returns True if the
|
||||
file is missing, unparsable, or either value mismatches.
|
||||
|
||||
Args:
|
||||
index_dir: Path to the search index directory
|
||||
@@ -291,9 +140,6 @@ def needs_rebuild(index_dir: Path) -> bool:
|
||||
if "language" not in data or data["language"] != settings.SEARCH_LANGUAGE:
|
||||
logger.info("Search index language changed - rebuilding.")
|
||||
return True
|
||||
if data.get("schema_fingerprint") != schema_fingerprint():
|
||||
logger.info("Search index schema fingerprint mismatch - rebuilding.")
|
||||
return True
|
||||
except ValueError:
|
||||
return True
|
||||
return False
|
||||
@@ -324,7 +170,6 @@ def _write_sentinels(index_dir: Path) -> None:
|
||||
{
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"language": settings.SEARCH_LANGUAGE,
|
||||
"schema_fingerprint": schema_fingerprint(),
|
||||
},
|
||||
),
|
||||
)
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from functools import cache
|
||||
from typing import Final
|
||||
|
||||
import tantivy
|
||||
@@ -72,7 +71,7 @@ def register_tokenizers(index: tantivy.Index, language: str | None) -> None:
|
||||
use fast=True and Tantivy requires fast-field tokenizers to exist
|
||||
even for documents that omit those fields.
|
||||
"""
|
||||
index.register_tokenizer("paperless_text", paperless_text_analyzer(language))
|
||||
index.register_tokenizer("paperless_text", _paperless_text(language))
|
||||
index.register_tokenizer("simple_analyzer", _simple_analyzer())
|
||||
index.register_tokenizer("bigram_analyzer", _bigram_analyzer())
|
||||
index.register_tokenizer("simple_search_analyzer", _simple_search_analyzer())
|
||||
@@ -80,7 +79,7 @@ def register_tokenizers(index: tantivy.Index, language: str | None) -> None:
|
||||
index.register_fast_field_tokenizer("simple_analyzer", _simple_analyzer())
|
||||
|
||||
|
||||
def paperless_text_analyzer(language: str | None) -> tantivy.TextAnalyzer:
|
||||
def _paperless_text(language: str | None) -> tantivy.TextAnalyzer:
|
||||
"""Main full-text tokenizer for content, title, etc: simple -> remove_long(129) -> lowercase -> ascii_fold [-> stemmer]"""
|
||||
builder = (
|
||||
tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.simple())
|
||||
@@ -101,54 +100,6 @@ def paperless_text_analyzer(language: str | None) -> tantivy.TextAnalyzer:
|
||||
return builder.build()
|
||||
|
||||
|
||||
@cache
|
||||
def _pattern_stemmer(language: str | None) -> tantivy.TextAnalyzer | None:
|
||||
"""The stemming tail of paperless_text_analyzer, over a whole literal run.
|
||||
|
||||
Same language gate and same Snowball stemmer paperless_text_analyzer
|
||||
applies at index time, so query patterns follow SEARCH_LANGUAGE. Returns
|
||||
None when that gate disables stemming; paperless_text_analyzer already
|
||||
warns about an unsupported language, so this stays quiet.
|
||||
|
||||
The raw tokenizer keeps the run whole (a wildcard literal is a fragment,
|
||||
not necessarily a word), and remove_long is kept so an over-long run is
|
||||
treated the same way the index treats it.
|
||||
"""
|
||||
if not language:
|
||||
return None
|
||||
tantivy_lang = _LANGUAGE_MAP.get(language.lower())
|
||||
if tantivy_lang is None:
|
||||
return None
|
||||
return (
|
||||
tantivy.TextAnalyzerBuilder(tantivy.Tokenizer.raw())
|
||||
.filter(tantivy.Filter.remove_long(_TOKEN_REMOVE_LONG_LIMIT))
|
||||
.filter(tantivy.Filter.stemmer(tantivy_lang))
|
||||
.build()
|
||||
)
|
||||
|
||||
|
||||
def stem_pattern_text(text: str, language: str | None) -> str:
|
||||
"""Stem an already lowercased/ascii-folded run the way index terms are.
|
||||
|
||||
Returns text unchanged when stemming is disabled for language, and also
|
||||
when the stem step does not yield exactly one token: remove_long drops a run
|
||||
past the length limit, leaving no stem to substitute. Falling back to the
|
||||
text as typed is the safe direction for a pattern prefix, since it can only
|
||||
be as narrow as it was before stemming was considered.
|
||||
|
||||
The raw tokenizer emits one token whatever the input and the stemmer is
|
||||
1-to-1, so only the zero-token case can fire today; the guard covers both
|
||||
counts so a tokenizer change cannot turn this into an IndexError.
|
||||
"""
|
||||
analyzer = _pattern_stemmer(language)
|
||||
if analyzer is None:
|
||||
return text
|
||||
tokens = analyzer.analyze(text)
|
||||
if len(tokens) != 1:
|
||||
return text
|
||||
return tokens[0]
|
||||
|
||||
|
||||
def _simple_analyzer() -> tantivy.TextAnalyzer:
|
||||
"""Tokenizer for shadow sort fields (title_sort, correspondent_sort, type_sort): simple -> lowercase -> ascii_fold."""
|
||||
return (
|
||||
|
||||
@@ -0,0 +1,610 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from datetime import UTC
|
||||
from datetime import datetime
|
||||
from datetime import timedelta
|
||||
from typing import TYPE_CHECKING
|
||||
from typing import TypeAlias
|
||||
|
||||
import regex
|
||||
from dateutil.relativedelta import relativedelta
|
||||
|
||||
from documents.search._dates import _DATE_KEYWORDS
|
||||
from documents.search._dates import _DATE_ONLY_FIELDS
|
||||
from documents.search._dates import _date_only_range
|
||||
from documents.search._dates import _datetime_range
|
||||
from documents.search._dates import _field_range_from_dates
|
||||
from documents.search._dates import _fmt
|
||||
from documents.search._dates import _precision_bounds
|
||||
from documents.search._dates import _utc_bounds_for_field
|
||||
|
||||
# Compiled regex that matches any known multi-word (or single-word) date keyword
|
||||
# at the start of a match position, longest alternatives first so "previous week"
|
||||
# wins over a hypothetical shorter "previous".
|
||||
_KEYWORD_VALUE_RE = regex.compile(
|
||||
"|".join(sorted((regex.escape(k) for k in _DATE_KEYWORDS), key=len, reverse=True)),
|
||||
regex.IGNORECASE,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from datetime import tzinfo
|
||||
|
||||
# TODO: this module translates date queries into Tantivy *string* syntax, which
|
||||
# forces a workaround for something Tantivy's string parser cannot express on
|
||||
# date fields: open-ended ranges use far-past/far-future string sentinels
|
||||
# (OPEN_LO/OPEN_HI). These can be replaced with a real tantivy.Query object
|
||||
# (Query.range_query(..., None) for open bounds) once tantivy-py accepts Python
|
||||
# datetimes in range_query/term_query on Date fields. That support exists on
|
||||
# tantivy-py master (PRs #655 + #666) but postdates the pinned 0.26.0 wheel, so
|
||||
# it is blocked only on a published release > 0.26.0 and a dependency bump.
|
||||
# (Unparsable dates now raise InvalidDateQuery -> HTTP 400 rather than using a
|
||||
# no-match string sentinel.)
|
||||
|
||||
# Fields that store exact, non-analyzed comma-joined tokens in the index and so
|
||||
# need explicit comma->AND expansion (Whoosh KEYWORD(commas=True) set).
|
||||
MULTI_VALUE_FIELDS = frozenset({"tag", "tag_id", "viewer_id"})
|
||||
|
||||
# Date fields whose values/ranges get rewritten to RFC3339 Tantivy ranges.
|
||||
DATE_FIELDS = frozenset({"created", "modified", "added"})
|
||||
|
||||
# Field aliases: Whoosh (v2) field names that were renamed in the Tantivy schema.
|
||||
# Preserved here so v2 queries using the old names continue to work without 400
|
||||
# errors instead of silently failing. Applied by _render to non-date field tokens.
|
||||
FIELD_ALIASES: dict[str, str] = {
|
||||
"type": "document_type",
|
||||
"type_id": "document_type_id",
|
||||
"path": "storage_path",
|
||||
"path_id": "storage_path_id",
|
||||
}
|
||||
|
||||
# Known schema fields: a comma immediately followed by ``<known>:`` is a clause
|
||||
# separator. Restricting to known fields prevents URL-like ``http:`` misfires.
|
||||
KNOWN_FIELDS = frozenset(
|
||||
{
|
||||
"title",
|
||||
"content",
|
||||
"correspondent",
|
||||
"document_type",
|
||||
"type", # v2 alias -> document_type
|
||||
"storage_path",
|
||||
"path", # v2 alias -> storage_path
|
||||
"tag",
|
||||
"tag_id",
|
||||
"correspondent_id",
|
||||
"document_type_id",
|
||||
"type_id", # v2 alias -> document_type_id
|
||||
"storage_path_id",
|
||||
"path_id", # v2 alias -> storage_path_id
|
||||
"owner_id",
|
||||
"viewer_id",
|
||||
"asn",
|
||||
"page_count",
|
||||
"num_notes",
|
||||
"created",
|
||||
"modified",
|
||||
"added",
|
||||
"original_filename",
|
||||
"checksum",
|
||||
"notes",
|
||||
"custom_fields",
|
||||
},
|
||||
)
|
||||
|
||||
_FIELD_RE = regex.compile(r"(?P<field>\w+):")
|
||||
|
||||
# Matches the TO separator inside a range bracket. Handles three forms:
|
||||
# middle: "lo TO hi" (either lo or hi may be empty)
|
||||
# trailing: "lo TO" (open upper bound)
|
||||
# leading: "TO hi" (open lower bound)
|
||||
# Bounds MAY contain internal spaces (e.g. "-7 days"), so we use .*? / .+?
|
||||
# and split on the whitespace-delimited " TO " / " to " separator.
|
||||
_RANGE_RE = regex.compile(
|
||||
r"^\s*(?P<lo>.*?)\s+[Tt][Oo]\s+(?P<hi>.+?)\s*$"
|
||||
r"|"
|
||||
r"^\s*(?P<lo2>.+?)\s+[Tt][Oo]\s*$"
|
||||
r"|"
|
||||
r"^\s*[Tt][Oo]\s+(?P<hi2>.+?)\s*$",
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class FieldValue:
|
||||
field: str
|
||||
value: str
|
||||
|
||||
|
||||
# Produced by the comma-resolution pass (not by scan()).
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class FieldValueList:
|
||||
field: str
|
||||
values: tuple[str, ...]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class FieldRange:
|
||||
field: str
|
||||
open: str
|
||||
lo: str
|
||||
hi: str
|
||||
close: str
|
||||
|
||||
|
||||
# Produced by the comma-resolution pass (not by scan()).
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Comma:
|
||||
pass
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Passthrough:
|
||||
raw: str
|
||||
|
||||
|
||||
Token: TypeAlias = FieldValue | FieldValueList | FieldRange | Comma | Passthrough
|
||||
|
||||
_CLOSE: dict[str, str] = {"[": "]", "{": "}"}
|
||||
|
||||
|
||||
def scan(query: str) -> list[Token]:
|
||||
"""
|
||||
Tokenize a raw query into date/comma-aware tokens, leaving everything else
|
||||
as verbatim ``Passthrough`` runs. Non-recursive: finds the first matching
|
||||
close bracket/quote. Nested brackets are not valid Tantivy range syntax and
|
||||
pass through verbatim on mismatch.
|
||||
"""
|
||||
tokens: list[Token] = []
|
||||
buf: list[str] = [] # accumulates passthrough chars
|
||||
i, n = 0, len(query)
|
||||
while i < n:
|
||||
matched = _match_field_token(query, i)
|
||||
if matched is None:
|
||||
buf.append(query[i])
|
||||
i += 1
|
||||
continue
|
||||
token, i = matched
|
||||
if buf and buf[-1] == ",":
|
||||
buf.pop()
|
||||
_flush(buf, tokens)
|
||||
tokens.append(Comma())
|
||||
else:
|
||||
_flush(buf, tokens)
|
||||
tokens.append(token)
|
||||
i = _maybe_comma(query, i, tokens)
|
||||
_flush(buf, tokens)
|
||||
return tokens
|
||||
|
||||
|
||||
def _flush(buf: list[str], tokens: list[Token]) -> None:
|
||||
"""Emit any accumulated passthrough characters as a single token."""
|
||||
if buf:
|
||||
tokens.append(Passthrough("".join(buf)))
|
||||
buf.clear()
|
||||
|
||||
|
||||
def _at_word_boundary(query: str, i: int) -> bool:
|
||||
"""A field token may begin only at the start or after a non-word character."""
|
||||
return i == 0 or not (query[i - 1].isalnum() or query[i - 1] == "_")
|
||||
|
||||
|
||||
def _match_field_token(query: str, i: int) -> tuple[Token, int] | None:
|
||||
"""
|
||||
If a known ``field:`` token starts at ``i``, consume it and return
|
||||
``(token, end_index)``; otherwise return None so the caller treats the
|
||||
character as passthrough. Handles both ``field:[range]`` and ``field:value``,
|
||||
and returns None when the range/value cannot be consumed.
|
||||
"""
|
||||
m = _FIELD_RE.match(query, i)
|
||||
if m is None or m.group("field") not in KNOWN_FIELDS:
|
||||
return None
|
||||
if not _at_word_boundary(query, i):
|
||||
return None
|
||||
field = m.group("field")
|
||||
j = m.end()
|
||||
if j < len(query) and query[j] in "[{":
|
||||
return _consume_range(query, j, field)
|
||||
consumed = _consume_field_value(query, field, j)
|
||||
if consumed is None:
|
||||
return None
|
||||
value, end = consumed
|
||||
return FieldValue(field, value), end
|
||||
|
||||
|
||||
def _consume_field_value(query: str, field: str, start: int) -> tuple[str, int] | None:
|
||||
"""
|
||||
Consume a field value starting at ``start``: a multi-word date keyword phrase
|
||||
(date fields only), or a bare/quoted value, then absorb any comma-joined
|
||||
continuation that is not a clause separator. ``resolve_commas`` later splits a
|
||||
multi-value field's joined value into a ``FieldValueList``; for other fields
|
||||
the comma stays literal.
|
||||
"""
|
||||
n = len(query)
|
||||
consumed = None
|
||||
if field in DATE_FIELDS:
|
||||
km = _KEYWORD_VALUE_RE.match(query, start)
|
||||
if km is not None and (km.end() >= n or query[km.end()] in " \t),"):
|
||||
consumed = (km.group(0), km.end())
|
||||
if consumed is None:
|
||||
consumed = _consume_value(query, start)
|
||||
if consumed is None:
|
||||
return None
|
||||
value, k = consumed
|
||||
while k < n and query[k] == ",":
|
||||
if _looks_like_known_field(query, k + 1):
|
||||
break # clause separator: left for _maybe_comma to emit a Comma()
|
||||
more = _consume_value(query, k + 1)
|
||||
if more is None:
|
||||
break
|
||||
value = f"{value},{more[0]}"
|
||||
k = more[1]
|
||||
return value, k
|
||||
|
||||
|
||||
def _consume_range(
|
||||
query: str,
|
||||
start: int,
|
||||
field: str,
|
||||
) -> tuple[FieldRange, int] | None:
|
||||
"""Consume ``[lo TO hi]`` / ``{lo TO hi}`` from ``start`` (the bracket)."""
|
||||
open_br = query[start]
|
||||
close_br = _CLOSE[open_br]
|
||||
end = query.find(close_br, start + 1)
|
||||
if end == -1:
|
||||
return None
|
||||
inner = query[start + 1 : end]
|
||||
m = _RANGE_RE.match(inner)
|
||||
if m is not None:
|
||||
if m.group("lo") is not None or m.group("hi") is not None:
|
||||
# Middle form: "lo TO hi" (either may be empty string)
|
||||
lo = (m.group("lo") or "").strip()
|
||||
hi = (m.group("hi") or "").strip()
|
||||
elif m.group("lo2") is not None:
|
||||
# Trailing form: "lo TO"
|
||||
lo = m.group("lo2").strip()
|
||||
hi = ""
|
||||
else:
|
||||
# Leading form: "TO hi"
|
||||
lo = ""
|
||||
hi = (m.group("hi2") or "").strip()
|
||||
else:
|
||||
lo, hi = inner.strip(), ""
|
||||
return FieldRange(field, open_br, lo, hi, close_br), end + 1
|
||||
|
||||
|
||||
def _consume_value(query: str, start: int) -> tuple[str, int] | None:
|
||||
"""Consume a bare or quoted field value from ``start``, stopping at comma."""
|
||||
n = len(query)
|
||||
if start >= n or query[start] in " \t":
|
||||
return None
|
||||
if query[start] in "\"'":
|
||||
quote = query[start]
|
||||
end = query.find(quote, start + 1)
|
||||
if end == -1:
|
||||
return None
|
||||
return query[start : end + 1], end + 1
|
||||
j = start
|
||||
while j < n and query[j] not in " \t),":
|
||||
j += 1
|
||||
return query[start:j], j
|
||||
|
||||
|
||||
def _looks_like_known_field(query: str, pos: int) -> bool:
|
||||
"""True if a known ``field:`` token starts at ``pos``."""
|
||||
m = _FIELD_RE.match(query, pos)
|
||||
return bool(m and m.group("field") in KNOWN_FIELDS)
|
||||
|
||||
|
||||
def _maybe_comma(query: str, i: int, tokens: list) -> int:
|
||||
"""If a clause-separator comma follows at ``i``, emit ``Comma()`` and advance."""
|
||||
if i < len(query) and query[i] == "," and _looks_like_known_field(query, i + 1):
|
||||
tokens.append(Comma())
|
||||
return i + 1
|
||||
return i
|
||||
|
||||
|
||||
def resolve_commas(tokens: list) -> list:
|
||||
"""
|
||||
Collapse value-list commas into ``FieldValueList`` and keep clause-separator
|
||||
commas as ``Comma``. (Clause-sep commas are already emitted by ``scan`` via
|
||||
the value-stop logic; this pass folds value-lists.)
|
||||
"""
|
||||
out: list = []
|
||||
for tok in tokens:
|
||||
if (
|
||||
isinstance(tok, FieldValue)
|
||||
and tok.field in MULTI_VALUE_FIELDS
|
||||
and "," in tok.value
|
||||
):
|
||||
values = tuple(v for v in tok.value.split(",") if v)
|
||||
out.append(FieldValueList(tok.field, values))
|
||||
else:
|
||||
out.append(tok)
|
||||
return out
|
||||
|
||||
|
||||
class SearchQueryError(ValueError):
|
||||
"""
|
||||
Base for user-fixable search query errors.
|
||||
|
||||
Carries a message safe to surface to the user (no internal details). The view
|
||||
layer catches this and returns an HTTP 400, so any future subclass (unknown
|
||||
field, malformed range, wrapped parser errors) gets the same treatment.
|
||||
"""
|
||||
|
||||
|
||||
class InvalidDateQuery(SearchQueryError):
|
||||
"""Raised when a date field value or range bound cannot be parsed."""
|
||||
|
||||
def __init__(self, field: str, value: str) -> None:
|
||||
self.field = field
|
||||
self.value = value
|
||||
super().__init__(f"Invalid date value {value!r} for field {field!r}.")
|
||||
|
||||
|
||||
_DIGITS_RE = regex.compile(r"^\d{4}(?:\d{2}){0,2}$")
|
||||
_ISO_RE = regex.compile(r"^\d{4}(?:-\d{2}(?:-\d{2})?)?$")
|
||||
|
||||
|
||||
def translate_scalar(field: str, value: str, tz: tzinfo) -> str:
|
||||
"""Translate a bare date-field value to a Tantivy range string."""
|
||||
bare = value.strip("\"'").lower()
|
||||
if bare in _DATE_KEYWORDS:
|
||||
if field in _DATE_ONLY_FIELDS:
|
||||
return f"{field}:{_date_only_range(bare, tz)}"
|
||||
return f"{field}:{_datetime_range(bare, tz)}"
|
||||
digits = value.replace("-", "")
|
||||
if _DIGITS_RE.match(value) or _ISO_RE.match(value):
|
||||
bounds = _precision_bounds(digits)
|
||||
if bounds is None:
|
||||
raise InvalidDateQuery(field, value)
|
||||
return _field_range_from_dates(field, bounds[0], bounds[1], tz)
|
||||
if regex.fullmatch(r"\d{14}", value):
|
||||
try:
|
||||
dt = datetime(
|
||||
int(value[0:4]),
|
||||
int(value[4:6]),
|
||||
int(value[6:8]),
|
||||
int(value[8:10]),
|
||||
int(value[10:12]),
|
||||
int(value[12:14]),
|
||||
tzinfo=UTC,
|
||||
)
|
||||
except ValueError:
|
||||
raise InvalidDateQuery(field, value) from None
|
||||
iso = _fmt(dt)
|
||||
return f"{field}:[{iso} TO {iso}]"
|
||||
# Unrecognized shape -> tell the user their date is malformed rather than
|
||||
# silently matching nothing or emitting invalid Tantivy syntax.
|
||||
raise InvalidDateQuery(field, value)
|
||||
|
||||
|
||||
# Open-bound sentinels for date ranges. These far-past/far-future strings allow
|
||||
# open-ended ranges to be expressed as Tantivy string queries until tantivy-py
|
||||
# exposes Query.range_query(..., None) on Date fields (see module TODO).
|
||||
OPEN_LO = "0001-01-01T00:00:00Z"
|
||||
OPEN_HI = "9999-12-31T23:59:59Z"
|
||||
|
||||
|
||||
# Matches compact now-offset tokens like now-7d, now+1h, now-30m.
|
||||
_NOW_COMPACT_RE = regex.compile(
|
||||
r"^now(?P<sign>[+-])(?P<n>\d+)(?P<unit>[dhm])$",
|
||||
regex.IGNORECASE,
|
||||
)
|
||||
|
||||
# Matches "±N <unit>" Whoosh-style offsets (e.g. -7 days, -1 week, +3 hours).
|
||||
# Whoosh's own date parser (qparser.dateparse.PlusMinus) additionally accepted
|
||||
# abbreviated unit spellings (e.g. "yrs", "yr", "y", "mos", "wks", "hrs", "mins",
|
||||
# "secs"); saved views/searches created under the old Whoosh backend can still
|
||||
# contain those tokens (e.g. "-999yrs"), so they are accepted here too and
|
||||
# normalized to a canonical unit via _UNIT_ALIASES below.
|
||||
_NOW_SPACED_RE = regex.compile(
|
||||
r"^(?P<sign>[+-])(?P<n>\d+)\s*"
|
||||
r"(?P<unit>years|year|yrs|yr|ys|y"
|
||||
r"|months|month|mons|mon|mos|mo"
|
||||
r"|weeks|week|wks|wk|ws|w"
|
||||
r"|days|day|dys|dy|ds|d"
|
||||
r"|hours|hour|hrs|hr|hs|h"
|
||||
r"|minutes|minute|mins|min|ms|m"
|
||||
r"|seconds|second|secs|sec|s)$",
|
||||
regex.IGNORECASE,
|
||||
)
|
||||
|
||||
# Maps every accepted unit spelling (including Whoosh-era abbreviations) to the
|
||||
# canonical unit name used as a key into the delta map in _resolve_relative_bound.
|
||||
_UNIT_ALIASES: dict[str, str] = {
|
||||
alias: canonical
|
||||
for canonical, aliases in {
|
||||
"year": ("years", "year", "yrs", "yr", "ys", "y"),
|
||||
"month": ("months", "month", "mons", "mon", "mos", "mo"),
|
||||
"week": ("weeks", "week", "wks", "wk", "ws", "w"),
|
||||
"day": ("days", "day", "dys", "dy", "ds", "d"),
|
||||
"hour": ("hours", "hour", "hrs", "hr", "hs", "h"),
|
||||
"minute": ("minutes", "minute", "mins", "min", "ms", "m"),
|
||||
"second": ("seconds", "second", "secs", "sec", "s"),
|
||||
}.items()
|
||||
for alias in aliases
|
||||
}
|
||||
|
||||
|
||||
def _resolve_relative_bound(token: str) -> datetime | None:
|
||||
"""
|
||||
Resolve a relative bound token to an exact UTC instant, or return None.
|
||||
|
||||
Supported forms:
|
||||
- ``now`` -> current UTC instant
|
||||
- ``now+/-<n>d/h/m`` -> now +/- timedelta (d=days, h=hours, m=minutes)
|
||||
- ``±N <unit>`` -> now +/- delta; month/year use relativedelta;
|
||||
unit also accepts Whoosh-era abbreviations
|
||||
(e.g. "yrs", "mos", "wks", "hrs", "mins", "secs")
|
||||
"""
|
||||
stripped = token.strip()
|
||||
low = stripped.lower()
|
||||
now = datetime.now(UTC)
|
||||
|
||||
if low == "now":
|
||||
return now
|
||||
|
||||
m = _NOW_COMPACT_RE.match(stripped)
|
||||
if m:
|
||||
sign = 1 if m.group("sign") == "+" else -1
|
||||
n = int(m.group("n"))
|
||||
unit = m.group("unit").lower()
|
||||
delta = (
|
||||
sign
|
||||
* {
|
||||
"d": timedelta(days=n),
|
||||
"h": timedelta(hours=n),
|
||||
"m": timedelta(minutes=n),
|
||||
}[unit]
|
||||
)
|
||||
return now + delta
|
||||
|
||||
m = _NOW_SPACED_RE.match(stripped)
|
||||
if m:
|
||||
sign = 1 if m.group("sign") == "+" else -1
|
||||
n = int(m.group("n"))
|
||||
unit = _UNIT_ALIASES[m.group("unit").lower()]
|
||||
delta_map: dict[str, timedelta | relativedelta] = {
|
||||
"second": timedelta(seconds=n),
|
||||
"minute": timedelta(minutes=n),
|
||||
"hour": timedelta(hours=n),
|
||||
"day": timedelta(days=n),
|
||||
"week": timedelta(weeks=n),
|
||||
"month": relativedelta(months=n),
|
||||
"year": relativedelta(years=n),
|
||||
}
|
||||
return now - delta_map[unit] if sign == -1 else now + delta_map[unit]
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def _bound_datetimes(
|
||||
field: str,
|
||||
token: str,
|
||||
tz: tzinfo,
|
||||
) -> tuple[datetime, datetime] | None:
|
||||
"""
|
||||
Return (floor_dt, ceil_dt) UTC datetimes for a single range bound token, or
|
||||
None if the token is unparsable. ``now`` and relative offsets resolve to the
|
||||
current instant (floor == ceil == that instant; no day-flooring).
|
||||
"""
|
||||
token = token.strip()
|
||||
|
||||
# Try relative/now forms first (before stripping hyphens which would mangle them).
|
||||
rel = _resolve_relative_bound(token)
|
||||
if rel is not None:
|
||||
return rel, rel
|
||||
|
||||
# Full ISO datetime token (contains "T"): parse directly and return an exact
|
||||
# instant (floor == ceil). Python 3.11+ datetime.fromisoformat accepts trailing Z.
|
||||
if "T" in token:
|
||||
try:
|
||||
dt = datetime.fromisoformat(token)
|
||||
# Ensure timezone-aware UTC result.
|
||||
dt = dt.replace(tzinfo=UTC) if dt.tzinfo is None else dt.astimezone(UTC)
|
||||
return dt, dt
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
digits = token.replace("-", "")
|
||||
bounds = _precision_bounds(digits)
|
||||
if bounds is None:
|
||||
return None
|
||||
start, end = bounds
|
||||
return _utc_bounds_for_field(field, start, end, tz)
|
||||
|
||||
|
||||
def _render(tok: Token, tz: tzinfo) -> str:
|
||||
"""Render a single token back to a Tantivy query string fragment."""
|
||||
if isinstance(tok, Passthrough):
|
||||
return tok.raw
|
||||
if isinstance(tok, Comma):
|
||||
return " AND "
|
||||
if isinstance(tok, FieldValueList):
|
||||
field = FIELD_ALIASES.get(tok.field, tok.field)
|
||||
return " AND ".join(f"{field}:{v}" for v in tok.values)
|
||||
if isinstance(tok, FieldValue):
|
||||
field = FIELD_ALIASES.get(tok.field, tok.field)
|
||||
if field in DATE_FIELDS:
|
||||
return translate_scalar(field, tok.value, tz)
|
||||
return f"{field}:{tok.value}"
|
||||
if isinstance(tok, FieldRange):
|
||||
field = FIELD_ALIASES.get(tok.field, tok.field)
|
||||
if field in DATE_FIELDS:
|
||||
return translate_range(field, tok.lo, tok.hi, tz)
|
||||
return f"{field}:{tok.open}{tok.lo} TO {tok.hi}{tok.close}"
|
||||
return "" # pragma: no cover
|
||||
|
||||
|
||||
# Post-render operator normalization patterns: collapse repeated whitespace and
|
||||
# strip spaced/trailing Tantivy boolean operators that would otherwise be invalid.
|
||||
_MULTI_SPACE_RE = regex.compile(r" {2,}")
|
||||
_TRAILING_OP_RE = regex.compile(r"\s+[-+]+\s*$")
|
||||
_SPACED_OP_RE = regex.compile(r"\s+[-+]\s+")
|
||||
|
||||
|
||||
def _normalize_operators(text: str) -> str:
|
||||
"""
|
||||
Collapse multiple spaces, strip trailing dangling operators, and replace
|
||||
spaced operators (`` - `` / `` + ``) with a single space.
|
||||
|
||||
Applied only to Passthrough fragments (the rendered output is scanned for
|
||||
operator artifacts outside bracketed ranges) via a post-render pass on the
|
||||
full rendered string. This preserves date ranges (``[... TO ...]``) verbatim
|
||||
while cleaning natural-language separators in the surrounding text.
|
||||
"""
|
||||
text = _MULTI_SPACE_RE.sub(" ", text)
|
||||
text = _TRAILING_OP_RE.sub("", text).strip()
|
||||
text = _SPACED_OP_RE.sub(" ", text).strip()
|
||||
return text
|
||||
|
||||
|
||||
def translate_query(raw: str, tz: tzinfo) -> str:
|
||||
"""Translate a raw Whoosh-style query into Tantivy-compatible syntax."""
|
||||
tokens = resolve_commas(scan(raw))
|
||||
rendered = "".join(_render(t, tz) for t in tokens)
|
||||
return _normalize_operators(rendered)
|
||||
|
||||
|
||||
def translate_range(field: str, lo: str, hi: str, tz: tzinfo) -> str:
|
||||
"""Translate a date-field ``[lo TO hi]`` range to a Tantivy ISO range string.
|
||||
|
||||
Handles partial-date bounds (YYYY, YYYYMM, YYYYMMDD, ISO dash variants),
|
||||
open bounds (empty string -> OPEN_LO/OPEN_HI), ``now``, and reversed ranges
|
||||
(swaps tokens before computing floor/ceil so the span is always correct).
|
||||
"""
|
||||
lo_s = lo.strip()
|
||||
hi_s = hi.strip()
|
||||
|
||||
# Parse both bounds to (floor, ceil) pairs when present.
|
||||
lo_pair: tuple[datetime, datetime] | None = None
|
||||
hi_pair: tuple[datetime, datetime] | None = None
|
||||
|
||||
if lo_s:
|
||||
lo_pair = _bound_datetimes(field, lo_s, tz)
|
||||
if lo_pair is None:
|
||||
raise InvalidDateQuery(field, lo_s)
|
||||
if hi_s:
|
||||
hi_pair = _bound_datetimes(field, hi_s, tz)
|
||||
if hi_pair is None:
|
||||
raise InvalidDateQuery(field, hi_s)
|
||||
|
||||
# Detect a reversed range: only swap when BOTH bounds are present.
|
||||
if lo_pair is not None and hi_pair is not None and lo_pair[0] > hi_pair[0]:
|
||||
lo_pair, hi_pair = hi_pair, lo_pair
|
||||
|
||||
lo_iso = _fmt(lo_pair[0]) if lo_pair is not None else OPEN_LO
|
||||
|
||||
# A bound resolves to (floor, ceil) where floor == ceil for an exact instant
|
||||
# (a full ISO datetime, "now", or a "+/-N unit" offset) and floor != ceil for
|
||||
# a coarser period token (year/month/day precision). Only the latter needs a
|
||||
# half-open close: its ceil is the start of the *next* period and must be
|
||||
# excluded, or that instant (e.g. the 1st of next month) wrongly matches.
|
||||
if hi_pair is not None:
|
||||
hi_iso = _fmt(hi_pair[1])
|
||||
hi_close = "]" if hi_pair[0] == hi_pair[1] else "}"
|
||||
else:
|
||||
hi_iso = OPEN_HI
|
||||
hi_close = "]"
|
||||
|
||||
return f"{field}:[{lo_iso} TO {hi_iso}{hi_close}"
|
||||
@@ -794,10 +794,12 @@ def cleanup_user_deletion(sender, instance: User | Group, **kwargs) -> None:
|
||||
def add_to_index(sender, document, **kwargs) -> None:
|
||||
from documents.search import get_backend
|
||||
|
||||
get_backend().add_or_update(
|
||||
document,
|
||||
effective_content=document.get_effective_content(),
|
||||
)
|
||||
# A newly consumed version is not searchable on its own, its content
|
||||
# becomes the effective_content of the root document
|
||||
if document.root_document_id:
|
||||
document = document.root_document
|
||||
|
||||
get_backend().add_or_update(document)
|
||||
|
||||
|
||||
def run_workflows_added(
|
||||
|
||||
@@ -64,6 +64,7 @@ from documents.signals.handlers import send_websocket_document_updated
|
||||
from documents.utils import IterWrapper
|
||||
from documents.utils import compute_checksum
|
||||
from documents.utils import identity
|
||||
from documents.versioning import annotate_effective_content
|
||||
from documents.workflows.utils import get_workflows_for_trigger
|
||||
from paperless.config import AIConfig
|
||||
from paperless.logging import consume_task_id
|
||||
@@ -114,10 +115,7 @@ def index_document(self, document_id: int) -> None:
|
||||
)
|
||||
return
|
||||
with get_backend().batch_update() as batch:
|
||||
batch.add_or_update(
|
||||
document,
|
||||
effective_content=document.get_effective_content(),
|
||||
)
|
||||
batch.add_or_update(document)
|
||||
|
||||
|
||||
@shared_task(
|
||||
@@ -312,7 +310,10 @@ def bulk_update_documents(document_ids) -> None:
|
||||
from documents.search import get_backend
|
||||
|
||||
document_ids = list(document_ids)
|
||||
documents = Document.objects.filter(id__in=document_ids)
|
||||
# Annotated so indexing below doesn't query the versions of each document
|
||||
documents = annotate_effective_content(
|
||||
Document.objects.filter(id__in=document_ids),
|
||||
)
|
||||
|
||||
for doc in documents:
|
||||
clear_document_caches(doc.pk)
|
||||
|
||||
@@ -1,11 +1,15 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import tempfile
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
from documents.search._backend import reset_backend
|
||||
from documents.search._schema import build_schema
|
||||
from documents.search._tokenizer import register_tokenizers
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Generator
|
||||
@@ -31,3 +35,11 @@ def backend() -> Generator[TantivyBackend, None, None]:
|
||||
finally:
|
||||
b.close()
|
||||
reset_backend()
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def index() -> tantivy.Index:
|
||||
"""A real Tantivy index for parse-acceptance tests (module scope for speed)."""
|
||||
idx = tantivy.Index(build_schema(), path=tempfile.mkdtemp())
|
||||
register_tokenizers(idx, "english")
|
||||
return idx
|
||||
|
||||
@@ -1,411 +0,0 @@
|
||||
"""Result-level acceptance corpus: real documents indexed via build_schema(),
|
||||
real queries run through parse_user_query(), matched-document-ID sets
|
||||
asserted — not intermediate ASTs or query strings. This is paperless-ngx's
|
||||
analogue of whoosh-compat's own tests/emitter/test_acceptance_e2e.py.
|
||||
|
||||
Supersedes test_query.py's TestParseUserQuery result-level cases.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC
|
||||
from datetime import datetime
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import time_machine
|
||||
from django.contrib.auth.models import User
|
||||
|
||||
from documents.models import CustomField
|
||||
from documents.models import CustomFieldInstance
|
||||
from documents.models import Document
|
||||
from documents.models import DocumentType
|
||||
from documents.models import Note
|
||||
from documents.models import StoragePath
|
||||
from documents.search._query import parse_user_query
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
FROZEN_NOW = datetime(2026, 6, 15, 12, 0, tzinfo=UTC)
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
"""Create a Document and index it in one step, for the common case
|
||||
where nothing needs to happen between the two (no related Note/
|
||||
CustomFieldInstance to attach first)."""
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def indexed_documents(backend: TantivyBackend) -> dict[str, int]:
|
||||
"""Index a small fixture set, return {label: doc_id} for corpus queries."""
|
||||
docs = {
|
||||
"invoice_2020": _index(
|
||||
backend,
|
||||
title="Invoice 2020",
|
||||
content="invoice total due",
|
||||
checksum="acc-invoice-2020",
|
||||
archive_serial_number=100,
|
||||
),
|
||||
"invoice_2021": _index(
|
||||
backend,
|
||||
title="Invoice 2021",
|
||||
content="invoice total due",
|
||||
checksum="acc-invoice-2021",
|
||||
archive_serial_number=101,
|
||||
),
|
||||
"invoice_2023": _index(
|
||||
backend,
|
||||
title="Invoice 2023",
|
||||
content="invoice total due",
|
||||
checksum="acc-invoice-2023",
|
||||
archive_serial_number=102,
|
||||
),
|
||||
"receipt_2022": _index(
|
||||
backend,
|
||||
title="Receipt 2022",
|
||||
content="receipt total due",
|
||||
checksum="acc-receipt-2022",
|
||||
archive_serial_number=103,
|
||||
),
|
||||
}
|
||||
return {label: doc.pk for label, doc in docs.items()}
|
||||
|
||||
|
||||
class TestIssue13568BracketWildcard:
|
||||
"""paperless-ngx#13568: title:202[0-3]* must keep its character class,
|
||||
not fold to a prefix query that silently drops it."""
|
||||
|
||||
def test_bracket_class_wildcard_matches_only_in_range_years(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_documents: dict[str, int],
|
||||
) -> None:
|
||||
# [0-1] (not [0-3]) is deliberate: the fixture's four years are
|
||||
# 2020/2021/2022/2023, i.e. their trailing digit is 0/1/2/3
|
||||
# respectively - a [0-3] class would match all four and the test
|
||||
# would pass even if the character class were silently dropped and
|
||||
# folded to an unconstrained "202*" prefix. [0-1] partitions the
|
||||
# fixture into a genuine in-range/out-of-range split.
|
||||
matched = _matched_ids(backend, "title:202[0-1]*")
|
||||
expected = {
|
||||
indexed_documents["invoice_2020"],
|
||||
indexed_documents["invoice_2021"],
|
||||
}
|
||||
assert matched == expected, (
|
||||
"title:202[0-1]* must match 2020/2021 titles and exclude 2022/2023 "
|
||||
"- if this matches everything, the wildcard's character class was "
|
||||
"silently dropped (issue #13568's original bug)"
|
||||
)
|
||||
|
||||
|
||||
class TestFieldBoosts:
|
||||
def test_title_boost_ranks_title_match_above_content_only_match(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
title_match = _index(
|
||||
backend,
|
||||
title="urgent",
|
||||
content="nothing else relevant",
|
||||
checksum="acc-boost-title",
|
||||
)
|
||||
_index(
|
||||
backend,
|
||||
title="nothing",
|
||||
content="urgent matter here",
|
||||
checksum="acc-boost-content",
|
||||
)
|
||||
query = parse_user_query(backend._index, "urgent", UTC)
|
||||
searcher = backend._index.searcher()
|
||||
results = searcher.search(query, limit=10)
|
||||
ranked_ids = [
|
||||
searcher.doc(addr).to_dict()["id"][0] for _score, addr in results.hits
|
||||
]
|
||||
assert ranked_ids[0] == title_match.pk
|
||||
|
||||
|
||||
class TestJsonSubpaths:
|
||||
def test_notes_user_matches_document_with_that_note_author(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
alice = User.objects.create_user(username="alice")
|
||||
doc_with_note = Document.objects.create(
|
||||
title="Has note",
|
||||
content="x",
|
||||
checksum="acc-note-with",
|
||||
)
|
||||
Note.objects.create(document=doc_with_note, user=alice, note="reminder")
|
||||
backend.add_or_update(doc_with_note)
|
||||
_index(backend, title="No note", content="x", checksum="acc-note-without")
|
||||
matched = _matched_ids(backend, "notes.user:alice")
|
||||
assert matched == {doc_with_note.pk}
|
||||
|
||||
def test_custom_fields_name_and_value_combine(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
field = CustomField.objects.create(
|
||||
name="Contract Number",
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
other_field = CustomField.objects.create(
|
||||
name="Other Field",
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
matching = Document.objects.create(
|
||||
title="Matching",
|
||||
content="x",
|
||||
checksum="acc-cf-matching",
|
||||
)
|
||||
CustomFieldInstance.objects.create(
|
||||
document=matching,
|
||||
field=field,
|
||||
value_text="policy",
|
||||
)
|
||||
backend.add_or_update(matching)
|
||||
non_matching = Document.objects.create(
|
||||
title="Non-matching",
|
||||
content="x",
|
||||
checksum="acc-cf-nonmatching",
|
||||
)
|
||||
CustomFieldInstance.objects.create(
|
||||
document=non_matching,
|
||||
field=other_field,
|
||||
value_text="policy",
|
||||
)
|
||||
backend.add_or_update(non_matching)
|
||||
matched = _matched_ids(
|
||||
backend,
|
||||
'custom_fields.name:"Contract Number" custom_fields.value:policy',
|
||||
)
|
||||
assert matched == {matching.pk}
|
||||
|
||||
|
||||
class TestUnregisteredIdFieldFoldsToLiteralText:
|
||||
"""tag_id, owner_id, etc. are intentionally excluded from the
|
||||
FieldRegistry - always internal index columns, never meant to be
|
||||
query-addressable. Prove an unregistered field folds to a literal
|
||||
text search that matches nothing, rather than erroring."""
|
||||
|
||||
def test_tag_id_query_matches_nothing(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_documents: dict[str, int],
|
||||
) -> None:
|
||||
matched = _matched_ids(backend, "tag_id:5")
|
||||
assert matched == set()
|
||||
|
||||
|
||||
class TestFuzzyBlendSurvivesWhooshGrammar:
|
||||
"""A query mixing whoosh-only grammar (a date keyword) with a typo'd
|
||||
free-text word must still fuzzy-match the intended document when
|
||||
ADVANCED_FUZZY_SEARCH_THRESHOLD is enabled. The fuzzy clause is built
|
||||
from the parsed query's free-text tokens (whoosh_compat's
|
||||
free_text_tokens), never from the raw query string, so whoosh grammar
|
||||
that tantivy's own parser rejects cannot knock the fuzzy clause out."""
|
||||
|
||||
def test_typo_fuzzy_matches_alongside_date_keyword(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
settings,
|
||||
) -> None:
|
||||
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.5
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
doc = _index(
|
||||
backend,
|
||||
title="Receipt March",
|
||||
content="receipt total due",
|
||||
checksum="fuzzy-blend-1",
|
||||
archive_serial_number=900,
|
||||
)
|
||||
# Sanity: the exact spelling matches through the exact clause.
|
||||
assert doc.pk in _matched_ids(backend, "added:today receipt")
|
||||
# The regression: the misspelling (one transposition) only
|
||||
# matches via the fuzzy clause, and "added:today" is
|
||||
# whoosh-only grammar tantivy's parser rejects, so raw-string
|
||||
# fuzzy parsing skips the clause entirely and this returns
|
||||
# nothing. The typo is deliberate; keep codespell away from it.
|
||||
typo_query = "added:today reciept" # codespell:ignore reciept
|
||||
assert doc.pk in _matched_ids(backend, typo_query)
|
||||
|
||||
def test_negated_words_do_not_fuzzy_match(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
settings,
|
||||
) -> None:
|
||||
# A term the user excluded must not resurface through the fuzzy
|
||||
# clause. The shape is chosen so this genuinely discriminates: the
|
||||
# indexed document contains the NOT'd word but NOT the positive
|
||||
# word, so nothing matches the exact clause, and a fuzzy string
|
||||
# naively built from ALL words (including the NOT'd one) would
|
||||
# make this document the sole hit, normalize its score to 1.0,
|
||||
# and survive any threshold. (A shape with an exact-matching
|
||||
# sibling document does NOT discriminate: normalization ranks the
|
||||
# resurfaced doc far below the exact match and the threshold cuts
|
||||
# it even for a naive implementation.)
|
||||
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.5
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
_index(
|
||||
backend,
|
||||
title="Receipt Archive",
|
||||
content="receipt archived stack",
|
||||
checksum="fuzzy-blend-2",
|
||||
archive_serial_number=901,
|
||||
)
|
||||
assert _matched_ids(backend, "added:today total NOT receipt") == set()
|
||||
|
||||
|
||||
class TestUnquotedDateKeywordPhrases:
|
||||
"""The unquoted spelling (added:previous month) is honored natively by
|
||||
whoosh-compat's own grammar for this closed phrase vocabulary — no
|
||||
app-level rewrite is involved. Pins that the historically supported
|
||||
spelling keeps working now that paperless no longer pre-quotes it."""
|
||||
|
||||
@pytest.fixture
|
||||
def period_documents(self, backend: TantivyBackend) -> dict[str, int]:
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
in_may = _index(
|
||||
backend,
|
||||
title="May Doc",
|
||||
content="statement",
|
||||
checksum="kw-may",
|
||||
archive_serial_number=910,
|
||||
added=datetime(2026, 5, 20, 12, 0, tzinfo=UTC),
|
||||
)
|
||||
in_june = _index(
|
||||
backend,
|
||||
title="June Doc",
|
||||
content="statement",
|
||||
checksum="kw-june",
|
||||
archive_serial_number=911,
|
||||
added=datetime(2026, 6, 10, 12, 0, tzinfo=UTC),
|
||||
)
|
||||
return {"in_may": in_may.pk, "in_june": in_june.pk}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
pytest.param("added:previous month", id="unquoted"),
|
||||
pytest.param('added:"previous month"', id="quoted"),
|
||||
pytest.param("added:Previous Month", id="unquoted-mixed-case"),
|
||||
],
|
||||
)
|
||||
def test_unquoted_matches_the_same_documents_as_quoted(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
period_documents: dict[str, int],
|
||||
query: str,
|
||||
) -> None:
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
assert _matched_ids(backend, query) == {period_documents["in_may"]}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
pytest.param("added:this month", id="this-month"),
|
||||
pytest.param("added:this year", id="this-year"),
|
||||
pytest.param("added:previous week", id="previous-week"),
|
||||
pytest.param("added:previous quarter", id="previous-quarter"),
|
||||
pytest.param("added:previous year", id="previous-year"),
|
||||
pytest.param("created:previous month", id="created-field"),
|
||||
pytest.param("modified:previous month", id="modified-field"),
|
||||
],
|
||||
)
|
||||
def test_every_phrase_and_date_field_parses_without_error(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
period_documents: dict[str, int],
|
||||
query: str,
|
||||
) -> None:
|
||||
# The whole vocabulary times every date field must at least parse
|
||||
# and search cleanly (no SearchQueryError -> no HTTP 400); exact
|
||||
# window semantics are whoosh-compat's, pinned in its own suite.
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
_matched_ids(backend, query)
|
||||
|
||||
def test_text_field_keyword_words_are_ordinary_text(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
period_documents: dict[str, int],
|
||||
) -> None:
|
||||
# "previous month" after a TEXT field (or unfielded) is ordinary
|
||||
# text, not a date phrase: a title actually containing the words
|
||||
# matches, and the date-window documents do not.
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
wordy = _index(
|
||||
backend,
|
||||
title="Notes from the previous month",
|
||||
content="meeting notes",
|
||||
checksum="kw-text",
|
||||
archive_serial_number=912,
|
||||
)
|
||||
assert _matched_ids(backend, "title:previous month") == {wordy.pk}
|
||||
|
||||
|
||||
class TestFieldAliases:
|
||||
"""type:/path: are registry aliases for document_type:/storage_path:.
|
||||
The only other alias coverage is parse-shape; these prove resolution
|
||||
end-to-end against a real index."""
|
||||
|
||||
def test_type_alias_and_canonical_name_match_the_same_document(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
invoice_type = DocumentType.objects.create(name="invoice")
|
||||
# Discriminating shape: document_type is itself a default search
|
||||
# field, so if alias resolution ever broke and "type:invoice"
|
||||
# demoted to unfielded text, the token would STILL match the typed
|
||||
# document through the field value. The decoy carries the query
|
||||
# word in content, so a demoted search matches BOTH documents and
|
||||
# the exact-set assertions fail. (The title avoids stemming to
|
||||
# "type": english stems Typed -> type.)
|
||||
typed = _index(
|
||||
backend,
|
||||
title="First",
|
||||
content="quarterly statement",
|
||||
checksum="alias-type-1",
|
||||
document_type=invoice_type,
|
||||
)
|
||||
_index(
|
||||
backend,
|
||||
title="Second",
|
||||
content="invoice mentioned in body",
|
||||
checksum="alias-type-2",
|
||||
)
|
||||
assert _matched_ids(backend, "type:invoice") == {typed.pk}
|
||||
assert _matched_ids(backend, "document_type:invoice") == {typed.pk}
|
||||
|
||||
def test_path_alias_and_canonical_name_match_the_same_document(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
archive = StoragePath.objects.create(name="archive", path="archive/{title}")
|
||||
stored = _index(
|
||||
backend,
|
||||
title="Stored",
|
||||
content="quarterly statement",
|
||||
checksum="alias-path-1",
|
||||
storage_path=archive,
|
||||
)
|
||||
# storage_path is NOT a default search field today, so a demoted
|
||||
# "path:archive" already matches nothing; the content decoy keeps
|
||||
# this test discriminating even if it ever joins the defaults.
|
||||
_index(
|
||||
backend,
|
||||
title="Loose",
|
||||
content="archive mentioned in body",
|
||||
checksum="alias-path-2",
|
||||
)
|
||||
assert _matched_ids(backend, "path:archive") == {stored.pk}
|
||||
assert _matched_ids(backend, "storage_path:archive") == {stored.pk}
|
||||
@@ -16,6 +16,7 @@ from documents.search._backend import TantivyBackend
|
||||
from documents.search._backend import WriteBatch
|
||||
from documents.search._backend import get_backend
|
||||
from documents.search._backend import reset_backend
|
||||
from documents.signals.handlers import add_to_index
|
||||
from documents.tests.factories import CorrespondentFactory
|
||||
from documents.tests.factories import DocumentFactory
|
||||
from documents.tests.factories import DocumentTypeFactory
|
||||
@@ -1030,6 +1031,81 @@ class TestHighlightHits:
|
||||
assert len(hits) == 0
|
||||
|
||||
|
||||
class TestVersionIndexing:
|
||||
"""
|
||||
GIVEN:
|
||||
- A root document whose new version has just been consumed, e.g. by
|
||||
the password removal workflow action
|
||||
WHEN:
|
||||
- The consumption finished signal is handled
|
||||
THEN:
|
||||
- The root document is indexed with the new version's content, since
|
||||
versions are not searchable on their own
|
||||
"""
|
||||
|
||||
def test_consumed_version_updates_root_entry(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
mocker: MockerFixture,
|
||||
) -> None:
|
||||
root = Document.objects.create(
|
||||
title="Statement",
|
||||
content="",
|
||||
checksum="VER1",
|
||||
pk=90,
|
||||
)
|
||||
backend.add_or_update(root)
|
||||
version = Document.objects.create(
|
||||
title="Statement",
|
||||
content="unprotected statement text",
|
||||
checksum="VER2",
|
||||
pk=91,
|
||||
root_document=root,
|
||||
version_index=1,
|
||||
)
|
||||
mocker.patch("documents.search.get_backend", return_value=backend)
|
||||
|
||||
add_to_index(sender=None, document=version)
|
||||
|
||||
assert backend.search_ids("unprotected", user=None) == [root.pk]
|
||||
|
||||
|
||||
class TestEffectiveContentIndexing:
|
||||
"""
|
||||
GIVEN:
|
||||
- A root document with a newer version
|
||||
WHEN:
|
||||
- The root document is indexed
|
||||
THEN:
|
||||
- The newest version's content is indexed, never the root's own
|
||||
outdated text
|
||||
"""
|
||||
|
||||
def test_root_is_indexed_with_latest_version_content(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
root = Document.objects.create(
|
||||
title="Statement",
|
||||
content="stale original text",
|
||||
checksum="EFF1",
|
||||
pk=95,
|
||||
)
|
||||
Document.objects.create(
|
||||
title="Statement",
|
||||
content="latest version text",
|
||||
checksum="EFF2",
|
||||
pk=96,
|
||||
root_document=root,
|
||||
version_index=1,
|
||||
)
|
||||
|
||||
backend.add_or_update(root)
|
||||
|
||||
assert backend.search_ids("latest", user=None) == [root.pk]
|
||||
assert backend.search_ids("stale", user=None) == []
|
||||
|
||||
|
||||
class TestIndexDirectoryGarbageCollection:
|
||||
"""Regression tests for Tantivy segment files leaking on disk when
|
||||
multiple long-lived worker processes (Granian/Celery) take turns writing
|
||||
|
||||
@@ -1,133 +0,0 @@
|
||||
"""The CJK bigram clause blended into QUERY-mode searches.
|
||||
|
||||
The clause exists so CJK runs are matchable at all (the default analyzers
|
||||
keep a whitespace-free CJK run as one indivisible token), but it must not
|
||||
widen the query beyond what the user asked for: a CJK term the query
|
||||
excludes, or restricts to one field, must not come back through it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestCjkClauseFollowsTheParsedQuery:
|
||||
def test_negated_cjk_term_is_excluded(self, backend: TantivyBackend) -> None:
|
||||
"""'invoice NOT 漢字' must not return the document containing 漢字."""
|
||||
with_cjk = _index(
|
||||
backend,
|
||||
title="Invoice A",
|
||||
content="invoice total 漢字",
|
||||
checksum="cjk-neg-1",
|
||||
)
|
||||
without_cjk = _index(
|
||||
backend,
|
||||
title="Invoice B",
|
||||
content="invoice total only",
|
||||
checksum="cjk-neg-2",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "invoice") == {with_cjk.pk, without_cjk.pk}
|
||||
assert _matched_ids(backend, "invoice NOT 漢字") == {without_cjk.pk}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("threshold", "expected"),
|
||||
[
|
||||
pytest.param(None, {"titled"}, id="fuzzy_off"),
|
||||
pytest.param(0.0, {"titled", "content_only"}, id="fuzzy_on"),
|
||||
],
|
||||
)
|
||||
def test_fielded_cjk_term_searches_only_that_field(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
settings: SettingsWrapper,
|
||||
threshold: float | None,
|
||||
expected: set[str],
|
||||
) -> None:
|
||||
"""'title:東京' must not match a document whose 東京 is in the content.
|
||||
|
||||
The CJK clause honours the field. The fuzzy clause, when enabled,
|
||||
does not: it contributes every free-text term UNFIELDED by design
|
||||
(see _try_parse_fuzzy_query), so it brings the content-only
|
||||
document back on its own 0.1-boosted terms. That is the documented
|
||||
trade-off, pinned here so it stays deliberate.
|
||||
"""
|
||||
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = threshold
|
||||
content_only = _index(
|
||||
backend,
|
||||
title="Tokyo report",
|
||||
content="東京都の人口は約1400万人です",
|
||||
checksum="cjk-field-1",
|
||||
)
|
||||
titled = _index(
|
||||
backend,
|
||||
title="東京都の報告書",
|
||||
content="an english summary",
|
||||
checksum="cjk-field-2",
|
||||
)
|
||||
pks = {"titled": titled.pk, "content_only": content_only.pk}
|
||||
|
||||
assert _matched_ids(backend, "東京") == set(pks.values())
|
||||
assert _matched_ids(backend, "title:東京") == {pks[label] for label in expected}
|
||||
|
||||
def test_cjk_on_a_non_default_field_builds_no_clause(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""A CJK term restricted to a field outside the default search fields
|
||||
has nothing to contribute to the bigram clause: 'notes:東京' must not
|
||||
fall back to matching 東京 in the content."""
|
||||
_index(
|
||||
backend,
|
||||
title="Tokyo report",
|
||||
content="東京都の人口は約1400万人です",
|
||||
checksum="cjk-notes-1",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "notes:東京") == set()
|
||||
|
||||
def test_bare_cjk_term_still_matches_every_default_field(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""The clause's reason for existing: an unfielded CJK run matches
|
||||
wherever it is indexed, and does so alongside a latin term."""
|
||||
in_content = _index(
|
||||
backend,
|
||||
title="report",
|
||||
content="本文に重要な情報",
|
||||
checksum="cjk-bare-1",
|
||||
)
|
||||
in_title = _index(
|
||||
backend,
|
||||
title="重要な報告書",
|
||||
content="english only",
|
||||
checksum="cjk-bare-2",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "重要") == {in_content.pk, in_title.pk}
|
||||
assert _matched_ids(backend, "重要 OR report") == {
|
||||
in_content.pk,
|
||||
in_title.pk,
|
||||
}
|
||||
@@ -1,75 +0,0 @@
|
||||
"""Whoosh's compact, separator-free date spelling, resolved end to end.
|
||||
|
||||
whoosh-compat owns both widths of this spelling and asserts both of each
|
||||
form's bounds directly: ``test_compact_numeric_datetime`` pins the 8-digit
|
||||
form as a whole calendar day (lower bound, upper bound and exclusivity), and
|
||||
``test_compact_numeric_datetime_full_width_is_a_single_second_instant`` pins
|
||||
the 14-digit form as one instant. The 14-digit form is kept here as the single
|
||||
representative because it is the one that exercises paperless's ``added``
|
||||
DATETIME fast field at full precision: the corpus separates a document at
|
||||
the named instant from one on the same calendar day at another hour and one
|
||||
on the next day at the same hour, so a query that degrades into a whole-day
|
||||
window, or drops the time of day, matches the wrong set rather than passing
|
||||
on a corpus that could not tell the difference.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC
|
||||
from datetime import datetime
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def docs(backend: TantivyBackend) -> dict[str, int]:
|
||||
return {
|
||||
"instant": _index(
|
||||
backend,
|
||||
title="On the instant",
|
||||
content="x",
|
||||
checksum="compact-date-instant",
|
||||
added=datetime(2005, 3, 4, 15, 30, tzinfo=UTC),
|
||||
).pk,
|
||||
"same_day": _index(
|
||||
backend,
|
||||
title="Same day, other hour",
|
||||
content="x",
|
||||
checksum="compact-date-same-day",
|
||||
added=datetime(2005, 3, 4, 9, 0, tzinfo=UTC),
|
||||
).pk,
|
||||
"next_day": _index(
|
||||
backend,
|
||||
title="Next day, same hour",
|
||||
content="x",
|
||||
checksum="compact-date-next-day",
|
||||
added=datetime(2005, 3, 5, 15, 30, tzinfo=UTC),
|
||||
).pk,
|
||||
}
|
||||
|
||||
|
||||
def test_fourteen_digits_is_a_single_instant(
|
||||
backend: TantivyBackend,
|
||||
docs: dict[str, int],
|
||||
) -> None:
|
||||
# same_day is what tells this apart from the 8-digit day-window form,
|
||||
# next_day from a form that ignored the time altogether.
|
||||
assert _matched_ids(backend, "added:20050304153000") == {docs["instant"]}
|
||||
@@ -1,72 +0,0 @@
|
||||
"""Pins the correctness gained by deleting the pre-parse
|
||||
_quote_date_keyword_phrases rewrite.
|
||||
|
||||
That rewrite matched date-keyword phrases (e.g. "previous month" after a
|
||||
date field) anywhere in the raw query string, including inside an
|
||||
unrelated quoted string, and inserted quotes mid-phrase there too — its
|
||||
own docstring gave ``title:"see added:previous month notes"`` as the
|
||||
example of what it corrupted. whoosh-compat's grammar accepts the same
|
||||
phrase vocabulary unquoted natively (see TestUnquotedDateKeywordPhrases
|
||||
in test_acceptance.py), so the rewrite was redundant everywhere it was
|
||||
safe and actively wrong everywhere it was not. This is the one case that
|
||||
tells the two apart: a literal title phrase that happens to contain
|
||||
"added:previous month" as running text.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestQuotedStringContainingDateKeywordText:
|
||||
"""A quoted title phrase containing the literal text
|
||||
"added:previous month" as running words must match on that literal
|
||||
text alone, never spill into an unfielded search for "previous" and
|
||||
"month" across the default search fields the way the deleted rewrite
|
||||
would have decomposed it into."""
|
||||
|
||||
def test_matches_only_the_literal_phrase(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
literal = _index(
|
||||
backend,
|
||||
title="see added:previous month notes",
|
||||
content="quarterly filing",
|
||||
checksum="dkp-literal",
|
||||
archive_serial_number=920,
|
||||
)
|
||||
# Under the deleted rewrite, this decoy would incorrectly match:
|
||||
# its title contains the "see added:" and " notes" fragments the
|
||||
# corrupted parse required as title phrases, and its content
|
||||
# supplies "previous" and "month" as the decomposed word-match
|
||||
# clauses the rewrite turned the middle of the phrase into.
|
||||
decoy = _index(
|
||||
backend,
|
||||
title="see added: quarterly report notes",
|
||||
content="we reviewed the previous statement about month end",
|
||||
checksum="dkp-decoy",
|
||||
archive_serial_number=921,
|
||||
)
|
||||
query = 'title:"see added:previous month notes"'
|
||||
assert _matched_ids(backend, query) == {literal.pk}
|
||||
assert decoy.pk not in _matched_ids(backend, query)
|
||||
@@ -1,83 +0,0 @@
|
||||
"""Date keyword phrases (``today``, etc.) resolved in a non-UTC timezone,
|
||||
end to end.
|
||||
|
||||
paperless's own ``tz=get_current_timezone()`` plumbing
|
||||
(``TantivyBackend._parse_query``) is exercised elsewhere only for
|
||||
relative *ranges* (``added:[-1 week to now]``, in
|
||||
documents/tests/test_api_search.py). This covers a date *keyword*
|
||||
(``today``), whose day boundary depends on the active timezone the same
|
||||
way but goes through whoosh-compat's DateParserPlugin resolution instead
|
||||
of an explicit range.
|
||||
|
||||
Discriminating shape: frozen at 2026-06-15T02:00 UTC, which is
|
||||
2026-06-14T22:00 in America/New_York -- still "today" (06-14) there, but
|
||||
already "today" (06-15) in UTC. Two documents pin both directions of the
|
||||
mistake a hardcoded-UTC bug would make:
|
||||
|
||||
- ``in_ny_today`` (added 2026-06-14T20:00 UTC = 2026-06-14T16:00 NY) is
|
||||
inside New York's "today" window and outside a naive UTC-calendar-day
|
||||
window. A ``tz``-ignoring bug would miss it.
|
||||
- ``in_utc_calendar_day_only`` (added 2026-06-15T10:00 UTC =
|
||||
2026-06-15T06:00 NY) is inside a naive UTC-calendar-day window but
|
||||
outside New York's actual "today" window. A ``tz``-ignoring bug would
|
||||
wrongly match it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC
|
||||
from datetime import datetime
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import time_machine
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
FROZEN_NOW = datetime(2026, 6, 15, 2, 0, tzinfo=UTC)
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestDateKeywordUsesTheActiveTimezone:
|
||||
def test_today_matches_the_new_york_calendar_day_not_the_utc_one(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
settings: SettingsWrapper,
|
||||
) -> None:
|
||||
settings.TIME_ZONE = "America/New_York"
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
in_ny_today = _index(
|
||||
backend,
|
||||
title="NY today",
|
||||
content="x",
|
||||
checksum="tz-keyword-ny-today",
|
||||
added=datetime(2026, 6, 14, 20, 0, tzinfo=UTC),
|
||||
)
|
||||
# Not captured: the exact-set assertion below already proves
|
||||
# this document (inside a naive UTC-calendar-day window, but
|
||||
# outside New York's actual "today") does not match.
|
||||
_index(
|
||||
backend,
|
||||
title="UTC calendar day only",
|
||||
content="x",
|
||||
checksum="tz-keyword-utc-calendar-day-only",
|
||||
added=datetime(2026, 6, 15, 10, 0, tzinfo=UTC),
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "added:today") == {in_ny_today.pk}
|
||||
@@ -1,20 +0,0 @@
|
||||
"""``_DEFAULT_SEARCH_FIELDS`` must stay a subset of the registered public
|
||||
field names.
|
||||
|
||||
Nothing enforced this before: a rename in PUBLIC_FIELDS not mirrored in
|
||||
``_DEFAULT_SEARCH_FIELDS`` (documents/search/_query.py) would 400 every
|
||||
unfielded search at request time, since ``index.parse_query`` and the
|
||||
fuzzy/CJK clause builders are handed a field name the schema no longer
|
||||
has.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from documents.search._fields import PUBLIC_FIELDS
|
||||
from documents.search._query import _DEFAULT_SEARCH_FIELDS
|
||||
|
||||
|
||||
class TestDefaultSearchFieldsAreRegistered:
|
||||
def test_every_default_search_field_is_a_public_field(self) -> None:
|
||||
public_field_names = {f.name for f in PUBLIC_FIELDS}
|
||||
assert set(_DEFAULT_SEARCH_FIELDS) <= public_field_names
|
||||
@@ -1,356 +0,0 @@
|
||||
"""Pins the search syntax that ``docs/usage.md`` promises users.
|
||||
|
||||
Every query here appears verbatim, or as a direct paraphrase, in the
|
||||
"Document searches" section of ``docs/usage.md``. Each case indexes real
|
||||
documents and asserts on matched document IDs rather than on the parsed
|
||||
query, because a query that parses cleanly is not necessarily a query that
|
||||
means what the documentation says it means: ``added:now`` parses without a
|
||||
single diagnostic and then matches nothing, because it resolves to an
|
||||
instant rather than to a span.
|
||||
|
||||
The negative cases matter as much as the positive ones. They pin the
|
||||
behaviours the docs explicitly warn about, so that if any of them ever
|
||||
starts working the warning can be removed deliberately rather than being
|
||||
left standing as a lie.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC
|
||||
from datetime import datetime
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import time_machine
|
||||
|
||||
from documents.models import Document
|
||||
from documents.models import Note
|
||||
from documents.models import Tag
|
||||
from documents.search._errors import InvalidDateQuery
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Generator
|
||||
|
||||
from django.contrib.auth.models import User
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
# A Monday, so that "next monday"/"last monday" land a clean week either side.
|
||||
FROZEN_NOW = datetime(2026, 6, 15, 12, 0, tzinfo=UTC)
|
||||
|
||||
# The checksum used in the docs' `checksum:` example.
|
||||
DOC_CHECKSUM = "9f86d081884c7d659a2feaa0c55ad015a3bf4f1b2b0b822cd15d6c15b0f00a08"
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestLogicalExpressions:
|
||||
@pytest.fixture
|
||||
def docs(self, backend: TantivyBackend) -> dict[str, int]:
|
||||
return {
|
||||
"secret": _index(
|
||||
backend,
|
||||
title="Invoice one",
|
||||
content="invoice secret contents",
|
||||
checksum="doc-syntax-secret",
|
||||
).pk,
|
||||
"plain": _index(
|
||||
backend,
|
||||
title="Invoice two",
|
||||
content="invoice ordinary contents",
|
||||
checksum="doc-syntax-plain",
|
||||
).pk,
|
||||
}
|
||||
|
||||
def test_not_excludes_a_term(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
docs: dict[str, int],
|
||||
) -> None:
|
||||
assert _matched_ids(backend, "invoice NOT secret") == {docs["plain"]}
|
||||
|
||||
def test_leading_hyphen_requires_the_term_instead_of_excluding_it(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
docs: dict[str, int],
|
||||
) -> None:
|
||||
# The docs warn about exactly this: separators are stripped at index
|
||||
# time, so "-secret" is the term "secret" and the query is an AND.
|
||||
assert _matched_ids(backend, "invoice -secret") == {docs["secret"]}
|
||||
|
||||
def test_or_inside_parentheses_matches_either_branch(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
docs: dict[str, int],
|
||||
) -> None:
|
||||
matched = _matched_ids(backend, "invoice AND (secret OR ordinary)")
|
||||
assert matched == {docs["secret"], docs["plain"]}
|
||||
|
||||
|
||||
class TestPhraseSearch:
|
||||
def test_quoted_phrase_requires_the_words_in_order(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
doc = _index(
|
||||
backend,
|
||||
title="Phrase",
|
||||
content="the quick brown fox jumps",
|
||||
checksum="doc-syntax-phrase",
|
||||
)
|
||||
assert _matched_ids(backend, '"quick brown fox"') == {doc.pk}
|
||||
assert _matched_ids(backend, '"brown quick fox"') == set()
|
||||
|
||||
|
||||
class TestTagCommaList:
|
||||
"""``tag:bills,unpaid`` is published syntax (docs/usage.md), so this checks
|
||||
that the documented spelling still returns what the docs promise: only the
|
||||
document carrying every listed tag.
|
||||
|
||||
It is deliberately not proof of paperless's field configuration, and must
|
||||
not be read as such. Removing ``comma_values`` from the ``tag`` FieldSpec
|
||||
leaves this test passing, because paperless's analyzer splits the literal
|
||||
value "bills,unpaid" into the same two tokens the value-list reading
|
||||
produces, so the two readings select the same documents. The registry fact
|
||||
-- that ``tag`` opts in and no other field does -- is observable only at
|
||||
the registry, and is owned by test_registry.py's
|
||||
``test_tag_is_comma_values``/``test_correspondent_is_not_comma_values``.
|
||||
"""
|
||||
|
||||
def test_comma_list_requires_every_listed_tag(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
bills = Tag.objects.create(name="bills")
|
||||
unpaid = Tag.objects.create(name="unpaid")
|
||||
archived = Tag.objects.create(name="archived")
|
||||
|
||||
both = Document.objects.create(
|
||||
title="Both tags",
|
||||
content="body",
|
||||
checksum="doc-syntax-tag-both",
|
||||
)
|
||||
both.tags.add(bills, unpaid)
|
||||
backend.add_or_update(both)
|
||||
|
||||
one = Document.objects.create(
|
||||
title="One tag",
|
||||
content="body",
|
||||
checksum="doc-syntax-tag-one",
|
||||
)
|
||||
one.tags.add(bills, archived)
|
||||
backend.add_or_update(one)
|
||||
|
||||
assert _matched_ids(backend, "tag:bills,unpaid") == {both.pk}
|
||||
assert _matched_ids(backend, "tag:bills") == {both.pk, one.pk}
|
||||
|
||||
|
||||
class TestArchiveMetadataFields:
|
||||
@pytest.fixture
|
||||
def doc(self, backend: TantivyBackend, admin_user: User) -> Document:
|
||||
doc = Document.objects.create(
|
||||
title="Metadata",
|
||||
content="body",
|
||||
checksum=DOC_CHECKSUM,
|
||||
archive_serial_number=100,
|
||||
page_count=12,
|
||||
original_filename="invoice.pdf",
|
||||
)
|
||||
Note.objects.create(document=doc, user=admin_user, note="a note")
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
"asn:100",
|
||||
"asn:[50 to 150]",
|
||||
"page_count:12",
|
||||
"page_count:[10 to 20]",
|
||||
"num_notes:1",
|
||||
"num_notes:[1 to 5]",
|
||||
"original_filename:invoice.pdf",
|
||||
f"checksum:{DOC_CHECKSUM}",
|
||||
"checksum:9f86d081*",
|
||||
# A checksum term is stored verbatim, but a checksum *pattern* is
|
||||
# lowercased before it is matched, which the docs now say outright
|
||||
# next to the "only a complete, lowercase checksum matches" rule
|
||||
# that the uppercase term in the negative list below pins.
|
||||
"checksum:9F86D081*",
|
||||
],
|
||||
)
|
||||
def test_documented_metadata_query_matches(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
doc: Document,
|
||||
query: str,
|
||||
) -> None:
|
||||
assert _matched_ids(backend, query) == {doc.pk}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
# The docs say only a complete, lowercase checksum matches.
|
||||
"checksum:9f86d081",
|
||||
f"checksum:{DOC_CHECKSUM.upper()}",
|
||||
],
|
||||
)
|
||||
def test_partial_or_uppercase_checksum_matches_nothing(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
doc: Document,
|
||||
query: str,
|
||||
) -> None:
|
||||
assert _matched_ids(backend, query) == set()
|
||||
|
||||
|
||||
class TestDocumentedDateForms:
|
||||
@pytest.fixture(autouse=True)
|
||||
def frozen_now(self) -> Generator[None, None, None]:
|
||||
with time_machine.travel(FROZEN_NOW, tick=False):
|
||||
yield
|
||||
|
||||
@pytest.fixture
|
||||
def dated(self, backend: TantivyBackend) -> dict[str, int]:
|
||||
stamps = {
|
||||
"today": datetime(2026, 6, 15, 9, 0, tzinfo=UTC),
|
||||
"yesterday": datetime(2026, 6, 14, 9, 0, tzinfo=UTC),
|
||||
"tomorrow": datetime(2026, 6, 16, 9, 0, tzinfo=UTC),
|
||||
"next_monday": datetime(2026, 6, 22, 10, 0, tzinfo=UTC),
|
||||
"last_monday": datetime(2026, 6, 8, 10, 0, tzinfo=UTC),
|
||||
"january": datetime(2026, 1, 10, 10, 0, tzinfo=UTC),
|
||||
"old": datetime(2005, 3, 4, 15, 30, tzinfo=UTC),
|
||||
}
|
||||
return {
|
||||
label: _index(
|
||||
backend,
|
||||
title=label,
|
||||
content="dated body",
|
||||
checksum=f"doc-syntax-date-{label}",
|
||||
added=stamp,
|
||||
).pk
|
||||
for label, stamp in stamps.items()
|
||||
}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("query", "label"),
|
||||
[
|
||||
("added:today", "today"),
|
||||
("added:yesterday", "yesterday"),
|
||||
("added:tomorrow", "tomorrow"),
|
||||
('added:"next monday"', "next_monday"),
|
||||
('added:"last monday"', "last_monday"),
|
||||
("added:january", "january"),
|
||||
("added:2005-03-04", "old"),
|
||||
("added:2005-03", "old"),
|
||||
("added:[2005-01-01 to 2005-12-31]", "old"),
|
||||
("added:[2005 to 2009]", "old"),
|
||||
# A full timestamp works, but only quoted when it stands alone,
|
||||
# and only unquoted when it is a range bound. The bare standalone
|
||||
# spelling is pinned as a non-match below.
|
||||
('added:"2005-03-04T15:30:00Z"', "old"),
|
||||
("added:[2005-03-04T09:00:00Z to 2005-03-04T17:00:00Z]", "old"),
|
||||
# A quoted range bound works when the quotes are single ones; the
|
||||
# double-quoted spelling is pinned as an error below.
|
||||
("added:['2005-03-04' to 2005-03-05]", "old"),
|
||||
],
|
||||
)
|
||||
def test_documented_date_form_matches_its_day_or_month(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
dated: dict[str, int],
|
||||
query: str,
|
||||
label: str,
|
||||
) -> None:
|
||||
assert _matched_ids(backend, query) == {dated[label]}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
# Zero-width: these resolve to a single instant, not a span, so
|
||||
# nothing in a realistic corpus lands on them. The docs warn
|
||||
# about them rather than presenting them as usable.
|
||||
"added:now",
|
||||
"added:noon",
|
||||
"added:midnight",
|
||||
# Quoting is what rescues the other multi-word date expressions,
|
||||
# so pin that it does not rescue these: the problem is the width
|
||||
# of the resulting range, not the way the value is delimited.
|
||||
# One quoted spelling is enough for that; which keyword sits
|
||||
# inside the quotes is grammar whoosh-compat owns.
|
||||
'added:"now"',
|
||||
# A relative offset, which the warning in the docs names by this
|
||||
# exact spelling. Standing alone it is an instant like the rest of
|
||||
# this list; the same offset used as a range bound is a real
|
||||
# window, pinned by the test below.
|
||||
'added:"-1 week"',
|
||||
],
|
||||
)
|
||||
def test_forms_the_docs_warn_about_match_nothing(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
dated: dict[str, int],
|
||||
query: str,
|
||||
) -> None:
|
||||
assert _matched_ids(backend, query) == set()
|
||||
|
||||
def test_bare_timestamp_is_rejected_rather_than_matching_nothing(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
dated: dict[str, int],
|
||||
) -> None:
|
||||
"""The bare, unquoted spelling of a full timestamp. The quoted and
|
||||
range-bound spellings pinned above do work and match this fixture's
|
||||
document; this one is a user-fixable error rather than an empty
|
||||
result set, so the docs tell the user to quote it.
|
||||
|
||||
The reported value is the prefix the date grammar could consume, not
|
||||
the whole of what the user typed: paperless's own message, not this
|
||||
value, is what has to carry the "quote it" guidance.
|
||||
"""
|
||||
with pytest.raises(InvalidDateQuery) as exc_info:
|
||||
_matched_ids(backend, "added:2005-03-04T15:30:00Z")
|
||||
assert exc_info.value.field == "added"
|
||||
assert exc_info.value.value == "2005-03-"
|
||||
|
||||
def test_relative_offset_as_a_range_bound_is_a_real_window(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
dated: dict[str, int],
|
||||
) -> None:
|
||||
"""The same offset that matches nothing on its own spans the last
|
||||
seven days as a lower bound. The docs say so, next to the warning
|
||||
about the standalone form, so both readings are pinned together.
|
||||
|
||||
"last_monday" is indexed at 2026-06-08T10:00, two hours before the
|
||||
window opens, so its exclusion is what shows the bound is the offset
|
||||
and not a whole-day rounding of it.
|
||||
"""
|
||||
assert _matched_ids(backend, "added:['-1 week' to now]") == {
|
||||
dated["today"],
|
||||
dated["yesterday"],
|
||||
}
|
||||
|
||||
def test_double_quoted_range_bound_is_rejected(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
dated: dict[str, int],
|
||||
) -> None:
|
||||
"""Quoting a range bound is allowed, but only with single quotes: the
|
||||
double-quoted spelling reaches the date grammar with its quotes still
|
||||
attached and is not a recognizable date. The docs say so, so pin which
|
||||
of the two quote characters is the one that fails.
|
||||
"""
|
||||
with pytest.raises(InvalidDateQuery) as exc_info:
|
||||
_matched_ids(backend, 'added:["2005-03-04" to 2005-03-05]')
|
||||
assert exc_info.value.value == '"2005-03-04"'
|
||||
@@ -1,221 +0,0 @@
|
||||
"""Diagnostics route by Cause, and user-facing messages are host-owned.
|
||||
|
||||
whoosh-compat documents ``Diagnostic.message`` as developer output with no
|
||||
stability guarantee, so it must never reach an HTTP response body.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from datetime import UTC
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
from whoosh_compat.errors import Diagnostic
|
||||
from whoosh_compat.errors import DiagnosticKind
|
||||
from whoosh_compat.errors import QueryError
|
||||
from whoosh_compat.errors import cause_for
|
||||
from whoosh_compat.fields import FieldKind
|
||||
from whoosh_compat.fields import FieldRef
|
||||
|
||||
from documents.search._errors import SearchQueryError
|
||||
from documents.search._query import _map_emit_error
|
||||
from documents.search._query import _single_diagnostic_to_error
|
||||
from documents.search._query import parse_user_query
|
||||
from documents.search._schema import build_schema
|
||||
from documents.search._tokenizer import register_tokenizers
|
||||
|
||||
pytestmark = pytest.mark.search
|
||||
|
||||
_LIBRARY_PROSE = "INTERNAL LIBRARY WORDING WITH raw tantivy detail"
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def query_index() -> tantivy.Index:
|
||||
"""An in-memory, unstemmed index; these tests only parse, never index."""
|
||||
idx = tantivy.Index(build_schema(), path=None)
|
||||
register_tokenizers(idx, "")
|
||||
return idx
|
||||
|
||||
|
||||
def _diagnostic(
|
||||
kind: DiagnosticKind,
|
||||
*,
|
||||
field: FieldRef | None = FieldRef("title"),
|
||||
field_kind: FieldKind | None = FieldKind.TEXT,
|
||||
) -> Diagnostic:
|
||||
"""A Diagnostic shaped like the emitter's, with the library's own
|
||||
kind -> cause mapping rather than a hand-picked cause."""
|
||||
return Diagnostic(
|
||||
kind=kind,
|
||||
cause=cause_for(kind),
|
||||
message=_LIBRARY_PROSE,
|
||||
field=field,
|
||||
field_kind=field_kind,
|
||||
)
|
||||
|
||||
|
||||
class TestEmitErrorRouting:
|
||||
"""Every Cause gets a distinguishable treatment, not just "a 400"."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"kind",
|
||||
[
|
||||
DiagnosticKind.BACKEND_REJECTED,
|
||||
DiagnosticKind.AST_INVALID_SHAPE,
|
||||
DiagnosticKind.AST_UNKNOWN_FIELD,
|
||||
],
|
||||
)
|
||||
def test_internal_cause_is_not_converted(self, kind: DiagnosticKind) -> None:
|
||||
"""A library defect must surface as a 500 monitoring can see, not a
|
||||
400 blaming the user."""
|
||||
error = QueryError(_diagnostic(kind))
|
||||
with pytest.raises(QueryError) as excinfo:
|
||||
_map_emit_error(error)
|
||||
assert excinfo.value is error
|
||||
|
||||
def test_misconfigured_cause_is_logged_and_becomes_a_400(
|
||||
self,
|
||||
caplog: pytest.LogCaptureFixture,
|
||||
) -> None:
|
||||
kind = DiagnosticKind.SCHEMA_FIELD_MISSING
|
||||
with caplog.at_level(logging.ERROR, logger="paperless.search"):
|
||||
error = _map_emit_error(
|
||||
QueryError(_diagnostic(kind, field=FieldRef("asn"))),
|
||||
)
|
||||
assert isinstance(error, SearchQueryError)
|
||||
errors = [r for r in caplog.records if r.levelno == logging.ERROR]
|
||||
assert len(errors) == 1
|
||||
assert "asn" in errors[0].getMessage()
|
||||
assert kind.name in errors[0].getMessage()
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"kind",
|
||||
[
|
||||
DiagnosticKind.TEXT_RANGE,
|
||||
DiagnosticKind.PATTERN_TOO_COMPLEX,
|
||||
DiagnosticKind.EXISTS_REQUIRES_FAST,
|
||||
],
|
||||
)
|
||||
def test_unsupported_cause_is_a_400_with_no_operator_log(
|
||||
self,
|
||||
kind: DiagnosticKind,
|
||||
caplog: pytest.LogCaptureFixture,
|
||||
) -> None:
|
||||
"""A query tantivy cannot run is the user's to fix; it must not page
|
||||
an operator the way a registry/schema mismatch does.
|
||||
|
||||
EXISTS_REQUIRES_FAST is nominally MISCONFIGURED but belongs here: it
|
||||
is decided from the registry's own FieldSpec, so it never reports a
|
||||
disagreement anyone could resolve."""
|
||||
with caplog.at_level(logging.WARNING, logger="paperless.search"):
|
||||
error = _map_emit_error(QueryError(_diagnostic(kind)))
|
||||
assert isinstance(error, SearchQueryError)
|
||||
assert caplog.records == []
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"kind",
|
||||
[
|
||||
DiagnosticKind.TEXT_RANGE,
|
||||
DiagnosticKind.PATTERN_TOO_COMPLEX,
|
||||
DiagnosticKind.EXISTS_REQUIRES_FAST,
|
||||
DiagnosticKind.SCHEMA_FIELD_MISSING,
|
||||
],
|
||||
)
|
||||
def test_user_facing_message_never_echoes_library_prose(
|
||||
self,
|
||||
kind: DiagnosticKind,
|
||||
) -> None:
|
||||
error = _map_emit_error(QueryError(_diagnostic(kind)))
|
||||
assert _LIBRARY_PROSE not in str(error)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"kind",
|
||||
[
|
||||
DiagnosticKind.TEXT_RANGE,
|
||||
DiagnosticKind.PATTERN_TOO_COMPLEX,
|
||||
DiagnosticKind.EXISTS_REQUIRES_FAST,
|
||||
DiagnosticKind.SCHEMA_FIELD_MISSING,
|
||||
],
|
||||
)
|
||||
def test_user_facing_message_names_the_field(
|
||||
self,
|
||||
kind: DiagnosticKind,
|
||||
) -> None:
|
||||
"""FieldRef.__str__ yields the canonical dotted name, including a
|
||||
JSON subpath, so every user-reachable emit kind can name it."""
|
||||
diagnostic = _diagnostic(
|
||||
kind,
|
||||
field=FieldRef("custom_fields", "value"),
|
||||
field_kind=FieldKind.JSON,
|
||||
)
|
||||
error = _map_emit_error(QueryError(diagnostic))
|
||||
assert "custom_fields.value" in str(error)
|
||||
|
||||
|
||||
class TestParseDiagnosticMessages:
|
||||
"""Parse-time diagnostics are host-worded too, off field_kind."""
|
||||
|
||||
def test_too_deep_is_a_400_without_library_prose(self) -> None:
|
||||
error = _single_diagnostic_to_error(
|
||||
_diagnostic(DiagnosticKind.TOO_DEEP, field=None, field_kind=None),
|
||||
)
|
||||
assert isinstance(error, SearchQueryError)
|
||||
assert _LIBRARY_PROSE not in str(error)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("kind", "field_kind"),
|
||||
[
|
||||
(DiagnosticKind.PATTERN_ON_NUMERIC, FieldKind.U64),
|
||||
(DiagnosticKind.PATTERN_ON_BOOLEAN_EXISTS, FieldKind.BOOLEAN_EXISTS),
|
||||
(DiagnosticKind.PATTERN_ON_SUBPATH, FieldKind.JSON),
|
||||
],
|
||||
)
|
||||
def test_pattern_on_kinds_name_the_field_and_its_kind(
|
||||
self,
|
||||
kind: DiagnosticKind,
|
||||
field_kind: FieldKind,
|
||||
) -> None:
|
||||
error = _single_diagnostic_to_error(
|
||||
_diagnostic(kind, field=FieldRef("asn"), field_kind=field_kind),
|
||||
)
|
||||
message = str(error)
|
||||
assert _LIBRARY_PROSE not in message
|
||||
assert "asn" in message
|
||||
assert field_kind.name.lower() in message
|
||||
|
||||
|
||||
class TestRealQueriesRouteCorrectly:
|
||||
"""The routing table against diagnostics emit() really produces."""
|
||||
|
||||
def test_text_range_is_a_400_naming_the_field(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
) -> None:
|
||||
with pytest.raises(SearchQueryError) as excinfo:
|
||||
parse_user_query(query_index, "title:[a to b]", UTC)
|
||||
assert "title" in str(excinfo.value)
|
||||
|
||||
def test_wildcard_on_a_numeric_field_is_a_400_naming_the_field(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
) -> None:
|
||||
with pytest.raises(SearchQueryError) as excinfo:
|
||||
parse_user_query(query_index, "asn:12*", UTC)
|
||||
assert "asn" in str(excinfo.value)
|
||||
|
||||
def test_internal_diagnostic_escapes_as_a_query_error(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""The one case with no query text that reaches it: emit() reporting
|
||||
a defect in itself must not be converted to a user-facing 400."""
|
||||
import documents.search._query as query_mod
|
||||
|
||||
def raise_internal(*args: object, **kwargs: object) -> None:
|
||||
raise QueryError(_diagnostic(DiagnosticKind.BACKEND_REJECTED))
|
||||
|
||||
monkeypatch.setattr(query_mod, "tantivy_emit", raise_internal)
|
||||
with pytest.raises(QueryError):
|
||||
parse_user_query(query_index, "invoice", UTC)
|
||||
@@ -1,92 +0,0 @@
|
||||
"""``field:*`` on a JSON field is user error, not an operator alert.
|
||||
|
||||
whoosh-compat classifies EXISTS_REQUIRES_FAST as MISCONFIGURED, and
|
||||
_map_emit_error used to route every MISCONFIGURED diagnostic to an ERROR log.
|
||||
But the kind is decided from the registry's own FieldSpec (kind plus fast)
|
||||
without consulting the index schema, and field_descriptors() builds the JSON
|
||||
fields non-fast deliberately, so nothing is misconfigured and no operator
|
||||
action can clear the condition. Any authenticated user could otherwise emit
|
||||
ERROR lines in a loop by repeating ``notes:*``.
|
||||
|
||||
SCHEMA_FIELD_MISSING, the other MISCONFIGURED kind, does compare the registry
|
||||
against the live schema, so it stays an ERROR.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from datetime import UTC
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
from whoosh_compat.errors import Diagnostic
|
||||
from whoosh_compat.errors import DiagnosticKind
|
||||
from whoosh_compat.errors import QueryError
|
||||
from whoosh_compat.errors import cause_for
|
||||
from whoosh_compat.fields import FieldKind
|
||||
from whoosh_compat.fields import FieldRef
|
||||
|
||||
from documents.search._errors import SearchQueryError
|
||||
from documents.search._query import _map_emit_error
|
||||
from documents.search._query import parse_user_query
|
||||
from documents.search._schema import build_schema
|
||||
from documents.search._tokenizer import register_tokenizers
|
||||
|
||||
pytestmark = pytest.mark.search
|
||||
|
||||
# Every spelling of "does this JSON field have a value" a user can type.
|
||||
EXISTS_QUERIES = [
|
||||
"notes:*",
|
||||
"notes.note:*",
|
||||
"notes.user:*",
|
||||
"custom_fields:*",
|
||||
"custom_fields.name:*",
|
||||
"custom_fields.value:*",
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def query_index() -> tantivy.Index:
|
||||
idx = tantivy.Index(build_schema(), path=None)
|
||||
register_tokenizers(idx, "")
|
||||
return idx
|
||||
|
||||
|
||||
class TestJsonExistsIsUserError:
|
||||
@pytest.mark.parametrize("query", EXISTS_QUERIES)
|
||||
def test_query_is_a_400_that_emits_no_error_log(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
caplog: pytest.LogCaptureFixture,
|
||||
query: str,
|
||||
) -> None:
|
||||
with caplog.at_level(logging.WARNING, logger="paperless.search"):
|
||||
with pytest.raises(SearchQueryError) as excinfo:
|
||||
parse_user_query(query_index, query, UTC)
|
||||
assert query.split(":", maxsplit=1)[0] in str(excinfo.value)
|
||||
assert [r for r in caplog.records if r.levelno >= logging.ERROR] == []
|
||||
|
||||
|
||||
class TestGenuineMisconfigurationStillLogs:
|
||||
def test_schema_field_missing_is_an_error_log(
|
||||
self,
|
||||
caplog: pytest.LogCaptureFixture,
|
||||
) -> None:
|
||||
"""The registry naming a field the index schema does not have is a
|
||||
real mismatch an operator can fix, so it keeps the alert."""
|
||||
kind = DiagnosticKind.SCHEMA_FIELD_MISSING
|
||||
error = QueryError(
|
||||
Diagnostic(
|
||||
kind=kind,
|
||||
cause=cause_for(kind),
|
||||
message="field 'asn' is not defined in the index schema",
|
||||
field=FieldRef("asn"),
|
||||
field_kind=FieldKind.U64,
|
||||
),
|
||||
)
|
||||
with caplog.at_level(logging.ERROR, logger="paperless.search"):
|
||||
mapped = _map_emit_error(error)
|
||||
assert isinstance(mapped, SearchQueryError)
|
||||
records = [r for r in caplog.records if r.levelno == logging.ERROR]
|
||||
assert len(records) == 1
|
||||
assert kind.name in records[0].getMessage()
|
||||
@@ -1,10 +0,0 @@
|
||||
from whoosh_compat import FieldKind
|
||||
|
||||
from documents.search._fields import PUBLIC_FIELDS
|
||||
|
||||
|
||||
class TestPublicFields:
|
||||
def test_json_fields_have_subpaths(self) -> None:
|
||||
for field in PUBLIC_FIELDS:
|
||||
if field.kind is FieldKind.JSON:
|
||||
assert field.subpaths, f"{field.name} is JSON but has no subpaths"
|
||||
@@ -1,174 +0,0 @@
|
||||
"""The words the fuzzy blend clause hands back to tantivy's parser.
|
||||
|
||||
The clause re-parses a word string through tantivy, which analyzes it
|
||||
again, so the words must be the query's raw text rather than the analyzed
|
||||
text (analysis is not idempotent), and must still be split into plain
|
||||
words so that hyphenated, dotted and quoted terms keep contributing.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pytest_django.fixtures import SettingsWrapper
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def fuzzy_enabled(settings: SettingsWrapper) -> None:
|
||||
"""Enable the fuzzy blend clause. The threshold doubles as a minimum
|
||||
score filter, so it is set to 0.0: every hit passes and the test sees
|
||||
the clause's matching behaviour, not the filter's."""
|
||||
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.0
|
||||
|
||||
|
||||
class TestFuzzyClauseWords:
|
||||
def test_a_stemmed_word_is_not_stemmed_a_second_time(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""'universities' stems to 'univers'; feeding that back to tantivy
|
||||
stems it again to 'univ', whose fuzzy prefix reaches unrelated
|
||||
words. The clause must stay wide enough for a typo and no wider."""
|
||||
wanted = _index(
|
||||
backend,
|
||||
title="A",
|
||||
content="universities of europe",
|
||||
checksum="fuzz-stem-1",
|
||||
)
|
||||
typo = _index(
|
||||
backend,
|
||||
title="B",
|
||||
content="universties of europe",
|
||||
checksum="fuzz-stem-2",
|
||||
)
|
||||
_index(
|
||||
backend,
|
||||
title="C",
|
||||
content="univalent chemical bonds",
|
||||
checksum="fuzz-stem-3",
|
||||
)
|
||||
_index(
|
||||
backend,
|
||||
title="D",
|
||||
content="unicycle repair manual",
|
||||
checksum="fuzz-stem-4",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "universities") == {wanted.pk, typo.pk}
|
||||
|
||||
def test_a_hyphenated_term_still_reaches_the_clause(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""'COVID-19' is one raw token: unless it is split into words, it
|
||||
carries characters the re-parse would read as grammar, is dropped,
|
||||
and the whole query loses its fuzzy clause."""
|
||||
misspelled = _index(
|
||||
backend,
|
||||
title="A",
|
||||
content="covidx testing results",
|
||||
checksum="fuzz-hyphen-1",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, "COVID-19") == {misspelled.pk}
|
||||
|
||||
def test_a_phrase_still_reaches_the_clause(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
"""A phrase is one raw token carrying a space, and is the whole
|
||||
query's only free text here."""
|
||||
near_miss = _index(
|
||||
backend,
|
||||
title="A",
|
||||
content="taxation reportage weekly",
|
||||
checksum="fuzz-phrase-1",
|
||||
)
|
||||
|
||||
assert _matched_ids(backend, '"tax reports"') == {near_miss.pk}
|
||||
|
||||
|
||||
class TestBooleanKeywordsInRawText:
|
||||
"""Tantivy's boolean keywords are word runs, so they survive the cut
|
||||
into words and its own parser reads them as grammar. Raw query text
|
||||
reaches that parser with its case intact, so a quoted phrase can carry
|
||||
them in."""
|
||||
|
||||
@pytest.fixture
|
||||
def corpus(self, backend: TantivyBackend) -> dict[str, int]:
|
||||
both = _index(
|
||||
backend,
|
||||
title="A",
|
||||
content="taxation reportage weekly",
|
||||
checksum="fuzz-kw-1",
|
||||
)
|
||||
tax_only = _index(
|
||||
backend,
|
||||
title="B",
|
||||
content="taxation only here",
|
||||
checksum="fuzz-kw-2",
|
||||
)
|
||||
report_only = _index(
|
||||
backend,
|
||||
title="C",
|
||||
content="reportage only here",
|
||||
checksum="fuzz-kw-3",
|
||||
)
|
||||
return {
|
||||
"both": both.pk,
|
||||
"tax_only": tax_only.pk,
|
||||
"report_only": report_only.pk,
|
||||
}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
pytest.param('"tax AND reports"', id="and"),
|
||||
pytest.param('"tax OR reports"', id="or"),
|
||||
pytest.param('"tax NOT reports"', id="not"),
|
||||
pytest.param('"tax IN reports"', id="in"),
|
||||
],
|
||||
)
|
||||
def test_a_keyword_inside_a_phrase_stays_an_ordinary_word(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
corpus: dict[str, int],
|
||||
query: str,
|
||||
) -> None:
|
||||
"""The phrase asks for three words, so the clause must stay the
|
||||
disjunction it is for '"tax reports"': AND must not turn it into a
|
||||
conjunction, NOT must not give it its own exclusion, IN must not
|
||||
fail the parse."""
|
||||
assert _matched_ids(backend, '"tax reports"') == set(corpus.values())
|
||||
assert _matched_ids(backend, query) == set(corpus.values())
|
||||
|
||||
def test_a_trailing_keyword_does_not_drop_the_clause(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
corpus: dict[str, int],
|
||||
) -> None:
|
||||
"""'tax AND' is a syntax error to tantivy's parser, which would
|
||||
cost the whole query its fuzzy clause."""
|
||||
assert _matched_ids(backend, '"tax AND"') == {
|
||||
corpus["both"],
|
||||
corpus["tax_only"],
|
||||
}
|
||||
@@ -1,192 +0,0 @@
|
||||
"""Regression coverage for the unguarded TEXT-mode highlight query.
|
||||
|
||||
parse_simple_text_highlight_query re-parses simple-search tokens through
|
||||
Tantivy's query-string parser to build a SnippetGenerator-compatible query.
|
||||
Simple-search tokens keep arbitrary punctuation (quotes, colons, brackets,
|
||||
slashes), so any token carrying Tantivy query grammar raised an unguarded
|
||||
ValueError. The search itself had already succeeded by the time this ran:
|
||||
only the highlight step failed, and with the DocumentViewSet.list
|
||||
exception handler narrowed elsewhere on this branch, that ValueError now
|
||||
reaches the client as a bare 500 rather than a 400.
|
||||
|
||||
Covers three angles:
|
||||
- the query builder itself: quoting each token as its own escaped phrase
|
||||
should let it parse instead of raising, for every failure mode a plain-
|
||||
text query can trigger (syntax error, unknown field, unsupported regex).
|
||||
- highlight_hits: even when a token still can't be expressed as a
|
||||
highlight query, the guard must fall back to a query that still
|
||||
produces usable highlight HTML, not silently empty ones.
|
||||
- the real API endpoint: pinning the previously-500 status to 200.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
from rest_framework import status
|
||||
|
||||
from documents.search._backend import SearchMode
|
||||
from documents.search._query import parse_simple_text_highlight_query
|
||||
from documents.search._schema import build_schema
|
||||
from documents.search._tokenizer import register_tokenizers
|
||||
from documents.tests.factories import DocumentFactory
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from rest_framework.test import APIClient
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
# Each spelling below trips a different Tantivy parser failure mode:
|
||||
# 'a"b' -> Syntax Error (unterminated quote)
|
||||
# foo:bar -> unknown field
|
||||
# (a -> Syntax Error (unbalanced group)
|
||||
# [a -> Syntax Error (unbalanced range)
|
||||
# /a/ -> Unsupported query (regex queries disallowed)
|
||||
_MALFORMED_QUERIES = [
|
||||
pytest.param('a"b', id="unterminated_quote"),
|
||||
pytest.param("foo:bar", id="unknown_field"),
|
||||
pytest.param("(a", id="unbalanced_group"),
|
||||
pytest.param("[a", id="unbalanced_range"),
|
||||
pytest.param("/a/", id="unsupported_regex"),
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def query_index() -> tantivy.Index:
|
||||
"""An in-memory, unstemmed index for parse-only tests."""
|
||||
schema = build_schema()
|
||||
idx = tantivy.Index(schema, path=None)
|
||||
register_tokenizers(idx, "")
|
||||
return idx
|
||||
|
||||
|
||||
class TestParseSimpleTextHighlightQueryDoesNotRaise:
|
||||
"""The query builder itself must tolerate Tantivy syntax in its tokens."""
|
||||
|
||||
@pytest.mark.parametrize("raw_query", _MALFORMED_QUERIES)
|
||||
def test_malformed_token_does_not_raise(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
raw_query: str,
|
||||
) -> None:
|
||||
assert isinstance(
|
||||
parse_simple_text_highlight_query(query_index, raw_query),
|
||||
tantivy.Query,
|
||||
)
|
||||
|
||||
|
||||
class TestHighlightHitsProducesUsableHighlights:
|
||||
"""highlight_hits must keep producing real <b>-wrapped snippet HTML for
|
||||
these queries, not merely avoid raising."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"raw_query",
|
||||
[*_MALFORMED_QUERIES, pytest.param("plain text", id="plain_text_sanity")],
|
||||
)
|
||||
def test_highlight_still_contains_matched_text(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
raw_query: str,
|
||||
) -> None:
|
||||
doc = DocumentFactory.create(
|
||||
title="probe",
|
||||
content=f"needle content containing {raw_query} literally here",
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
hits = backend.highlight_hits(
|
||||
raw_query,
|
||||
[doc.pk],
|
||||
search_mode=SearchMode.TEXT,
|
||||
)
|
||||
|
||||
assert len(hits) == 1
|
||||
highlights = hits[0]["highlights"]
|
||||
assert "content" in highlights, (
|
||||
f"Expected a content highlight for {raw_query!r}, got: {highlights!r}"
|
||||
)
|
||||
assert "<b>" in highlights["content"], (
|
||||
f"Highlight for {raw_query!r} carries no matched-term markup: "
|
||||
f"{highlights['content']!r}"
|
||||
)
|
||||
|
||||
|
||||
class TestHighlightGuardDiscriminatesOnValueError:
|
||||
"""The guard added to highlight_hits must catch exactly ValueError, the
|
||||
same shape as the sibling notes_text guard, and let anything else
|
||||
through -- so a real library defect is never mistaken for a harmless
|
||||
syntax error."""
|
||||
|
||||
def test_non_value_error_is_not_swallowed(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
import documents.search._backend as backend_mod
|
||||
|
||||
def raise_runtime_error(*args: object, **kwargs: object) -> object:
|
||||
raise RuntimeError("synthetic bug, unrelated to query syntax")
|
||||
|
||||
monkeypatch.setattr(
|
||||
backend_mod,
|
||||
"parse_simple_text_highlight_query",
|
||||
raise_runtime_error,
|
||||
)
|
||||
|
||||
doc = DocumentFactory.create(title="probe", content="anything here")
|
||||
backend.add_or_update(doc)
|
||||
|
||||
with pytest.raises(RuntimeError):
|
||||
backend.highlight_hits(
|
||||
"anything",
|
||||
[doc.pk],
|
||||
search_mode=SearchMode.TEXT,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("_search_index")
|
||||
class TestApiNoLongerReturns500:
|
||||
"""Pins the actual regression: a matching TEXT-mode search whose query
|
||||
string carries Tantivy syntax must return results, not a server error."""
|
||||
|
||||
@pytest.mark.parametrize("raw_query", _MALFORMED_QUERIES)
|
||||
def test_malformed_text_query_returns_200(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
raw_query: str,
|
||||
) -> None:
|
||||
from documents.search import get_backend
|
||||
|
||||
doc = DocumentFactory.create(
|
||||
title="probe",
|
||||
content=f"needle content containing {raw_query} literally here",
|
||||
)
|
||||
get_backend().add_or_update(doc)
|
||||
|
||||
response = admin_client.get(f"/api/documents/?text={raw_query}")
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
assert response.data["count"] == 1
|
||||
|
||||
def test_plain_text_query_still_returns_200(
|
||||
self,
|
||||
admin_client: APIClient,
|
||||
) -> None:
|
||||
"""Sanity check: the guard must not mask a total failure of the
|
||||
ordinary highlight path."""
|
||||
from documents.search import get_backend
|
||||
|
||||
doc = DocumentFactory.create(
|
||||
title="probe",
|
||||
content="needle content containing plain text literally here",
|
||||
)
|
||||
get_backend().add_or_update(doc)
|
||||
|
||||
response = admin_client.get("/api/documents/?text=plain text")
|
||||
|
||||
assert response.status_code == status.HTTP_200_OK
|
||||
assert response.data["count"] == 1
|
||||
@@ -1,148 +0,0 @@
|
||||
"""Bare notes:/custom_fields: prefix resolution.
|
||||
|
||||
"notes:foo"/"custom_fields:foo" were valid fielded searches before the
|
||||
whoosh-compat migration. The registry only exposes them as JSON subpaths, so
|
||||
each JSON FieldSpec declares a default subpath (SubpathSpec(default=True)):
|
||||
notes: resolves to notes.note:, custom_fields: resolves to
|
||||
custom_fields.value:. This replaced an earlier regex-based rewrite
|
||||
(_rewrite_bare_json_field_prefixes) that ran on the raw query string before
|
||||
parsing and was blind to quoting, so a phrase like
|
||||
content:"payment notes: none" was silently corrupted into a notes-field
|
||||
search and matched nothing. Resolving the default subpath inside the parser
|
||||
instead means quoting is already understood by the time it happens.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
from django.contrib.auth.models import User
|
||||
|
||||
from documents.models import CustomField
|
||||
from documents.models import CustomFieldInstance
|
||||
from documents.models import Document
|
||||
from documents.models import Note
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
def _index(backend: TantivyBackend, **kwargs: object) -> Document:
|
||||
doc = Document.objects.create(**kwargs)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestBareJsonFieldPrefixes:
|
||||
def test_bare_notes_prefix_searches_note_text(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
alice = User.objects.create_user(username="alice")
|
||||
with_note = Document.objects.create(
|
||||
title="Has note",
|
||||
content="x",
|
||||
checksum="bare-notes-with",
|
||||
)
|
||||
Note.objects.create(document=with_note, user=alice, note="crocodile")
|
||||
backend.add_or_update(with_note)
|
||||
# This document's CONTENT contains the words a demoted text search
|
||||
# would match; it must NOT match once the prefix addresses notes.
|
||||
_index(
|
||||
backend,
|
||||
title="Notes about things",
|
||||
content="notes crocodile mention",
|
||||
checksum="bare-notes-decoy",
|
||||
)
|
||||
assert _matched_ids(backend, "notes:crocodile") == {with_note.pk}
|
||||
|
||||
def test_bare_custom_fields_prefix_searches_values(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
field = CustomField.objects.create(
|
||||
name="Policy Number",
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
with_value = Document.objects.create(
|
||||
title="Has field",
|
||||
content="x",
|
||||
checksum="bare-cf-with",
|
||||
)
|
||||
CustomFieldInstance.objects.create(
|
||||
document=with_value,
|
||||
field=field,
|
||||
value_text="crocodile",
|
||||
)
|
||||
backend.add_or_update(with_value)
|
||||
_index(
|
||||
backend,
|
||||
title="Custom things",
|
||||
content="custom fields crocodile",
|
||||
checksum="bare-cf-decoy",
|
||||
)
|
||||
assert _matched_ids(backend, "custom_fields:crocodile") == {with_value.pk}
|
||||
|
||||
def test_subpath_spellings_are_untouched(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
bob = User.objects.create_user(username="bob")
|
||||
doc = Document.objects.create(
|
||||
title="Bob note",
|
||||
content="x",
|
||||
checksum="bare-subpath",
|
||||
)
|
||||
Note.objects.create(document=doc, user=bob, note="remark")
|
||||
backend.add_or_update(doc)
|
||||
assert _matched_ids(backend, "notes.user:bob") == {doc.pk}
|
||||
assert _matched_ids(backend, "notes.note:remark") == {doc.pk}
|
||||
|
||||
|
||||
class TestQuotedPhraseContainingNotesColonIsNotCorrupted:
|
||||
"""The regex rewrite this migration removes was blind to quoting: it
|
||||
matched "notes:" anywhere in the raw query string, including inside an
|
||||
already-quoted phrase on an unrelated field, silently turning
|
||||
content:"payment notes: none" into a notes-field search that matched
|
||||
nothing. Resolving the default subpath during parsing (which is
|
||||
quote-aware) fixes this."""
|
||||
|
||||
def test_quoted_phrase_with_notes_colon_matches_by_content(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
target = _index(
|
||||
backend,
|
||||
title="Statement",
|
||||
content="payment notes: none",
|
||||
checksum="quoted-phrase-notes-colon",
|
||||
)
|
||||
assert _matched_ids(
|
||||
backend,
|
||||
'content:"payment notes: none"',
|
||||
) == {target.pk}
|
||||
|
||||
def test_quoted_phrase_matches_the_same_document_unquoted(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
# Same document, phrasing without the colon: this proves the fix is
|
||||
# about quote-awareness, not about the words themselves being
|
||||
# unsearchable.
|
||||
target = _index(
|
||||
backend,
|
||||
title="Statement",
|
||||
content="payment notes none",
|
||||
checksum="quoted-phrase-no-colon",
|
||||
)
|
||||
assert _matched_ids(
|
||||
backend,
|
||||
'content:"payment notes none"',
|
||||
) == {target.pk}
|
||||
@@ -1,83 +0,0 @@
|
||||
"""Every declared JSON subpath must actually be written to the index.
|
||||
|
||||
PUBLIC_FIELDS declares each JSON field's subpaths (e.g. ``notes`` ->
|
||||
{"user", "note"}), but nothing coupled that declaration to what
|
||||
``_backend.py``'s document builder actually writes into the JSON blob at
|
||||
index time. A subpath declared but never written would be
|
||||
queryable-but-always-empty -- syntactically valid, silently matching
|
||||
nothing -- with no test failure anywhere.
|
||||
|
||||
This indexes one real document carrying values for every JSON field
|
||||
(a Note, a CustomFieldInstance) and inspects the document's own stored
|
||||
JSON payload, rather than running field-specific queries: that way a
|
||||
future JSON field's subpaths are covered automatically, without a new
|
||||
per-subpath query having to be added by hand each time.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
from django.contrib.auth.models import User
|
||||
from whoosh_compat import FieldKind
|
||||
|
||||
from documents.models import CustomField
|
||||
from documents.models import CustomFieldInstance
|
||||
from documents.models import Document
|
||||
from documents.models import Note
|
||||
from documents.search._fields import PUBLIC_FIELDS
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
class TestJsonSubpathsAreWrittenAtIndexTime:
|
||||
def test_every_declared_json_subpath_appears_in_the_stored_document(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
) -> None:
|
||||
user = User.objects.create_user(username="completeness-user")
|
||||
field = CustomField.objects.create(
|
||||
name="Completeness Field",
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
doc = Document.objects.create(
|
||||
title="Completeness doc",
|
||||
content="x",
|
||||
checksum="json-subpath-completeness",
|
||||
)
|
||||
Note.objects.create(document=doc, user=user, note="a note")
|
||||
CustomFieldInstance.objects.create(
|
||||
document=doc,
|
||||
field=field,
|
||||
value_text="a value",
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
index = backend._index
|
||||
searcher = index.searcher()
|
||||
hits = searcher.search(
|
||||
tantivy.Query.term_query(index.schema, "id", doc.pk),
|
||||
limit=1,
|
||||
).hits
|
||||
assert hits, "the document was not indexed"
|
||||
stored = searcher.doc(hits[0][1]).to_dict()
|
||||
|
||||
json_fields = [f for f in PUBLIC_FIELDS if f.kind is FieldKind.JSON]
|
||||
assert json_fields, "no JSON fields declared - fixture is stale"
|
||||
for field_spec in json_fields:
|
||||
stored_values = stored.get(field_spec.name)
|
||||
assert stored_values, (
|
||||
f"{field_spec.name} was not written to the index at all"
|
||||
)
|
||||
written_keys = stored_values[0].keys()
|
||||
for subpath in field_spec.subpaths:
|
||||
assert subpath in written_keys, (
|
||||
f"{field_spec.name}.{subpath} is declared in PUBLIC_FIELDS "
|
||||
"but _backend.py's document builder never writes it - it "
|
||||
"would be queryable but always empty"
|
||||
)
|
||||
@@ -1,92 +0,0 @@
|
||||
"""Wildcard patterns on KEYWORD fields must stay literal.
|
||||
|
||||
``checksum`` is the only KEYWORD field: it is indexed with the raw tokenizer,
|
||||
so its terms are never lowercased, folded or stemmed. Running its wildcard
|
||||
patterns through the stemming normalizer rewrote hex prefixes ("ceded" ->
|
||||
"cede") and returned documents whose checksum did not start with what the user
|
||||
typed, which for an identity field is a wrong answer.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
from documents.search._registry import get_field_registry
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from whoosh_compat import FieldRegistry
|
||||
from whoosh_compat import PatternNormalizer
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
CEDEF00D = "cedef00ddeadbeef0123456789abcdef01234567"
|
||||
CEDEDEAD = "cededeadbeef567801234567" + "89abcdef01234567"
|
||||
|
||||
|
||||
def _normalizer(registry: FieldRegistry, name: str) -> PatternNormalizer:
|
||||
ref = registry.make_ref(name)
|
||||
assert ref is not None
|
||||
resolved = registry.resolve(ref)
|
||||
assert resolved is not None
|
||||
assert resolved.spec.pattern_normalizer is not None
|
||||
return resolved.spec.pattern_normalizer
|
||||
|
||||
|
||||
class TestKeywordPatternNormalizer:
|
||||
@pytest.mark.parametrize(
|
||||
"run",
|
||||
[
|
||||
pytest.param("ceded", id="stems_to_cede"),
|
||||
pytest.param("added", id="stems_to_ad"),
|
||||
pytest.param("cafed", id="stems_to_cafe"),
|
||||
],
|
||||
)
|
||||
def test_keyword_runs_are_folded_not_stemmed(self, run: str) -> None:
|
||||
"""One form, the run as typed: a KEYWORD pattern must never be widened
|
||||
to a stem, which would return checksums that do not start with what
|
||||
the user typed."""
|
||||
normalize = _normalizer(get_field_registry("en"), "checksum")
|
||||
assert normalize(run) == run
|
||||
|
||||
def test_text_runs_still_offer_their_stem(self) -> None:
|
||||
"""A TEXT field offers the stem alongside the typed run, so a term
|
||||
matching either one is reachable."""
|
||||
normalize = _normalizer(get_field_registry("en"), "title")
|
||||
assert tuple(normalize("Running")) == ("running", "run")
|
||||
|
||||
|
||||
class TestChecksumPrefixQueries:
|
||||
@pytest.fixture
|
||||
def indexed(self, backend: TantivyBackend) -> None:
|
||||
for i, checksum in enumerate((CEDEF00D, CEDEDEAD)):
|
||||
doc = Document.objects.create(
|
||||
title=f"Checksum doc {i}",
|
||||
content="invoices for the quarter",
|
||||
checksum=checksum,
|
||||
archive_serial_number=940 + i,
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
def _ids(self, backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
def test_prefix_matches_only_the_document_that_starts_with_it(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed: None,
|
||||
) -> None:
|
||||
matched = self._ids(backend, "checksum:ceded*")
|
||||
expected = Document.objects.get(checksum=CEDEDEAD).pk
|
||||
assert matched == {expected}
|
||||
|
||||
def test_text_prefix_still_reaches_the_stemmed_index(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed: None,
|
||||
) -> None:
|
||||
assert len(self._ids(backend, "invoice*")) == 2
|
||||
@@ -1,233 +0,0 @@
|
||||
"""Wildcard patterns must match a stemmed index.
|
||||
|
||||
Query patterns are normalized but were not stemmed, while index terms are
|
||||
stemmed, so the natural spelling of a prefix search matched nothing:
|
||||
``invoice*`` found no document although ``invoic*`` did. v2's index was
|
||||
UNSTEMMED (whoosh ``TEXT()`` defaults to ``StandardAnalyzer``), so this
|
||||
regressed against both baselines.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from documents.models import Document
|
||||
from documents.search._registry import _make_pattern_normalizer
|
||||
from documents.search._tokenizer import ascii_fold
|
||||
from documents.search._tokenizer import paperless_text_analyzer
|
||||
from documents.search._tokenizer import stem_pattern_text
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from whoosh_compat import PatternNormalizer
|
||||
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
CONTENT = (
|
||||
"invoice total due for electricity from both companies, "
|
||||
"payments made to the university library, copies attached"
|
||||
)
|
||||
|
||||
|
||||
def _matched_ids(backend: TantivyBackend, query: str) -> set[int]:
|
||||
return set(backend.search_ids(query, user=None))
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def indexed_doc(backend: TantivyBackend) -> Document:
|
||||
doc = Document.objects.create(
|
||||
title="Invoice 2020 productname",
|
||||
content=CONTENT,
|
||||
checksum="pattern-stemming-1",
|
||||
archive_serial_number=900,
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
return doc
|
||||
|
||||
|
||||
class TestPrefixStemming:
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
"invoice*",
|
||||
"electricity*",
|
||||
"companies*",
|
||||
"payments*",
|
||||
"library*",
|
||||
"title:Invoice*",
|
||||
],
|
||||
)
|
||||
def test_full_word_prefix_matches_its_stem(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
query: str,
|
||||
) -> None:
|
||||
assert _matched_ids(backend, query) == {indexed_doc.id}
|
||||
|
||||
@pytest.mark.parametrize("query", ["invoic*", "electr*", "payment*"])
|
||||
def test_already_stemmed_prefix_still_matches(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
query: str,
|
||||
) -> None:
|
||||
assert _matched_ids(backend, query) == {indexed_doc.id}
|
||||
|
||||
@pytest.mark.parametrize("query", ["univers*", "librar*"])
|
||||
def test_partial_prefix_reaches_the_stemmed_term(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
query: str,
|
||||
) -> None:
|
||||
"""A prefix shorter than a whole word still matches, and neither of
|
||||
these needs the two-alternative path to do it.
|
||||
|
||||
Measured under "en": the stemmer leaves "librar" alone, so it has one
|
||||
form, and that form is a prefix of the "librari" the index holds for
|
||||
"library". "univers" stems to the *shorter* "univ", and the run as
|
||||
typed and its stem are both prefixes of the "univers" the index holds
|
||||
for "university". The case where the two forms genuinely diverge, and
|
||||
only one of them matches, is
|
||||
test_stem_substitution_reaches_both_the_inflection_and_the_compound.
|
||||
"""
|
||||
assert _matched_ids(backend, query) == {indexed_doc.id}
|
||||
|
||||
def test_full_word_reaches_the_stem_but_a_fragment_of_it_does_not(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
) -> None:
|
||||
"""The alternatives widen recall without turning a wildcard into a
|
||||
prefix search over the original text.
|
||||
|
||||
"university" is stored as "univers". The stem of "universities" is
|
||||
that same "univers", so the longer word matches; "universit" is a
|
||||
prefix of neither its own stem nor the stored term, so the *shorter*
|
||||
fragment matches nothing. usage.md names this pair, so a reader told
|
||||
that `universit*` fails is also told which spelling works.
|
||||
"""
|
||||
assert _matched_ids(backend, "universities*") == {indexed_doc.id}
|
||||
assert _matched_ids(backend, "universit*") == set()
|
||||
|
||||
def test_pattern_past_the_stem_boundary_is_documented_not_fixed(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
) -> None:
|
||||
"""produ*name cannot match a stemmed index ("productname" is indexed as
|
||||
"productnam"); usage.md must not advertise it. Pinned so the limitation
|
||||
is deliberate, not accidental."""
|
||||
assert _matched_ids(backend, "produ*name") == set()
|
||||
|
||||
def test_stem_substitution_reaches_both_the_inflection_and_the_compound(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
) -> None:
|
||||
"""English stemming substitutes as well as truncates: "copy" and
|
||||
"copies" both index as "copi", while "copyright" keeps its literal "y".
|
||||
Neither form is a prefix of the other, so no single normalized string
|
||||
reaches both. The run is therefore emitted as a disjunction of the
|
||||
folded and stemmed forms, and "copy*" reaches the base word, its
|
||||
inflections and the compound alike.
|
||||
"""
|
||||
compound = Document.objects.create(
|
||||
title="Copyright notice",
|
||||
content="copyright notice for the work",
|
||||
checksum="pattern-stemming-2",
|
||||
archive_serial_number=901,
|
||||
)
|
||||
backend.add_or_update(compound)
|
||||
|
||||
assert _matched_ids(backend, "copy*") == {indexed_doc.id, compound.id}
|
||||
assert _matched_ids(backend, "copyright*") == {compound.id}
|
||||
|
||||
|
||||
class TestStemsMatchTheIndexAnalyzer:
|
||||
"""stem_pattern_text rebuilds paperless_text_analyzer's stemming tail rather
|
||||
than sharing it, so a filter added to the index analyzer alone would silently
|
||||
stop patterns from reaching the terms it produces.
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"language",
|
||||
["en", "de", "fr", "es", "sv", None, "klingon"],
|
||||
)
|
||||
@pytest.mark.parametrize(
|
||||
"word",
|
||||
["Copies", "copyright", "Companies", "Invoices", "laufen", "casas", "Straße"],
|
||||
)
|
||||
def test_stem_equals_the_index_term(self, word: str, language: str | None) -> None:
|
||||
indexed = paperless_text_analyzer(language).analyze(word)[0]
|
||||
assert stem_pattern_text(ascii_fold(word.lower()), language) == indexed
|
||||
|
||||
|
||||
def _forms(normalize: PatternNormalizer, text: str) -> tuple[str, ...]:
|
||||
"""The distinct forms a term may match, in order, the way the emitter reads
|
||||
the normalizer's answer (see whoosh_compat.PatternNormalizer)."""
|
||||
result = normalize(text)
|
||||
if isinstance(result, str):
|
||||
return (result,)
|
||||
return tuple(dict.fromkeys(result))
|
||||
|
||||
|
||||
class TestPatternNormalizer:
|
||||
@pytest.mark.parametrize(
|
||||
("text", "expected"),
|
||||
[
|
||||
("Invoice", ("invoice", "invoic")),
|
||||
("companies", ("companies", "compani")),
|
||||
# y -> i is a substitution, so both forms are needed: the index
|
||||
# holds "librari" for "library" and "library" for "librarian".
|
||||
("library", ("library", "librari")),
|
||||
# A run the stemmer leaves alone collapses back to one form, so it
|
||||
# costs exactly the one regex branch it did before.
|
||||
("invoic", ("invoic",)),
|
||||
("Universit", ("universit",)),
|
||||
("Café", ("cafe",)),
|
||||
],
|
||||
)
|
||||
def test_offers_the_typed_run_and_its_stem(
|
||||
self,
|
||||
text: str,
|
||||
expected: tuple[str, ...],
|
||||
) -> None:
|
||||
assert _forms(_make_pattern_normalizer("en"), text) == expected
|
||||
|
||||
def test_run_that_yields_no_token_falls_back_to_the_typed_run(self) -> None:
|
||||
"""A run past the remove_long limit analyzes to zero tokens, so there is
|
||||
no stem to offer and only the folded run remains."""
|
||||
over_long = "invoices" * 20
|
||||
assert _forms(_make_pattern_normalizer("en"), over_long) == (over_long,)
|
||||
|
||||
@pytest.mark.parametrize("language", [None, "klingon"])
|
||||
def test_unstemmed_language_folds_only(self, language: str | None) -> None:
|
||||
"""With no stemmer configured, or one this build has no stemmer for, the
|
||||
index holds surface forms and the pattern must keep them too."""
|
||||
assert _forms(_make_pattern_normalizer(language), "Invoices") == ("invoices",)
|
||||
|
||||
@pytest.mark.parametrize("char", ["a", "Z", "é"])
|
||||
def test_a_single_character_collapses_to_one_folded_form(self, char: str) -> None:
|
||||
"""A bracket class body is normalized one character at a time and the
|
||||
answer is used only when it is a single one-character form, so a
|
||||
stemmer that changed a lone character would silently disable folding
|
||||
inside classes."""
|
||||
forms = _forms(_make_pattern_normalizer("en"), char)
|
||||
assert len(forms) == 1
|
||||
assert len(forms[0]) == 1
|
||||
|
||||
|
||||
class TestBracketClassStillFolds:
|
||||
def test_class_body_matches_case_insensitively(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
indexed_doc: Document,
|
||||
) -> None:
|
||||
"""The class body is folded per character, which the alternatives
|
||||
contract preserves only because a lone character stems to itself."""
|
||||
assert _matched_ids(backend, "title:[IP]nvoice*") == {indexed_doc.id}
|
||||
@@ -1,148 +0,0 @@
|
||||
"""Permission filtering must hold against the real indexed document shape.
|
||||
|
||||
Only three of the index's unsigned ``*_id`` columns are load-bearing:
|
||||
``owner_id``, ``viewer_id`` and ``viewer_group_id``, all read by
|
||||
build_permission_filter. The rest (correspondent/document_type/storage_path/tag
|
||||
ids) were written on every document and read by nothing, and were dropped.
|
||||
|
||||
These tests index real Documents through the backend's own document builder and
|
||||
assert result-level visibility per user, so a mistake about which columns are
|
||||
load-bearing shows up as documents leaking across users rather than as a passing
|
||||
unit test over a hand-built index.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
from django.contrib.auth.models import Group
|
||||
from django.contrib.auth.models import User
|
||||
from guardian.shortcuts import assign_perm
|
||||
|
||||
from documents.models import Correspondent
|
||||
from documents.models import Document
|
||||
from documents.models import DocumentType
|
||||
from documents.models import StoragePath
|
||||
from documents.models import Tag
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from documents.search._backend import TantivyBackend
|
||||
|
||||
pytestmark = [pytest.mark.search, pytest.mark.django_db]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def owner() -> User:
|
||||
return User.objects.create_user(username="owner")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def stranger() -> User:
|
||||
return User.objects.create_user(username="stranger")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def viewer() -> User:
|
||||
return User.objects.create_user(username="viewer")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def group_member() -> User:
|
||||
user = User.objects.create_user(username="group_member")
|
||||
user.groups.add(Group.objects.create(name="accounting"))
|
||||
return user
|
||||
|
||||
|
||||
class TestPermissionFilteringOnIndexedDocuments:
|
||||
def test_unowned_document_is_visible_to_everyone(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
stranger: User,
|
||||
) -> None:
|
||||
doc = Document.objects.create(
|
||||
title="Public Invoice",
|
||||
content="invoice total due",
|
||||
checksum="perm-unowned",
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
assert backend.search_ids("invoice", user=stranger) == [doc.pk]
|
||||
|
||||
def test_owned_document_is_visible_only_to_its_owner(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
owner: User,
|
||||
stranger: User,
|
||||
) -> None:
|
||||
doc = Document.objects.create(
|
||||
title="Private Invoice",
|
||||
content="invoice total due",
|
||||
checksum="perm-owned",
|
||||
owner=owner,
|
||||
)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
assert backend.search_ids("invoice", user=owner) == [doc.pk]
|
||||
assert backend.search_ids("invoice", user=stranger) == []
|
||||
|
||||
def test_explicitly_shared_document_is_visible_to_the_viewer(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
owner: User,
|
||||
viewer: User,
|
||||
stranger: User,
|
||||
) -> None:
|
||||
doc = Document.objects.create(
|
||||
title="Shared Invoice",
|
||||
content="invoice total due",
|
||||
checksum="perm-shared-user",
|
||||
owner=owner,
|
||||
)
|
||||
assign_perm("view_document", viewer, doc)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
assert backend.search_ids("invoice", user=viewer) == [doc.pk]
|
||||
assert backend.search_ids("invoice", user=stranger) == []
|
||||
|
||||
def test_group_shared_document_is_visible_to_group_members(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
owner: User,
|
||||
group_member: User,
|
||||
stranger: User,
|
||||
) -> None:
|
||||
doc = Document.objects.create(
|
||||
title="Group Invoice",
|
||||
content="invoice total due",
|
||||
checksum="perm-shared-group",
|
||||
owner=owner,
|
||||
)
|
||||
assign_perm("view_document", group_member.groups.first(), doc)
|
||||
backend.add_or_update(doc)
|
||||
|
||||
assert backend.search_ids("invoice", user=group_member) == [doc.pk]
|
||||
assert backend.search_ids("invoice", user=stranger) == []
|
||||
|
||||
def test_metadata_does_not_widen_visibility(
|
||||
self,
|
||||
backend: TantivyBackend,
|
||||
owner: User,
|
||||
stranger: User,
|
||||
) -> None:
|
||||
"""A document carrying correspondent/type/storage-path/tag metadata is
|
||||
still filtered by owner alone."""
|
||||
doc = Document.objects.create(
|
||||
title="Tagged Invoice",
|
||||
content="invoice total due",
|
||||
checksum="perm-metadata",
|
||||
owner=owner,
|
||||
correspondent=Correspondent.objects.create(name="ACME"),
|
||||
document_type=DocumentType.objects.create(name="Bill"),
|
||||
storage_path=StoragePath.objects.create(name="Archive", path="archive/"),
|
||||
)
|
||||
doc.tags.add(Tag.objects.create(name="paid"))
|
||||
backend.add_or_update(doc)
|
||||
|
||||
assert backend.search_ids("invoice", user=owner) == [doc.pk]
|
||||
assert backend.search_ids("invoice", user=stranger) == []
|
||||
@@ -1,96 +1,448 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from datetime import UTC
|
||||
from datetime import datetime
|
||||
from datetime import tzinfo
|
||||
from typing import TYPE_CHECKING
|
||||
from zoneinfo import ZoneInfo
|
||||
|
||||
import pytest
|
||||
import tantivy
|
||||
import time_machine
|
||||
|
||||
from documents.search._backend import build_permission_filter
|
||||
from documents.search._errors import InvalidDateQuery
|
||||
from documents.search._errors import InvalidNumberQuery
|
||||
from documents.search._errors import MultipleSearchQueryErrors
|
||||
from documents.search._errors import SearchQueryError
|
||||
from documents.search._dates import _date_only_range
|
||||
from documents.search._dates import _datetime_range
|
||||
from documents.search._query import build_permission_filter
|
||||
from documents.search._query import parse_simple_text_highlight_query
|
||||
from documents.search._query import parse_user_query
|
||||
from documents.search._schema import build_schema
|
||||
from documents.search._tokenizer import register_tokenizers
|
||||
from documents.search._translate import InvalidDateQuery
|
||||
from documents.search._translate import translate_query
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from django.contrib.auth.base_user import AbstractBaseUser
|
||||
|
||||
pytestmark = pytest.mark.search
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def query_index() -> tantivy.Index:
|
||||
"""An in-memory, unstemmed index shared read-only across this module's
|
||||
parse-only tests (none of them index documents)."""
|
||||
schema = build_schema()
|
||||
idx = tantivy.Index(schema, path=None)
|
||||
register_tokenizers(idx, "")
|
||||
return idx
|
||||
EASTERN = ZoneInfo("America/New_York") # UTC-5 / UTC-4 (DST)
|
||||
AUCKLAND = ZoneInfo("Pacific/Auckland") # UTC+13 in southern-hemisphere summer
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def populated_index() -> tantivy.Index:
|
||||
"""An index holding one document, so a query matching nothing is
|
||||
distinguishable from one matching everything."""
|
||||
idx = tantivy.Index(build_schema(), path=None)
|
||||
register_tokenizers(idx, "")
|
||||
writer = idx.writer()
|
||||
doc = tantivy.Document()
|
||||
doc.add_unsigned("id", 1)
|
||||
doc.add_text("content", "needle in indexed content")
|
||||
writer.add_document(doc)
|
||||
writer.commit()
|
||||
idx.reload()
|
||||
return idx
|
||||
def _range(result: str, field: str) -> tuple[str, str]:
|
||||
# Half-open period ranges close with "}" (exclusive); exact-instant ranges
|
||||
# (full ISO datetimes, "now", relative offsets) close with "]" (inclusive).
|
||||
m = re.search(rf"{field}:\[(.+?) TO (.+?)[\]}}]", result)
|
||||
assert m, f"No range for {field!r} in: {result!r}"
|
||||
return m.group(1), m.group(2)
|
||||
|
||||
|
||||
def _highlight_hit_count(index: tantivy.Index, raw_query: str) -> int:
|
||||
query = parse_simple_text_highlight_query(index, raw_query)
|
||||
return index.searcher().search(query, limit=1).count
|
||||
class TestCreatedDateField:
|
||||
"""
|
||||
created is a Django DateField: indexed as midnight UTC of the local calendar
|
||||
date. No offset arithmetic needed - the local calendar date is what matters.
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("tz", "expected_lo", "expected_hi"),
|
||||
[
|
||||
pytest.param(UTC, "2026-03-28T00:00:00Z", "2026-03-29T00:00:00Z", id="utc"),
|
||||
pytest.param(
|
||||
EASTERN,
|
||||
"2026-03-28T00:00:00Z",
|
||||
"2026-03-29T00:00:00Z",
|
||||
id="eastern_same_calendar_date",
|
||||
),
|
||||
],
|
||||
)
|
||||
@time_machine.travel(datetime(2026, 3, 28, 15, 30, tzinfo=UTC), tick=False)
|
||||
def test_today(self, tz: tzinfo, expected_lo: str, expected_hi: str) -> None:
|
||||
lo, hi = _range(translate_query("created:today", tz), "created")
|
||||
assert lo == expected_lo
|
||||
assert hi == expected_hi
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 28, 3, 0, tzinfo=UTC), tick=False)
|
||||
def test_today_auckland_ahead_of_utc(self) -> None:
|
||||
# UTC 03:00 -> Auckland (UTC+13) = 16:00 same date; local date = 2026-03-28
|
||||
lo, _ = _range(
|
||||
translate_query("created:today", AUCKLAND),
|
||||
"created",
|
||||
)
|
||||
assert lo == "2026-03-28T00:00:00Z"
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("field", "keyword", "expected_lo", "expected_hi"),
|
||||
[
|
||||
pytest.param(
|
||||
"created",
|
||||
"yesterday",
|
||||
"2026-03-27T00:00:00Z",
|
||||
"2026-03-28T00:00:00Z",
|
||||
id="yesterday",
|
||||
),
|
||||
pytest.param(
|
||||
"created",
|
||||
"previous week",
|
||||
"2026-03-16T00:00:00Z",
|
||||
"2026-03-23T00:00:00Z",
|
||||
id="previous_week",
|
||||
),
|
||||
pytest.param(
|
||||
"created",
|
||||
"this month",
|
||||
"2026-03-01T00:00:00Z",
|
||||
"2026-04-01T00:00:00Z",
|
||||
id="this_month",
|
||||
),
|
||||
pytest.param(
|
||||
"created",
|
||||
"previous month",
|
||||
"2026-02-01T00:00:00Z",
|
||||
"2026-03-01T00:00:00Z",
|
||||
id="previous_month",
|
||||
),
|
||||
pytest.param(
|
||||
"created",
|
||||
"this year",
|
||||
"2026-01-01T00:00:00Z",
|
||||
"2027-01-01T00:00:00Z",
|
||||
id="this_year",
|
||||
),
|
||||
pytest.param(
|
||||
"created",
|
||||
"previous year",
|
||||
"2025-01-01T00:00:00Z",
|
||||
"2026-01-01T00:00:00Z",
|
||||
id="previous_year",
|
||||
),
|
||||
],
|
||||
)
|
||||
@time_machine.travel(datetime(2026, 3, 28, 15, 0, tzinfo=UTC), tick=False)
|
||||
def test_date_keywords(
|
||||
self,
|
||||
field: str,
|
||||
keyword: str,
|
||||
expected_lo: str,
|
||||
expected_hi: str,
|
||||
) -> None:
|
||||
# 2026-03-28 is Saturday; Mon-Sun week calculation built into expectations
|
||||
query = f"{field}:{keyword}"
|
||||
lo, hi = _range(translate_query(query, UTC), field)
|
||||
assert lo == expected_lo
|
||||
assert hi == expected_hi
|
||||
|
||||
@time_machine.travel(datetime(2026, 12, 15, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_this_month_december_wraps_to_next_year(self) -> None:
|
||||
# December: next month must roll over to January 1 of next year
|
||||
lo, hi = _range(
|
||||
translate_query("created:this month", UTC),
|
||||
"created",
|
||||
)
|
||||
assert lo == "2026-12-01T00:00:00Z"
|
||||
assert hi == "2027-01-01T00:00:00Z"
|
||||
|
||||
@time_machine.travel(datetime(2026, 1, 15, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_last_month_january_wraps_to_previous_year(self) -> None:
|
||||
# January: last month must roll back to December 1 of previous year
|
||||
lo, hi = _range(
|
||||
translate_query("created:previous month", UTC),
|
||||
"created",
|
||||
)
|
||||
assert lo == "2025-12-01T00:00:00Z"
|
||||
assert hi == "2026-01-01T00:00:00Z"
|
||||
|
||||
@time_machine.travel(datetime(2026, 7, 15, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_previous_quarter(self) -> None:
|
||||
lo, hi = _range(
|
||||
translate_query('created:"previous quarter"', UTC),
|
||||
"created",
|
||||
)
|
||||
assert lo == "2026-04-01T00:00:00Z"
|
||||
assert hi == "2026-07-01T00:00:00Z"
|
||||
|
||||
def test_unknown_keyword_raises(self) -> None:
|
||||
with pytest.raises(ValueError, match="Unknown keyword"):
|
||||
_date_only_range("bogus_keyword", UTC)
|
||||
|
||||
|
||||
class TestDateTimeFields:
|
||||
"""
|
||||
added/modified store full UTC datetimes. Natural keywords must convert
|
||||
the local day boundaries to UTC - timezone offset arithmetic IS required.
|
||||
"""
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 28, 15, 30, tzinfo=UTC), tick=False)
|
||||
def test_added_today_eastern(self) -> None:
|
||||
# EDT = UTC-4; local midnight 2026-03-28 00:00 EDT = 2026-03-28 04:00 UTC
|
||||
lo, hi = _range(translate_query("added:today", EASTERN), "added")
|
||||
assert lo == "2026-03-28T04:00:00Z"
|
||||
assert hi == "2026-03-29T04:00:00Z"
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 29, 2, 0, tzinfo=UTC), tick=False)
|
||||
def test_added_today_auckland_midnight_crossing(self) -> None:
|
||||
# UTC 02:00 on 2026-03-29 -> Auckland (UTC+13) = 2026-03-29 15:00 local
|
||||
# Auckland midnight = UTC 2026-03-28 11:00
|
||||
lo, hi = _range(translate_query("added:today", AUCKLAND), "added")
|
||||
assert lo == "2026-03-28T11:00:00Z"
|
||||
assert hi == "2026-03-29T11:00:00Z"
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 28, 15, 0, tzinfo=UTC), tick=False)
|
||||
def test_modified_today_utc(self) -> None:
|
||||
lo, hi = _range(
|
||||
translate_query("modified:today", UTC),
|
||||
"modified",
|
||||
)
|
||||
assert lo == "2026-03-28T00:00:00Z"
|
||||
assert hi == "2026-03-29T00:00:00Z"
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("keyword", "expected_lo", "expected_hi"),
|
||||
[
|
||||
pytest.param(
|
||||
"yesterday",
|
||||
"2026-03-27T00:00:00Z",
|
||||
"2026-03-28T00:00:00Z",
|
||||
id="yesterday",
|
||||
),
|
||||
pytest.param(
|
||||
"previous week",
|
||||
"2026-03-16T00:00:00Z",
|
||||
"2026-03-23T00:00:00Z",
|
||||
id="previous_week",
|
||||
),
|
||||
pytest.param(
|
||||
"this month",
|
||||
"2026-03-01T00:00:00Z",
|
||||
"2026-04-01T00:00:00Z",
|
||||
id="this_month",
|
||||
),
|
||||
pytest.param(
|
||||
"previous month",
|
||||
"2026-02-01T00:00:00Z",
|
||||
"2026-03-01T00:00:00Z",
|
||||
id="previous_month",
|
||||
),
|
||||
pytest.param(
|
||||
"this year",
|
||||
"2026-01-01T00:00:00Z",
|
||||
"2027-01-01T00:00:00Z",
|
||||
id="this_year",
|
||||
),
|
||||
pytest.param(
|
||||
"previous year",
|
||||
"2025-01-01T00:00:00Z",
|
||||
"2026-01-01T00:00:00Z",
|
||||
id="previous_year",
|
||||
),
|
||||
],
|
||||
)
|
||||
@time_machine.travel(datetime(2026, 3, 28, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_datetime_keywords_utc(
|
||||
self,
|
||||
keyword: str,
|
||||
expected_lo: str,
|
||||
expected_hi: str,
|
||||
) -> None:
|
||||
# 2026-03-28 is Saturday; weekday()==5 so Monday=2026-03-23
|
||||
lo, hi = _range(translate_query(f"added:{keyword}", UTC), "added")
|
||||
assert lo == expected_lo
|
||||
assert hi == expected_hi
|
||||
|
||||
@time_machine.travel(datetime(2026, 12, 15, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_this_month_december_wraps_to_next_year(self) -> None:
|
||||
# December: next month wraps to January of next year
|
||||
lo, hi = _range(translate_query("added:this month", UTC), "added")
|
||||
assert lo == "2026-12-01T00:00:00Z"
|
||||
assert hi == "2027-01-01T00:00:00Z"
|
||||
|
||||
@time_machine.travel(datetime(2026, 1, 15, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_last_month_january_wraps_to_previous_year(self) -> None:
|
||||
# January: last month wraps back to December of previous year
|
||||
lo, hi = _range(
|
||||
translate_query("added:previous month", UTC),
|
||||
"added",
|
||||
)
|
||||
assert lo == "2025-12-01T00:00:00Z"
|
||||
assert hi == "2026-01-01T00:00:00Z"
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("query", "expected_lo", "expected_hi"),
|
||||
[
|
||||
pytest.param(
|
||||
'added:"previous quarter"',
|
||||
"2026-04-01T00:00:00Z",
|
||||
"2026-07-01T00:00:00Z",
|
||||
id="quoted_previous_quarter",
|
||||
),
|
||||
pytest.param(
|
||||
"added:previous month",
|
||||
"2026-06-01T00:00:00Z",
|
||||
"2026-07-01T00:00:00Z",
|
||||
id="bare_previous_month",
|
||||
),
|
||||
pytest.param(
|
||||
"added:this month",
|
||||
"2026-07-01T00:00:00Z",
|
||||
"2026-08-01T00:00:00Z",
|
||||
id="bare_this_month",
|
||||
),
|
||||
],
|
||||
)
|
||||
@time_machine.travel(datetime(2026, 7, 15, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_legacy_natural_language_aliases(
|
||||
self,
|
||||
query: str,
|
||||
expected_lo: str,
|
||||
expected_hi: str,
|
||||
) -> None:
|
||||
lo, hi = _range(translate_query(query, UTC), "added")
|
||||
assert lo == expected_lo
|
||||
assert hi == expected_hi
|
||||
|
||||
def test_unknown_keyword_raises(self) -> None:
|
||||
with pytest.raises(ValueError, match="Unknown keyword"):
|
||||
_datetime_range("bogus_keyword", UTC)
|
||||
|
||||
|
||||
class TestWhooshQueryRewriting:
|
||||
"""All Whoosh query syntax variants must be rewritten to ISO 8601 before Tantivy parses them."""
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 28, 15, 0, tzinfo=UTC), tick=False)
|
||||
def test_compact_date_shim_rewrites_to_iso(self) -> None:
|
||||
result = translate_query("created:20240115120000", UTC)
|
||||
assert "2024-01-15" in result
|
||||
assert "20240115120000" not in result
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 28, 15, 0, tzinfo=UTC), tick=False)
|
||||
def test_relative_range_shim_removes_now(self) -> None:
|
||||
result = translate_query("added:[now-7d TO now]", UTC)
|
||||
assert "now" not in result
|
||||
assert "2026-03-" in result
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 28, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_bracket_minus_7_days(self) -> None:
|
||||
lo, hi = _range(
|
||||
translate_query("added:[-7 days to now]", UTC),
|
||||
"added",
|
||||
)
|
||||
assert lo == "2026-03-21T12:00:00Z"
|
||||
assert hi == "2026-03-28T12:00:00Z"
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 28, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_bracket_minus_1_week(self) -> None:
|
||||
lo, hi = _range(
|
||||
translate_query("added:[-1 week to now]", UTC),
|
||||
"added",
|
||||
)
|
||||
assert lo == "2026-03-21T12:00:00Z"
|
||||
assert hi == "2026-03-28T12:00:00Z"
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 28, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_bracket_minus_1_month_uses_relativedelta(self) -> None:
|
||||
# relativedelta(months=1) from 2026-03-28 = 2026-02-28 (not 29)
|
||||
lo, hi = _range(
|
||||
translate_query("created:[-1 month to now]", UTC),
|
||||
"created",
|
||||
)
|
||||
assert lo == "2026-02-28T12:00:00Z"
|
||||
assert hi == "2026-03-28T12:00:00Z"
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 28, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_bracket_minus_1_year(self) -> None:
|
||||
lo, hi = _range(
|
||||
translate_query("modified:[-1 year to now]", UTC),
|
||||
"modified",
|
||||
)
|
||||
assert lo == "2025-03-28T12:00:00Z"
|
||||
assert hi == "2026-03-28T12:00:00Z"
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 28, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_bracket_plural_unit_hours(self) -> None:
|
||||
lo, hi = _range(
|
||||
translate_query("added:[-3 hours to now]", UTC),
|
||||
"added",
|
||||
)
|
||||
assert lo == "2026-03-28T09:00:00Z"
|
||||
assert hi == "2026-03-28T12:00:00Z"
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 28, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_bracket_case_insensitive(self) -> None:
|
||||
result = translate_query("added:[-1 WEEK TO NOW]", UTC)
|
||||
assert "now" not in result.lower()
|
||||
lo, hi = _range(result, "added")
|
||||
assert lo == "2026-03-21T12:00:00Z"
|
||||
assert hi == "2026-03-28T12:00:00Z"
|
||||
|
||||
@time_machine.travel(datetime(2026, 3, 28, 12, 0, tzinfo=UTC), tick=False)
|
||||
def test_relative_range_swaps_bounds_when_lo_exceeds_hi(self) -> None:
|
||||
# [now+1h TO now-1h] has lo > hi before substitution; they must be swapped
|
||||
lo, hi = _range(
|
||||
translate_query("added:[now+1h TO now-1h]", UTC),
|
||||
"added",
|
||||
)
|
||||
assert lo == "2026-03-28T11:00:00Z"
|
||||
assert hi == "2026-03-28T13:00:00Z"
|
||||
|
||||
def test_8digit_created_date_field_always_uses_utc_midnight(self) -> None:
|
||||
# created is a DateField: boundaries are always UTC midnight, no TZ offset
|
||||
result = translate_query("created:20231201", EASTERN)
|
||||
lo, hi = _range(result, "created")
|
||||
assert lo == "2023-12-01T00:00:00Z"
|
||||
assert hi == "2023-12-02T00:00:00Z"
|
||||
|
||||
def test_8digit_added_datetime_field_converts_local_midnight_to_utc(self) -> None:
|
||||
# added is DateTimeField: midnight Dec 1 Eastern (EST = UTC-5) = 05:00 UTC
|
||||
result = translate_query("added:20231201", EASTERN)
|
||||
lo, hi = _range(result, "added")
|
||||
assert lo == "2023-12-01T05:00:00Z"
|
||||
assert hi == "2023-12-02T05:00:00Z"
|
||||
|
||||
def test_8digit_modified_datetime_field_converts_local_midnight_to_utc(
|
||||
self,
|
||||
) -> None:
|
||||
result = translate_query("modified:20231201", EASTERN)
|
||||
lo, hi = _range(result, "modified")
|
||||
assert lo == "2023-12-01T05:00:00Z"
|
||||
assert hi == "2023-12-02T05:00:00Z"
|
||||
|
||||
def test_8digit_invalid_date_raises(self) -> None:
|
||||
# The translation pipeline raises InvalidDateQuery for unparsable dates
|
||||
# (e.g. month=13) so the API can surface a 400 telling the user the date
|
||||
# is malformed instead of silently returning zero results.
|
||||
with pytest.raises(InvalidDateQuery) as exc_info:
|
||||
translate_query("added:20231340", UTC)
|
||||
assert exc_info.value.field == "added"
|
||||
assert exc_info.value.value == "20231340"
|
||||
|
||||
|
||||
class TestParseUserQuery:
|
||||
"""parse_user_query runs the full preprocessing pipeline."""
|
||||
|
||||
@pytest.fixture
|
||||
def query_index(self) -> tantivy.Index:
|
||||
schema = build_schema()
|
||||
idx = tantivy.Index(schema, path=None)
|
||||
register_tokenizers(idx, "")
|
||||
return idx
|
||||
|
||||
def test_returns_tantivy_query(self, query_index: tantivy.Index) -> None:
|
||||
assert isinstance(parse_user_query(query_index, "invoice", UTC), tantivy.Query)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"raw_query",
|
||||
[
|
||||
pytest.param("invoice", id="plain_text"),
|
||||
pytest.param("created:today", id="date_keyword"),
|
||||
pytest.param("created:[2005 to 2009]", id="whoosh_date_range"),
|
||||
pytest.param('added:"previous month"', id="quoted_date_phrase"),
|
||||
pytest.param("title:202[0-1]*", id="bracket_class_wildcard"),
|
||||
],
|
||||
)
|
||||
def test_fuzzy_mode_does_not_raise(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
settings,
|
||||
raw_query: str,
|
||||
) -> None:
|
||||
# These are all valid whoosh grammar that tantivy's own query parser
|
||||
# (used only by the fuzzy blend clause) cannot parse; the fuzzy
|
||||
# clause must degrade gracefully instead of raising and failing the
|
||||
# whole query. See _try_parse_fuzzy_query.
|
||||
settings.ADVANCED_FUZZY_SEARCH_THRESHOLD = 0.5
|
||||
assert isinstance(parse_user_query(query_index, raw_query, UTC), tantivy.Query)
|
||||
assert isinstance(parse_user_query(query_index, "invoice", UTC), tantivy.Query)
|
||||
|
||||
def test_date_keyword_resolves_without_raising(
|
||||
def test_date_rewriting_applied_before_tantivy_parse(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
) -> None:
|
||||
# whoosh-compat's DateParserPlugin resolves "today" against the AST
|
||||
# directly (no string rewrite to an ISO range happens anywhere in
|
||||
# this pipeline); the emitted tantivy query must still build cleanly.
|
||||
# created:today must be rewritten to an ISO range before Tantivy parses it;
|
||||
# if passed raw, Tantivy would reject "today" as an invalid date value
|
||||
with time_machine.travel(datetime(2026, 3, 28, 12, 0, tzinfo=UTC), tick=False):
|
||||
q = parse_user_query(query_index, "created:today", UTC)
|
||||
assert isinstance(q, tantivy.Query)
|
||||
@@ -114,58 +466,302 @@ class TestParseUserQuery:
|
||||
) -> None:
|
||||
assert isinstance(parse_user_query(query_index, raw_query, UTC), tantivy.Query)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"raw_query",
|
||||
[
|
||||
# Partial date scalar (year only)
|
||||
pytest.param("created:2020", id="created_year_scalar"),
|
||||
# 8-digit compact date range in brackets
|
||||
pytest.param(
|
||||
"created:[20200101 TO 20201231]",
|
||||
id="created_8digit_bracket_range",
|
||||
),
|
||||
# Comma-separated field + date range (Whoosh v2 multi-clause syntax)
|
||||
pytest.param(
|
||||
"title:x,created:[2020 TO 2021]",
|
||||
id="title_comma_created_range",
|
||||
),
|
||||
# Field alias: type -> document_type
|
||||
pytest.param("type:invoice", id="type_alias"),
|
||||
# Multi-word date keyword
|
||||
pytest.param("created:previous week", id="created_previous_week"),
|
||||
# Full ISO datetime range
|
||||
pytest.param(
|
||||
"created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z]",
|
||||
id="created_iso_range",
|
||||
),
|
||||
# Comma-separated ISO ranges (Whoosh v2 syntax)
|
||||
pytest.param(
|
||||
"created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z],"
|
||||
"added:[2026-05-01T00:00:00Z TO 2026-06-01T00:00:00Z]",
|
||||
id="comma_iso_ranges",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_advanced_search_queries_do_not_raise(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
raw_query: str,
|
||||
) -> None:
|
||||
"""
|
||||
End-to-end: queries that the frontend sends must parse without raising.
|
||||
|
||||
This tests the full pipeline: translate_query -> tantivy parse_query.
|
||||
Equivalent to asserting HTTP 200 (not 400) for each query form.
|
||||
"""
|
||||
with time_machine.travel(datetime(2026, 6, 15, 12, 0, tzinfo=UTC), tick=False):
|
||||
assert isinstance(
|
||||
parse_user_query(query_index, raw_query, UTC),
|
||||
tantivy.Query,
|
||||
)
|
||||
|
||||
def test_invalid_date_propagates_not_swallowed(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
) -> None:
|
||||
# parse_user_query never falls back to the raw query string on a parse
|
||||
# error — a bad date diagnostic from whoosh-compat always maps to an
|
||||
# InvalidDateQuery and must propagate, so the view can return a 400
|
||||
# instead of silently parsing the raw (invalid) date.
|
||||
# parse_user_query falls back to the raw query on unexpected translation
|
||||
# errors, but an InvalidDateQuery is intentional and must propagate so the
|
||||
# view can return a 400 instead of silently parsing the raw (invalid) date.
|
||||
with pytest.raises(InvalidDateQuery) as exc_info:
|
||||
parse_user_query(query_index, "created:202023", UTC)
|
||||
assert exc_info.value.field == "created"
|
||||
assert exc_info.value.value == "202023"
|
||||
|
||||
def test_invalid_number_raises_invalid_number_query(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
) -> None:
|
||||
with pytest.raises(InvalidNumberQuery) as exc_info:
|
||||
parse_user_query(query_index, "asn:notanumber", UTC)
|
||||
assert exc_info.value.field == "asn"
|
||||
assert exc_info.value.value == "notanumber"
|
||||
|
||||
def test_multiple_bad_fields_raise_multiple_search_query_errors(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
) -> None:
|
||||
with pytest.raises(MultipleSearchQueryErrors) as exc_info:
|
||||
parse_user_query(
|
||||
query_index,
|
||||
"created:notadate AND asn:notanumber",
|
||||
UTC,
|
||||
)
|
||||
assert len(exc_info.value.errors) == 2
|
||||
kinds = {type(e) for e in exc_info.value.errors}
|
||||
assert kinds == {InvalidDateQuery, InvalidNumberQuery}
|
||||
class TestYearRangeRewriting:
|
||||
"""Whoosh-style year-only date ranges must be rewritten to ISO 8601."""
|
||||
|
||||
def test_unregistered_id_field_folds_to_literal_text_not_error(
|
||||
@pytest.mark.parametrize(
|
||||
("query", "field", "expected_lo", "expected_hi"),
|
||||
[
|
||||
pytest.param(
|
||||
"created:[2020 TO 2020]",
|
||||
"created",
|
||||
"2020-01-01T00:00:00Z",
|
||||
"2021-01-01T00:00:00Z",
|
||||
id="single_year_created",
|
||||
),
|
||||
pytest.param(
|
||||
"created:[2018 TO 2021]",
|
||||
"created",
|
||||
"2018-01-01T00:00:00Z",
|
||||
"2022-01-01T00:00:00Z",
|
||||
id="multi_year_range_created",
|
||||
),
|
||||
pytest.param(
|
||||
"added:[2022 TO 2023]",
|
||||
"added",
|
||||
"2022-01-01T00:00:00Z",
|
||||
"2024-01-01T00:00:00Z",
|
||||
id="added_field",
|
||||
),
|
||||
pytest.param(
|
||||
"modified:[2021 TO 2021]",
|
||||
"modified",
|
||||
"2021-01-01T00:00:00Z",
|
||||
"2022-01-01T00:00:00Z",
|
||||
id="modified_field",
|
||||
),
|
||||
pytest.param(
|
||||
"created:[2020 to 2020]",
|
||||
"created",
|
||||
"2020-01-01T00:00:00Z",
|
||||
"2021-01-01T00:00:00Z",
|
||||
id="lowercase_to_keyword",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_year_range_rewritten(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
query: str,
|
||||
field: str,
|
||||
expected_lo: str,
|
||||
expected_hi: str,
|
||||
) -> None:
|
||||
# tag_id is intentionally excluded from the FieldRegistry — whoosh-compat
|
||||
# parity leniency folds it into literal text, not a diagnostic/400.
|
||||
# A result-level assertion that this fold actually matches nothing
|
||||
# against real documents lives in
|
||||
# test_acceptance.py::TestUnregisteredIdFieldFoldsToLiteralText.
|
||||
q = parse_user_query(query_index, "tag_id:5", UTC)
|
||||
assert isinstance(q, tantivy.Query)
|
||||
result = translate_query(query, UTC)
|
||||
lo, hi = _range(result, field)
|
||||
assert lo == expected_lo
|
||||
assert hi == expected_hi
|
||||
|
||||
def test_reversed_year_range_is_swapped(self) -> None:
|
||||
# A reversed range must not yield lo > hi, which Tantivy treats as an
|
||||
# empty range (silently zero results). The bounds are swapped instead.
|
||||
result = translate_query("created:[2025 TO 2020]", UTC)
|
||||
lo, hi = _range(result, "created")
|
||||
assert lo == "2020-01-01T00:00:00Z"
|
||||
assert hi == "2026-01-01T00:00:00Z"
|
||||
|
||||
def test_year_range_in_complex_boolean_query(self) -> None:
|
||||
query = "tag:steuer AND (title:2020 OR (NOT title:2019 AND NOT title:2018 AND created:[2020 TO 2020]))"
|
||||
result = translate_query(query, UTC)
|
||||
lo, hi = _range(result, "created")
|
||||
assert lo == "2020-01-01T00:00:00Z"
|
||||
assert hi == "2021-01-01T00:00:00Z"
|
||||
assert "title:2020" in result
|
||||
assert "title:2019" in result
|
||||
assert "title:2018" in result
|
||||
|
||||
def test_already_iso_date_range_passes_through_unchanged(self) -> None:
|
||||
original = "created:[2020-01-01T00:00:00Z TO 2021-01-01T00:00:00Z]"
|
||||
assert translate_query(original, UTC) == original
|
||||
|
||||
def test_8digit_in_brackets_not_matched_as_year_range(self) -> None:
|
||||
# [YYYYMMDD TO YYYYMMDD]: the translation layer converts 8-digit bounds to
|
||||
# ISO day ranges. 20200101 -> 2020-01-01T00:00:00Z (lo of that day);
|
||||
# 20201231 -> the ceil of Dec 31 = 2021-01-01T00:00:00Z (exclusive end).
|
||||
# This is the correct and accepted behavior: old compact form becomes a
|
||||
# proper Tantivy-parseable ISO range.
|
||||
original = "created:[20200101 TO 20201231]"
|
||||
result = translate_query(original, UTC)
|
||||
lo, hi = _range(result, "created")
|
||||
assert lo == "2020-01-01T00:00:00Z"
|
||||
assert hi == "2021-01-01T00:00:00Z"
|
||||
|
||||
|
||||
class TestNonDateFieldsNotRewritten:
|
||||
"""Date rewriters must only fire on the date fields (created/modified/added).
|
||||
|
||||
Integer fields like asn/id/page_count and unknown fields would otherwise be
|
||||
rewritten into date ranges and rejected by Tantivy as type mismatches.
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
pytest.param("asn:20240101", id="asn_8digit"),
|
||||
pytest.param("id:20240101", id="id_8digit"),
|
||||
pytest.param("page_count:12345678", id="page_count_8digit"),
|
||||
pytest.param("num_notes:20231201", id="num_notes_8digit"),
|
||||
],
|
||||
)
|
||||
def test_8digit_on_integer_field_passes_through_unchanged(self, query: str) -> None:
|
||||
assert translate_query(query, EASTERN) == query
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
pytest.param("asn:[2000 TO 2024]", id="asn_year_range"),
|
||||
pytest.param("id:[2000 TO 2024]", id="id_year_range"),
|
||||
pytest.param("page_count:[2000 TO 2024]", id="page_count_year_range"),
|
||||
],
|
||||
)
|
||||
def test_year_range_on_integer_field_passes_through_unchanged(
|
||||
self,
|
||||
query: str,
|
||||
) -> None:
|
||||
assert translate_query(query, UTC) == query
|
||||
|
||||
def test_unknown_field_keyword_passes_through_unchanged(self) -> None:
|
||||
# foobar is not a date field: 'foobar:today' must not become a date range,
|
||||
# which Tantivy would otherwise reject as an unknown/typed field.
|
||||
assert translate_query("foobar:today", UTC) == "foobar:today"
|
||||
|
||||
|
||||
class TestPassthrough:
|
||||
"""Queries without field prefixes or unrelated content pass through unchanged."""
|
||||
|
||||
def test_bare_keyword_no_field_prefix_unchanged(self) -> None:
|
||||
# Bare 'today' with no field: prefix passes through unchanged
|
||||
result = translate_query("bank statement today", UTC)
|
||||
assert "today" in result
|
||||
|
||||
def test_unrelated_query_unchanged(self) -> None:
|
||||
assert translate_query("title:invoice", UTC) == "title:invoice"
|
||||
|
||||
|
||||
class TestNormalizeQuery:
|
||||
"""translate_query expands comma-separated values and collapses whitespace."""
|
||||
|
||||
def test_normalize_expands_comma_separated_tags(self) -> None:
|
||||
assert translate_query("tag:foo,bar", UTC) == "tag:foo AND tag:bar"
|
||||
|
||||
def test_normalize_comma_between_range_expressions(self) -> None:
|
||||
# Comma-separated field range expressions (Whoosh v2 syntax) must be
|
||||
# converted to AND so Tantivy does not receive an invalid comma.
|
||||
q = "created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z],added:[2026-05-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
|
||||
assert translate_query(q, UTC) == (
|
||||
"created:[2026-01-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
|
||||
" AND "
|
||||
"added:[2026-05-01T00:00:00Z TO 2026-06-01T00:00:00Z]"
|
||||
)
|
||||
|
||||
def test_normalize_expands_three_values(self) -> None:
|
||||
assert (
|
||||
translate_query("tag:foo,bar,baz", UTC) == "tag:foo AND tag:bar AND tag:baz"
|
||||
)
|
||||
|
||||
def test_normalize_collapses_whitespace(self) -> None:
|
||||
assert translate_query("bank statement", UTC) == "bank statement"
|
||||
|
||||
def test_normalize_no_commas_unchanged(self) -> None:
|
||||
assert translate_query("bank statement", UTC) == "bank statement"
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("raw", "expected"),
|
||||
[
|
||||
pytest.param(
|
||||
"h52.1 - kurzsichtigkeit",
|
||||
"h52.1 kurzsichtigkeit",
|
||||
id="icd_code_dash_description",
|
||||
),
|
||||
pytest.param(
|
||||
"H52.1 - asd",
|
||||
"H52.1 asd",
|
||||
id="icd_code_uppercase_dash",
|
||||
),
|
||||
pytest.param(
|
||||
"h52.1 -",
|
||||
"h52.1",
|
||||
id="trailing_minus",
|
||||
),
|
||||
pytest.param(
|
||||
". -",
|
||||
".",
|
||||
id="dot_trailing_minus",
|
||||
),
|
||||
pytest.param(
|
||||
"h52. -",
|
||||
"h52.",
|
||||
id="partial_code_trailing_minus",
|
||||
),
|
||||
pytest.param(
|
||||
"foo - bar - baz",
|
||||
"foo bar baz",
|
||||
id="multiple_dashes",
|
||||
),
|
||||
pytest.param(
|
||||
"foo + bar",
|
||||
"foo bar",
|
||||
id="spaced_plus_operator",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_normalize_strips_dangling_operators(self, raw: str, expected: str) -> None:
|
||||
assert translate_query(raw, UTC) == expected
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"query",
|
||||
[
|
||||
pytest.param("term -other", id="adjacent_not_operator"),
|
||||
pytest.param("-term", id="leading_not_operator"),
|
||||
pytest.param("+term", id="leading_must_operator"),
|
||||
pytest.param("foo -bar +baz", id="mixed_adjacent_operators"),
|
||||
],
|
||||
)
|
||||
def test_normalize_preserves_valid_operators(self, query: str) -> None:
|
||||
assert translate_query(query, UTC) == query
|
||||
|
||||
|
||||
class TestParseSimpleTextHighlightQuery:
|
||||
"""parse_simple_text_highlight_query must not raise on natural-language queries."""
|
||||
|
||||
@pytest.fixture
|
||||
def query_index(self) -> tantivy.Index:
|
||||
schema = build_schema()
|
||||
idx = tantivy.Index(schema, path=None)
|
||||
register_tokenizers(idx, "")
|
||||
return idx
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"raw_query",
|
||||
[
|
||||
@@ -187,25 +783,16 @@ class TestParseSimpleTextHighlightQuery:
|
||||
tantivy.Query,
|
||||
)
|
||||
|
||||
def test_a_real_token_matches_the_corpus(
|
||||
self,
|
||||
populated_index: tantivy.Index,
|
||||
) -> None:
|
||||
"""Without this, an empty corpus would make the two assertions below
|
||||
pass for a query that matches every document."""
|
||||
assert _highlight_hit_count(populated_index, "needle") == 1
|
||||
def test_empty_query_returns_empty_query(self, query_index: tantivy.Index) -> None:
|
||||
result = parse_simple_text_highlight_query(query_index, "")
|
||||
assert isinstance(result, tantivy.Query)
|
||||
|
||||
def test_empty_query_matches_no_document(
|
||||
def test_all_operators_returns_empty_query(
|
||||
self,
|
||||
populated_index: tantivy.Index,
|
||||
query_index: tantivy.Index,
|
||||
) -> None:
|
||||
assert _highlight_hit_count(populated_index, "") == 0
|
||||
|
||||
def test_all_operators_query_matches_no_document(
|
||||
self,
|
||||
populated_index: tantivy.Index,
|
||||
) -> None:
|
||||
assert _highlight_hit_count(populated_index, "- +") == 0
|
||||
result = parse_simple_text_highlight_query(query_index, "- +")
|
||||
assert isinstance(result, tantivy.Query)
|
||||
|
||||
|
||||
class TestPermissionFilter:
|
||||
@@ -297,52 +884,3 @@ class TestPermissionFilter:
|
||||
user = django_user_model(pk=20)
|
||||
perm = build_permission_filter(perm_index.schema, user)
|
||||
assert perm_index.searcher().search(perm, limit=10).count == 1 # only unowned
|
||||
|
||||
|
||||
class TestSearchQueryErrors:
|
||||
def test_invalid_date_query_is_a_search_query_error(self) -> None:
|
||||
err = InvalidDateQuery("created", "notadate")
|
||||
assert isinstance(err, SearchQueryError)
|
||||
assert err.field == "created"
|
||||
assert err.value == "notadate"
|
||||
assert "created" in str(err)
|
||||
assert "notadate" in str(err)
|
||||
|
||||
def test_invalid_number_query_is_a_search_query_error(self) -> None:
|
||||
err = InvalidNumberQuery("asn", "notanumber")
|
||||
assert isinstance(err, SearchQueryError)
|
||||
assert err.field == "asn"
|
||||
assert err.value == "notanumber"
|
||||
assert "asn" in str(err)
|
||||
assert "notanumber" in str(err)
|
||||
|
||||
def test_multiple_search_query_errors_aggregates(self) -> None:
|
||||
sub_errors = [
|
||||
InvalidDateQuery("created", "notadate"),
|
||||
InvalidNumberQuery("asn", "notanumber"),
|
||||
]
|
||||
err = MultipleSearchQueryErrors(sub_errors)
|
||||
assert isinstance(err, SearchQueryError)
|
||||
assert err.errors == tuple(sub_errors)
|
||||
assert "created" in str(err)
|
||||
assert "asn" in str(err)
|
||||
|
||||
|
||||
class TestEmitErrorContract:
|
||||
"""A QueryError from emit() surfaces as a SearchQueryError (HTTP 400).
|
||||
|
||||
The Cause-based routing table itself is covered in test_error_routing.py.
|
||||
"""
|
||||
|
||||
def test_exists_requires_fast_gets_the_user_facing_rewrite(
|
||||
self,
|
||||
query_index: tantivy.Index,
|
||||
) -> None:
|
||||
# whoosh-compat's own message advises a host-side fast=True config
|
||||
# change the user can't act on, so this checks OUR wording, not
|
||||
# whoosh-compat's (that's its own test suite's job now).
|
||||
with pytest.raises(SearchQueryError) as exc_info:
|
||||
parse_user_query(query_index, "notes.user:*", UTC)
|
||||
assert str(exc_info.value) == (
|
||||
"Existence searches (field:*) are not supported for field 'notes.user'."
|
||||
)
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user