mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-08-12 05:43:18 +00:00
Compare commits
34
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6380933b96 | ||
|
|
9de12ff1c5 | ||
|
|
d24fcb92b7 | ||
|
|
4df2656623 | ||
|
|
dd5d93e84a | ||
|
|
013b8dbc32 | ||
|
|
c61ea2b398 | ||
|
|
9d673645b8 | ||
|
|
4b1ca62349 | ||
|
|
93e6d25bb7 | ||
|
|
116cced6a6 | ||
|
|
1ce7d62b66 | ||
|
|
44a822cb10 | ||
|
|
6df70e62f7 | ||
|
|
c478d7bb5d | ||
|
|
ef54e0564a | ||
|
|
67628401c8 | ||
|
|
5c96b38f4e | ||
|
|
21ac856e3f | ||
|
|
37d6b02ebc | ||
|
|
7456b52e84 | ||
|
|
6a392ea099 | ||
|
|
2032ad1341 | ||
|
|
cc8fee91c4 | ||
|
|
a5d46a883e | ||
|
|
b0e1793093 | ||
|
|
7e466d1f71 | ||
|
|
7b69a178c0 | ||
|
|
22cd13a8a9 | ||
|
|
72a4676be0 | ||
|
|
6673144d23 | ||
|
|
994a84cf92 | ||
|
|
654ce5d8f3 | ||
|
|
5d5e9b6db4 |
+2
-1
@@ -301,7 +301,8 @@ The following methods are supported:
|
|||||||
- `delete`
|
- `delete`
|
||||||
- No `parameters` required
|
- No `parameters` required
|
||||||
- `reprocess`
|
- `reprocess`
|
||||||
- No `parameters` required
|
- Optional `parameters`: `{ "remote_ocr": true }` to send the documents to the
|
||||||
|
remote OCR engine, see [Remote OCR](usage.md#remote-ocr). Defaults to false.
|
||||||
- `set_permissions`
|
- `set_permissions`
|
||||||
- Requires `parameters`:
|
- Requires `parameters`:
|
||||||
- `"set_permissions": PERMISSIONS_OBJ` (see format [above](#permissions)) and / or
|
- `"set_permissions": PERMISSIONS_OBJ` (see format [above](#permissions)) and / or
|
||||||
|
|||||||
+16
-5
@@ -948,11 +948,10 @@ for display in the web interface.
|
|||||||
|
|
||||||
!!! note
|
!!! note
|
||||||
|
|
||||||
The **remote OCR parser** (Azure AI) also honors this setting: when
|
The **remote OCR parser** (Azure AI) always produces a searchable
|
||||||
no archive is requested (`never`, or `auto` with a born-digital PDF),
|
PDF and stores it as the archive copy, regardless of this setting.
|
||||||
the remote engine is skipped entirely and locally-extracted text is
|
`ARCHIVE_FILE_GENERATION=never` has no effect when the remote
|
||||||
used instead, avoiding an unnecessary API call and a duplicate text
|
parser handles a document.
|
||||||
layer.
|
|
||||||
|
|
||||||
#### [`PAPERLESS_OCR_CLEAN=<mode>`](#PAPERLESS_OCR_CLEAN) {#PAPERLESS_OCR_CLEAN}
|
#### [`PAPERLESS_OCR_CLEAN=<mode>`](#PAPERLESS_OCR_CLEAN) {#PAPERLESS_OCR_CLEAN}
|
||||||
|
|
||||||
@@ -2048,6 +2047,18 @@ password. All of these options come from their similarly-named [Django settings]
|
|||||||
|
|
||||||
Defaults to None.
|
Defaults to None.
|
||||||
|
|
||||||
|
#### [`PAPERLESS_REMOTE_OCR_MODE=<str>`](#PAPERLESS_REMOTE_OCR_MODE) {#PAPERLESS_REMOTE_OCR_MODE}
|
||||||
|
|
||||||
|
: Which documents are sent to the remote OCR engine.
|
||||||
|
|
||||||
|
- `always`: every document of a supported file type is sent to the remote
|
||||||
|
engine, bypassing the local OCR engine.
|
||||||
|
- `workflow_only`: documents are processed locally unless a workflow
|
||||||
|
explicitly enables remote OCR for them, letting you use the remote engine
|
||||||
|
selectively.
|
||||||
|
|
||||||
|
Defaults to "always".
|
||||||
|
|
||||||
## AI {#ai}
|
## AI {#ai}
|
||||||
|
|
||||||
#### [`PAPERLESS_AI_ENABLED=<bool>`](#PAPERLESS_AI_ENABLED) {#PAPERLESS_AI_ENABLED}
|
#### [`PAPERLESS_AI_ENABLED=<bool>`](#PAPERLESS_AI_ENABLED) {#PAPERLESS_AI_ENABLED}
|
||||||
|
|||||||
@@ -456,6 +456,20 @@ def score(
|
|||||||
return 10
|
return 10
|
||||||
```
|
```
|
||||||
|
|
||||||
|
**Remote services**
|
||||||
|
|
||||||
|
If your parser sends document content to a remote service, declare it:
|
||||||
|
|
||||||
|
```python
|
||||||
|
class MyCustomParser:
|
||||||
|
uses_remote_service = True
|
||||||
|
```
|
||||||
|
|
||||||
|
Paperless-ngx excludes such parsers when the document being consumed has not
|
||||||
|
been marked for remote processing, so users can keep remote OCR off by default
|
||||||
|
and enable it selectively with a workflow. Parsers that do not declare the
|
||||||
|
attribute are treated as fully local and are always considered.
|
||||||
|
|
||||||
**Archive and rendition flags**
|
**Archive and rendition flags**
|
||||||
|
|
||||||
```python
|
```python
|
||||||
|
|||||||
@@ -187,11 +187,10 @@ PAPERLESS_ARCHIVE_FILE_GENERATION=auto
|
|||||||
|
|
||||||
### Remote OCR parser
|
### Remote OCR parser
|
||||||
|
|
||||||
If you use the **remote OCR parser** (Azure AI), `ARCHIVE_FILE_GENERATION` is
|
If you use the **remote OCR parser** (Azure AI), note that it always produces a
|
||||||
honored the same way as for the local engine: when no archive is requested
|
searchable PDF and stores it as the archive copy. `ARCHIVE_FILE_GENERATION=never`
|
||||||
(`never`, or `auto` with a born-digital PDF), the remote engine is skipped
|
has no effect for documents handled by the remote parser - the archive is produced
|
||||||
entirely and locally-extracted text is used instead, avoiding an unnecessary
|
unconditionally by the remote engine.
|
||||||
API call and a duplicate text layer.
|
|
||||||
|
|
||||||
## Search Index (Whoosh -> Tantivy)
|
## Search Index (Whoosh -> Tantivy)
|
||||||
|
|
||||||
|
|||||||
+52
-4
@@ -576,9 +576,7 @@ The following workflow action types are available:
|
|||||||
- Tags, correspondent, document type and storage path
|
- Tags, correspondent, document type and storage path
|
||||||
- Document owner
|
- Document owner
|
||||||
- View and / or edit permissions to users or groups
|
- View and / or edit permissions to users or groups
|
||||||
- Custom fields, optionally with a value. If no value is set, the field is only added to the
|
- Custom fields. Note that no value for the field will be set
|
||||||
document and any value it may already have is left untouched. If a value is set, it will
|
|
||||||
overwrite an existing value of that field on the document.
|
|
||||||
|
|
||||||
##### Removal {#workflow-action-removal}
|
##### Removal {#workflow-action-removal}
|
||||||
|
|
||||||
@@ -650,6 +648,48 @@ happened while it was still encrypted, that original version will likewise be mi
|
|||||||
**Current limitation**: Passwords are stored as a simple list without descriptions. To handle
|
**Current limitation**: Passwords are stored as a simple list without descriptions. To handle
|
||||||
multiple PDF types with different passwords, create separate workflows for each use case.
|
multiple PDF types with different passwords, create separate workflows for each use case.
|
||||||
|
|
||||||
|
##### Remote OCR {#workflow-action-remote-ocr}
|
||||||
|
|
||||||
|
"Remote OCR" actions send the document to the configured remote OCR engine instead of processing it
|
||||||
|
locally. To use remote OCR selectively, set the [remote OCR mode](configuration.md#PAPERLESS_REMOTE_OCR_MODE)
|
||||||
|
to `workflow_only` then add this action to a workflow that matches only the documents you
|
||||||
|
want sent to the remote engine. See [Remote OCR](#remote-ocr) for the engine setup. The action only works with
|
||||||
|
a **Consumption Started** trigger.
|
||||||
|
|
||||||
|
The action takes no options, its presence is what enables remote OCR for a matching document.
|
||||||
|
|
||||||
|
If the remote engine is not configured, or does not support the document's file type, the document is
|
||||||
|
processed locally instead and a warning is written to the log.
|
||||||
|
|
||||||
|
##### Apply AI Suggestions {#workflow-action-apply-ai-suggestions}
|
||||||
|
|
||||||
|
"Apply AI Suggestions" actions ask the configured AI service for title and metadata suggestions,
|
||||||
|
the same as the AI suggestions shown on the document detail page, except applied automatically and in bulk.
|
||||||
|
It requires [AI features](configuration.md#ai) to be enabled. You can specify:
|
||||||
|
|
||||||
|
- Which suggestions to apply: title, tags, correspondent, document type, storage path and / or created
|
||||||
|
date. Suggestions for fields you did not select are discarded.
|
||||||
|
- Whether to create missing items. By default only tags, correspondents and document types that
|
||||||
|
already exist are assigned and any other suggestion is dropped. With this enabled, suggested items
|
||||||
|
that do not exist are created. Storage paths are never created.
|
||||||
|
- Whether to overwrite existing values. By default a field is only filled in if it is currently empty.
|
||||||
|
Note that documents almost always already have a title and created date, so if you select those you
|
||||||
|
will usually want to enable this too. Tags are an exception: suggested tags are always added and
|
||||||
|
never replace the document's existing tags.
|
||||||
|
|
||||||
|
The action works with every trigger **except Consumption Started**, because suggestions are made from
|
||||||
|
the document's text, which does not exist until after the document has been processed.
|
||||||
|
|
||||||
|
Because the query to the AI service is slow, the action is queued and runs in the background rather
|
||||||
|
than as part of the workflow run itself. The document is updated once the suggestions come back.
|
||||||
|
|
||||||
|
!!! warning
|
||||||
|
|
||||||
|
Every matching document results in a query to the AI service, which may incur costs and have privacy
|
||||||
|
implications. Queries can be slow, so a workflow matching a large number of documents can occupy the
|
||||||
|
task queue, and delay consumption of new documents, etc. Consider narrowing the trigger filters,
|
||||||
|
running in small batches and / or increasing workers.
|
||||||
|
|
||||||
#### Workflow placeholders
|
#### Workflow placeholders
|
||||||
|
|
||||||
Titles and webhook payloads can be generated by workflows using [Jinja templates](https://jinja.palletsprojects.com/en/3.1.x/templates/).
|
Titles and webhook payloads can be generated by workflows using [Jinja templates](https://jinja.palletsprojects.com/en/3.1.x/templates/).
|
||||||
@@ -1086,11 +1126,19 @@ Paperless-ngx supports performing OCR on documents using remote services. At the
|
|||||||
[Microsoft's Azure "Document Intelligence" service](https://azure.microsoft.com/en-us/products/ai-services/ai-document-intelligence).
|
[Microsoft's Azure "Document Intelligence" service](https://azure.microsoft.com/en-us/products/ai-services/ai-document-intelligence).
|
||||||
This is of course a paid service (with a free tier) which requires an Azure account and subscription. Azure AI is not affiliated with
|
This is of course a paid service (with a free tier) which requires an Azure account and subscription. Azure AI is not affiliated with
|
||||||
Paperless-ngx in any way. When enabled, Paperless-ngx will automatically send appropriate documents to Azure for OCR processing, bypassing
|
Paperless-ngx in any way. When enabled, Paperless-ngx will automatically send appropriate documents to Azure for OCR processing, bypassing
|
||||||
the local OCR engine. See the [configuration](configuration.md#PAPERLESS_REMOTE_OCR_ENGINE) options for more details.
|
the local OCR engine. See the [configuration](configuration.md#PAPERLESS_REMOTE_OCR_ENGINE) options for more details. These
|
||||||
|
settings can be supplied as environment variables or via **Application Configuration**.
|
||||||
|
|
||||||
Additionally, when using a commercial service with this feature, consider both potential costs as well as any associated file size
|
Additionally, when using a commercial service with this feature, consider both potential costs as well as any associated file size
|
||||||
or page limitations (e.g. with a free tier).
|
or page limitations (e.g. with a free tier).
|
||||||
|
|
||||||
|
By default, every document of a supported file type is sent to the remote engine. To use it more selectively, set the
|
||||||
|
[remote OCR mode](configuration.md#PAPERLESS_REMOTE_OCR_MODE) to `workflow_only`. Documents are then processed locally
|
||||||
|
unless a [remote OCR workflow action](#workflow-action-remote-ocr) enables it for them, so you can limit the remote
|
||||||
|
engine to particular documents.
|
||||||
|
|
||||||
|
Setting the mode to `workflow_only` also allows the **Reprocess** actions to selectively use remote OCR for individual documents.
|
||||||
|
|
||||||
## Architecture
|
## Architecture
|
||||||
|
|
||||||
Paperless-ngx consists of the following components:
|
Paperless-ngx consists of the following components:
|
||||||
|
|||||||
+8
-8
@@ -599,7 +599,7 @@
|
|||||||
</context-group>
|
</context-group>
|
||||||
<context-group purpose="location">
|
<context-group purpose="location">
|
||||||
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.html</context>
|
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.html</context>
|
||||||
<context context-type="linenumber">85,86</context>
|
<context context-type="linenumber">84,85</context>
|
||||||
</context-group>
|
</context-group>
|
||||||
</trans-unit>
|
</trans-unit>
|
||||||
<trans-unit id="3768927257183755959" datatype="html">
|
<trans-unit id="3768927257183755959" datatype="html">
|
||||||
@@ -670,7 +670,7 @@
|
|||||||
</context-group>
|
</context-group>
|
||||||
<context-group purpose="location">
|
<context-group purpose="location">
|
||||||
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.html</context>
|
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.html</context>
|
||||||
<context context-type="linenumber">86,87</context>
|
<context context-type="linenumber">85,86</context>
|
||||||
</context-group>
|
</context-group>
|
||||||
</trans-unit>
|
</trans-unit>
|
||||||
<trans-unit id="5079885666748292382" datatype="html">
|
<trans-unit id="5079885666748292382" datatype="html">
|
||||||
@@ -9907,7 +9907,7 @@
|
|||||||
</context-group>
|
</context-group>
|
||||||
<context-group purpose="location">
|
<context-group purpose="location">
|
||||||
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.ts</context>
|
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.ts</context>
|
||||||
<context context-type="linenumber">314</context>
|
<context context-type="linenumber">296</context>
|
||||||
</context-group>
|
</context-group>
|
||||||
</trans-unit>
|
</trans-unit>
|
||||||
<trans-unit id="2620006875434695386" datatype="html">
|
<trans-unit id="2620006875434695386" datatype="html">
|
||||||
@@ -10234,7 +10234,7 @@
|
|||||||
</context-group>
|
</context-group>
|
||||||
<context-group purpose="location">
|
<context-group purpose="location">
|
||||||
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.ts</context>
|
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.ts</context>
|
||||||
<context context-type="linenumber">308</context>
|
<context context-type="linenumber">290</context>
|
||||||
</context-group>
|
</context-group>
|
||||||
</trans-unit>
|
</trans-unit>
|
||||||
<trans-unit id="3501895737484542570" datatype="html">
|
<trans-unit id="3501895737484542570" datatype="html">
|
||||||
@@ -10346,28 +10346,28 @@
|
|||||||
<source>Saved view "<x id="PH" equiv-text="savedView.name"/>" deleted.</source>
|
<source>Saved view "<x id="PH" equiv-text="savedView.name"/>" deleted.</source>
|
||||||
<context-group purpose="location">
|
<context-group purpose="location">
|
||||||
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.ts</context>
|
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.ts</context>
|
||||||
<context context-type="linenumber">178</context>
|
<context context-type="linenumber">160</context>
|
||||||
</context-group>
|
</context-group>
|
||||||
</trans-unit>
|
</trans-unit>
|
||||||
<trans-unit id="1660419335376265526" datatype="html">
|
<trans-unit id="1660419335376265526" datatype="html">
|
||||||
<source>Views saved successfully.</source>
|
<source>Views saved successfully.</source>
|
||||||
<context-group purpose="location">
|
<context-group purpose="location">
|
||||||
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.ts</context>
|
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.ts</context>
|
||||||
<context context-type="linenumber">255</context>
|
<context context-type="linenumber">237</context>
|
||||||
</context-group>
|
</context-group>
|
||||||
</trans-unit>
|
</trans-unit>
|
||||||
<trans-unit id="1699877326523238632" datatype="html">
|
<trans-unit id="1699877326523238632" datatype="html">
|
||||||
<source>Error while saving views.</source>
|
<source>Error while saving views.</source>
|
||||||
<context-group purpose="location">
|
<context-group purpose="location">
|
||||||
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.ts</context>
|
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.ts</context>
|
||||||
<context context-type="linenumber">260</context>
|
<context context-type="linenumber">242</context>
|
||||||
</context-group>
|
</context-group>
|
||||||
</trans-unit>
|
</trans-unit>
|
||||||
<trans-unit id="4919025779187821586" datatype="html">
|
<trans-unit id="4919025779187821586" datatype="html">
|
||||||
<source>Note: Sharing saved views does not share the underlying documents.</source>
|
<source>Note: Sharing saved views does not share the underlying documents.</source>
|
||||||
<context-group purpose="location">
|
<context-group purpose="location">
|
||||||
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.ts</context>
|
<context context-type="sourcefile">src/app/components/manage/saved-views/saved-views.component.ts</context>
|
||||||
<context context-type="linenumber">296</context>
|
<context context-type="linenumber">278</context>
|
||||||
</context-group>
|
</context-group>
|
||||||
</trans-unit>
|
</trans-unit>
|
||||||
<trans-unit id="1229748338333965418" datatype="html">
|
<trans-unit id="1229748338333965418" datatype="html">
|
||||||
|
|||||||
@@ -14,43 +14,48 @@
|
|||||||
<a ngbNavLink>{{category}}</a>
|
<a ngbNavLink>{{category}}</a>
|
||||||
<ng-template ngbNavContent>
|
<ng-template ngbNavContent>
|
||||||
<div class="p-3">
|
<div class="p-3">
|
||||||
<div class="row row-cols-1 row-cols-md-2 row-cols-lg-3 g-2">
|
@for (section of getCategorySections(category); track section) {
|
||||||
@for (option of getCategoryOptions(category); track option.key) {
|
@if (section) {
|
||||||
<div class="col">
|
<h5 class="mt-4 mb-3">{{section}}</h5>
|
||||||
<div class="card bg-light">
|
}
|
||||||
<div class="card-body">
|
<div class="row row-cols-1 row-cols-md-2 row-cols-lg-3 g-2">
|
||||||
<div class="card-title d-flex align-items-center">
|
@for (option of getCategoryOptions(category, section); track option.key) {
|
||||||
<h6 class="mb-0">
|
<div class="col">
|
||||||
{{option.title}}
|
<div class="card bg-light">
|
||||||
</h6>
|
<div class="card-body">
|
||||||
<a class="btn btn-sm btn-link" title="Read the documentation about this setting" i18n-title [href]="getDocsUrl(option.config_key)" target="_blank" referrerpolicy="no-referrer">
|
<div class="card-title d-flex align-items-center">
|
||||||
<i-bs name="info-circle"></i-bs>
|
<h6 class="mb-0">
|
||||||
</a>
|
{{option.title}}
|
||||||
@if (isSet(option.key)) {
|
</h6>
|
||||||
<button type="button" class="btn btn-sm btn-link text-danger ms-auto pe-0" title="Reset" i18n-title (click)="resetOption(option.key)">
|
<a class="btn btn-sm btn-link" title="Read the documentation about this setting" i18n-title [href]="getDocsUrl(option.config_key)" target="_blank" referrerpolicy="no-referrer">
|
||||||
<i-bs class="me-1" name="x"></i-bs><ng-container i18n>Reset</ng-container>
|
<i-bs name="info-circle"></i-bs>
|
||||||
</button>
|
</a>
|
||||||
|
@if (isSet(option.key)) {
|
||||||
|
<button type="button" class="btn btn-sm btn-link text-danger ms-auto pe-0" title="Reset" i18n-title (click)="resetOption(option.key)">
|
||||||
|
<i-bs class="me-1" name="x"></i-bs><ng-container i18n>Reset</ng-container>
|
||||||
|
</button>
|
||||||
|
}
|
||||||
|
</div>
|
||||||
|
<div class="mb-n3">
|
||||||
|
@switch (option.type) {
|
||||||
|
@case (ConfigOptionType.Select) { <pngx-input-select [formControlName]="option.key" [error]="errors[option.key]" [items]="option.choices" [allowNull]="true"></pngx-input-select> }
|
||||||
|
@case (ConfigOptionType.Number) { <pngx-input-number [formControlName]="option.key" [error]="errors[option.key]" [showAdd]="false"></pngx-input-number> }
|
||||||
|
@case (ConfigOptionType.Boolean) { <pngx-input-switch [formControlName]="option.key" [error]="errors[option.key]" [showUnsetNote]="true" [horizontal]="true" title="Enable" i18n-title></pngx-input-switch> }
|
||||||
|
@case (ConfigOptionType.String) { <pngx-input-text [formControlName]="option.key" [error]="errors[option.key]"></pngx-input-text> }
|
||||||
|
@case (ConfigOptionType.JSON) { <pngx-input-text [formControlName]="option.key" [error]="errors[option.key]"></pngx-input-text> }
|
||||||
|
@case (ConfigOptionType.File) { <pngx-input-file [formControlName]="option.key" (upload)="uploadFile($event, option.key)" [error]="errors[option.key]"></pngx-input-file> }
|
||||||
|
@case (ConfigOptionType.Password) { <pngx-input-password [formControlName]="option.key" [error]="errors[option.key]"></pngx-input-password> }
|
||||||
|
}
|
||||||
|
</div>
|
||||||
|
@if (option.note) {
|
||||||
|
<div class="form-text fst-italic">{{option.note}}</div>
|
||||||
}
|
}
|
||||||
</div>
|
</div>
|
||||||
<div class="mb-n3">
|
|
||||||
@switch (option.type) {
|
|
||||||
@case (ConfigOptionType.Select) { <pngx-input-select [formControlName]="option.key" [error]="errors[option.key]" [items]="option.choices" [allowNull]="true"></pngx-input-select> }
|
|
||||||
@case (ConfigOptionType.Number) { <pngx-input-number [formControlName]="option.key" [error]="errors[option.key]" [showAdd]="false"></pngx-input-number> }
|
|
||||||
@case (ConfigOptionType.Boolean) { <pngx-input-switch [formControlName]="option.key" [error]="errors[option.key]" [showUnsetNote]="true" [horizontal]="true" title="Enable" i18n-title></pngx-input-switch> }
|
|
||||||
@case (ConfigOptionType.String) { <pngx-input-text [formControlName]="option.key" [error]="errors[option.key]"></pngx-input-text> }
|
|
||||||
@case (ConfigOptionType.JSON) { <pngx-input-text [formControlName]="option.key" [error]="errors[option.key]"></pngx-input-text> }
|
|
||||||
@case (ConfigOptionType.File) { <pngx-input-file [formControlName]="option.key" (upload)="uploadFile($event, option.key)" [error]="errors[option.key]"></pngx-input-file> }
|
|
||||||
@case (ConfigOptionType.Password) { <pngx-input-password [formControlName]="option.key" [error]="errors[option.key]"></pngx-input-password> }
|
|
||||||
}
|
|
||||||
</div>
|
|
||||||
@if (option.note) {
|
|
||||||
<div class="form-text fst-italic">{{option.note}}</div>
|
|
||||||
}
|
|
||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
</div>
|
}
|
||||||
}
|
</div>
|
||||||
</div>
|
}
|
||||||
</div>
|
</div>
|
||||||
</ng-template>
|
</ng-template>
|
||||||
</li>
|
</li>
|
||||||
|
|||||||
@@ -8,7 +8,11 @@ import { NgbModule } from '@ng-bootstrap/ng-bootstrap'
|
|||||||
import { NgSelectModule } from '@ng-select/ng-select'
|
import { NgSelectModule } from '@ng-select/ng-select'
|
||||||
import { NgxBootstrapIconsModule, allIcons } from 'ngx-bootstrap-icons'
|
import { NgxBootstrapIconsModule, allIcons } from 'ngx-bootstrap-icons'
|
||||||
import { of, throwError } from 'rxjs'
|
import { of, throwError } from 'rxjs'
|
||||||
import { OutputTypeConfig } from 'src/app/data/paperless-config'
|
import {
|
||||||
|
ConfigCategory,
|
||||||
|
ConfigSection,
|
||||||
|
OutputTypeConfig,
|
||||||
|
} from 'src/app/data/paperless-config'
|
||||||
import { ConfigService } from 'src/app/services/config.service'
|
import { ConfigService } from 'src/app/services/config.service'
|
||||||
import { SettingsService } from 'src/app/services/settings.service'
|
import { SettingsService } from 'src/app/services/settings.service'
|
||||||
import { ToastService } from 'src/app/services/toast.service'
|
import { ToastService } from 'src/app/services/toast.service'
|
||||||
@@ -158,4 +162,24 @@ describe('ConfigComponent', () => {
|
|||||||
component.resetOption('barcodes_enabled')
|
component.resetOption('barcodes_enabled')
|
||||||
expect(component.configForm.get('barcodes_enabled').value).toBeNull()
|
expect(component.configForm.get('barcodes_enabled').value).toBeNull()
|
||||||
})
|
})
|
||||||
|
|
||||||
|
it('should group options into sections within a category, or not', () => {
|
||||||
|
const sections = component.getCategorySections(ConfigCategory.OCR)
|
||||||
|
expect(sections).toEqual([null, ConfigSection.RemoteOCR])
|
||||||
|
expect(
|
||||||
|
component
|
||||||
|
.getCategoryOptions(ConfigCategory.OCR)
|
||||||
|
.map((option) => option.key)
|
||||||
|
).toContain('output_type')
|
||||||
|
expect(
|
||||||
|
component
|
||||||
|
.getCategoryOptions(ConfigCategory.OCR, ConfigSection.RemoteOCR)
|
||||||
|
.map((option) => option.key)
|
||||||
|
).toEqual([
|
||||||
|
'remote_ocr_engine',
|
||||||
|
'remote_ocr_api_key',
|
||||||
|
'remote_ocr_endpoint',
|
||||||
|
'remote_ocr_mode',
|
||||||
|
])
|
||||||
|
})
|
||||||
})
|
})
|
||||||
|
|||||||
@@ -74,8 +74,20 @@ export class ConfigComponent
|
|||||||
return Object.values(ConfigCategory)
|
return Object.values(ConfigCategory)
|
||||||
}
|
}
|
||||||
|
|
||||||
getCategoryOptions(category: string): ConfigOption[] {
|
getCategorySections(category: string): string[] {
|
||||||
return PaperlessConfigOptions.filter((o) => o.category === category)
|
return [
|
||||||
|
...new Set(
|
||||||
|
PaperlessConfigOptions.filter((o) => o.category === category).map(
|
||||||
|
(o) => o.section ?? null // null means no section
|
||||||
|
)
|
||||||
|
),
|
||||||
|
]
|
||||||
|
}
|
||||||
|
|
||||||
|
getCategoryOptions(category: string, section: string = null): ConfigOption[] {
|
||||||
|
return PaperlessConfigOptions.filter(
|
||||||
|
(o) => o.category === category && (o.section ?? null) === section
|
||||||
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
initialConfig: PaperlessConfig
|
initialConfig: PaperlessConfig
|
||||||
|
|||||||
+28
@@ -0,0 +1,28 @@
|
|||||||
|
<div class="modal-header">
|
||||||
|
<h4 class="modal-title" id="modal-basic-title">{{title}}</h4>
|
||||||
|
<button type="button" class="btn-close" aria-label="Close" (click)="cancel()">
|
||||||
|
</button>
|
||||||
|
</div>
|
||||||
|
<div class="modal-body">
|
||||||
|
@if (messageBold) {
|
||||||
|
<p class="text-break"><b>{{messageBold}}</b></p>
|
||||||
|
}
|
||||||
|
@if (message) {
|
||||||
|
<p class="mb-0 text-break" [innerHTML]="message"></p>
|
||||||
|
}
|
||||||
|
@if (showRemoteOcr) {
|
||||||
|
<div class="form-check mt-3">
|
||||||
|
<input class="form-check-input" type="checkbox" id="reprocessRemoteOcr" [(ngModel)]="remoteOcr" />
|
||||||
|
<label class="form-check-label" for="reprocessRemoteOcr" i18n>Use remote OCR</label>
|
||||||
|
<div class="form-text" i18n>Sends the document to the configured remote OCR service, which may incur costs.</div>
|
||||||
|
</div>
|
||||||
|
}
|
||||||
|
</div>
|
||||||
|
<div class="modal-footer">
|
||||||
|
<button type="button" class="btn" [class]="cancelBtnClass" (click)="cancel()" [disabled]="!buttonsEnabled">
|
||||||
|
<span class="d-inline-block" style="padding-bottom: 1px;">{{cancelBtnCaption}}</span>
|
||||||
|
</button>
|
||||||
|
<button type="button" class="btn" [class]="btnClass" (click)="confirm()" [disabled]="!confirmButtonEnabled || !buttonsEnabled">
|
||||||
|
{{btnCaption}}
|
||||||
|
</button>
|
||||||
|
</div>
|
||||||
+72
@@ -0,0 +1,72 @@
|
|||||||
|
import { provideHttpClient, withInterceptorsFromDi } from '@angular/common/http'
|
||||||
|
import { provideHttpClientTesting } from '@angular/common/http/testing'
|
||||||
|
import { ComponentFixture, TestBed } from '@angular/core/testing'
|
||||||
|
import { NgbActiveModal } from '@ng-bootstrap/ng-bootstrap'
|
||||||
|
import { RemoteOCRModeConfig } from 'src/app/data/paperless-config'
|
||||||
|
import { SETTINGS_KEYS } from 'src/app/data/ui-settings'
|
||||||
|
import { SettingsService } from 'src/app/services/settings.service'
|
||||||
|
import { ReprocessConfirmDialogComponent } from './reprocess-confirm-dialog.component'
|
||||||
|
|
||||||
|
describe('ReprocessConfirmDialogComponent', () => {
|
||||||
|
let component: ReprocessConfirmDialogComponent
|
||||||
|
let fixture: ComponentFixture<ReprocessConfirmDialogComponent>
|
||||||
|
let settingsService: SettingsService
|
||||||
|
|
||||||
|
const createComponent = (configured: boolean, mode: string) => {
|
||||||
|
settingsService.set(SETTINGS_KEYS.REMOTE_OCR_CONFIGURED, configured)
|
||||||
|
settingsService.set(SETTINGS_KEYS.REMOTE_OCR_MODE, mode)
|
||||||
|
|
||||||
|
fixture = TestBed.createComponent(ReprocessConfirmDialogComponent)
|
||||||
|
component = fixture.componentInstance
|
||||||
|
fixture.detectChanges()
|
||||||
|
}
|
||||||
|
|
||||||
|
beforeEach(async () => {
|
||||||
|
TestBed.configureTestingModule({
|
||||||
|
providers: [
|
||||||
|
NgbActiveModal,
|
||||||
|
provideHttpClient(withInterceptorsFromDi()),
|
||||||
|
provideHttpClientTesting(),
|
||||||
|
],
|
||||||
|
imports: [ReprocessConfirmDialogComponent],
|
||||||
|
}).compileComponents()
|
||||||
|
|
||||||
|
settingsService = TestBed.inject(SettingsService)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should not request remote OCR by default', () => {
|
||||||
|
createComponent(true, RemoteOCRModeConfig.WORKFLOW_ONLY)
|
||||||
|
|
||||||
|
expect(component.remoteOcr).toBeFalsy()
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should not offer remote OCR when no engine is configured', () => {
|
||||||
|
createComponent(false, RemoteOCRModeConfig.WORKFLOW_ONLY)
|
||||||
|
|
||||||
|
expect(component.showRemoteOcr).toBeFalsy()
|
||||||
|
expect(
|
||||||
|
fixture.nativeElement.querySelector('#reprocessRemoteOcr')
|
||||||
|
).toBeNull()
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should not offer remote OCR when it already handles every document', () => {
|
||||||
|
createComponent(true, RemoteOCRModeConfig.ALWAYS)
|
||||||
|
|
||||||
|
expect(component.showRemoteOcr).toBeFalsy()
|
||||||
|
expect(
|
||||||
|
fixture.nativeElement.querySelector('#reprocessRemoteOcr')
|
||||||
|
).toBeNull()
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should offer remote OCR when configured and selective', () => {
|
||||||
|
createComponent(true, RemoteOCRModeConfig.WORKFLOW_ONLY)
|
||||||
|
|
||||||
|
expect(component.showRemoteOcr).toBeTruthy()
|
||||||
|
const checkbox = fixture.nativeElement.querySelector('#reprocessRemoteOcr')
|
||||||
|
expect(checkbox).not.toBeNull()
|
||||||
|
|
||||||
|
checkbox.click()
|
||||||
|
fixture.detectChanges()
|
||||||
|
expect(component.remoteOcr).toBeTruthy()
|
||||||
|
})
|
||||||
|
})
|
||||||
+20
@@ -0,0 +1,20 @@
|
|||||||
|
import { Component, inject } from '@angular/core'
|
||||||
|
import { FormsModule } from '@angular/forms'
|
||||||
|
import { SettingsService } from 'src/app/services/settings.service'
|
||||||
|
import { ConfirmDialogComponent } from '../confirm-dialog.component'
|
||||||
|
|
||||||
|
@Component({
|
||||||
|
selector: 'pngx-reprocess-confirm-dialog',
|
||||||
|
templateUrl: './reprocess-confirm-dialog.component.html',
|
||||||
|
imports: [FormsModule],
|
||||||
|
})
|
||||||
|
export class ReprocessConfirmDialogComponent extends ConfirmDialogComponent {
|
||||||
|
private settings = inject(SettingsService)
|
||||||
|
|
||||||
|
remoteOcr: boolean = false
|
||||||
|
|
||||||
|
public get showRemoteOcr(): boolean {
|
||||||
|
// Hidden when it is not configured, or when it already handles every document anyway.
|
||||||
|
return this.settings.remoteOCRIsSelectable
|
||||||
|
}
|
||||||
|
}
|
||||||
+46
@@ -455,6 +455,52 @@
|
|||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
}
|
}
|
||||||
|
@case (WorkflowActionType.RemoteOcr) {
|
||||||
|
<div class="row">
|
||||||
|
<div class="col">
|
||||||
|
<p class="text-muted small" i18n>The document will be sent to the configured remote OCR service. May incur costs.</p>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
}
|
||||||
|
@case (WorkflowActionType.ApplyAiSuggestions) {
|
||||||
|
<div class="row">
|
||||||
|
<div class="col">
|
||||||
|
<p class="text-muted small" i18n>The document will be sent to the configured AI service for suggestions. Consider costs and privacy.</p>
|
||||||
|
<pngx-input-select
|
||||||
|
i18n-title
|
||||||
|
title="Apply suggestions for"
|
||||||
|
[items]="aiSuggestionFieldOptions"
|
||||||
|
[multiple]="true"
|
||||||
|
formControlName="ai_suggestion_fields"
|
||||||
|
[error]="error?.actions?.[i]?.ai_suggestion_fields"
|
||||||
|
hint="Suggestions for fields that are not selected are discarded."
|
||||||
|
i18n-hint
|
||||||
|
></pngx-input-select>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
<div class="row">
|
||||||
|
<div class="col-md-6">
|
||||||
|
<pngx-input-switch
|
||||||
|
[horizontal]="true"
|
||||||
|
i18n-title
|
||||||
|
title="Create missing items"
|
||||||
|
formControlName="ai_create_missing"
|
||||||
|
hint="Create suggested tags, correspondents and document types that do not exist yet."
|
||||||
|
i18n-hint
|
||||||
|
></pngx-input-switch>
|
||||||
|
</div>
|
||||||
|
<div class="col-md-6">
|
||||||
|
<pngx-input-switch
|
||||||
|
[horizontal]="true"
|
||||||
|
i18n-title
|
||||||
|
title="Overwrite existing values"
|
||||||
|
formControlName="ai_overwrite_existing"
|
||||||
|
hint="Apply suggestions even if the document already has a value. Tags are always added, never replaced."
|
||||||
|
i18n-hint
|
||||||
|
></pngx-input-switch>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
}
|
||||||
}
|
}
|
||||||
</div>
|
</div>
|
||||||
</ng-template>
|
</ng-template>
|
||||||
|
|||||||
+228
-3
@@ -22,6 +22,7 @@ import {
|
|||||||
} from 'src/app/data/matching-model'
|
} from 'src/app/data/matching-model'
|
||||||
import { Workflow } from 'src/app/data/workflow'
|
import { Workflow } from 'src/app/data/workflow'
|
||||||
import {
|
import {
|
||||||
|
AISuggestionField,
|
||||||
WorkflowAction,
|
WorkflowAction,
|
||||||
WorkflowActionType,
|
WorkflowActionType,
|
||||||
} from 'src/app/data/workflow-action'
|
} from 'src/app/data/workflow-action'
|
||||||
@@ -29,6 +30,7 @@ import {
|
|||||||
DocumentSource,
|
DocumentSource,
|
||||||
WorkflowTriggerType,
|
WorkflowTriggerType,
|
||||||
} from 'src/app/data/workflow-trigger'
|
} from 'src/app/data/workflow-trigger'
|
||||||
|
import { SETTINGS_KEYS } from 'src/app/data/ui-settings'
|
||||||
import { IfOwnerDirective } from 'src/app/directives/if-owner.directive'
|
import { IfOwnerDirective } from 'src/app/directives/if-owner.directive'
|
||||||
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
|
import { IfPermissionsDirective } from 'src/app/directives/if-permissions.directive'
|
||||||
import { CorrespondentService } from 'src/app/services/rest/correspondent.service'
|
import { CorrespondentService } from 'src/app/services/rest/correspondent.service'
|
||||||
@@ -48,6 +50,7 @@ import { TagsComponent } from '../../input/tags/tags.component'
|
|||||||
import { TextComponent } from '../../input/text/text.component'
|
import { TextComponent } from '../../input/text/text.component'
|
||||||
import { EditDialogMode } from '../edit-dialog.component'
|
import { EditDialogMode } from '../edit-dialog.component'
|
||||||
import {
|
import {
|
||||||
|
AI_SUGGESTION_FIELD_OPTIONS,
|
||||||
DOCUMENT_SOURCE_OPTIONS,
|
DOCUMENT_SOURCE_OPTIONS,
|
||||||
SCHEDULE_DATE_FIELD_OPTIONS,
|
SCHEDULE_DATE_FIELD_OPTIONS,
|
||||||
TriggerFilterType,
|
TriggerFilterType,
|
||||||
@@ -224,7 +227,12 @@ describe('WorkflowEditDialogComponent', () => {
|
|||||||
).toEqual('Document Added')
|
).toEqual('Document Added')
|
||||||
expect(component.getTriggerTypeOptionName(null)).toEqual('')
|
expect(component.getTriggerTypeOptionName(null)).toEqual('')
|
||||||
expect(component.sourceOptions).toEqual(DOCUMENT_SOURCE_OPTIONS)
|
expect(component.sourceOptions).toEqual(DOCUMENT_SOURCE_OPTIONS)
|
||||||
expect(component.actionTypeOptions).toEqual(WORKFLOW_ACTION_OPTIONS)
|
// Remote OCR is absent until the workflow has a consumption trigger
|
||||||
|
expect(component.actionTypeOptions).toEqual(
|
||||||
|
WORKFLOW_ACTION_OPTIONS.filter(
|
||||||
|
(a) => a.id !== WorkflowActionType.RemoteOcr
|
||||||
|
)
|
||||||
|
)
|
||||||
expect(
|
expect(
|
||||||
component.getActionTypeOptionName(WorkflowActionType.Assignment)
|
component.getActionTypeOptionName(WorkflowActionType.Assignment)
|
||||||
).toEqual('Assignment')
|
).toEqual('Assignment')
|
||||||
@@ -233,14 +241,231 @@ describe('WorkflowEditDialogComponent', () => {
|
|||||||
SCHEDULE_DATE_FIELD_OPTIONS
|
SCHEDULE_DATE_FIELD_OPTIONS
|
||||||
)
|
)
|
||||||
|
|
||||||
// Email disabled
|
// Email, remote OCR and AI all disabled
|
||||||
jest.spyOn(settingsService, 'get').mockReturnValue(false)
|
jest.spyOn(settingsService, 'get').mockReturnValue(false)
|
||||||
component.ngOnInit()
|
component.ngOnInit()
|
||||||
expect(component.actionTypeOptions).toEqual(
|
expect(component.actionTypeOptions).toEqual(
|
||||||
WORKFLOW_ACTION_OPTIONS.filter((a) => a.id !== WorkflowActionType.Email)
|
WORKFLOW_ACTION_OPTIONS.filter(
|
||||||
|
(a) =>
|
||||||
|
a.id !== WorkflowActionType.Email &&
|
||||||
|
a.id !== WorkflowActionType.RemoteOcr &&
|
||||||
|
a.id !== WorkflowActionType.ApplyAiSuggestions
|
||||||
|
)
|
||||||
)
|
)
|
||||||
})
|
})
|
||||||
|
|
||||||
|
it('should offer remote OCR only for consumption workflows', () => {
|
||||||
|
jest.spyOn(settingsService, 'get').mockReturnValue(true)
|
||||||
|
|
||||||
|
// A consumption trigger makes the action reachable
|
||||||
|
component.object = {
|
||||||
|
name: 'Workflow 1',
|
||||||
|
order: 0,
|
||||||
|
enabled: true,
|
||||||
|
triggers: [{ type: WorkflowTriggerType.Consumption }],
|
||||||
|
actions: [],
|
||||||
|
} as Workflow
|
||||||
|
component.ngOnInit()
|
||||||
|
expect(component.actionTypeOptions.map((a) => a.id)).toContain(
|
||||||
|
WorkflowActionType.RemoteOcr
|
||||||
|
)
|
||||||
|
|
||||||
|
// Any other trigger type runs after the document has been parsed
|
||||||
|
component.object = {
|
||||||
|
name: 'Workflow 2',
|
||||||
|
order: 0,
|
||||||
|
enabled: true,
|
||||||
|
triggers: [{ type: WorkflowTriggerType.DocumentAdded }],
|
||||||
|
actions: [],
|
||||||
|
} as Workflow
|
||||||
|
component.ngOnInit()
|
||||||
|
expect(component.actionTypeOptions.map((a) => a.id)).not.toContain(
|
||||||
|
WorkflowActionType.RemoteOcr
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should offer remote OCR on a trigger added to a new workflow', () => {
|
||||||
|
jest.spyOn(settingsService, 'get').mockReturnValue(true)
|
||||||
|
component.ngOnInit()
|
||||||
|
|
||||||
|
// Nothing for the action to apply to yet
|
||||||
|
expect(component.actionTypeOptions.map((a) => a.id)).not.toContain(
|
||||||
|
WorkflowActionType.RemoteOcr
|
||||||
|
)
|
||||||
|
|
||||||
|
// addTrigger creates the form field with emitEvent false, so the options
|
||||||
|
// have to be computed on read rather than cached from valueChanges
|
||||||
|
component.addTrigger()
|
||||||
|
expect(component.actionTypeOptions.map((a) => a.id)).toContain(
|
||||||
|
WorkflowActionType.RemoteOcr
|
||||||
|
)
|
||||||
|
|
||||||
|
// Switching that trigger to a type that runs after parsing removes it
|
||||||
|
component.triggerFields
|
||||||
|
.at(0)
|
||||||
|
.get('type')
|
||||||
|
.setValue(WorkflowTriggerType.DocumentAdded)
|
||||||
|
expect(component.actionTypeOptions.map((a) => a.id)).not.toContain(
|
||||||
|
WorkflowActionType.RemoteOcr
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should keep remote OCR listed when an action already uses it', () => {
|
||||||
|
jest.spyOn(settingsService, 'get').mockReturnValue(true)
|
||||||
|
|
||||||
|
// Otherwise changing the trigger would silently blank the selection
|
||||||
|
component.object = {
|
||||||
|
name: 'Workflow 1',
|
||||||
|
order: 0,
|
||||||
|
enabled: true,
|
||||||
|
triggers: [{ type: WorkflowTriggerType.DocumentAdded }],
|
||||||
|
actions: [{ type: WorkflowActionType.RemoteOcr }],
|
||||||
|
} as Workflow
|
||||||
|
component.ngOnInit()
|
||||||
|
|
||||||
|
expect(component.actionTypeOptions.map((a) => a.id)).toContain(
|
||||||
|
WorkflowActionType.RemoteOcr
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should not offer remote OCR when no engine is configured', () => {
|
||||||
|
jest
|
||||||
|
.spyOn(settingsService, 'get')
|
||||||
|
.mockImplementation((key) => key !== SETTINGS_KEYS.REMOTE_OCR_CONFIGURED)
|
||||||
|
|
||||||
|
component.object = {
|
||||||
|
name: 'Workflow 1',
|
||||||
|
order: 0,
|
||||||
|
enabled: true,
|
||||||
|
triggers: [{ type: WorkflowTriggerType.Consumption }],
|
||||||
|
actions: [],
|
||||||
|
} as Workflow
|
||||||
|
component.ngOnInit()
|
||||||
|
|
||||||
|
expect(component.actionTypeOptions.map((a) => a.id)).not.toContain(
|
||||||
|
WorkflowActionType.RemoteOcr
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should offer apply AI suggestions unless every trigger is consumption', () => {
|
||||||
|
jest.spyOn(settingsService, 'get').mockReturnValue(true)
|
||||||
|
|
||||||
|
// Consumption runs before the document has been parsed, so there would be
|
||||||
|
// no content to make suggestions from
|
||||||
|
component.object = {
|
||||||
|
name: 'Workflow 1',
|
||||||
|
order: 0,
|
||||||
|
enabled: true,
|
||||||
|
triggers: [{ type: WorkflowTriggerType.Consumption }],
|
||||||
|
actions: [],
|
||||||
|
} as Workflow
|
||||||
|
component.ngOnInit()
|
||||||
|
expect(component.actionTypeOptions.map((a) => a.id)).not.toContain(
|
||||||
|
WorkflowActionType.ApplyAiSuggestions
|
||||||
|
)
|
||||||
|
|
||||||
|
// A second, usable trigger is enough
|
||||||
|
component.object = {
|
||||||
|
name: 'Workflow 2',
|
||||||
|
order: 0,
|
||||||
|
enabled: true,
|
||||||
|
triggers: [
|
||||||
|
{ type: WorkflowTriggerType.Consumption },
|
||||||
|
{ type: WorkflowTriggerType.DocumentAdded },
|
||||||
|
],
|
||||||
|
actions: [],
|
||||||
|
} as Workflow
|
||||||
|
component.ngOnInit()
|
||||||
|
expect(component.actionTypeOptions.map((a) => a.id)).toContain(
|
||||||
|
WorkflowActionType.ApplyAiSuggestions
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should keep apply AI suggestions listed when an action already uses it', () => {
|
||||||
|
jest.spyOn(settingsService, 'get').mockReturnValue(true)
|
||||||
|
|
||||||
|
// Otherwise changing the trigger would silently blank the selection
|
||||||
|
component.object = {
|
||||||
|
name: 'Workflow 1',
|
||||||
|
order: 0,
|
||||||
|
enabled: true,
|
||||||
|
triggers: [{ type: WorkflowTriggerType.Consumption }],
|
||||||
|
actions: [{ type: WorkflowActionType.ApplyAiSuggestions }],
|
||||||
|
} as Workflow
|
||||||
|
component.ngOnInit()
|
||||||
|
|
||||||
|
expect(component.actionTypeOptions.map((a) => a.id)).toContain(
|
||||||
|
WorkflowActionType.ApplyAiSuggestions
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should not offer apply AI suggestions when AI is disabled', () => {
|
||||||
|
jest
|
||||||
|
.spyOn(settingsService, 'get')
|
||||||
|
.mockImplementation((key) => key !== SETTINGS_KEYS.AI_ENABLED)
|
||||||
|
|
||||||
|
component.object = {
|
||||||
|
name: 'Workflow 1',
|
||||||
|
order: 0,
|
||||||
|
enabled: true,
|
||||||
|
triggers: [{ type: WorkflowTriggerType.DocumentAdded }],
|
||||||
|
actions: [],
|
||||||
|
} as Workflow
|
||||||
|
component.ngOnInit()
|
||||||
|
|
||||||
|
expect(component.actionTypeOptions.map((a) => a.id)).not.toContain(
|
||||||
|
WorkflowActionType.ApplyAiSuggestions
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should create form fields for apply AI suggestions options', () => {
|
||||||
|
component.object = {
|
||||||
|
name: 'Workflow 1',
|
||||||
|
order: 0,
|
||||||
|
enabled: true,
|
||||||
|
triggers: [{ type: WorkflowTriggerType.DocumentAdded }],
|
||||||
|
actions: [
|
||||||
|
{
|
||||||
|
type: WorkflowActionType.ApplyAiSuggestions,
|
||||||
|
ai_suggestion_fields: [
|
||||||
|
AISuggestionField.Title,
|
||||||
|
AISuggestionField.Tags,
|
||||||
|
],
|
||||||
|
ai_create_missing: true,
|
||||||
|
ai_overwrite_existing: true,
|
||||||
|
},
|
||||||
|
],
|
||||||
|
} as Workflow
|
||||||
|
component.ngOnInit()
|
||||||
|
|
||||||
|
const action = component.actionFields.at(0)
|
||||||
|
expect(action.get('ai_suggestion_fields').value).toEqual([
|
||||||
|
AISuggestionField.Title,
|
||||||
|
AISuggestionField.Tags,
|
||||||
|
])
|
||||||
|
expect(action.get('ai_create_missing').value).toBeTruthy()
|
||||||
|
expect(action.get('ai_overwrite_existing').value).toBeTruthy()
|
||||||
|
expect(component.aiSuggestionFieldOptions).toEqual(
|
||||||
|
AI_SUGGESTION_FIELD_OPTIONS
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should default apply AI suggestions options on a new action', () => {
|
||||||
|
component.object = {
|
||||||
|
name: 'Workflow 1',
|
||||||
|
order: 0,
|
||||||
|
enabled: true,
|
||||||
|
triggers: [{ type: WorkflowTriggerType.DocumentAdded }],
|
||||||
|
actions: [],
|
||||||
|
} as Workflow
|
||||||
|
component.addAction()
|
||||||
|
|
||||||
|
const action = component.actionFields.at(component.actionFields.length - 1)
|
||||||
|
expect(action.get('ai_suggestion_fields').value).toEqual([])
|
||||||
|
expect(action.get('ai_create_missing').value).toBeFalsy()
|
||||||
|
expect(action.get('ai_overwrite_existing').value).toBeFalsy()
|
||||||
|
})
|
||||||
|
|
||||||
it('should support add and remove triggers and actions', () => {
|
it('should support add and remove triggers and actions', () => {
|
||||||
component.object = workflow
|
component.object = workflow
|
||||||
component.addTrigger()
|
component.addTrigger()
|
||||||
|
|||||||
+102
-10
@@ -30,6 +30,7 @@ import { StoragePath } from 'src/app/data/storage-path'
|
|||||||
import { SETTINGS_KEYS } from 'src/app/data/ui-settings'
|
import { SETTINGS_KEYS } from 'src/app/data/ui-settings'
|
||||||
import { Workflow } from 'src/app/data/workflow'
|
import { Workflow } from 'src/app/data/workflow'
|
||||||
import {
|
import {
|
||||||
|
AISuggestionField,
|
||||||
WorkflowAction,
|
WorkflowAction,
|
||||||
WorkflowActionType,
|
WorkflowActionType,
|
||||||
} from 'src/app/data/workflow-action'
|
} from 'src/app/data/workflow-action'
|
||||||
@@ -148,6 +149,41 @@ export const WORKFLOW_ACTION_OPTIONS = [
|
|||||||
id: WorkflowActionType.MoveToTrash,
|
id: WorkflowActionType.MoveToTrash,
|
||||||
name: $localize`Move to trash`,
|
name: $localize`Move to trash`,
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
id: WorkflowActionType.RemoteOcr,
|
||||||
|
name: $localize`Remote OCR`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
id: WorkflowActionType.ApplyAiSuggestions,
|
||||||
|
name: $localize`Apply AI suggestions`,
|
||||||
|
},
|
||||||
|
]
|
||||||
|
|
||||||
|
export const AI_SUGGESTION_FIELD_OPTIONS = [
|
||||||
|
{
|
||||||
|
id: AISuggestionField.Title,
|
||||||
|
name: $localize`Title`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
id: AISuggestionField.Tags,
|
||||||
|
name: $localize`Tags`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
id: AISuggestionField.Correspondent,
|
||||||
|
name: $localize`Correspondent`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
id: AISuggestionField.DocumentType,
|
||||||
|
name: $localize`Document type`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
id: AISuggestionField.StoragePath,
|
||||||
|
name: $localize`Storage path`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
id: AISuggestionField.Created,
|
||||||
|
name: $localize`Created date`,
|
||||||
|
},
|
||||||
]
|
]
|
||||||
|
|
||||||
export enum TriggerFilterType {
|
export enum TriggerFilterType {
|
||||||
@@ -504,8 +540,6 @@ export class WorkflowEditDialogComponent
|
|||||||
|
|
||||||
expandedItem: number = null
|
expandedItem: number = null
|
||||||
|
|
||||||
readonly allowedActionTypes = signal([])
|
|
||||||
|
|
||||||
private readonly triggerFilterOptionsMap = new WeakMap<
|
private readonly triggerFilterOptionsMap = new WeakMap<
|
||||||
FormArray,
|
FormArray,
|
||||||
TriggerFilterOption[]
|
TriggerFilterOption[]
|
||||||
@@ -548,13 +582,58 @@ export class WorkflowEditDialogComponent
|
|||||||
this.checkRemovalActionFields.bind(this)
|
this.checkRemovalActionFields.bind(this)
|
||||||
)
|
)
|
||||||
this.checkRemovalActionFields(this.objectForm.value)
|
this.checkRemovalActionFields(this.objectForm.value)
|
||||||
this.allowedActionTypes.set(
|
}
|
||||||
this.settingsService.get(SETTINGS_KEYS.EMAIL_ENABLED)
|
|
||||||
? WORKFLOW_ACTION_OPTIONS
|
private allowedActionTypes: typeof WORKFLOW_ACTION_OPTIONS = null
|
||||||
: WORKFLOW_ACTION_OPTIONS.filter(
|
|
||||||
(a) => a.id !== WorkflowActionType.Email
|
private getAllowedActionTypes() {
|
||||||
)
|
let allowed = WORKFLOW_ACTION_OPTIONS
|
||||||
)
|
|
||||||
|
if (!this.settingsService.get(SETTINGS_KEYS.EMAIL_ENABLED)) {
|
||||||
|
allowed = allowed.filter((a) => a.id !== WorkflowActionType.Email)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Remote OCR is decided before the document is parsed, so it is only
|
||||||
|
// offered for workflows that run at consumption.
|
||||||
|
const formWorkflow: Workflow = this.objectForm?.value
|
||||||
|
const remoteOcrUsable =
|
||||||
|
this.settingsService.get(SETTINGS_KEYS.REMOTE_OCR_CONFIGURED) &&
|
||||||
|
(formWorkflow?.triggers?.some(
|
||||||
|
(trigger) => trigger.type === WorkflowTriggerType.Consumption
|
||||||
|
) ||
|
||||||
|
formWorkflow?.actions?.some(
|
||||||
|
(action) => action.type === WorkflowActionType.RemoteOcr
|
||||||
|
))
|
||||||
|
if (!remoteOcrUsable) {
|
||||||
|
allowed = allowed.filter((a) => a.id !== WorkflowActionType.RemoteOcr)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Only available after consumption. Unlike remote OCR this is hidden only
|
||||||
|
// once every trigger is consumption, so it stays offered on a workflow
|
||||||
|
// that has no triggers yet.
|
||||||
|
const aiSuggestionsUsable =
|
||||||
|
this.settingsService.get(SETTINGS_KEYS.AI_ENABLED) &&
|
||||||
|
(!formWorkflow?.triggers?.length ||
|
||||||
|
formWorkflow.triggers.some(
|
||||||
|
(trigger) => trigger.type !== WorkflowTriggerType.Consumption
|
||||||
|
) ||
|
||||||
|
formWorkflow.actions?.some(
|
||||||
|
(action) => action.type === WorkflowActionType.ApplyAiSuggestions
|
||||||
|
))
|
||||||
|
if (!aiSuggestionsUsable) {
|
||||||
|
allowed = allowed.filter(
|
||||||
|
(a) => a.id !== WorkflowActionType.ApplyAiSuggestions
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
if (
|
||||||
|
this.allowedActionTypes?.length === allowed.length &&
|
||||||
|
this.allowedActionTypes.every((a, i) => a.id === allowed[i].id)
|
||||||
|
) {
|
||||||
|
return this.allowedActionTypes
|
||||||
|
}
|
||||||
|
this.allowedActionTypes = allowed
|
||||||
|
return allowed
|
||||||
}
|
}
|
||||||
|
|
||||||
private checkRemovalActionFields(formWorkflow: Workflow) {
|
private checkRemovalActionFields(formWorkflow: Workflow) {
|
||||||
@@ -1198,6 +1277,11 @@ export class WorkflowEditDialogComponent
|
|||||||
passwords: new FormControl(
|
passwords: new FormControl(
|
||||||
this.formatPasswords(action.passwords ?? [])
|
this.formatPasswords(action.passwords ?? [])
|
||||||
),
|
),
|
||||||
|
ai_suggestion_fields: new FormControl(
|
||||||
|
action.ai_suggestion_fields ?? []
|
||||||
|
),
|
||||||
|
ai_create_missing: new FormControl(!!action.ai_create_missing),
|
||||||
|
ai_overwrite_existing: new FormControl(!!action.ai_overwrite_existing),
|
||||||
}),
|
}),
|
||||||
{ emitEvent }
|
{ emitEvent }
|
||||||
)
|
)
|
||||||
@@ -1279,13 +1363,18 @@ export class WorkflowEditDialogComponent
|
|||||||
|
|
||||||
get actionTypeOptions() {
|
get actionTypeOptions() {
|
||||||
this.settingsService.trackChanges()
|
this.settingsService.trackChanges()
|
||||||
return this.allowedActionTypes()
|
// Computed on read rather than cached
|
||||||
|
return this.getAllowedActionTypes()
|
||||||
}
|
}
|
||||||
|
|
||||||
getActionTypeOptionName(type: WorkflowActionType): string {
|
getActionTypeOptionName(type: WorkflowActionType): string {
|
||||||
return this.actionTypeOptions.find((t) => t.id === type)?.name ?? ''
|
return this.actionTypeOptions.find((t) => t.id === type)?.name ?? ''
|
||||||
}
|
}
|
||||||
|
|
||||||
|
get aiSuggestionFieldOptions() {
|
||||||
|
return AI_SUGGESTION_FIELD_OPTIONS
|
||||||
|
}
|
||||||
|
|
||||||
addAction() {
|
addAction() {
|
||||||
if (!this.object) {
|
if (!this.object) {
|
||||||
this.object = Object.assign({}, this.objectForm.value)
|
this.object = Object.assign({}, this.objectForm.value)
|
||||||
@@ -1339,6 +1428,9 @@ export class WorkflowEditDialogComponent
|
|||||||
include_document: false,
|
include_document: false,
|
||||||
},
|
},
|
||||||
passwords: [],
|
passwords: [],
|
||||||
|
ai_suggestion_fields: [],
|
||||||
|
ai_create_missing: false,
|
||||||
|
ai_overwrite_existing: false,
|
||||||
}
|
}
|
||||||
this.object.actions.push(action)
|
this.object.actions.push(action)
|
||||||
this.createActionField(action)
|
this.createActionField(action)
|
||||||
|
|||||||
+4
-4
@@ -52,10 +52,10 @@ describe('CustomFieldsValuesComponent', () => {
|
|||||||
})
|
})
|
||||||
|
|
||||||
it('should set selectedFields and map values correctly', () => {
|
it('should set selectedFields and map values correctly', () => {
|
||||||
component.value = { 1: 'value1', 3: 0, 4: false }
|
component.value = { 1: 'value1' }
|
||||||
component.selectedFields = [1, 2, 3, 4]
|
component.selectedFields = [1, 2]
|
||||||
expect(component.selectedFields).toEqual([1, 2, 3, 4])
|
expect(component.selectedFields).toEqual([1, 2])
|
||||||
expect(component.value).toEqual({ 1: 'value1', 2: null, 3: 0, 4: false })
|
expect(component.value).toEqual({ 1: 'value1', 2: null })
|
||||||
})
|
})
|
||||||
|
|
||||||
it('should return the correct custom field by id', () => {
|
it('should return the correct custom field by id', () => {
|
||||||
|
|||||||
+1
-1
@@ -77,7 +77,7 @@ export class CustomFieldsValuesComponent extends AbstractInputComponent<Object>
|
|||||||
this._selectedFields = newFields
|
this._selectedFields = newFields
|
||||||
// map the selected fields to an object with field_id as key and value as value
|
// map the selected fields to an object with field_id as key and value as value
|
||||||
this.value = newFields.reduce((acc, fieldId) => {
|
this.value = newFields.reduce((acc, fieldId) => {
|
||||||
acc[fieldId] = this.value?.[fieldId] ?? null
|
acc[fieldId] = this.value?.[fieldId] || null
|
||||||
return acc
|
return acc
|
||||||
}, {})
|
}, {})
|
||||||
this.onChange(this.value)
|
this.onChange(this.value)
|
||||||
|
|||||||
@@ -963,12 +963,24 @@ describe('DocumentDetailComponent', () => {
|
|||||||
component.reprocess()
|
component.reprocess()
|
||||||
const modalCloseSpy = jest.spyOn(openModal, 'close')
|
const modalCloseSpy = jest.spyOn(openModal, 'close')
|
||||||
openModal.componentInstance.confirmClicked.next()
|
openModal.componentInstance.confirmClicked.next()
|
||||||
expect(reprocessSpy).toHaveBeenCalledWith({ documents: [doc.id] })
|
expect(reprocessSpy).toHaveBeenCalledWith({ documents: [doc.id] }, false)
|
||||||
expect(modalSpy).toHaveBeenCalled()
|
expect(modalSpy).toHaveBeenCalled()
|
||||||
expect(toastSpy).toHaveBeenCalled()
|
expect(toastSpy).toHaveBeenCalled()
|
||||||
expect(modalCloseSpy).toHaveBeenCalled()
|
expect(modalCloseSpy).toHaveBeenCalled()
|
||||||
})
|
})
|
||||||
|
|
||||||
|
it('should pass remote OCR choice when reprocessing', () => {
|
||||||
|
initNormally()
|
||||||
|
const reprocessSpy = jest.spyOn(documentService, 'reprocessDocuments')
|
||||||
|
reprocessSpy.mockReturnValue(of(true))
|
||||||
|
let openModal: NgbModalRef
|
||||||
|
modalService.activeInstances.subscribe((modal) => (openModal = modal[0]))
|
||||||
|
component.reprocess()
|
||||||
|
openModal.componentInstance.remoteOcr = true
|
||||||
|
openModal.componentInstance.confirmClicked.next()
|
||||||
|
expect(reprocessSpy).toHaveBeenCalledWith({ documents: [doc.id] }, true)
|
||||||
|
})
|
||||||
|
|
||||||
it('should show error if redo ocr call fails', () => {
|
it('should show error if redo ocr call fails', () => {
|
||||||
initNormally()
|
initNormally()
|
||||||
const reprocessSpy = jest.spyOn(documentService, 'reprocessDocuments')
|
const reprocessSpy = jest.spyOn(documentService, 'reprocessDocuments')
|
||||||
|
|||||||
@@ -97,6 +97,7 @@ import { ISODateAdapter } from 'src/app/utils/ngb-iso-date-adapter'
|
|||||||
import * as UTIF from 'utif'
|
import * as UTIF from 'utif'
|
||||||
import { DocumentDetailFieldID } from '../admin/settings/settings.component'
|
import { DocumentDetailFieldID } from '../admin/settings/settings.component'
|
||||||
import { ConfirmDialogComponent } from '../common/confirm-dialog/confirm-dialog.component'
|
import { ConfirmDialogComponent } from '../common/confirm-dialog/confirm-dialog.component'
|
||||||
|
import { ReprocessConfirmDialogComponent } from '../common/confirm-dialog/reprocess-confirm-dialog/reprocess-confirm-dialog.component'
|
||||||
import { PasswordRemovalConfirmDialogComponent } from '../common/confirm-dialog/password-removal-confirm-dialog/password-removal-confirm-dialog.component'
|
import { PasswordRemovalConfirmDialogComponent } from '../common/confirm-dialog/password-removal-confirm-dialog/password-removal-confirm-dialog.component'
|
||||||
import { CustomFieldsDropdownComponent } from '../common/custom-fields-dropdown/custom-fields-dropdown.component'
|
import { CustomFieldsDropdownComponent } from '../common/custom-fields-dropdown/custom-fields-dropdown.component'
|
||||||
import { CorrespondentEditDialogComponent } from '../common/edit-dialog/correspondent-edit-dialog/correspondent-edit-dialog.component'
|
import { CorrespondentEditDialogComponent } from '../common/edit-dialog/correspondent-edit-dialog/correspondent-edit-dialog.component'
|
||||||
@@ -1402,7 +1403,7 @@ export class DocumentDetailComponent
|
|||||||
}
|
}
|
||||||
|
|
||||||
reprocess() {
|
reprocess() {
|
||||||
let modal = this.modalService.open(ConfirmDialogComponent, {
|
let modal = this.modalService.open(ReprocessConfirmDialogComponent, {
|
||||||
backdrop: 'static',
|
backdrop: 'static',
|
||||||
})
|
})
|
||||||
modal.componentInstance.title = $localize`Reprocess confirm`
|
modal.componentInstance.title = $localize`Reprocess confirm`
|
||||||
@@ -1413,7 +1414,10 @@ export class DocumentDetailComponent
|
|||||||
modal.componentInstance.confirmClicked.subscribe(() => {
|
modal.componentInstance.confirmClicked.subscribe(() => {
|
||||||
modal.componentInstance.buttonsEnabled = false
|
modal.componentInstance.buttonsEnabled = false
|
||||||
this.documentsService
|
this.documentsService
|
||||||
.reprocessDocuments({ documents: [this.document().id] })
|
.reprocessDocuments(
|
||||||
|
{ documents: [this.document().id] },
|
||||||
|
modal.componentInstance.remoteOcr
|
||||||
|
)
|
||||||
.subscribe({
|
.subscribe({
|
||||||
next: () => {
|
next: () => {
|
||||||
this.toastService.showInfo(
|
this.toastService.showInfo(
|
||||||
|
|||||||
@@ -1122,6 +1122,7 @@ describe('BulkEditorComponent', () => {
|
|||||||
req.flush(true)
|
req.flush(true)
|
||||||
expect(req.request.body).toEqual({
|
expect(req.request.body).toEqual({
|
||||||
documents: [3, 4],
|
documents: [3, 4],
|
||||||
|
remote_ocr: false,
|
||||||
})
|
})
|
||||||
httpTestingController.match(
|
httpTestingController.match(
|
||||||
`${environment.apiBaseUrl}documents/?page=1&page_size=50&ordering=-created&truncate_content=true&include_selection_data=true`
|
`${environment.apiBaseUrl}documents/?page=1&page_size=50&ordering=-created&truncate_content=true&include_selection_data=true`
|
||||||
|
|||||||
@@ -51,6 +51,7 @@ import { ToastService } from 'src/app/services/toast.service'
|
|||||||
import { flattenTags } from 'src/app/utils/flatten-tags'
|
import { flattenTags } from 'src/app/utils/flatten-tags'
|
||||||
import { queryParamsFromFilterRules } from 'src/app/utils/query-params'
|
import { queryParamsFromFilterRules } from 'src/app/utils/query-params'
|
||||||
import { MergeConfirmDialogComponent } from '../../common/confirm-dialog/merge-confirm-dialog/merge-confirm-dialog.component'
|
import { MergeConfirmDialogComponent } from '../../common/confirm-dialog/merge-confirm-dialog/merge-confirm-dialog.component'
|
||||||
|
import { ReprocessConfirmDialogComponent } from '../../common/confirm-dialog/reprocess-confirm-dialog/reprocess-confirm-dialog.component'
|
||||||
import { RotateConfirmDialogComponent } from '../../common/confirm-dialog/rotate-confirm-dialog/rotate-confirm-dialog.component'
|
import { RotateConfirmDialogComponent } from '../../common/confirm-dialog/rotate-confirm-dialog/rotate-confirm-dialog.component'
|
||||||
import { CorrespondentEditDialogComponent } from '../../common/edit-dialog/correspondent-edit-dialog/correspondent-edit-dialog.component'
|
import { CorrespondentEditDialogComponent } from '../../common/edit-dialog/correspondent-edit-dialog/correspondent-edit-dialog.component'
|
||||||
import { CustomFieldEditDialogComponent } from '../../common/edit-dialog/custom-field-edit-dialog/custom-field-edit-dialog.component'
|
import { CustomFieldEditDialogComponent } from '../../common/edit-dialog/custom-field-edit-dialog/custom-field-edit-dialog.component'
|
||||||
@@ -909,7 +910,7 @@ export class BulkEditorComponent
|
|||||||
}
|
}
|
||||||
|
|
||||||
reprocessSelected() {
|
reprocessSelected() {
|
||||||
let modal = this.modalService.open(ConfirmDialogComponent, {
|
let modal = this.modalService.open(ReprocessConfirmDialogComponent, {
|
||||||
backdrop: 'static',
|
backdrop: 'static',
|
||||||
})
|
})
|
||||||
modal.componentInstance.title = $localize`Reprocess confirm`
|
modal.componentInstance.title = $localize`Reprocess confirm`
|
||||||
@@ -923,7 +924,10 @@ export class BulkEditorComponent
|
|||||||
modal.componentInstance.buttonsEnabled = false
|
modal.componentInstance.buttonsEnabled = false
|
||||||
this.executeDocumentAction(
|
this.executeDocumentAction(
|
||||||
modal,
|
modal,
|
||||||
this.documentService.reprocessDocuments(this.getSelectionQuery())
|
this.documentService.reprocessDocuments(
|
||||||
|
this.getSelectionQuery(),
|
||||||
|
modal.componentInstance.remoteOcr
|
||||||
|
)
|
||||||
)
|
)
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -7,7 +7,7 @@
|
|||||||
</pngx-page-header>
|
</pngx-page-header>
|
||||||
<form [formGroup]="savedViewsForm" (ngSubmit)="save()">
|
<form [formGroup]="savedViewsForm" (ngSubmit)="save()">
|
||||||
<ul class="list-group mb-3" formGroupName="savedViews">
|
<ul class="list-group mb-3" formGroupName="savedViews">
|
||||||
@for (view of pagedSavedViews(); track view) {
|
@for (view of savedViews(); track view) {
|
||||||
<li class="list-group-item py-3">
|
<li class="list-group-item py-3">
|
||||||
<div [formGroupName]="view.id">
|
<div [formGroupName]="view.id">
|
||||||
<div class="row">
|
<div class="row">
|
||||||
@@ -81,11 +81,6 @@
|
|||||||
}
|
}
|
||||||
</ul>
|
</ul>
|
||||||
|
|
||||||
<div class="d-flex align-items-center mb-3">
|
<button type="button" (click)="reset()" class="btn btn-outline-secondary mb-2" [disabled]="(isDirty$ | async) === false" i18n>Cancel</button>
|
||||||
<button type="button" (click)="reset()" class="btn btn-outline-secondary mb-2" [disabled]="(isDirty$ | async) === false" i18n>Cancel</button>
|
<button type="submit" class="btn btn-primary ms-2 mb-2" [disabled]="(isDirty$ | async) === false" i18n>Save</button>
|
||||||
<button type="submit" class="btn btn-primary ms-2 mb-2" [disabled]="(isDirty$ | async) === false" i18n>Save</button>
|
|
||||||
@if (savedViews()?.length > pageSize) {
|
|
||||||
<ngb-pagination class="ms-auto" [pageSize]="pageSize" [collectionSize]="savedViews().length" [page]="page()" [maxSize]="5" (pageChange)="page.set($event)" size="sm" aria-label="Pagination"></ngb-pagination>
|
|
||||||
}
|
|
||||||
</div>
|
|
||||||
</form>
|
</form>
|
||||||
|
|||||||
@@ -4,7 +4,6 @@ import { provideHttpClientTesting } from '@angular/common/http/testing'
|
|||||||
import { signal } from '@angular/core'
|
import { signal } from '@angular/core'
|
||||||
import { ComponentFixture, TestBed } from '@angular/core/testing'
|
import { ComponentFixture, TestBed } from '@angular/core/testing'
|
||||||
import { FormsModule, ReactiveFormsModule } from '@angular/forms'
|
import { FormsModule, ReactiveFormsModule } from '@angular/forms'
|
||||||
import { By } from '@angular/platform-browser'
|
|
||||||
import { NgbModal, NgbModule } from '@ng-bootstrap/ng-bootstrap'
|
import { NgbModal, NgbModule } from '@ng-bootstrap/ng-bootstrap'
|
||||||
import { NgxBootstrapIconsModule, allIcons } from 'ngx-bootstrap-icons'
|
import { NgxBootstrapIconsModule, allIcons } from 'ngx-bootstrap-icons'
|
||||||
import { Subject, of, throwError } from 'rxjs'
|
import { Subject, of, throwError } from 'rxjs'
|
||||||
@@ -223,44 +222,6 @@ describe('SavedViewsComponent', () => {
|
|||||||
).toEqual(view.show_on_dashboard)
|
).toEqual(view.show_on_dashboard)
|
||||||
})
|
})
|
||||||
|
|
||||||
it('should page saved views, clamp the page if views are removed', () => {
|
|
||||||
const manyViews = Array.from({ length: 30 }, (_, i) => ({
|
|
||||||
id: i + 1,
|
|
||||||
name: `view${i + 1}`,
|
|
||||||
})) as SavedView[]
|
|
||||||
const listSpy = jest.spyOn(savedViewService, 'list').mockReturnValue(
|
|
||||||
of({
|
|
||||||
all: manyViews.map((v) => v.id),
|
|
||||||
count: manyViews.length,
|
|
||||||
results: manyViews.concat([]),
|
|
||||||
})
|
|
||||||
)
|
|
||||||
component.ngOnInit()
|
|
||||||
fixture.detectChanges()
|
|
||||||
expect(listSpy).toHaveBeenCalledWith(1, 100000, null, false, {
|
|
||||||
full_perms: true,
|
|
||||||
})
|
|
||||||
expect(component.pagedSavedViews()).toHaveLength(25)
|
|
||||||
expect(fixture.debugElement.query(By.css('ngb-pagination'))).not.toBeNull()
|
|
||||||
// all views have controls, not just the current page
|
|
||||||
expect(
|
|
||||||
Object.keys(component.savedViewsForm.get('savedViews').value)
|
|
||||||
).toHaveLength(30)
|
|
||||||
|
|
||||||
component.page.set(2)
|
|
||||||
expect(component.pagedSavedViews()).toHaveLength(5)
|
|
||||||
|
|
||||||
listSpy.mockReturnValue(
|
|
||||||
of({
|
|
||||||
all: manyViews.slice(0, 25).map((v) => v.id),
|
|
||||||
count: 25,
|
|
||||||
results: manyViews.slice(0, 25),
|
|
||||||
})
|
|
||||||
)
|
|
||||||
component.ngOnInit()
|
|
||||||
expect(component.page()).toEqual(1)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should support editing permissions', () => {
|
it('should support editing permissions', () => {
|
||||||
const confirmClicked = new Subject<any>()
|
const confirmClicked = new Subject<any>()
|
||||||
const modalRef = {
|
const modalRef = {
|
||||||
|
|||||||
@@ -1,19 +1,12 @@
|
|||||||
import { AsyncPipe } from '@angular/common'
|
import { AsyncPipe } from '@angular/common'
|
||||||
import {
|
import { Component, OnDestroy, OnInit, inject, signal } from '@angular/core'
|
||||||
Component,
|
|
||||||
OnDestroy,
|
|
||||||
OnInit,
|
|
||||||
computed,
|
|
||||||
inject,
|
|
||||||
signal,
|
|
||||||
} from '@angular/core'
|
|
||||||
import {
|
import {
|
||||||
FormControl,
|
FormControl,
|
||||||
FormGroup,
|
FormGroup,
|
||||||
FormsModule,
|
FormsModule,
|
||||||
ReactiveFormsModule,
|
ReactiveFormsModule,
|
||||||
} from '@angular/forms'
|
} from '@angular/forms'
|
||||||
import { NgbModal, NgbPaginationModule } from '@ng-bootstrap/ng-bootstrap'
|
import { NgbModal } from '@ng-bootstrap/ng-bootstrap'
|
||||||
import { dirtyCheck } from '@ngneat/dirty-check-forms'
|
import { dirtyCheck } from '@ngneat/dirty-check-forms'
|
||||||
import { NgxBootstrapIconsModule } from 'ngx-bootstrap-icons'
|
import { NgxBootstrapIconsModule } from 'ngx-bootstrap-icons'
|
||||||
import { BehaviorSubject, Observable, of, switchMap, takeUntil } from 'rxjs'
|
import { BehaviorSubject, Observable, of, switchMap, takeUntil } from 'rxjs'
|
||||||
@@ -49,7 +42,6 @@ import { LoadingComponentWithPermissions } from '../../loading-component/loading
|
|||||||
FormsModule,
|
FormsModule,
|
||||||
ReactiveFormsModule,
|
ReactiveFormsModule,
|
||||||
AsyncPipe,
|
AsyncPipe,
|
||||||
NgbPaginationModule,
|
|
||||||
NgxBootstrapIconsModule,
|
NgxBootstrapIconsModule,
|
||||||
],
|
],
|
||||||
})
|
})
|
||||||
@@ -66,14 +58,6 @@ export class SavedViewsComponent
|
|||||||
DisplayMode = DisplayMode
|
DisplayMode = DisplayMode
|
||||||
|
|
||||||
readonly savedViews = signal<SavedView[]>(undefined)
|
readonly savedViews = signal<SavedView[]>(undefined)
|
||||||
readonly page = signal(1)
|
|
||||||
public readonly pageSize = 25
|
|
||||||
// All views are loaded at init, so paging is only for display
|
|
||||||
readonly pagedSavedViews = computed(() => {
|
|
||||||
const start = (this.page() - 1) * this.pageSize
|
|
||||||
return this.savedViews()?.slice(start, start + this.pageSize)
|
|
||||||
})
|
|
||||||
|
|
||||||
private savedViewsGroup = new FormGroup({})
|
private savedViewsGroup = new FormGroup({})
|
||||||
public savedViewsForm: FormGroup = new FormGroup({
|
public savedViewsForm: FormGroup = new FormGroup({
|
||||||
savedViews: this.savedViewsGroup,
|
savedViews: this.savedViewsGroup,
|
||||||
@@ -100,11 +84,9 @@ export class SavedViewsComponent
|
|||||||
private reloadViews(): void {
|
private reloadViews(): void {
|
||||||
this.loading.set(true)
|
this.loading.set(true)
|
||||||
this.savedViewService
|
this.savedViewService
|
||||||
.list(1, 100000, null, false, { full_perms: true })
|
.list(null, null, null, false, { full_perms: true })
|
||||||
.subscribe((r) => {
|
.subscribe((r) => {
|
||||||
this.savedViews.set(r.results)
|
this.savedViews.set(r.results)
|
||||||
const pageCount = Math.ceil(r.results.length / this.pageSize)
|
|
||||||
this.page.update((page) => Math.min(page, Math.max(1, pageCount)))
|
|
||||||
this.initialize()
|
this.initialize()
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -54,6 +54,10 @@ export const ConfigCategory = {
|
|||||||
AI: $localize`AI Settings`,
|
AI: $localize`AI Settings`,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export const ConfigSection = {
|
||||||
|
RemoteOCR: $localize`Remote OCR`,
|
||||||
|
}
|
||||||
|
|
||||||
export const LLMEmbeddingBackendConfig = {
|
export const LLMEmbeddingBackendConfig = {
|
||||||
OPENAI_LIKE: 'openai-like',
|
OPENAI_LIKE: 'openai-like',
|
||||||
HUGGINGFACE: 'huggingface',
|
HUGGINGFACE: 'huggingface',
|
||||||
@@ -65,6 +69,15 @@ export const LLMBackendConfig = {
|
|||||||
OLLAMA: 'ollama',
|
OLLAMA: 'ollama',
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export const RemoteOCREngineConfig = {
|
||||||
|
AZURE_AI: 'azureai',
|
||||||
|
}
|
||||||
|
|
||||||
|
export const RemoteOCRModeConfig = {
|
||||||
|
ALWAYS: 'always',
|
||||||
|
WORKFLOW_ONLY: 'workflow_only',
|
||||||
|
}
|
||||||
|
|
||||||
export interface ConfigOption {
|
export interface ConfigOption {
|
||||||
key: string
|
key: string
|
||||||
title: string
|
title: string
|
||||||
@@ -72,6 +85,7 @@ export interface ConfigOption {
|
|||||||
choices?: Array<{ id: string; name: string }>
|
choices?: Array<{ id: string; name: string }>
|
||||||
config_key?: string
|
config_key?: string
|
||||||
category: string
|
category: string
|
||||||
|
section?: string
|
||||||
note?: string
|
note?: string
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -181,6 +195,43 @@ export const PaperlessConfigOptions: ConfigOption[] = [
|
|||||||
config_key: 'PAPERLESS_OCR_USER_ARGS',
|
config_key: 'PAPERLESS_OCR_USER_ARGS',
|
||||||
category: ConfigCategory.OCR,
|
category: ConfigCategory.OCR,
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
key: 'remote_ocr_engine',
|
||||||
|
title: $localize`Remote OCR Engine`,
|
||||||
|
type: ConfigOptionType.Select,
|
||||||
|
choices: mapToItems(RemoteOCREngineConfig),
|
||||||
|
config_key: 'PAPERLESS_REMOTE_OCR_ENGINE',
|
||||||
|
category: ConfigCategory.OCR,
|
||||||
|
section: ConfigSection.RemoteOCR,
|
||||||
|
note: $localize`Enabling remote OCR sends documents to a third-party service for processing. Consider the privacy implications as well as potential costs before enabling.`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
key: 'remote_ocr_api_key',
|
||||||
|
title: $localize`Remote OCR API Key`,
|
||||||
|
type: ConfigOptionType.Password,
|
||||||
|
config_key: 'PAPERLESS_REMOTE_OCR_API_KEY',
|
||||||
|
category: ConfigCategory.OCR,
|
||||||
|
section: ConfigSection.RemoteOCR,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
key: 'remote_ocr_endpoint',
|
||||||
|
title: $localize`Remote OCR Endpoint`,
|
||||||
|
type: ConfigOptionType.String,
|
||||||
|
config_key: 'PAPERLESS_REMOTE_OCR_ENDPOINT',
|
||||||
|
category: ConfigCategory.OCR,
|
||||||
|
section: ConfigSection.RemoteOCR,
|
||||||
|
note: $localize`Required when using the Azure AI engine.`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
key: 'remote_ocr_mode',
|
||||||
|
title: $localize`Remote OCR Mode`,
|
||||||
|
type: ConfigOptionType.Select,
|
||||||
|
choices: mapToItems(RemoteOCRModeConfig),
|
||||||
|
config_key: 'PAPERLESS_REMOTE_OCR_MODE',
|
||||||
|
category: ConfigCategory.OCR,
|
||||||
|
section: ConfigSection.RemoteOCR,
|
||||||
|
note: $localize`Which documents are sent to the remote engine. Use 'workflow_only' to keep remote OCR off unless a workflow enables it for a document.`,
|
||||||
|
},
|
||||||
{
|
{
|
||||||
key: 'app_logo',
|
key: 'app_logo',
|
||||||
title: $localize`Application Logo`,
|
title: $localize`Application Logo`,
|
||||||
@@ -398,6 +449,10 @@ export interface PaperlessConfig extends ObjectWithId {
|
|||||||
barcode_enable_tag: boolean
|
barcode_enable_tag: boolean
|
||||||
barcode_tag_mapping: object
|
barcode_tag_mapping: object
|
||||||
barcode_tag_split: boolean
|
barcode_tag_split: boolean
|
||||||
|
remote_ocr_engine: string
|
||||||
|
remote_ocr_api_key: string
|
||||||
|
remote_ocr_endpoint: string
|
||||||
|
remote_ocr_mode: string
|
||||||
ai_enabled: boolean
|
ai_enabled: boolean
|
||||||
llm_embedding_backend: string
|
llm_embedding_backend: string
|
||||||
llm_embedding_model: string
|
llm_embedding_model: string
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
import { PdfEditorEditMode } from '../components/common/pdf-editor/pdf-editor-edit-mode'
|
import { PdfEditorEditMode } from '../components/common/pdf-editor/pdf-editor-edit-mode'
|
||||||
import { PdfZoomScale } from '../components/common/pdf-viewer/pdf-viewer.types'
|
import { PdfZoomScale } from '../components/common/pdf-viewer/pdf-viewer.types'
|
||||||
|
import { RemoteOCRModeConfig } from './paperless-config'
|
||||||
import { User } from './user'
|
import { User } from './user'
|
||||||
|
|
||||||
export interface UiSettings {
|
export interface UiSettings {
|
||||||
@@ -94,6 +95,8 @@ export const SETTINGS_KEYS = {
|
|||||||
OUTLOOK_OAUTH_URL: 'outlook_oauth_url',
|
OUTLOOK_OAUTH_URL: 'outlook_oauth_url',
|
||||||
EMAIL_ENABLED: 'email_enabled',
|
EMAIL_ENABLED: 'email_enabled',
|
||||||
AI_ENABLED: 'ai_enabled',
|
AI_ENABLED: 'ai_enabled',
|
||||||
|
REMOTE_OCR_CONFIGURED: 'remote_ocr:configured',
|
||||||
|
REMOTE_OCR_MODE: 'remote_ocr:mode',
|
||||||
}
|
}
|
||||||
|
|
||||||
export const SETTINGS: UiSetting[] = [
|
export const SETTINGS: UiSetting[] = [
|
||||||
@@ -347,4 +350,14 @@ export const SETTINGS: UiSetting[] = [
|
|||||||
type: 'string',
|
type: 'string',
|
||||||
default: PdfEditorEditMode.Create,
|
default: PdfEditorEditMode.Create,
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
key: SETTINGS_KEYS.REMOTE_OCR_CONFIGURED,
|
||||||
|
type: 'boolean',
|
||||||
|
default: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
key: SETTINGS_KEYS.REMOTE_OCR_MODE,
|
||||||
|
type: 'string',
|
||||||
|
default: RemoteOCRModeConfig.ALWAYS,
|
||||||
|
},
|
||||||
]
|
]
|
||||||
|
|||||||
@@ -7,6 +7,18 @@ export enum WorkflowActionType {
|
|||||||
Webhook = 4,
|
Webhook = 4,
|
||||||
PasswordRemoval = 5,
|
PasswordRemoval = 5,
|
||||||
MoveToTrash = 6,
|
MoveToTrash = 6,
|
||||||
|
RemoteOcr = 7,
|
||||||
|
ApplyAiSuggestions = 8,
|
||||||
|
}
|
||||||
|
|
||||||
|
// see src/documents/models.py AISuggestionField
|
||||||
|
export enum AISuggestionField {
|
||||||
|
Title = 'title',
|
||||||
|
Tags = 'tags',
|
||||||
|
Correspondent = 'correspondent',
|
||||||
|
DocumentType = 'document_type',
|
||||||
|
StoragePath = 'storage_path',
|
||||||
|
Created = 'created',
|
||||||
}
|
}
|
||||||
|
|
||||||
export interface WorkflowActionEmail extends ObjectWithId {
|
export interface WorkflowActionEmail extends ObjectWithId {
|
||||||
@@ -101,4 +113,10 @@ export interface WorkflowAction extends ObjectWithId {
|
|||||||
webhook?: WorkflowActionWebhook
|
webhook?: WorkflowActionWebhook
|
||||||
|
|
||||||
passwords?: string[]
|
passwords?: string[]
|
||||||
|
|
||||||
|
ai_suggestion_fields?: AISuggestionField[]
|
||||||
|
|
||||||
|
ai_create_missing?: boolean
|
||||||
|
|
||||||
|
ai_overwrite_existing?: boolean
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -284,6 +284,21 @@ describe(`DocumentService`, () => {
|
|||||||
expect(req.request.method).toEqual('POST')
|
expect(req.request.method).toEqual('POST')
|
||||||
expect(req.request.body).toEqual({
|
expect(req.request.body).toEqual({
|
||||||
documents: ids,
|
documents: ids,
|
||||||
|
remote_ocr: false,
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should request remote OCR when reprocessing with it enabled', () => {
|
||||||
|
const ids = [1, 2, 3]
|
||||||
|
subscription = service
|
||||||
|
.reprocessDocuments({ documents: ids }, true)
|
||||||
|
.subscribe()
|
||||||
|
const req = httpTestingController.expectOne(
|
||||||
|
`${environment.apiBaseUrl}${endpoint}/reprocess/`
|
||||||
|
)
|
||||||
|
expect(req.request.body).toEqual({
|
||||||
|
documents: ids,
|
||||||
|
remote_ocr: true,
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
|
|||||||
@@ -349,9 +349,13 @@ export class DocumentService extends AbstractPaperlessService<Document> {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
reprocessDocuments(selection: DocumentSelectionQuery) {
|
reprocessDocuments(
|
||||||
|
selection: DocumentSelectionQuery,
|
||||||
|
remoteOcr: boolean = false
|
||||||
|
) {
|
||||||
return this.http.post(this.getResourceUrl(null, 'reprocess'), {
|
return this.http.post(this.getResourceUrl(null, 'reprocess'), {
|
||||||
...selection,
|
...selection,
|
||||||
|
remote_ocr: remoteOcr,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -13,6 +13,7 @@ import { environment } from 'src/environments/environment'
|
|||||||
import { CustomFieldDataType } from '../data/custom-field'
|
import { CustomFieldDataType } from '../data/custom-field'
|
||||||
import { DEFAULT_DISPLAY_FIELDS, DisplayField } from '../data/document'
|
import { DEFAULT_DISPLAY_FIELDS, DisplayField } from '../data/document'
|
||||||
import { SavedView } from '../data/saved-view'
|
import { SavedView } from '../data/saved-view'
|
||||||
|
import { RemoteOCRModeConfig } from '../data/paperless-config'
|
||||||
import { SETTINGS_KEYS, UiSettings } from '../data/ui-settings'
|
import { SETTINGS_KEYS, UiSettings } from '../data/ui-settings'
|
||||||
import { PermissionsService } from './permissions.service'
|
import { PermissionsService } from './permissions.service'
|
||||||
import { CustomFieldsService } from './rest/custom-fields.service'
|
import { CustomFieldsService } from './rest/custom-fields.service'
|
||||||
@@ -434,4 +435,26 @@ describe('SettingsService', () => {
|
|||||||
).name
|
).name
|
||||||
).toEqual(customFields[0].name)
|
).toEqual(customFields[0].name)
|
||||||
})
|
})
|
||||||
|
it('should offer remote OCR only when configured and selective', () => {
|
||||||
|
settingsService.set(SETTINGS_KEYS.REMOTE_OCR_CONFIGURED, false)
|
||||||
|
settingsService.set(
|
||||||
|
SETTINGS_KEYS.REMOTE_OCR_MODE,
|
||||||
|
RemoteOCRModeConfig.WORKFLOW_ONLY
|
||||||
|
)
|
||||||
|
expect(settingsService.remoteOCRIsSelectable).toBeFalsy()
|
||||||
|
|
||||||
|
// configured, but already handling every document
|
||||||
|
settingsService.set(SETTINGS_KEYS.REMOTE_OCR_CONFIGURED, true)
|
||||||
|
settingsService.set(
|
||||||
|
SETTINGS_KEYS.REMOTE_OCR_MODE,
|
||||||
|
RemoteOCRModeConfig.ALWAYS
|
||||||
|
)
|
||||||
|
expect(settingsService.remoteOCRIsSelectable).toBeFalsy()
|
||||||
|
|
||||||
|
settingsService.set(
|
||||||
|
SETTINGS_KEYS.REMOTE_OCR_MODE,
|
||||||
|
RemoteOCRModeConfig.WORKFLOW_ONLY
|
||||||
|
)
|
||||||
|
expect(settingsService.remoteOCRIsSelectable).toBeTruthy()
|
||||||
|
})
|
||||||
})
|
})
|
||||||
|
|||||||
@@ -19,6 +19,7 @@ import {
|
|||||||
} from 'src/app/utils/color'
|
} from 'src/app/utils/color'
|
||||||
import { DEFAULT_APP_TITLE, environment } from 'src/environments/environment'
|
import { DEFAULT_APP_TITLE, environment } from 'src/environments/environment'
|
||||||
import { DEFAULT_DISPLAY_FIELDS, DisplayField } from '../data/document'
|
import { DEFAULT_DISPLAY_FIELDS, DisplayField } from '../data/document'
|
||||||
|
import { RemoteOCRModeConfig } from '../data/paperless-config'
|
||||||
import { SavedView } from '../data/saved-view'
|
import { SavedView } from '../data/saved-view'
|
||||||
import {
|
import {
|
||||||
PAPERLESS_GREEN_HEX,
|
PAPERLESS_GREEN_HEX,
|
||||||
@@ -687,6 +688,17 @@ export class SettingsService {
|
|||||||
return this.settingIsSet(SETTINGS_KEYS.UPDATE_CHECKING_ENABLED)
|
return this.settingIsSet(SETTINGS_KEYS.UPDATE_CHECKING_ENABLED)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Offering remote OCR as a choice only makes sense when an engine
|
||||||
|
* is configured but is not already handling every document.
|
||||||
|
*/
|
||||||
|
get remoteOCRIsSelectable(): boolean {
|
||||||
|
return (
|
||||||
|
this.get(SETTINGS_KEYS.REMOTE_OCR_CONFIGURED) &&
|
||||||
|
this.get(SETTINGS_KEYS.REMOTE_OCR_MODE) !== RemoteOCRModeConfig.ALWAYS
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
offerTour(): boolean {
|
offerTour(): boolean {
|
||||||
return this.dashboardIsEmpty() && !this.get(SETTINGS_KEYS.TOUR_COMPLETE)
|
return this.dashboardIsEmpty() && !this.get(SETTINGS_KEYS.TOUR_COMPLETE)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -394,10 +394,16 @@ def delete(doc_ids: list[int]) -> Literal["OK"]:
|
|||||||
return "OK"
|
return "OK"
|
||||||
|
|
||||||
|
|
||||||
def reprocess(doc_ids: list[int]) -> Literal["OK"]:
|
def reprocess(doc_ids: list[int], *, remote_ocr: bool = False) -> Literal["OK"]:
|
||||||
|
"""
|
||||||
|
Re-run parsing for the given documents.
|
||||||
|
|
||||||
|
Consumption workflows do not run here, so ``remote_ocr`` is how the user
|
||||||
|
asks for the remote engine when it is not configured to handle everything.
|
||||||
|
"""
|
||||||
for document_id in doc_ids:
|
for document_id in doc_ids:
|
||||||
update_document_content_maybe_archive_file.apply_async(
|
update_document_content_maybe_archive_file.apply_async(
|
||||||
kwargs={"document_id": document_id},
|
kwargs={"document_id": document_id, "remote_ocr": remote_ocr},
|
||||||
headers={"trigger_source": PaperlessTask.TriggerSource.MANUAL},
|
headers={"trigger_source": PaperlessTask.TriggerSource.MANUAL},
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -53,6 +53,7 @@ from documents.utils import copy_basic_file_stats
|
|||||||
from documents.utils import copy_file_with_basic_stats
|
from documents.utils import copy_file_with_basic_stats
|
||||||
from documents.utils import run_subprocess
|
from documents.utils import run_subprocess
|
||||||
from paperless.config import OcrConfig
|
from paperless.config import OcrConfig
|
||||||
|
from paperless.config import RemoteOCRConfig
|
||||||
from paperless.models import ArchiveFileGenerationChoices
|
from paperless.models import ArchiveFileGenerationChoices
|
||||||
from paperless.parsers import ParserContext
|
from paperless.parsers import ParserContext
|
||||||
from paperless.parsers import ParserProtocol
|
from paperless.parsers import ParserProtocol
|
||||||
@@ -451,12 +452,19 @@ class ConsumerPlugin(
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
self.log.error(f"Error attempting to clean PDF: {e}")
|
self.log.error(f"Error attempting to clean PDF: {e}")
|
||||||
|
|
||||||
|
# Workflows have already run at this point, so the metadata knows
|
||||||
|
# whether this document was singled out for remote OCR
|
||||||
|
allow_remote = (
|
||||||
|
self.metadata.remote_ocr or RemoteOCRConfig().remote_ocr_by_default
|
||||||
|
)
|
||||||
|
|
||||||
# Based on the mime type, get the parser for that type
|
# Based on the mime type, get the parser for that type
|
||||||
parser_class: type[ParserProtocol] | None = (
|
parser_class: type[ParserProtocol] | None = (
|
||||||
get_parser_registry().get_parser_for_file(
|
get_parser_registry().get_parser_for_file(
|
||||||
mime_type,
|
mime_type,
|
||||||
self.filename,
|
self.filename,
|
||||||
self.working_copy,
|
self.working_copy,
|
||||||
|
allow_remote=allow_remote,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
if not parser_class:
|
if not parser_class:
|
||||||
@@ -465,6 +473,16 @@ class ConsumerPlugin(
|
|||||||
f"Unsupported mime type {mime_type}",
|
f"Unsupported mime type {mime_type}",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if self.metadata.remote_ocr and not getattr(
|
||||||
|
parser_class,
|
||||||
|
"uses_remote_service",
|
||||||
|
False,
|
||||||
|
):
|
||||||
|
self.log.warning(
|
||||||
|
"Remote OCR was requested for this document but no remote "
|
||||||
|
"parser is available for it, processing locally instead.",
|
||||||
|
)
|
||||||
|
|
||||||
# Notify all listeners that we're going to do some work.
|
# Notify all listeners that we're going to do some work.
|
||||||
|
|
||||||
document_consumption_started.send(
|
document_consumption_started.send(
|
||||||
|
|||||||
@@ -34,6 +34,7 @@ class DocumentMetadataOverrides:
|
|||||||
skip_asn_if_exists: bool = False
|
skip_asn_if_exists: bool = False
|
||||||
version_label: str | None = None
|
version_label: str | None = None
|
||||||
actor_id: int | None = None
|
actor_id: int | None = None
|
||||||
|
remote_ocr: bool = False
|
||||||
|
|
||||||
def update(self, other: "DocumentMetadataOverrides") -> "DocumentMetadataOverrides":
|
def update(self, other: "DocumentMetadataOverrides") -> "DocumentMetadataOverrides":
|
||||||
"""
|
"""
|
||||||
@@ -57,6 +58,8 @@ class DocumentMetadataOverrides:
|
|||||||
self.actor_id = other.actor_id
|
self.actor_id = other.actor_id
|
||||||
if other.skip_asn_if_exists:
|
if other.skip_asn_if_exists:
|
||||||
self.skip_asn_if_exists = True
|
self.skip_asn_if_exists = True
|
||||||
|
if other.remote_ocr:
|
||||||
|
self.remote_ocr = True
|
||||||
if other.version_label is not None:
|
if other.version_label is not None:
|
||||||
self.version_label = other.version_label
|
self.version_label = other.version_label
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,30 @@
|
|||||||
|
# Generated by Django 5.2.16 on 2026-08-10 17:27
|
||||||
|
|
||||||
|
from django.db import migrations
|
||||||
|
from django.db import models
|
||||||
|
|
||||||
|
|
||||||
|
class Migration(migrations.Migration):
|
||||||
|
dependencies = [
|
||||||
|
("documents", "0022_add_perf_indexes"),
|
||||||
|
]
|
||||||
|
|
||||||
|
operations = [
|
||||||
|
migrations.AlterField(
|
||||||
|
model_name="workflowaction",
|
||||||
|
name="type",
|
||||||
|
field=models.PositiveSmallIntegerField(
|
||||||
|
choices=[
|
||||||
|
(1, "Assignment"),
|
||||||
|
(2, "Removal"),
|
||||||
|
(3, "Email"),
|
||||||
|
(4, "Webhook"),
|
||||||
|
(5, "Password removal"),
|
||||||
|
(6, "Move to trash"),
|
||||||
|
(7, "Remote OCR"),
|
||||||
|
],
|
||||||
|
default=1,
|
||||||
|
verbose_name="Workflow Action Type",
|
||||||
|
),
|
||||||
|
),
|
||||||
|
]
|
||||||
@@ -0,0 +1,84 @@
|
|||||||
|
# Generated by Django 5.2.16 on 2026-08-10 18:26
|
||||||
|
|
||||||
|
from django.db import migrations
|
||||||
|
from django.db import models
|
||||||
|
|
||||||
|
|
||||||
|
class Migration(migrations.Migration):
|
||||||
|
dependencies = [
|
||||||
|
("documents", "0023_alter_workflowaction_type"),
|
||||||
|
]
|
||||||
|
|
||||||
|
operations = [
|
||||||
|
migrations.AddField(
|
||||||
|
model_name="workflowaction",
|
||||||
|
name="ai_create_missing",
|
||||||
|
field=models.BooleanField(
|
||||||
|
default=False,
|
||||||
|
help_text="Create suggested tags, correspondents, document types and storage paths that do not already exist instead of skipping them.",
|
||||||
|
verbose_name="create missing objects",
|
||||||
|
),
|
||||||
|
),
|
||||||
|
migrations.AddField(
|
||||||
|
model_name="workflowaction",
|
||||||
|
name="ai_overwrite_existing",
|
||||||
|
field=models.BooleanField(
|
||||||
|
default=False,
|
||||||
|
help_text="Apply suggestions even if the document already has a value for that field. Tags are always added to, never replaced.",
|
||||||
|
verbose_name="overwrite existing values",
|
||||||
|
),
|
||||||
|
),
|
||||||
|
migrations.AddField(
|
||||||
|
model_name="workflowaction",
|
||||||
|
name="ai_suggestion_fields",
|
||||||
|
field=models.JSONField(
|
||||||
|
blank=True,
|
||||||
|
help_text="Which of the AI-suggested fields to apply to the document.",
|
||||||
|
null=True,
|
||||||
|
verbose_name="AI suggestion fields",
|
||||||
|
),
|
||||||
|
),
|
||||||
|
migrations.AlterField(
|
||||||
|
model_name="workflowaction",
|
||||||
|
name="type",
|
||||||
|
field=models.PositiveSmallIntegerField(
|
||||||
|
choices=[
|
||||||
|
(1, "Assignment"),
|
||||||
|
(2, "Removal"),
|
||||||
|
(3, "Email"),
|
||||||
|
(4, "Webhook"),
|
||||||
|
(5, "Password removal"),
|
||||||
|
(6, "Move to trash"),
|
||||||
|
(7, "Remote OCR"),
|
||||||
|
(8, "Apply AI suggestions"),
|
||||||
|
],
|
||||||
|
default=1,
|
||||||
|
verbose_name="Workflow Action Type",
|
||||||
|
),
|
||||||
|
),
|
||||||
|
migrations.AlterField(
|
||||||
|
model_name="paperlesstask",
|
||||||
|
name="task_type",
|
||||||
|
field=models.CharField(
|
||||||
|
choices=[
|
||||||
|
("consume_file", "Consume File"),
|
||||||
|
("train_classifier", "Train Classifier"),
|
||||||
|
("sanity_check", "Sanity Check"),
|
||||||
|
("index_optimize", "Index Optimize"),
|
||||||
|
("mail_fetch", "Mail Fetch"),
|
||||||
|
("llm_index", "LLM Index"),
|
||||||
|
("empty_trash", "Empty Trash"),
|
||||||
|
("check_workflows", "Check Workflows"),
|
||||||
|
("bulk_update", "Bulk Update"),
|
||||||
|
("reprocess_document", "Reprocess Document"),
|
||||||
|
("build_share_link", "Build Share Link"),
|
||||||
|
("bulk_delete", "Bulk Delete"),
|
||||||
|
("apply_ai_suggestions", "Apply AI Suggestions"),
|
||||||
|
],
|
||||||
|
db_index=True,
|
||||||
|
help_text="The kind of work being performed",
|
||||||
|
max_length=50,
|
||||||
|
verbose_name="Task Type",
|
||||||
|
),
|
||||||
|
),
|
||||||
|
]
|
||||||
@@ -695,6 +695,7 @@ class PaperlessTask(ModelWithOwner):
|
|||||||
REPROCESS_DOCUMENT = "reprocess_document", _("Reprocess Document")
|
REPROCESS_DOCUMENT = "reprocess_document", _("Reprocess Document")
|
||||||
BUILD_SHARE_LINK = "build_share_link", _("Build Share Link")
|
BUILD_SHARE_LINK = "build_share_link", _("Build Share Link")
|
||||||
BULK_DELETE = "bulk_delete", _("Bulk Delete")
|
BULK_DELETE = "bulk_delete", _("Bulk Delete")
|
||||||
|
APPLY_AI_SUGGESTIONS = "apply_ai_suggestions", _("Apply AI Suggestions")
|
||||||
|
|
||||||
COMPLETE_STATUSES = (
|
COMPLETE_STATUSES = (
|
||||||
Status.SUCCESS,
|
Status.SUCCESS,
|
||||||
@@ -1599,6 +1600,22 @@ class WorkflowAction(models.Model):
|
|||||||
6,
|
6,
|
||||||
_("Move to trash"),
|
_("Move to trash"),
|
||||||
)
|
)
|
||||||
|
REMOTE_OCR = (
|
||||||
|
7,
|
||||||
|
_("Remote OCR"),
|
||||||
|
)
|
||||||
|
APPLY_AI_SUGGESTIONS = (
|
||||||
|
8,
|
||||||
|
_("Apply AI suggestions"),
|
||||||
|
)
|
||||||
|
|
||||||
|
class AISuggestionField(models.TextChoices):
|
||||||
|
TITLE = ("title", _("Title"))
|
||||||
|
TAGS = ("tags", _("Tags"))
|
||||||
|
CORRESPONDENT = ("correspondent", _("Correspondent"))
|
||||||
|
DOCUMENT_TYPE = ("document_type", _("Document type"))
|
||||||
|
STORAGE_PATH = ("storage_path", _("Storage path"))
|
||||||
|
CREATED = ("created", _("Created date"))
|
||||||
|
|
||||||
type = models.PositiveSmallIntegerField(
|
type = models.PositiveSmallIntegerField(
|
||||||
_("Workflow Action Type"),
|
_("Workflow Action Type"),
|
||||||
@@ -1837,6 +1854,33 @@ class WorkflowAction(models.Model):
|
|||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
ai_suggestion_fields = models.JSONField(
|
||||||
|
_("AI suggestion fields"),
|
||||||
|
null=True,
|
||||||
|
blank=True,
|
||||||
|
help_text=_(
|
||||||
|
"Which of the AI-suggested fields to apply to the document.",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
ai_create_missing = models.BooleanField(
|
||||||
|
_("create missing objects"),
|
||||||
|
default=False,
|
||||||
|
help_text=_(
|
||||||
|
"Create suggested tags, correspondents, document types and storage "
|
||||||
|
"paths that do not already exist instead of skipping them.",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
ai_overwrite_existing = models.BooleanField(
|
||||||
|
_("overwrite existing values"),
|
||||||
|
default=False,
|
||||||
|
help_text=_(
|
||||||
|
"Apply suggestions even if the document already has a value for that "
|
||||||
|
"field. Tags are always added to, never replaced.",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
class Meta:
|
class Meta:
|
||||||
verbose_name = _("workflow action")
|
verbose_name = _("workflow action")
|
||||||
verbose_name_plural = _("workflow actions")
|
verbose_name_plural = _("workflow actions")
|
||||||
|
|||||||
@@ -1744,7 +1744,7 @@ class DeleteDocumentsSerializer(DocumentSelectionSerializer):
|
|||||||
|
|
||||||
|
|
||||||
class ReprocessDocumentsSerializer(DocumentSelectionSerializer):
|
class ReprocessDocumentsSerializer(DocumentSelectionSerializer):
|
||||||
pass
|
remote_ocr = serializers.BooleanField(required=False, default=False)
|
||||||
|
|
||||||
|
|
||||||
class BulkEditSerializer(
|
class BulkEditSerializer(
|
||||||
@@ -2086,6 +2086,13 @@ class BulkEditSerializer(
|
|||||||
f"Page {op['page']} is out of bounds for document with {doc.page_count} pages.",
|
f"Page {op['page']} is out of bounds for document with {doc.page_count} pages.",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def _validate_parameters_reprocess(self, parameters) -> None:
|
||||||
|
if "remote_ocr" in parameters:
|
||||||
|
if not isinstance(parameters["remote_ocr"], bool):
|
||||||
|
raise serializers.ValidationError("remote_ocr must be a boolean")
|
||||||
|
else:
|
||||||
|
parameters["remote_ocr"] = False
|
||||||
|
|
||||||
def validate_parameters_remove_password(self, parameters):
|
def validate_parameters_remove_password(self, parameters):
|
||||||
if "password" not in parameters:
|
if "password" not in parameters:
|
||||||
raise serializers.ValidationError("password not specified")
|
raise serializers.ValidationError("password not specified")
|
||||||
@@ -2150,6 +2157,8 @@ class BulkEditSerializer(
|
|||||||
self._validate_parameters_edit_pdf(parameters, attrs["documents"][0])
|
self._validate_parameters_edit_pdf(parameters, attrs["documents"][0])
|
||||||
elif method == bulk_edit.remove_password:
|
elif method == bulk_edit.remove_password:
|
||||||
self.validate_parameters_remove_password(parameters)
|
self.validate_parameters_remove_password(parameters)
|
||||||
|
elif method == bulk_edit.reprocess:
|
||||||
|
self._validate_parameters_reprocess(parameters)
|
||||||
|
|
||||||
return attrs
|
return attrs
|
||||||
|
|
||||||
@@ -3175,6 +3184,9 @@ class WorkflowActionSerializer(serializers.ModelSerializer[WorkflowAction]):
|
|||||||
"email",
|
"email",
|
||||||
"webhook",
|
"webhook",
|
||||||
"passwords",
|
"passwords",
|
||||||
|
"ai_suggestion_fields",
|
||||||
|
"ai_create_missing",
|
||||||
|
"ai_overwrite_existing",
|
||||||
]
|
]
|
||||||
|
|
||||||
def validate(self, attrs):
|
def validate(self, attrs):
|
||||||
@@ -3213,13 +3225,6 @@ class WorkflowActionSerializer(serializers.ModelSerializer[WorkflowAction]):
|
|||||||
{"assign_title": f'Invalid f-string detected: "{e.args[0]}"'},
|
{"assign_title": f'Invalid f-string detected: "{e.args[0]}"'},
|
||||||
)
|
)
|
||||||
|
|
||||||
if attrs.get("assign_custom_fields_values"):
|
|
||||||
# Empty strings treated as None to avoid unexpected behavior
|
|
||||||
attrs["assign_custom_fields_values"] = {
|
|
||||||
field_id: (None if value == "" else value)
|
|
||||||
for field_id, value in attrs["assign_custom_fields_values"].items()
|
|
||||||
}
|
|
||||||
|
|
||||||
if (
|
if (
|
||||||
"type" in attrs
|
"type" in attrs
|
||||||
and attrs["type"] == WorkflowAction.WorkflowActionType.EMAIL
|
and attrs["type"] == WorkflowAction.WorkflowActionType.EMAIL
|
||||||
@@ -3255,6 +3260,23 @@ class WorkflowActionSerializer(serializers.ModelSerializer[WorkflowAction]):
|
|||||||
"Passwords are required for password removal actions",
|
"Passwords are required for password removal actions",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if (
|
||||||
|
"type" in attrs
|
||||||
|
and attrs["type"] == WorkflowAction.WorkflowActionType.APPLY_AI_SUGGESTIONS
|
||||||
|
):
|
||||||
|
fields = attrs.get("ai_suggestion_fields")
|
||||||
|
valid_fields = set(WorkflowAction.AISuggestionField.values)
|
||||||
|
if (
|
||||||
|
fields is None
|
||||||
|
or not isinstance(fields, list)
|
||||||
|
or len(fields) == 0
|
||||||
|
or any(field not in valid_fields for field in fields)
|
||||||
|
):
|
||||||
|
raise serializers.ValidationError(
|
||||||
|
"At least one valid field is required for apply AI "
|
||||||
|
f"suggestions actions, options are: {sorted(valid_fields)}",
|
||||||
|
)
|
||||||
|
|
||||||
return attrs
|
return attrs
|
||||||
|
|
||||||
|
|
||||||
@@ -3275,6 +3297,40 @@ class WorkflowSerializer(serializers.ModelSerializer[Workflow]):
|
|||||||
"actions",
|
"actions",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
def validate(self, attrs):
|
||||||
|
attrs = super().validate(attrs)
|
||||||
|
|
||||||
|
triggers = attrs.get("triggers") or []
|
||||||
|
actions = attrs.get("actions") or []
|
||||||
|
|
||||||
|
# Remote OCR can only work with consumption triggers
|
||||||
|
if any(
|
||||||
|
action.get("type") == WorkflowAction.WorkflowActionType.REMOTE_OCR
|
||||||
|
for action in actions
|
||||||
|
) and not any(
|
||||||
|
trigger.get("type") == WorkflowTrigger.WorkflowTriggerType.CONSUMPTION
|
||||||
|
for trigger in triggers
|
||||||
|
):
|
||||||
|
raise serializers.ValidationError(
|
||||||
|
"Remote OCR actions require a consumption started trigger",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Suggestions are made from the document content, which does not exist
|
||||||
|
# until after consumption has finished
|
||||||
|
if any(
|
||||||
|
action.get("type") == WorkflowAction.WorkflowActionType.APPLY_AI_SUGGESTIONS
|
||||||
|
for action in actions
|
||||||
|
) and not any(
|
||||||
|
trigger.get("type") != WorkflowTrigger.WorkflowTriggerType.CONSUMPTION
|
||||||
|
for trigger in triggers
|
||||||
|
):
|
||||||
|
raise serializers.ValidationError(
|
||||||
|
"Apply AI suggestions actions require a trigger other than "
|
||||||
|
"consumption started",
|
||||||
|
)
|
||||||
|
|
||||||
|
return attrs
|
||||||
|
|
||||||
def update_triggers_and_actions(
|
def update_triggers_and_actions(
|
||||||
self,
|
self,
|
||||||
instance: Workflow,
|
instance: Workflow,
|
||||||
|
|||||||
@@ -971,6 +971,39 @@ def run_workflows(
|
|||||||
)
|
)
|
||||||
elif action.type == WorkflowAction.WorkflowActionType.MOVE_TO_TRASH:
|
elif action.type == WorkflowAction.WorkflowActionType.MOVE_TO_TRASH:
|
||||||
has_move_to_trash_action = True
|
has_move_to_trash_action = True
|
||||||
|
elif action.type == WorkflowAction.WorkflowActionType.REMOTE_OCR:
|
||||||
|
if use_overrides and overrides:
|
||||||
|
overrides.remote_ocr = True
|
||||||
|
else:
|
||||||
|
# If a workflow has a consumption trigger *and* another type,
|
||||||
|
# the document has already been parsed by the time the other one fires
|
||||||
|
logger.debug(
|
||||||
|
"Remote OCR action only applies to consumption "
|
||||||
|
"triggers, ignoring",
|
||||||
|
extra={"group": logging_group},
|
||||||
|
)
|
||||||
|
elif (
|
||||||
|
action.type
|
||||||
|
== WorkflowAction.WorkflowActionType.APPLY_AI_SUGGESTIONS
|
||||||
|
):
|
||||||
|
if use_overrides:
|
||||||
|
# The document has not been parsed yet, so there is no
|
||||||
|
# content for the LLM to make suggestions from
|
||||||
|
logger.debug(
|
||||||
|
"Apply AI suggestions action does not apply to "
|
||||||
|
"consumption triggers, ignoring",
|
||||||
|
extra={"group": logging_group},
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
# Queued rather than run sync
|
||||||
|
from documents.tasks import apply_ai_suggestions
|
||||||
|
|
||||||
|
# kwargs so the PaperlessTask record can note the
|
||||||
|
# document, see _extract_input_data
|
||||||
|
apply_ai_suggestions.delay(
|
||||||
|
action_id=action.pk,
|
||||||
|
document_id=document.pk,
|
||||||
|
)
|
||||||
|
|
||||||
if not use_overrides:
|
if not use_overrides:
|
||||||
# limit title to 128 characters
|
# limit title to 128 characters
|
||||||
@@ -1026,6 +1059,7 @@ TRACKED_TASKS: dict[str, PaperlessTask.TaskType] = {
|
|||||||
"documents.tasks.update_document_content_maybe_archive_file": PaperlessTask.TaskType.REPROCESS_DOCUMENT,
|
"documents.tasks.update_document_content_maybe_archive_file": PaperlessTask.TaskType.REPROCESS_DOCUMENT,
|
||||||
"documents.tasks.build_share_link_bundle": PaperlessTask.TaskType.BUILD_SHARE_LINK,
|
"documents.tasks.build_share_link_bundle": PaperlessTask.TaskType.BUILD_SHARE_LINK,
|
||||||
"documents.bulk_edit.delete": PaperlessTask.TaskType.BULK_DELETE,
|
"documents.bulk_edit.delete": PaperlessTask.TaskType.BULK_DELETE,
|
||||||
|
"documents.tasks.apply_ai_suggestions": PaperlessTask.TaskType.APPLY_AI_SUGGESTIONS,
|
||||||
}
|
}
|
||||||
|
|
||||||
_CELERY_STATE_TO_STATUS: dict[str, PaperlessTask.Status] = {
|
_CELERY_STATE_TO_STATUS: dict[str, PaperlessTask.Status] = {
|
||||||
@@ -1079,6 +1113,12 @@ def _extract_input_data(
|
|||||||
return {"account_ids": account_ids}
|
return {"account_ids": account_ids}
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
|
if task_type == PaperlessTask.TaskType.APPLY_AI_SUGGESTIONS:
|
||||||
|
document_id = task_kwargs.get("document_id")
|
||||||
|
if document_id is not None:
|
||||||
|
return {"document_id": document_id}
|
||||||
|
return {}
|
||||||
|
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+49
-1
@@ -66,6 +66,7 @@ from documents.utils import compute_checksum
|
|||||||
from documents.utils import identity
|
from documents.utils import identity
|
||||||
from documents.workflows.utils import get_workflows_for_trigger
|
from documents.workflows.utils import get_workflows_for_trigger
|
||||||
from paperless.config import AIConfig
|
from paperless.config import AIConfig
|
||||||
|
from paperless.config import RemoteOCRConfig
|
||||||
from paperless.logging import consume_task_id
|
from paperless.logging import consume_task_id
|
||||||
from paperless.parsers import ParserContext
|
from paperless.parsers import ParserContext
|
||||||
from paperless.parsers.registry import get_parser_registry
|
from paperless.parsers.registry import get_parser_registry
|
||||||
@@ -337,10 +338,17 @@ def bulk_update_documents(document_ids) -> None:
|
|||||||
|
|
||||||
|
|
||||||
@shared_task
|
@shared_task
|
||||||
def update_document_content_maybe_archive_file(document_id) -> None:
|
def update_document_content_maybe_archive_file(
|
||||||
|
document_id,
|
||||||
|
*,
|
||||||
|
remote_ocr: bool = False,
|
||||||
|
) -> None:
|
||||||
"""
|
"""
|
||||||
Re-creates OCR content and thumbnail for a document, and archive file if
|
Re-creates OCR content and thumbnail for a document, and archive file if
|
||||||
it exists.
|
it exists.
|
||||||
|
|
||||||
|
Remote OCR is used only when the engine is configured to handle everything
|
||||||
|
or if explicitly asked for via ``remote_ocr``.
|
||||||
"""
|
"""
|
||||||
document = Document.objects.get(id=document_id)
|
document = Document.objects.get(id=document_id)
|
||||||
|
|
||||||
@@ -350,6 +358,7 @@ def update_document_content_maybe_archive_file(document_id) -> None:
|
|||||||
mime_type,
|
mime_type,
|
||||||
document.original_filename or "",
|
document.original_filename or "",
|
||||||
document.source_path,
|
document.source_path,
|
||||||
|
allow_remote=remote_ocr or RemoteOCRConfig().remote_ocr_by_default,
|
||||||
)
|
)
|
||||||
|
|
||||||
if not parser_class:
|
if not parser_class:
|
||||||
@@ -704,6 +713,45 @@ def llmindex_index(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@shared_task(
|
||||||
|
bind=True,
|
||||||
|
autoretry_for=(Exception,),
|
||||||
|
max_retries=3,
|
||||||
|
retry_backoff=60,
|
||||||
|
retry_backoff_max=600,
|
||||||
|
retry_jitter=True,
|
||||||
|
)
|
||||||
|
def apply_ai_suggestions(self, action_id: int, document_id: int) -> None:
|
||||||
|
"""
|
||||||
|
Deferred "apply AI suggestions" workflow action.
|
||||||
|
"""
|
||||||
|
from documents.models import WorkflowAction
|
||||||
|
from documents.workflows.ai import apply_ai_suggestions_to_document
|
||||||
|
|
||||||
|
try:
|
||||||
|
action = WorkflowAction.objects.get(pk=action_id)
|
||||||
|
document = Document.objects.select_related("owner").get(pk=document_id)
|
||||||
|
except (WorkflowAction.DoesNotExist, Document.DoesNotExist):
|
||||||
|
logger.warning(
|
||||||
|
"Workflow action %s or document %s no longer exists, "
|
||||||
|
"not applying AI suggestions",
|
||||||
|
action_id,
|
||||||
|
document_id,
|
||||||
|
)
|
||||||
|
return
|
||||||
|
|
||||||
|
if not apply_ai_suggestions_to_document(action, document):
|
||||||
|
return
|
||||||
|
|
||||||
|
# No document_updated signal to avoid loop
|
||||||
|
clear_document_caches(document.pk)
|
||||||
|
index_document.delay(document.pk)
|
||||||
|
|
||||||
|
ai_config = AIConfig()
|
||||||
|
if ai_config.llm_index_enabled:
|
||||||
|
update_document_in_llm_index.apply_async(kwargs={"document": document})
|
||||||
|
|
||||||
|
|
||||||
@shared_task
|
@shared_task
|
||||||
def update_document_in_llm_index(document) -> None:
|
def update_document_in_llm_index(document) -> None:
|
||||||
llm_index_add_or_update_document(document)
|
llm_index_add_or_update_document(document)
|
||||||
|
|||||||
@@ -72,6 +72,10 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
|
|||||||
"barcode_enable_tag": None,
|
"barcode_enable_tag": None,
|
||||||
"barcode_tag_mapping": None,
|
"barcode_tag_mapping": None,
|
||||||
"barcode_tag_split": None,
|
"barcode_tag_split": None,
|
||||||
|
"remote_ocr_engine": None,
|
||||||
|
"remote_ocr_api_key": None,
|
||||||
|
"remote_ocr_endpoint": None,
|
||||||
|
"remote_ocr_mode": None,
|
||||||
"ai_enabled": False,
|
"ai_enabled": False,
|
||||||
"llm_embedding_backend": None,
|
"llm_embedding_backend": None,
|
||||||
"llm_embedding_model": None,
|
"llm_embedding_model": None,
|
||||||
@@ -870,6 +874,49 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
|
|||||||
config.refresh_from_db()
|
config.refresh_from_db()
|
||||||
self.assertEqual(config.llm_api_key, None)
|
self.assertEqual(config.llm_api_key, None)
|
||||||
|
|
||||||
|
def test_update_remote_ocr_api_key(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- Existing config with remote_ocr_api_key specified
|
||||||
|
WHEN:
|
||||||
|
- API to update remote_ocr_api_key is called with all *s
|
||||||
|
- API to update remote_ocr_api_key is called with empty string
|
||||||
|
THEN:
|
||||||
|
- remote_ocr_api_key is unchanged
|
||||||
|
- remote_ocr_api_key is set to None
|
||||||
|
"""
|
||||||
|
config = ApplicationConfiguration.objects.first()
|
||||||
|
assert config is not None
|
||||||
|
config.remote_ocr_api_key = "1234567890"
|
||||||
|
config.save()
|
||||||
|
|
||||||
|
# Test with all *
|
||||||
|
response = self.client.patch(
|
||||||
|
f"{self.ENDPOINT}1/",
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"remote_ocr_api_key": "*" * 32,
|
||||||
|
},
|
||||||
|
),
|
||||||
|
content_type="application/json",
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||||
|
config.refresh_from_db()
|
||||||
|
self.assertEqual(config.remote_ocr_api_key, "1234567890")
|
||||||
|
# Test with empty string
|
||||||
|
response = self.client.patch(
|
||||||
|
f"{self.ENDPOINT}1/",
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"remote_ocr_api_key": "",
|
||||||
|
},
|
||||||
|
),
|
||||||
|
content_type="application/json",
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||||
|
config.refresh_from_db()
|
||||||
|
self.assertEqual(config.remote_ocr_api_key, None)
|
||||||
|
|
||||||
def test_enable_ai_index_triggers_update(self) -> None:
|
def test_enable_ai_index_triggers_update(self) -> None:
|
||||||
"""
|
"""
|
||||||
GIVEN:
|
GIVEN:
|
||||||
|
|||||||
@@ -532,7 +532,29 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
|
|||||||
m.assert_called_once()
|
m.assert_called_once()
|
||||||
args, kwargs = m.call_args
|
args, kwargs = m.call_args
|
||||||
self.assertEqual(args[0], [self.doc1.id])
|
self.assertEqual(args[0], [self.doc1.id])
|
||||||
self.assertEqual(len(kwargs), 0)
|
self.assertEqual(kwargs, {"remote_ocr": False})
|
||||||
|
|
||||||
|
@mock.patch("documents.views.bulk_edit.reprocess")
|
||||||
|
def test_reprocess_documents_endpoint_remote_ocr(self, m) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- API data to reprocess a document with remote OCR requested
|
||||||
|
WHEN:
|
||||||
|
- API is called
|
||||||
|
THEN:
|
||||||
|
- reprocess is called with remote_ocr=True
|
||||||
|
"""
|
||||||
|
self.setup_mock(m, "reprocess")
|
||||||
|
response = self.client.post(
|
||||||
|
"/api/documents/reprocess/",
|
||||||
|
json.dumps({"documents": [self.doc1.id], "remote_ocr": True}),
|
||||||
|
content_type="application/json",
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||||
|
m.assert_called_once()
|
||||||
|
args, kwargs = m.call_args
|
||||||
|
self.assertEqual(args[0], [self.doc1.id])
|
||||||
|
self.assertEqual(kwargs, {"remote_ocr": True})
|
||||||
|
|
||||||
@mock.patch("documents.serialisers.bulk_edit.set_storage_path")
|
@mock.patch("documents.serialisers.bulk_edit.set_storage_path")
|
||||||
def test_api_set_storage_path(self, m) -> None:
|
def test_api_set_storage_path(self, m) -> None:
|
||||||
@@ -1553,6 +1575,29 @@ class TestBulkEditAPI(DirectoriesMixin, APITestCase):
|
|||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def test_legacy_bulk_edit_reprocess_invalid_remote_ocr(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- The deprecated bulk_edit endpoint with a non-boolean remote_ocr
|
||||||
|
WHEN:
|
||||||
|
- API is called
|
||||||
|
THEN:
|
||||||
|
- The request is rejected rather than passed through to the task
|
||||||
|
"""
|
||||||
|
response = self.client.post(
|
||||||
|
"/api/documents/bulk_edit/",
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"documents": [self.doc1.id],
|
||||||
|
"method": "reprocess",
|
||||||
|
"parameters": {"remote_ocr": "yes please"},
|
||||||
|
},
|
||||||
|
),
|
||||||
|
content_type="application/json",
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||||
|
|
||||||
@mock.patch("documents.views.bulk_edit.edit_pdf")
|
@mock.patch("documents.views.bulk_edit.edit_pdf")
|
||||||
def test_edit_pdf(self, m) -> None:
|
def test_edit_pdf(self, m) -> None:
|
||||||
self.setup_mock(m, "edit_pdf")
|
self.setup_mock(m, "edit_pdf")
|
||||||
|
|||||||
@@ -60,6 +60,10 @@ class TestApiUiSettings(DirectoriesMixin, APITestCase):
|
|||||||
},
|
},
|
||||||
"email_enabled": False,
|
"email_enabled": False,
|
||||||
"ai_enabled": False,
|
"ai_enabled": False,
|
||||||
|
"remote_ocr": {
|
||||||
|
"configured": False,
|
||||||
|
"mode": "always",
|
||||||
|
},
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -154,6 +158,50 @@ class TestApiUiSettings(DirectoriesMixin, APITestCase):
|
|||||||
str(response.data["settings"]),
|
str(response.data["settings"]),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
@override_settings(
|
||||||
|
REMOTE_OCR_ENGINE="azureai",
|
||||||
|
REMOTE_OCR_API_KEY="somekey",
|
||||||
|
REMOTE_OCR_ENDPOINT="https://example.cognitiveservices.azure.com",
|
||||||
|
REMOTE_OCR_MODE="workflow_only",
|
||||||
|
)
|
||||||
|
def test_settings_reports_remote_ocr_when_configured(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A fully configured remote OCR engine in workflow_only mode
|
||||||
|
WHEN:
|
||||||
|
- The ui_settings endpoint is called
|
||||||
|
THEN:
|
||||||
|
- The UI is told remote OCR is available and selective, so it can
|
||||||
|
offer it where it would actually change something
|
||||||
|
"""
|
||||||
|
response = self.client.get(self.ENDPOINT, format="json")
|
||||||
|
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||||
|
self.assertEqual(
|
||||||
|
response.data["settings"]["remote_ocr"],
|
||||||
|
{"configured": True, "mode": "workflow_only"},
|
||||||
|
)
|
||||||
|
|
||||||
|
@override_settings(
|
||||||
|
REMOTE_OCR_ENGINE="azureai",
|
||||||
|
REMOTE_OCR_API_KEY=None,
|
||||||
|
REMOTE_OCR_ENDPOINT=None,
|
||||||
|
)
|
||||||
|
def test_settings_reports_remote_ocr_incompletely_configured(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An engine named but missing its endpoint and API key
|
||||||
|
WHEN:
|
||||||
|
- The ui_settings endpoint is called
|
||||||
|
THEN:
|
||||||
|
- It is reported as not configured, matching what the parser
|
||||||
|
registry will actually do
|
||||||
|
"""
|
||||||
|
response = self.client.get(self.ENDPOINT, format="json")
|
||||||
|
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||||
|
self.assertFalse(response.data["settings"]["remote_ocr"]["configured"])
|
||||||
|
|
||||||
@override_settings(
|
@override_settings(
|
||||||
OAUTH_CALLBACK_BASE_URL="http://localhost:8000",
|
OAUTH_CALLBACK_BASE_URL="http://localhost:8000",
|
||||||
GMAIL_OAUTH_CLIENT_ID="abc123",
|
GMAIL_OAUTH_CLIENT_ID="abc123",
|
||||||
|
|||||||
@@ -390,6 +390,221 @@ class TestApiWorkflows(DirectoriesMixin, APITestCase):
|
|||||||
|
|
||||||
self.assertEqual(Workflow.objects.count(), 1)
|
self.assertEqual(Workflow.objects.count(), 1)
|
||||||
|
|
||||||
|
def test_api_create_remote_ocr_action_requires_consumption_trigger(
|
||||||
|
self,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- API request to create a workflow with a remote OCR action
|
||||||
|
- No consumption started trigger, so the action could never run
|
||||||
|
WHEN:
|
||||||
|
- API is called
|
||||||
|
THEN:
|
||||||
|
- Correct HTTP 400 response
|
||||||
|
- No objects are created
|
||||||
|
"""
|
||||||
|
existing_count = Workflow.objects.count()
|
||||||
|
|
||||||
|
response = self.client.post(
|
||||||
|
self.ENDPOINT,
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"name": "Remote OCR too late",
|
||||||
|
"order": 1,
|
||||||
|
"triggers": [
|
||||||
|
{
|
||||||
|
"type": WorkflowTrigger.WorkflowTriggerType.DOCUMENT_ADDED,
|
||||||
|
},
|
||||||
|
],
|
||||||
|
"actions": [
|
||||||
|
{
|
||||||
|
"type": WorkflowAction.WorkflowActionType.REMOTE_OCR,
|
||||||
|
},
|
||||||
|
],
|
||||||
|
},
|
||||||
|
),
|
||||||
|
content_type="application/json",
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||||
|
self.assertEqual(Workflow.objects.count(), existing_count)
|
||||||
|
|
||||||
|
def test_api_create_remote_ocr_action_with_consumption_trigger(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- API request to create a workflow with a remote OCR action
|
||||||
|
- A consumption started trigger alongside another trigger type
|
||||||
|
WHEN:
|
||||||
|
- API is called
|
||||||
|
THEN:
|
||||||
|
- The workflow is created, the action applies to consumption only
|
||||||
|
"""
|
||||||
|
response = self.client.post(
|
||||||
|
self.ENDPOINT,
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"name": "Remote OCR on consume",
|
||||||
|
"order": 1,
|
||||||
|
"triggers": [
|
||||||
|
{
|
||||||
|
"type": WorkflowTrigger.WorkflowTriggerType.CONSUMPTION,
|
||||||
|
"filter_filename": "*.pdf",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"type": WorkflowTrigger.WorkflowTriggerType.DOCUMENT_ADDED,
|
||||||
|
},
|
||||||
|
],
|
||||||
|
"actions": [
|
||||||
|
{
|
||||||
|
"type": WorkflowAction.WorkflowActionType.REMOTE_OCR,
|
||||||
|
},
|
||||||
|
],
|
||||||
|
},
|
||||||
|
),
|
||||||
|
content_type="application/json",
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_201_CREATED)
|
||||||
|
|
||||||
|
def _post_ai_suggestions_workflow(self, *, trigger_types, action: dict):
|
||||||
|
def trigger(trigger_type):
|
||||||
|
# consumption triggers require a filter of their own
|
||||||
|
if trigger_type == WorkflowTrigger.WorkflowTriggerType.CONSUMPTION:
|
||||||
|
return {"type": trigger_type, "filter_filename": "*.pdf"}
|
||||||
|
return {"type": trigger_type}
|
||||||
|
|
||||||
|
return self.client.post(
|
||||||
|
self.ENDPOINT,
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"name": "Apply AI suggestions",
|
||||||
|
"order": 1,
|
||||||
|
"triggers": [trigger(t) for t in trigger_types],
|
||||||
|
"actions": [
|
||||||
|
{
|
||||||
|
"type": WorkflowAction.WorkflowActionType.APPLY_AI_SUGGESTIONS,
|
||||||
|
**action,
|
||||||
|
},
|
||||||
|
],
|
||||||
|
},
|
||||||
|
),
|
||||||
|
content_type="application/json",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_api_create_apply_ai_suggestions_action(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- API request to create a workflow with an apply AI suggestions
|
||||||
|
action and a valid set of fields
|
||||||
|
WHEN:
|
||||||
|
- API is called
|
||||||
|
THEN:
|
||||||
|
- The workflow is created with the chosen options
|
||||||
|
"""
|
||||||
|
response = self._post_ai_suggestions_workflow(
|
||||||
|
trigger_types=[WorkflowTrigger.WorkflowTriggerType.DOCUMENT_ADDED],
|
||||||
|
action={
|
||||||
|
"ai_suggestion_fields": ["title", "tags", "correspondent"],
|
||||||
|
"ai_create_missing": True,
|
||||||
|
"ai_overwrite_existing": True,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_201_CREATED)
|
||||||
|
action = Workflow.objects.get(name="Apply AI suggestions").actions.first()
|
||||||
|
self.assertEqual(
|
||||||
|
action.ai_suggestion_fields,
|
||||||
|
["title", "tags", "correspondent"],
|
||||||
|
)
|
||||||
|
self.assertTrue(action.ai_create_missing)
|
||||||
|
self.assertTrue(action.ai_overwrite_existing)
|
||||||
|
|
||||||
|
def test_api_create_apply_ai_suggestions_action_requires_fields(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- API request to create an apply AI suggestions action with no
|
||||||
|
fields selected, which could never do anything
|
||||||
|
WHEN:
|
||||||
|
- API is called
|
||||||
|
THEN:
|
||||||
|
- Correct HTTP 400 response
|
||||||
|
- No objects are created
|
||||||
|
"""
|
||||||
|
existing_count = Workflow.objects.count()
|
||||||
|
|
||||||
|
response = self._post_ai_suggestions_workflow(
|
||||||
|
trigger_types=[WorkflowTrigger.WorkflowTriggerType.DOCUMENT_ADDED],
|
||||||
|
action={"ai_suggestion_fields": []},
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||||
|
self.assertEqual(Workflow.objects.count(), existing_count)
|
||||||
|
|
||||||
|
def test_api_create_apply_ai_suggestions_action_rejects_unknown_field(
|
||||||
|
self,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- API request to create an apply AI suggestions action naming a
|
||||||
|
field that does not exist
|
||||||
|
WHEN:
|
||||||
|
- API is called
|
||||||
|
THEN:
|
||||||
|
- Correct HTTP 400 response
|
||||||
|
"""
|
||||||
|
response = self._post_ai_suggestions_workflow(
|
||||||
|
trigger_types=[WorkflowTrigger.WorkflowTriggerType.DOCUMENT_ADDED],
|
||||||
|
action={"ai_suggestion_fields": ["title", "not_a_field"]},
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||||
|
|
||||||
|
def test_api_create_apply_ai_suggestions_action_rejects_consumption_only(
|
||||||
|
self,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- API request to create an apply AI suggestions action whose only
|
||||||
|
trigger is consumption started, so there is no document content
|
||||||
|
to make suggestions from yet
|
||||||
|
WHEN:
|
||||||
|
- API is called
|
||||||
|
THEN:
|
||||||
|
- Correct HTTP 400 response
|
||||||
|
- No objects are created
|
||||||
|
"""
|
||||||
|
existing_count = Workflow.objects.count()
|
||||||
|
|
||||||
|
response = self._post_ai_suggestions_workflow(
|
||||||
|
trigger_types=[WorkflowTrigger.WorkflowTriggerType.CONSUMPTION],
|
||||||
|
action={"ai_suggestion_fields": ["title"]},
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||||
|
self.assertEqual(Workflow.objects.count(), existing_count)
|
||||||
|
|
||||||
|
def test_api_create_apply_ai_suggestions_action_allows_extra_consumption_trigger(
|
||||||
|
self,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- API request to create an apply AI suggestions action with a
|
||||||
|
consumption trigger alongside a usable one
|
||||||
|
WHEN:
|
||||||
|
- API is called
|
||||||
|
THEN:
|
||||||
|
- The workflow is created, the action applies to the other trigger
|
||||||
|
"""
|
||||||
|
response = self._post_ai_suggestions_workflow(
|
||||||
|
trigger_types=[
|
||||||
|
WorkflowTrigger.WorkflowTriggerType.CONSUMPTION,
|
||||||
|
WorkflowTrigger.WorkflowTriggerType.DOCUMENT_ADDED,
|
||||||
|
],
|
||||||
|
action={"ai_suggestion_fields": ["title"]},
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(response.status_code, status.HTTP_201_CREATED)
|
||||||
|
|
||||||
def test_api_create_workflow_trigger_action_empty_fields(self) -> None:
|
def test_api_create_workflow_trigger_action_empty_fields(self) -> None:
|
||||||
"""
|
"""
|
||||||
GIVEN:
|
GIVEN:
|
||||||
@@ -422,11 +637,6 @@ class TestApiWorkflows(DirectoriesMixin, APITestCase):
|
|||||||
json.dumps(
|
json.dumps(
|
||||||
{
|
{
|
||||||
"assign_title": "",
|
"assign_title": "",
|
||||||
"assign_custom_fields": [self.cf1.id, self.cf2.id],
|
|
||||||
"assign_custom_fields_values": {
|
|
||||||
str(self.cf1.id): "",
|
|
||||||
str(self.cf2.id): 0,
|
|
||||||
},
|
|
||||||
},
|
},
|
||||||
),
|
),
|
||||||
content_type="application/json",
|
content_type="application/json",
|
||||||
@@ -434,10 +644,6 @@ class TestApiWorkflows(DirectoriesMixin, APITestCase):
|
|||||||
self.assertEqual(response.status_code, status.HTTP_201_CREATED)
|
self.assertEqual(response.status_code, status.HTTP_201_CREATED)
|
||||||
action = WorkflowAction.objects.get(id=response.data["id"])
|
action = WorkflowAction.objects.get(id=response.data["id"])
|
||||||
self.assertIsNone(action.assign_title)
|
self.assertIsNone(action.assign_title)
|
||||||
self.assertEqual(
|
|
||||||
action.assign_custom_fields_values,
|
|
||||||
{str(self.cf1.id): None, str(self.cf2.id): 0},
|
|
||||||
)
|
|
||||||
|
|
||||||
response = self.client.post(
|
response = self.client.post(
|
||||||
self.ENDPOINT_TRIGGERS,
|
self.ENDPOINT_TRIGGERS,
|
||||||
|
|||||||
@@ -1782,3 +1782,56 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
|||||||
|
|
||||||
self.assertIn("wrong password", str(exc.exception))
|
self.assertIn("wrong password", str(exc.exception))
|
||||||
self.assertIn("Error removing password from document", cm.output[0])
|
self.assertIn("Error removing password from document", cm.output[0])
|
||||||
|
|
||||||
|
|
||||||
|
class TestBulkEditReprocess(DirectoriesMixin, TestCase):
|
||||||
|
def setUp(self) -> None:
|
||||||
|
super().setUp()
|
||||||
|
|
||||||
|
self.doc = Document.objects.create(
|
||||||
|
title="test",
|
||||||
|
checksum="A",
|
||||||
|
mime_type="application/pdf",
|
||||||
|
)
|
||||||
|
|
||||||
|
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file")
|
||||||
|
def test_reprocess_defaults_to_local(self, mock_task: mock.Mock) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A reprocess request that says nothing about remote OCR
|
||||||
|
WHEN:
|
||||||
|
- reprocess is called
|
||||||
|
THEN:
|
||||||
|
- The task is queued without asking for the remote engine
|
||||||
|
"""
|
||||||
|
result = bulk_edit.reprocess([self.doc.id])
|
||||||
|
|
||||||
|
self.assertEqual(result, "OK")
|
||||||
|
mock_task.apply_async.assert_called_once()
|
||||||
|
_, kwargs = mock_task.apply_async.call_args
|
||||||
|
self.assertEqual(
|
||||||
|
kwargs["kwargs"],
|
||||||
|
{"document_id": self.doc.id, "remote_ocr": False},
|
||||||
|
)
|
||||||
|
|
||||||
|
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file")
|
||||||
|
def test_reprocess_passes_remote_ocr(self, mock_task: mock.Mock) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A reprocess request that explicitly asks for remote OCR
|
||||||
|
WHEN:
|
||||||
|
- reprocess is called
|
||||||
|
THEN:
|
||||||
|
- The request is forwarded to the task for every document
|
||||||
|
"""
|
||||||
|
other = Document.objects.create(
|
||||||
|
title="test2",
|
||||||
|
checksum="B",
|
||||||
|
mime_type="application/pdf",
|
||||||
|
)
|
||||||
|
|
||||||
|
bulk_edit.reprocess([self.doc.id, other.id], remote_ocr=True)
|
||||||
|
|
||||||
|
self.assertEqual(mock_task.apply_async.call_count, 2)
|
||||||
|
for call in mock_task.apply_async.call_args_list:
|
||||||
|
self.assertTrue(call.kwargs["kwargs"]["remote_ocr"])
|
||||||
|
|||||||
@@ -1559,6 +1559,72 @@ class PostConsumeTestCase(DirectoriesMixin, GetConsumerMixin, TestCase):
|
|||||||
consumer.run_post_consume_script(doc)
|
consumer.run_post_consume_script(doc)
|
||||||
|
|
||||||
|
|
||||||
|
class TestConsumerRemoteOCR(
|
||||||
|
DirectoriesMixin,
|
||||||
|
FileSystemAssertsMixin,
|
||||||
|
GetConsumerMixin,
|
||||||
|
TestCase,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
The consumer resolves the remote OCR mode and the per-document request from
|
||||||
|
workflows into the allow_remote flag it hands to the parser registry.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def setUp(self) -> None:
|
||||||
|
super().setUp()
|
||||||
|
|
||||||
|
patcher = mock.patch("documents.consumer.get_parser_registry")
|
||||||
|
self.mock_registry = patcher.start()
|
||||||
|
self.mock_registry.return_value.get_parser_for_file.return_value = DummyParser
|
||||||
|
self.addCleanup(patcher.stop)
|
||||||
|
|
||||||
|
def _consume(self, *, overrides: DocumentMetadataOverrides | None = None) -> bool:
|
||||||
|
src = (
|
||||||
|
Path(__file__).parent
|
||||||
|
/ "samples"
|
||||||
|
/ "documents"
|
||||||
|
/ "originals"
|
||||||
|
/ "0000001.pdf"
|
||||||
|
)
|
||||||
|
dst = self.dirs.scratch_dir / "sample.pdf"
|
||||||
|
shutil.copy(src, dst)
|
||||||
|
|
||||||
|
with self.get_consumer(dst, overrides=overrides) as consumer:
|
||||||
|
consumer.run()
|
||||||
|
|
||||||
|
_, kwargs = self.mock_registry.return_value.get_parser_for_file.call_args
|
||||||
|
return kwargs["allow_remote"]
|
||||||
|
|
||||||
|
@override_settings(REMOTE_OCR_MODE="always")
|
||||||
|
def test_always_mode_allows_remote(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN: Remote OCR mode is 'always'.
|
||||||
|
WHEN: A document is consumed without any workflow asking for it.
|
||||||
|
THEN: The registry is allowed to pick the remote parser.
|
||||||
|
"""
|
||||||
|
self.assertTrue(self._consume())
|
||||||
|
|
||||||
|
@override_settings(REMOTE_OCR_MODE="workflow_only")
|
||||||
|
def test_workflow_only_mode_denies_remote_by_default(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN: Remote OCR mode is 'workflow_only'.
|
||||||
|
WHEN: A document is consumed and nothing asked for remote OCR.
|
||||||
|
THEN: The remote parser is excluded.
|
||||||
|
"""
|
||||||
|
self.assertFalse(self._consume())
|
||||||
|
|
||||||
|
@override_settings(REMOTE_OCR_MODE="workflow_only")
|
||||||
|
def test_workflow_only_mode_allows_remote_when_requested(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN: Remote OCR mode is 'workflow_only'.
|
||||||
|
WHEN: A workflow set remote_ocr on the metadata overrides.
|
||||||
|
THEN: The registry is allowed to pick the remote parser.
|
||||||
|
"""
|
||||||
|
self.assertTrue(
|
||||||
|
self._consume(overrides=DocumentMetadataOverrides(remote_ocr=True)),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
class TestMetadataOverrides(TestCase):
|
class TestMetadataOverrides(TestCase):
|
||||||
def test_update_skip_asn_if_exists(self) -> None:
|
def test_update_skip_asn_if_exists(self) -> None:
|
||||||
base = DocumentMetadataOverrides()
|
base = DocumentMetadataOverrides()
|
||||||
@@ -1566,6 +1632,20 @@ class TestMetadataOverrides(TestCase):
|
|||||||
base.update(incoming)
|
base.update(incoming)
|
||||||
self.assertTrue(base.skip_asn_if_exists)
|
self.assertTrue(base.skip_asn_if_exists)
|
||||||
|
|
||||||
|
def test_update_remote_ocr(self) -> None:
|
||||||
|
base = DocumentMetadataOverrides()
|
||||||
|
base.update(DocumentMetadataOverrides(remote_ocr=True))
|
||||||
|
self.assertTrue(base.remote_ocr)
|
||||||
|
|
||||||
|
def test_update_remote_ocr_is_not_unset(self) -> None:
|
||||||
|
"""
|
||||||
|
A later workflow that says nothing must not undo an earlier one that
|
||||||
|
asked for remote OCR.
|
||||||
|
"""
|
||||||
|
base = DocumentMetadataOverrides(remote_ocr=True)
|
||||||
|
base.update(DocumentMetadataOverrides())
|
||||||
|
self.assertTrue(base.remote_ocr)
|
||||||
|
|
||||||
def test_update_actor_and_version_label(self) -> None:
|
def test_update_actor_and_version_label(self) -> None:
|
||||||
base = DocumentMetadataOverrides(
|
base = DocumentMetadataOverrides(
|
||||||
actor_id=1,
|
actor_id=1,
|
||||||
|
|||||||
@@ -385,6 +385,25 @@ class TestTaskFailureHandler:
|
|||||||
task_failure_handler(task_id=None, exception=ValueError("x"), traceback=None)
|
task_failure_handler(task_id=None, exception=ValueError("x"), traceback=None)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.django_db
|
||||||
|
class TestApplyAiSuggestionsTracking:
|
||||||
|
def test_records_the_document_it_is_for(self) -> None:
|
||||||
|
"""
|
||||||
|
The action queues one task per document, so the tracked record notes
|
||||||
|
which document it is for -- otherwise a bulk run is an indistinguishable
|
||||||
|
wall of identical entries in the tasks list.
|
||||||
|
"""
|
||||||
|
task_id = send_publish(
|
||||||
|
"documents.tasks.apply_ai_suggestions",
|
||||||
|
(),
|
||||||
|
{"action_id": 1, "document_id": 42},
|
||||||
|
)
|
||||||
|
|
||||||
|
task = PaperlessTask.objects.get(task_id=task_id)
|
||||||
|
assert task.task_type == PaperlessTask.TaskType.APPLY_AI_SUGGESTIONS
|
||||||
|
assert task.input_data == {"document_id": 42}
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
@pytest.mark.django_db
|
||||||
class TestTaskRevokedHandler:
|
class TestTaskRevokedHandler:
|
||||||
def test_marks_task_revoked(self, mocker: pytest_mock.MockerFixture) -> None:
|
def test_marks_task_revoked(self, mocker: pytest_mock.MockerFixture) -> None:
|
||||||
|
|||||||
@@ -14,6 +14,7 @@ from documents.models import Correspondent
|
|||||||
from documents.models import Document
|
from documents.models import Document
|
||||||
from documents.models import DocumentType
|
from documents.models import DocumentType
|
||||||
from documents.models import Tag
|
from documents.models import Tag
|
||||||
|
from documents.models import WorkflowAction
|
||||||
from documents.sanity_checker import SanityCheckFailedException
|
from documents.sanity_checker import SanityCheckFailedException
|
||||||
from documents.sanity_checker import SanityCheckMessages
|
from documents.sanity_checker import SanityCheckMessages
|
||||||
from documents.tests.test_classifier import dummy_preprocess
|
from documents.tests.test_classifier import dummy_preprocess
|
||||||
@@ -287,6 +288,45 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
|
|||||||
self.assertNotEqual(Document.objects.get(pk=doc.pk).content, "test")
|
self.assertNotEqual(Document.objects.get(pk=doc.pk).content, "test")
|
||||||
|
|
||||||
|
|
||||||
|
class TestUpdateContentRemoteOCR(DirectoriesMixin, TestCase):
|
||||||
|
"""
|
||||||
|
Consumption workflows do not run on reprocess, so the remote parser is
|
||||||
|
used only in 'always' mode or when the caller explicitly asks for it.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def setUp(self) -> None:
|
||||||
|
super().setUp()
|
||||||
|
|
||||||
|
patcher = mock.patch("documents.tasks.get_parser_registry")
|
||||||
|
self.mock_registry = patcher.start()
|
||||||
|
self.mock_registry.return_value.get_parser_for_file.return_value = None
|
||||||
|
self.addCleanup(patcher.stop)
|
||||||
|
|
||||||
|
self.doc = Document.objects.create(
|
||||||
|
title="test",
|
||||||
|
content="my document",
|
||||||
|
checksum="wow",
|
||||||
|
mime_type="application/pdf",
|
||||||
|
)
|
||||||
|
|
||||||
|
def _allow_remote(self, **kwargs) -> bool:
|
||||||
|
tasks.update_document_content_maybe_archive_file(self.doc.pk, **kwargs)
|
||||||
|
_, call_kwargs = self.mock_registry.return_value.get_parser_for_file.call_args
|
||||||
|
return call_kwargs["allow_remote"]
|
||||||
|
|
||||||
|
@override_settings(REMOTE_OCR_MODE="always")
|
||||||
|
def test_always_mode_allows_remote(self) -> None:
|
||||||
|
self.assertTrue(self._allow_remote())
|
||||||
|
|
||||||
|
@override_settings(REMOTE_OCR_MODE="workflow_only")
|
||||||
|
def test_workflow_only_mode_denies_remote_by_default(self) -> None:
|
||||||
|
self.assertFalse(self._allow_remote())
|
||||||
|
|
||||||
|
@override_settings(REMOTE_OCR_MODE="workflow_only")
|
||||||
|
def test_workflow_only_mode_allows_remote_when_requested(self) -> None:
|
||||||
|
self.assertTrue(self._allow_remote(remote_ocr=True))
|
||||||
|
|
||||||
|
|
||||||
class TestAIIndex(DirectoriesMixin, TestCase):
|
class TestAIIndex(DirectoriesMixin, TestCase):
|
||||||
@override_settings(
|
@override_settings(
|
||||||
AI_ENABLED=True,
|
AI_ENABLED=True,
|
||||||
@@ -408,3 +448,110 @@ class TestAIIndex(DirectoriesMixin, TestCase):
|
|||||||
rebuild=False,
|
rebuild=False,
|
||||||
document_ids=doc_ids,
|
document_ids=doc_ids,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TestApplyAISuggestionsTask(DirectoriesMixin, TestCase):
|
||||||
|
def setUp(self) -> None:
|
||||||
|
super().setUp()
|
||||||
|
self.doc = Document.objects.create(
|
||||||
|
title="doc",
|
||||||
|
content="content",
|
||||||
|
checksum="apply-ai-suggestions",
|
||||||
|
)
|
||||||
|
self.action = WorkflowAction.objects.create(
|
||||||
|
type=WorkflowAction.WorkflowActionType.APPLY_AI_SUGGESTIONS,
|
||||||
|
ai_suggestion_fields=[WorkflowAction.AISuggestionField.TITLE],
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_reindexes_without_sending_document_updated(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An apply AI suggestions action that changes the document
|
||||||
|
WHEN:
|
||||||
|
- The task runs
|
||||||
|
THEN:
|
||||||
|
- The search index and caches are refreshed directly, deliberately
|
||||||
|
not via the document_updated signal: that re-runs updated
|
||||||
|
workflows, which for this action means queueing another LLM
|
||||||
|
query for a document it just changed, forever
|
||||||
|
"""
|
||||||
|
with (
|
||||||
|
mock.patch(
|
||||||
|
"documents.workflows.ai.apply_ai_suggestions_to_document",
|
||||||
|
return_value=["title"],
|
||||||
|
),
|
||||||
|
mock.patch("documents.tasks.index_document") as index_document,
|
||||||
|
mock.patch("documents.tasks.clear_document_caches") as clear_caches,
|
||||||
|
mock.patch("documents.tasks.document_updated") as document_updated,
|
||||||
|
):
|
||||||
|
tasks.apply_ai_suggestions(self.action.pk, self.doc.pk)
|
||||||
|
|
||||||
|
index_document.delay.assert_called_once_with(self.doc.pk)
|
||||||
|
clear_caches.assert_called_once_with(self.doc.pk)
|
||||||
|
document_updated.send.assert_not_called()
|
||||||
|
|
||||||
|
def test_no_changes_skips_reindex(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An apply AI suggestions action that changes nothing
|
||||||
|
WHEN:
|
||||||
|
- The task runs
|
||||||
|
THEN:
|
||||||
|
- No reindexing work is queued
|
||||||
|
"""
|
||||||
|
with (
|
||||||
|
mock.patch(
|
||||||
|
"documents.workflows.ai.apply_ai_suggestions_to_document",
|
||||||
|
return_value=[],
|
||||||
|
),
|
||||||
|
mock.patch("documents.tasks.index_document") as index_document,
|
||||||
|
):
|
||||||
|
tasks.apply_ai_suggestions(self.action.pk, self.doc.pk)
|
||||||
|
|
||||||
|
index_document.delay.assert_not_called()
|
||||||
|
|
||||||
|
@override_settings(AI_ENABLED=True, LLM_EMBEDDING_BACKEND="huggingface")
|
||||||
|
def test_updates_llm_index_when_enabled(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An apply AI suggestions action that changes the document
|
||||||
|
- The LLM index is enabled
|
||||||
|
WHEN:
|
||||||
|
- The task runs
|
||||||
|
THEN:
|
||||||
|
- The document is updated in the LLM index too
|
||||||
|
"""
|
||||||
|
with (
|
||||||
|
mock.patch(
|
||||||
|
"documents.workflows.ai.apply_ai_suggestions_to_document",
|
||||||
|
return_value=["title"],
|
||||||
|
),
|
||||||
|
mock.patch("documents.tasks.index_document"),
|
||||||
|
mock.patch(
|
||||||
|
"documents.tasks.update_document_in_llm_index",
|
||||||
|
) as update_in_llm_index,
|
||||||
|
):
|
||||||
|
tasks.apply_ai_suggestions(self.action.pk, self.doc.pk)
|
||||||
|
|
||||||
|
update_in_llm_index.apply_async.assert_called_once()
|
||||||
|
|
||||||
|
def test_deleted_document_is_a_noop(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A document that was deleted between the workflow running and the
|
||||||
|
queued task starting
|
||||||
|
WHEN:
|
||||||
|
- The task runs
|
||||||
|
THEN:
|
||||||
|
- It logs and exits rather than raising
|
||||||
|
"""
|
||||||
|
with (
|
||||||
|
mock.patch(
|
||||||
|
"documents.workflows.ai.apply_ai_suggestions_to_document",
|
||||||
|
) as apply_suggestions,
|
||||||
|
self.assertLogs("paperless.tasks", level="WARNING") as cm,
|
||||||
|
):
|
||||||
|
tasks.apply_ai_suggestions(self.action.pk, self.doc.pk + 1000)
|
||||||
|
|
||||||
|
apply_suggestions.assert_not_called()
|
||||||
|
self.assertIn("no longer exists", "".join(cm.output))
|
||||||
|
|||||||
@@ -31,7 +31,9 @@ from documents.file_handling import create_source_path_directory
|
|||||||
from documents.file_handling import generate_filename
|
from documents.file_handling import generate_filename
|
||||||
from documents.file_handling import generate_unique_filename
|
from documents.file_handling import generate_unique_filename
|
||||||
from documents.signals.handlers import run_workflows
|
from documents.signals.handlers import run_workflows
|
||||||
|
from documents.workflows.ai import apply_ai_suggestions_to_document
|
||||||
from documents.workflows.webhooks import send_webhook
|
from documents.workflows.webhooks import send_webhook
|
||||||
|
from paperless_ai.exceptions import LLMTimeoutError
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
from django.db.models import QuerySet
|
from django.db.models import QuerySet
|
||||||
@@ -2000,55 +2002,6 @@ class TestWorkflows(
|
|||||||
r"Doc added in \w{3,}",
|
r"Doc added in \w{3,}",
|
||||||
) # Match any 3-letter month name
|
) # Match any 3-letter month name
|
||||||
|
|
||||||
def test_document_updated_workflow_existing_custom_field_empty_value(self) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- Existing workflow with UPDATED trigger and action that assigns a custom field
|
|
||||||
with an empty value
|
|
||||||
WHEN:
|
|
||||||
- Document is updated that already contains the field with a value
|
|
||||||
THEN:
|
|
||||||
- The existing value is left untouched, see GH #13627
|
|
||||||
"""
|
|
||||||
trigger = WorkflowTrigger.objects.create(
|
|
||||||
type=WorkflowTrigger.WorkflowTriggerType.DOCUMENT_UPDATED,
|
|
||||||
filter_has_document_type=self.dt,
|
|
||||||
)
|
|
||||||
action = WorkflowAction.objects.create()
|
|
||||||
action.assign_custom_fields.add(self.cf1)
|
|
||||||
action.assign_custom_fields_values = {self.cf1.pk: ""}
|
|
||||||
action.save()
|
|
||||||
w = Workflow.objects.create(
|
|
||||||
name="Workflow 1",
|
|
||||||
order=0,
|
|
||||||
)
|
|
||||||
w.triggers.add(trigger)
|
|
||||||
w.actions.add(action)
|
|
||||||
w.save()
|
|
||||||
|
|
||||||
doc = Document.objects.create(
|
|
||||||
title="sample test",
|
|
||||||
correspondent=self.c,
|
|
||||||
original_filename="sample.pdf",
|
|
||||||
)
|
|
||||||
CustomFieldInstance.objects.create(
|
|
||||||
document=doc,
|
|
||||||
field=self.cf1,
|
|
||||||
value_text="existing value",
|
|
||||||
)
|
|
||||||
|
|
||||||
superuser = User.objects.create_superuser("superuser")
|
|
||||||
self.client.force_authenticate(user=superuser)
|
|
||||||
|
|
||||||
self.client.patch(
|
|
||||||
f"/api/documents/{doc.id}/",
|
|
||||||
{"document_type": self.dt.id},
|
|
||||||
format="json",
|
|
||||||
)
|
|
||||||
|
|
||||||
doc.refresh_from_db()
|
|
||||||
self.assertEqual(doc.custom_fields.get(field=self.cf1).value, "existing value")
|
|
||||||
|
|
||||||
def test_document_updated_workflow_existing_custom_field(self) -> None:
|
def test_document_updated_workflow_existing_custom_field(self) -> None:
|
||||||
"""
|
"""
|
||||||
GIVEN:
|
GIVEN:
|
||||||
@@ -5409,3 +5362,493 @@ class TestDateWorkflowLocalization(
|
|||||||
document = Document.objects.first()
|
document = Document.objects.first()
|
||||||
assert document is not None
|
assert document is not None
|
||||||
assert document.title == expected_title
|
assert document.title == expected_title
|
||||||
|
|
||||||
|
|
||||||
|
class TestRemoteOCRWorkflowAction(DirectoriesMixin, SampleDirMixin, APITestCase):
|
||||||
|
def _make_workflow(self, trigger_type) -> None:
|
||||||
|
trigger = WorkflowTrigger.objects.create(type=trigger_type)
|
||||||
|
action = WorkflowAction.objects.create(
|
||||||
|
type=WorkflowAction.WorkflowActionType.REMOTE_OCR,
|
||||||
|
)
|
||||||
|
w = Workflow.objects.create(name="Remote OCR", order=0)
|
||||||
|
w.triggers.add(trigger)
|
||||||
|
w.actions.add(action)
|
||||||
|
w.save()
|
||||||
|
|
||||||
|
def test_consumption_trigger_requests_remote_ocr(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A consumption workflow with a remote OCR action
|
||||||
|
WHEN:
|
||||||
|
- A matching document is consumed
|
||||||
|
THEN:
|
||||||
|
- The overrides ask for remote OCR, which is what the consumer
|
||||||
|
reads when choosing a parser
|
||||||
|
"""
|
||||||
|
self._make_workflow(WorkflowTrigger.WorkflowTriggerType.CONSUMPTION)
|
||||||
|
|
||||||
|
test_file = shutil.copy(
|
||||||
|
self.SAMPLE_DIR / "simple.pdf",
|
||||||
|
self.dirs.scratch_dir / "simple.pdf",
|
||||||
|
)
|
||||||
|
overrides = DocumentMetadataOverrides()
|
||||||
|
|
||||||
|
run_workflows(
|
||||||
|
WorkflowTrigger.WorkflowTriggerType.CONSUMPTION,
|
||||||
|
ConsumableDocument(
|
||||||
|
source=DocumentSource.ConsumeFolder,
|
||||||
|
original_file=test_file,
|
||||||
|
),
|
||||||
|
overrides=overrides,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertTrue(overrides.remote_ocr)
|
||||||
|
|
||||||
|
def test_other_trigger_types_are_ignored(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A workflow with a remote OCR action that also has a
|
||||||
|
non-consumption trigger, which is a valid combination
|
||||||
|
WHEN:
|
||||||
|
- The non-consumption trigger fires
|
||||||
|
THEN:
|
||||||
|
- The action is skipped, since the document has already been
|
||||||
|
parsed by this point
|
||||||
|
"""
|
||||||
|
trigger = WorkflowTrigger.objects.create(
|
||||||
|
type=WorkflowTrigger.WorkflowTriggerType.CONSUMPTION,
|
||||||
|
)
|
||||||
|
updated_trigger = WorkflowTrigger.objects.create(
|
||||||
|
type=WorkflowTrigger.WorkflowTriggerType.DOCUMENT_UPDATED,
|
||||||
|
)
|
||||||
|
action = WorkflowAction.objects.create(
|
||||||
|
type=WorkflowAction.WorkflowActionType.REMOTE_OCR,
|
||||||
|
)
|
||||||
|
w = Workflow.objects.create(name="Remote OCR", order=0)
|
||||||
|
w.triggers.add(trigger, updated_trigger)
|
||||||
|
w.actions.add(action)
|
||||||
|
w.save()
|
||||||
|
|
||||||
|
doc = Document.objects.create(
|
||||||
|
title="sample test",
|
||||||
|
original_filename="sample.pdf",
|
||||||
|
)
|
||||||
|
|
||||||
|
with self.assertLogs("paperless.handlers", level="DEBUG") as cm:
|
||||||
|
run_workflows(
|
||||||
|
WorkflowTrigger.WorkflowTriggerType.DOCUMENT_UPDATED,
|
||||||
|
doc,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertIn("only applies to consumption triggers", "".join(cm.output))
|
||||||
|
|
||||||
|
|
||||||
|
SUGGESTIONS = {
|
||||||
|
"title": "Suggested Title",
|
||||||
|
"tags": ["Existing Tag", "Suggested Tag"],
|
||||||
|
"correspondents": ["Existing Correspondent", "Suggested Correspondent"],
|
||||||
|
"document_types": ["Suggested Document Type"],
|
||||||
|
"storage_paths": ["Suggested Storage Path"],
|
||||||
|
"dates": ["2024-03-05"],
|
||||||
|
}
|
||||||
|
|
||||||
|
ALL_SUGGESTION_FIELDS = [
|
||||||
|
WorkflowAction.AISuggestionField.TITLE,
|
||||||
|
WorkflowAction.AISuggestionField.TAGS,
|
||||||
|
WorkflowAction.AISuggestionField.CORRESPONDENT,
|
||||||
|
WorkflowAction.AISuggestionField.DOCUMENT_TYPE,
|
||||||
|
WorkflowAction.AISuggestionField.STORAGE_PATH,
|
||||||
|
WorkflowAction.AISuggestionField.CREATED,
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
@override_settings(AI_ENABLED=True)
|
||||||
|
class TestApplyAISuggestionsWorkflowAction(
|
||||||
|
DirectoriesMixin,
|
||||||
|
SampleDirMixin,
|
||||||
|
APITestCase,
|
||||||
|
):
|
||||||
|
def setUp(self) -> None:
|
||||||
|
super().setUp()
|
||||||
|
self.user = User.objects.create(username="ai-user")
|
||||||
|
self.doc = Document.objects.create(
|
||||||
|
title="original.pdf",
|
||||||
|
content="the document content",
|
||||||
|
checksum="ai-suggestions-checksum",
|
||||||
|
mime_type="application/pdf",
|
||||||
|
created=datetime.date(2020, 1, 1),
|
||||||
|
owner=self.user,
|
||||||
|
)
|
||||||
|
|
||||||
|
def make_action(self, **kwargs) -> WorkflowAction:
|
||||||
|
return WorkflowAction.objects.create(
|
||||||
|
type=WorkflowAction.WorkflowActionType.APPLY_AI_SUGGESTIONS,
|
||||||
|
ai_suggestion_fields=kwargs.pop(
|
||||||
|
"ai_suggestion_fields",
|
||||||
|
ALL_SUGGESTION_FIELDS,
|
||||||
|
),
|
||||||
|
**kwargs,
|
||||||
|
)
|
||||||
|
|
||||||
|
def make_workflow(self, action: WorkflowAction, trigger_type) -> Workflow:
|
||||||
|
trigger = WorkflowTrigger.objects.create(type=trigger_type)
|
||||||
|
w = Workflow.objects.create(name="Apply AI suggestions", order=0)
|
||||||
|
w.triggers.add(trigger)
|
||||||
|
w.actions.add(action)
|
||||||
|
w.save()
|
||||||
|
return w
|
||||||
|
|
||||||
|
def apply(self, action: WorkflowAction) -> list[str]:
|
||||||
|
with mock.patch(
|
||||||
|
"documents.workflows.ai.get_ai_document_classification",
|
||||||
|
return_value=SUGGESTIONS,
|
||||||
|
):
|
||||||
|
changed = apply_ai_suggestions_to_document(action, self.doc)
|
||||||
|
self.doc.refresh_from_db()
|
||||||
|
return changed
|
||||||
|
|
||||||
|
def test_document_added_trigger_queues_task(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A document added workflow with an apply AI suggestions action
|
||||||
|
WHEN:
|
||||||
|
- A matching document is added
|
||||||
|
THEN:
|
||||||
|
- The work is queued rather than run inline, so a slow LLM query
|
||||||
|
cannot stall the rest of the workflow run
|
||||||
|
"""
|
||||||
|
action = self.make_action()
|
||||||
|
self.make_workflow(action, WorkflowTrigger.WorkflowTriggerType.DOCUMENT_ADDED)
|
||||||
|
|
||||||
|
with mock.patch("documents.tasks.apply_ai_suggestions.delay") as delay:
|
||||||
|
run_workflows(
|
||||||
|
WorkflowTrigger.WorkflowTriggerType.DOCUMENT_ADDED,
|
||||||
|
self.doc,
|
||||||
|
)
|
||||||
|
|
||||||
|
delay.assert_called_once_with(action_id=action.pk, document_id=self.doc.pk)
|
||||||
|
|
||||||
|
def test_consumption_trigger_is_ignored(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A workflow with an apply AI suggestions action and a consumption
|
||||||
|
trigger alongside a valid one
|
||||||
|
WHEN:
|
||||||
|
- The consumption trigger fires
|
||||||
|
THEN:
|
||||||
|
- The action is skipped, since the document has not been parsed
|
||||||
|
yet and so has no content to make suggestions from
|
||||||
|
"""
|
||||||
|
action = self.make_action()
|
||||||
|
w = self.make_workflow(
|
||||||
|
action,
|
||||||
|
WorkflowTrigger.WorkflowTriggerType.DOCUMENT_ADDED,
|
||||||
|
)
|
||||||
|
w.triggers.add(
|
||||||
|
WorkflowTrigger.objects.create(
|
||||||
|
type=WorkflowTrigger.WorkflowTriggerType.CONSUMPTION,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
test_file = shutil.copy(
|
||||||
|
self.SAMPLE_DIR / "simple.pdf",
|
||||||
|
self.dirs.scratch_dir / "simple.pdf",
|
||||||
|
)
|
||||||
|
|
||||||
|
with (
|
||||||
|
mock.patch("documents.tasks.apply_ai_suggestions.delay") as delay,
|
||||||
|
self.assertLogs("paperless.handlers", level="DEBUG") as cm,
|
||||||
|
):
|
||||||
|
run_workflows(
|
||||||
|
WorkflowTrigger.WorkflowTriggerType.CONSUMPTION,
|
||||||
|
ConsumableDocument(
|
||||||
|
source=DocumentSource.ConsumeFolder,
|
||||||
|
original_file=test_file,
|
||||||
|
),
|
||||||
|
overrides=DocumentMetadataOverrides(),
|
||||||
|
)
|
||||||
|
|
||||||
|
delay.assert_not_called()
|
||||||
|
self.assertIn("does not apply to consumption triggers", "".join(cm.output))
|
||||||
|
|
||||||
|
def test_no_selected_fields_does_nothing(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An action with no suggestion fields selected
|
||||||
|
WHEN:
|
||||||
|
- The action is applied
|
||||||
|
THEN:
|
||||||
|
- Nothing is changed and it is logged
|
||||||
|
"""
|
||||||
|
action = self.make_action(ai_suggestion_fields=[])
|
||||||
|
|
||||||
|
with self.assertLogs("paperless.workflows.ai", level="WARNING") as cm:
|
||||||
|
changed = self.apply(action)
|
||||||
|
|
||||||
|
self.assertEqual(changed, [])
|
||||||
|
self.assertIn("no AI suggestion fields selected", "".join(cm.output))
|
||||||
|
|
||||||
|
@override_settings(AI_ENABLED=False)
|
||||||
|
def test_ai_disabled_does_nothing(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An action on an install where AI has since been disabled
|
||||||
|
WHEN:
|
||||||
|
- The action is applied
|
||||||
|
THEN:
|
||||||
|
- Nothing is changed and it is logged
|
||||||
|
"""
|
||||||
|
action = self.make_action()
|
||||||
|
|
||||||
|
with self.assertLogs("paperless.workflows.ai", level="ERROR") as cm:
|
||||||
|
changed = self.apply(action)
|
||||||
|
|
||||||
|
self.assertEqual(changed, [])
|
||||||
|
self.assertIn("AI is not enabled", "".join(cm.output))
|
||||||
|
|
||||||
|
def test_invalid_configuration_leaves_document_untouched(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An AI backend that is misconfigured
|
||||||
|
WHEN:
|
||||||
|
- The action is applied
|
||||||
|
THEN:
|
||||||
|
- The failure is logged and the document is left alone. It is not
|
||||||
|
re-raised, because retrying will not fix a bad configuration
|
||||||
|
"""
|
||||||
|
action = self.make_action()
|
||||||
|
|
||||||
|
with (
|
||||||
|
mock.patch(
|
||||||
|
"documents.workflows.ai.get_ai_document_classification",
|
||||||
|
side_effect=ValueError("nope"),
|
||||||
|
),
|
||||||
|
self.assertLogs("paperless.workflows.ai", level="ERROR") as cm,
|
||||||
|
):
|
||||||
|
changed = apply_ai_suggestions_to_document(action, self.doc)
|
||||||
|
|
||||||
|
self.assertEqual(changed, [])
|
||||||
|
self.doc.refresh_from_db()
|
||||||
|
self.assertEqual(self.doc.title, "original.pdf")
|
||||||
|
self.assertIn("Invalid AI configuration", "".join(cm.output))
|
||||||
|
|
||||||
|
def test_transient_llm_failure_is_raised_for_retry(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An LLM backend that times out, or rate limits the request
|
||||||
|
WHEN:
|
||||||
|
- The action is applied
|
||||||
|
THEN:
|
||||||
|
- The error propagates so the queued task can back off and retry,
|
||||||
|
rather than silently dropping this document's suggestions
|
||||||
|
"""
|
||||||
|
action = self.make_action()
|
||||||
|
|
||||||
|
with (
|
||||||
|
mock.patch(
|
||||||
|
"documents.workflows.ai.get_ai_document_classification",
|
||||||
|
side_effect=LLMTimeoutError(),
|
||||||
|
),
|
||||||
|
self.assertRaises(LLMTimeoutError),
|
||||||
|
):
|
||||||
|
apply_ai_suggestions_to_document(action, self.doc)
|
||||||
|
|
||||||
|
self.doc.refresh_from_db()
|
||||||
|
self.assertEqual(self.doc.title, "original.pdf")
|
||||||
|
|
||||||
|
def test_only_matching_objects_are_applied(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An action without create missing, and only some of the suggested
|
||||||
|
objects existing
|
||||||
|
WHEN:
|
||||||
|
- The action is applied
|
||||||
|
THEN:
|
||||||
|
- Only the existing objects are assigned, unmatched suggestions are
|
||||||
|
dropped rather than creating anything
|
||||||
|
"""
|
||||||
|
tag = Tag.objects.create(name="Existing Tag", owner=self.user)
|
||||||
|
correspondent = Correspondent.objects.create(
|
||||||
|
name="Existing Correspondent",
|
||||||
|
owner=self.user,
|
||||||
|
)
|
||||||
|
action = self.make_action(ai_overwrite_existing=True)
|
||||||
|
|
||||||
|
changed = self.apply(action)
|
||||||
|
|
||||||
|
self.assertEqual(self.doc.correspondent, correspondent)
|
||||||
|
self.assertEqual(list(self.doc.tags.all()), [tag])
|
||||||
|
# Nothing matched for these and create missing is off
|
||||||
|
self.assertIsNone(self.doc.document_type)
|
||||||
|
self.assertIsNone(self.doc.storage_path)
|
||||||
|
self.assertNotIn("document_type", changed)
|
||||||
|
self.assertEqual(Tag.objects.count(), 1)
|
||||||
|
self.assertEqual(Correspondent.objects.count(), 1)
|
||||||
|
|
||||||
|
def test_create_missing_creates_objects_owned_by_document_owner(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An action with create missing enabled
|
||||||
|
WHEN:
|
||||||
|
- The action is applied and suggestions match nothing
|
||||||
|
THEN:
|
||||||
|
- Tags, correspondents and document types are created, owned by the
|
||||||
|
document owner so they stay private to them
|
||||||
|
- Storage paths are never created, since a path template cannot be
|
||||||
|
inferred from a name
|
||||||
|
"""
|
||||||
|
action = self.make_action(
|
||||||
|
ai_create_missing=True,
|
||||||
|
ai_overwrite_existing=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
changed = self.apply(action)
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
sorted(t.name for t in self.doc.tags.all()),
|
||||||
|
["Existing Tag", "Suggested Tag"],
|
||||||
|
)
|
||||||
|
self.assertEqual(self.doc.correspondent.name, "Existing Correspondent")
|
||||||
|
self.assertEqual(self.doc.correspondent.owner, self.user)
|
||||||
|
self.assertEqual(self.doc.document_type.name, "Suggested Document Type")
|
||||||
|
self.assertEqual(self.doc.document_type.owner, self.user)
|
||||||
|
|
||||||
|
self.assertIsNone(self.doc.storage_path)
|
||||||
|
self.assertFalse(StoragePath.objects.exists())
|
||||||
|
self.assertNotIn("storage_path", changed)
|
||||||
|
|
||||||
|
def test_overwrite_disabled_keeps_existing_values(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An action without overwrite existing
|
||||||
|
- A document that already has a title, created date and
|
||||||
|
correspondent
|
||||||
|
WHEN:
|
||||||
|
- The action is applied
|
||||||
|
THEN:
|
||||||
|
- The existing values are kept, only the empty document type is
|
||||||
|
filled in
|
||||||
|
"""
|
||||||
|
existing = Correspondent.objects.create(name="Mine", owner=self.user)
|
||||||
|
self.doc.correspondent = existing
|
||||||
|
self.doc.save()
|
||||||
|
action = self.make_action(ai_create_missing=True)
|
||||||
|
|
||||||
|
changed = self.apply(action)
|
||||||
|
|
||||||
|
self.assertEqual(self.doc.title, "original.pdf")
|
||||||
|
self.assertEqual(self.doc.created, datetime.date(2020, 1, 1))
|
||||||
|
self.assertEqual(self.doc.correspondent, existing)
|
||||||
|
self.assertEqual(self.doc.document_type.name, "Suggested Document Type")
|
||||||
|
self.assertNotIn("title", changed)
|
||||||
|
self.assertNotIn("correspondent", changed)
|
||||||
|
|
||||||
|
def test_overwrite_enabled_replaces_existing_values(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An action with overwrite existing
|
||||||
|
- A document that already has a title and created date
|
||||||
|
WHEN:
|
||||||
|
- The action is applied
|
||||||
|
THEN:
|
||||||
|
- The suggested values replace them
|
||||||
|
"""
|
||||||
|
action = self.make_action(
|
||||||
|
ai_create_missing=True,
|
||||||
|
ai_overwrite_existing=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
changed = self.apply(action)
|
||||||
|
|
||||||
|
self.assertEqual(self.doc.title, "Suggested Title")
|
||||||
|
self.assertEqual(self.doc.created, datetime.date(2024, 3, 5))
|
||||||
|
self.assertIn("title", changed)
|
||||||
|
self.assertIn("created", changed)
|
||||||
|
|
||||||
|
def test_tags_are_added_not_replaced(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A document that already has a tag unrelated to the suggestions
|
||||||
|
WHEN:
|
||||||
|
- The action is applied with overwrite existing enabled
|
||||||
|
THEN:
|
||||||
|
- The existing tag is kept, since suggested tags are always
|
||||||
|
additive regardless of the overwrite setting
|
||||||
|
"""
|
||||||
|
kept = Tag.objects.create(name="Do Not Remove", owner=self.user)
|
||||||
|
self.doc.tags.add(kept)
|
||||||
|
Tag.objects.create(name="Existing Tag", owner=self.user)
|
||||||
|
action = self.make_action(ai_overwrite_existing=True)
|
||||||
|
|
||||||
|
self.apply(action)
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
sorted(t.name for t in self.doc.tags.all()),
|
||||||
|
["Do Not Remove", "Existing Tag"],
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_unselected_fields_are_untouched(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An action that only selects the title
|
||||||
|
WHEN:
|
||||||
|
- The action is applied
|
||||||
|
THEN:
|
||||||
|
- Only the title changes, even though the LLM suggested everything
|
||||||
|
"""
|
||||||
|
action = self.make_action(
|
||||||
|
ai_suggestion_fields=[WorkflowAction.AISuggestionField.TITLE],
|
||||||
|
ai_create_missing=True,
|
||||||
|
ai_overwrite_existing=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
changed = self.apply(action)
|
||||||
|
|
||||||
|
self.assertEqual(changed, ["title"])
|
||||||
|
self.assertEqual(self.doc.title, "Suggested Title")
|
||||||
|
self.assertEqual(self.doc.tags.count(), 0)
|
||||||
|
self.assertIsNone(self.doc.correspondent)
|
||||||
|
self.assertEqual(self.doc.created, datetime.date(2020, 1, 1))
|
||||||
|
|
||||||
|
def test_another_users_private_objects_are_not_matched(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A suggested tag name that exists, but is owned by someone else
|
||||||
|
WHEN:
|
||||||
|
- The action is applied
|
||||||
|
THEN:
|
||||||
|
- It is not assigned, because the document owner cannot see it
|
||||||
|
"""
|
||||||
|
other = User.objects.create(username="someone-else")
|
||||||
|
Tag.objects.create(name="Existing Tag", owner=other)
|
||||||
|
action = self.make_action(
|
||||||
|
ai_suggestion_fields=[WorkflowAction.AISuggestionField.TAGS],
|
||||||
|
)
|
||||||
|
|
||||||
|
self.apply(action)
|
||||||
|
|
||||||
|
self.assertEqual(self.doc.tags.count(), 0)
|
||||||
|
|
||||||
|
def test_unparsable_dates_are_skipped(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- Suggested dates that are not all valid
|
||||||
|
WHEN:
|
||||||
|
- The action is applied
|
||||||
|
THEN:
|
||||||
|
- The first usable date is applied and the rest ignored
|
||||||
|
"""
|
||||||
|
action = self.make_action(
|
||||||
|
ai_suggestion_fields=[WorkflowAction.AISuggestionField.CREATED],
|
||||||
|
ai_overwrite_existing=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
with mock.patch(
|
||||||
|
"documents.workflows.ai.get_ai_document_classification",
|
||||||
|
return_value={**SUGGESTIONS, "dates": ["not a date", "2019-07-04"]},
|
||||||
|
):
|
||||||
|
changed = apply_ai_suggestions_to_document(action, self.doc)
|
||||||
|
|
||||||
|
self.doc.refresh_from_db()
|
||||||
|
self.assertEqual(changed, ["created"])
|
||||||
|
self.assertEqual(self.doc.created, datetime.date(2019, 7, 4))
|
||||||
|
|||||||
+17
-17
@@ -236,12 +236,15 @@ from paperless import version
|
|||||||
from paperless.celery import app as celery_app
|
from paperless.celery import app as celery_app
|
||||||
from paperless.config import AIConfig
|
from paperless.config import AIConfig
|
||||||
from paperless.config import GeneralConfig
|
from paperless.config import GeneralConfig
|
||||||
|
from paperless.config import RemoteOCRConfig
|
||||||
from paperless.models import ApplicationConfiguration
|
from paperless.models import ApplicationConfiguration
|
||||||
from paperless.parsers.registry import get_parser_registry
|
from paperless.parsers.registry import get_parser_registry
|
||||||
|
from paperless.parsers.remote import RemoteEngineConfig
|
||||||
from paperless.serialisers import GroupSerializer
|
from paperless.serialisers import GroupSerializer
|
||||||
from paperless.serialisers import UserSerializer
|
from paperless.serialisers import UserSerializer
|
||||||
from paperless.views import StandardPagination
|
from paperless.views import StandardPagination
|
||||||
from paperless_ai.ai_classifier import get_ai_document_classification
|
from paperless_ai.ai_classifier import get_ai_document_classification
|
||||||
|
from paperless_ai.ai_classifier import get_llm_output_language
|
||||||
from paperless_ai.chat import stream_chat_with_documents
|
from paperless_ai.chat import stream_chat_with_documents
|
||||||
from paperless_ai.exceptions import LLMTimeoutError
|
from paperless_ai.exceptions import LLMTimeoutError
|
||||||
from paperless_ai.matching import extract_unmatched_names
|
from paperless_ai.matching import extract_unmatched_names
|
||||||
@@ -653,20 +656,6 @@ class TagViewSet(PermissionsAwareDocumentCountMixin, ModelViewSet[Tag]):
|
|||||||
update_document_parent_tags(tag, new_parent)
|
update_document_parent_tags(tag, new_parent)
|
||||||
|
|
||||||
|
|
||||||
def _get_llm_output_language(ai_config: AIConfig, request) -> str | None:
|
|
||||||
output_language = ai_config.llm_output_language
|
|
||||||
if (
|
|
||||||
not output_language
|
|
||||||
and hasattr(request.user, "ui_settings")
|
|
||||||
and isinstance(
|
|
||||||
request.user.ui_settings.settings,
|
|
||||||
dict,
|
|
||||||
)
|
|
||||||
):
|
|
||||||
output_language = request.user.ui_settings.settings.get("language")
|
|
||||||
return output_language
|
|
||||||
|
|
||||||
|
|
||||||
@extend_schema_view(**generate_object_with_permissions_schema(DocumentTypeSerializer))
|
@extend_schema_view(**generate_object_with_permissions_schema(DocumentTypeSerializer))
|
||||||
class DocumentTypeViewSet(
|
class DocumentTypeViewSet(
|
||||||
PermissionsAwareDocumentCountMixin,
|
PermissionsAwareDocumentCountMixin,
|
||||||
@@ -1528,7 +1517,10 @@ class DocumentViewSet(
|
|||||||
if not ai_config.ai_enabled:
|
if not ai_config.ai_enabled:
|
||||||
return HttpResponseBadRequest("AI is required for this feature")
|
return HttpResponseBadRequest("AI is required for this feature")
|
||||||
|
|
||||||
output_language = _get_llm_output_language(ai_config=ai_config, request=request)
|
output_language = get_llm_output_language(
|
||||||
|
ai_config=ai_config,
|
||||||
|
user=request.user,
|
||||||
|
)
|
||||||
llm_cache_backend = ":".join(
|
llm_cache_backend = ":".join(
|
||||||
part
|
part
|
||||||
for part in (
|
for part in (
|
||||||
@@ -2267,13 +2259,16 @@ class ChatStreamingView(GenericAPIView[Any]):
|
|||||||
if not has_perms_owner_aware(request.user, "view_document", document):
|
if not has_perms_owner_aware(request.user, "view_document", document):
|
||||||
return HttpResponseForbidden("Insufficient permissions")
|
return HttpResponseForbidden("Insufficient permissions")
|
||||||
|
|
||||||
documents = Document.objects.filter(pk=document.pk)
|
documents = [document]
|
||||||
else:
|
else:
|
||||||
documents = Document.objects.filter(
|
documents = Document.objects.filter(
|
||||||
id__in=permitted_document_ids(request.user),
|
id__in=permitted_document_ids(request.user),
|
||||||
)
|
)
|
||||||
|
|
||||||
output_language = _get_llm_output_language(ai_config=ai_config, request=request)
|
output_language = get_llm_output_language(
|
||||||
|
ai_config=ai_config,
|
||||||
|
user=request.user,
|
||||||
|
)
|
||||||
|
|
||||||
response = StreamingHttpResponse(
|
response = StreamingHttpResponse(
|
||||||
stream_chat_with_documents(
|
stream_chat_with_documents(
|
||||||
@@ -4010,6 +4005,11 @@ class UiSettingsView(GenericAPIView[Any]):
|
|||||||
|
|
||||||
ui_settings["auditlog_enabled"] = settings.AUDIT_LOG_ENABLED
|
ui_settings["auditlog_enabled"] = settings.AUDIT_LOG_ENABLED
|
||||||
|
|
||||||
|
ui_settings["remote_ocr"] = {
|
||||||
|
"configured": RemoteEngineConfig.from_app_config().engine_is_valid(),
|
||||||
|
"mode": RemoteOCRConfig().remote_ocr_mode,
|
||||||
|
}
|
||||||
|
|
||||||
if settings.GMAIL_OAUTH_ENABLED or settings.OUTLOOK_OAUTH_ENABLED:
|
if settings.GMAIL_OAUTH_ENABLED or settings.OUTLOOK_OAUTH_ENABLED:
|
||||||
manager = PaperlessMailOAuth2Manager()
|
manager = PaperlessMailOAuth2Manager()
|
||||||
if settings.GMAIL_OAUTH_ENABLED:
|
if settings.GMAIL_OAUTH_ENABLED:
|
||||||
|
|||||||
@@ -0,0 +1,241 @@
|
|||||||
|
import logging
|
||||||
|
from datetime import date
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from django.contrib.auth.models import User
|
||||||
|
|
||||||
|
from documents.models import Correspondent
|
||||||
|
from documents.models import Document
|
||||||
|
from documents.models import DocumentType
|
||||||
|
from documents.models import StoragePath
|
||||||
|
from documents.models import Tag
|
||||||
|
from documents.models import WorkflowAction
|
||||||
|
from paperless.config import AIConfig
|
||||||
|
from paperless_ai.ai_classifier import get_ai_document_classification
|
||||||
|
from paperless_ai.ai_classifier import get_llm_output_language
|
||||||
|
from paperless_ai.matching import extract_unmatched_names
|
||||||
|
from paperless_ai.matching import match_correspondents_by_name
|
||||||
|
from paperless_ai.matching import match_document_types_by_name
|
||||||
|
from paperless_ai.matching import match_storage_paths_by_name
|
||||||
|
from paperless_ai.matching import match_tags_by_name
|
||||||
|
|
||||||
|
logger = logging.getLogger("paperless.workflows.ai")
|
||||||
|
|
||||||
|
AISuggestionField = WorkflowAction.AISuggestionField
|
||||||
|
|
||||||
|
# Tags use m2m relation instead
|
||||||
|
DIRECT_FIELDS: dict[str, str] = {
|
||||||
|
AISuggestionField.TITLE: "title",
|
||||||
|
AISuggestionField.CORRESPONDENT: "correspondent",
|
||||||
|
AISuggestionField.DOCUMENT_TYPE: "document_type",
|
||||||
|
AISuggestionField.STORAGE_PATH: "storage_path",
|
||||||
|
AISuggestionField.CREATED: "created",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_date(dates: list[str]) -> date | None:
|
||||||
|
"""
|
||||||
|
First usable date out of the suggestions, which are expected as
|
||||||
|
YYYY-MM-DD. Document.created is a DateField, so only one can be applied.
|
||||||
|
"""
|
||||||
|
for value in dates:
|
||||||
|
try:
|
||||||
|
return datetime.strptime(value, "%Y-%m-%d").date()
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
logger.debug("Ignoring unparsable suggested date %s", value)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_object(
|
||||||
|
model,
|
||||||
|
names: list[str],
|
||||||
|
matched: list,
|
||||||
|
*,
|
||||||
|
create_missing: bool,
|
||||||
|
owner: User | None,
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Single object from a suggestion list. The best match if there was one, else
|
||||||
|
optionally a newly-created object. StoragePaths are excluded.
|
||||||
|
"""
|
||||||
|
if matched:
|
||||||
|
return matched[0]
|
||||||
|
|
||||||
|
if not create_missing or model is StoragePath:
|
||||||
|
return None
|
||||||
|
|
||||||
|
unmatched = extract_unmatched_names(names, matched)
|
||||||
|
if not unmatched:
|
||||||
|
return None
|
||||||
|
|
||||||
|
# (name, owner) is what MatchingModel is unique on
|
||||||
|
obj, created = model.objects.get_or_create(
|
||||||
|
name=unmatched[0][:128],
|
||||||
|
owner=owner,
|
||||||
|
)
|
||||||
|
if created:
|
||||||
|
logger.info("Created %s '%s' from AI suggestion", model.__name__, obj.name)
|
||||||
|
return obj
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_tags(
|
||||||
|
names: list[str],
|
||||||
|
matched: list[Tag],
|
||||||
|
*,
|
||||||
|
create_missing: bool,
|
||||||
|
owner: User | None,
|
||||||
|
) -> list[Tag]:
|
||||||
|
"""
|
||||||
|
Matched tags, plus newly created ones if create_missing is set.
|
||||||
|
"""
|
||||||
|
tags = list(matched)
|
||||||
|
if not create_missing:
|
||||||
|
return tags
|
||||||
|
|
||||||
|
for name in extract_unmatched_names(names, matched):
|
||||||
|
tag, created = Tag.objects.get_or_create(
|
||||||
|
name=name[:128],
|
||||||
|
owner=owner,
|
||||||
|
)
|
||||||
|
if created:
|
||||||
|
logger.info("Created tag '%s' from AI suggestion", tag.name)
|
||||||
|
tags.append(tag)
|
||||||
|
return tags
|
||||||
|
|
||||||
|
|
||||||
|
def apply_ai_suggestions_to_document(
|
||||||
|
action: WorkflowAction,
|
||||||
|
document: Document,
|
||||||
|
logging_group=None,
|
||||||
|
) -> list[str]:
|
||||||
|
"""
|
||||||
|
Get suggestions about `document` and write the chosen fields.
|
||||||
|
|
||||||
|
Returns the names of the fields that were actually changed.
|
||||||
|
"""
|
||||||
|
selected = set(action.ai_suggestion_fields or [])
|
||||||
|
if not selected:
|
||||||
|
logger.warning(
|
||||||
|
"Workflow action %s has no AI suggestion fields selected, skipping",
|
||||||
|
action.pk,
|
||||||
|
extra={"group": logging_group},
|
||||||
|
)
|
||||||
|
return []
|
||||||
|
|
||||||
|
ai_config = AIConfig()
|
||||||
|
if not ai_config.ai_enabled:
|
||||||
|
logger.error(
|
||||||
|
"AI is not enabled, cannot apply AI suggestions for document %s",
|
||||||
|
document.pk,
|
||||||
|
extra={"group": logging_group},
|
||||||
|
)
|
||||||
|
return []
|
||||||
|
|
||||||
|
# Workflows run without a user, so we use the document owner
|
||||||
|
owner = document.owner
|
||||||
|
|
||||||
|
try:
|
||||||
|
suggestions = get_ai_document_classification(
|
||||||
|
document,
|
||||||
|
owner,
|
||||||
|
get_llm_output_language(ai_config, owner),
|
||||||
|
)
|
||||||
|
except ValueError:
|
||||||
|
# A bad AI config will not fix itself, so swallow it rather than
|
||||||
|
# letting the caller retry. Timeouts, rate limits, network errors etc
|
||||||
|
# propagate so the queued task can back off and try again.
|
||||||
|
logger.exception(
|
||||||
|
"Invalid AI configuration, cannot get suggestions for document %s",
|
||||||
|
document.pk,
|
||||||
|
extra={"group": logging_group},
|
||||||
|
)
|
||||||
|
return []
|
||||||
|
|
||||||
|
overwrite = action.ai_overwrite_existing
|
||||||
|
create_missing = action.ai_create_missing
|
||||||
|
updated_fields: list[str] = []
|
||||||
|
|
||||||
|
def should_set(field: str) -> bool:
|
||||||
|
# The field is selected and (overwrite or it's empty)
|
||||||
|
return field in selected and (
|
||||||
|
overwrite or getattr(document, DIRECT_FIELDS[field]) in (None, "")
|
||||||
|
)
|
||||||
|
|
||||||
|
if should_set(AISuggestionField.TITLE):
|
||||||
|
title = (suggestions.get("title") or "").strip()
|
||||||
|
if title:
|
||||||
|
# title is capped at 128 characters
|
||||||
|
document.title = title[:128]
|
||||||
|
updated_fields.append("title")
|
||||||
|
|
||||||
|
if should_set(AISuggestionField.CORRESPONDENT):
|
||||||
|
names = suggestions.get("correspondents", [])
|
||||||
|
correspondent = resolve_object(
|
||||||
|
Correspondent,
|
||||||
|
names,
|
||||||
|
match_correspondents_by_name(names, owner),
|
||||||
|
create_missing=create_missing,
|
||||||
|
owner=owner,
|
||||||
|
)
|
||||||
|
if correspondent:
|
||||||
|
document.correspondent = correspondent
|
||||||
|
updated_fields.append("correspondent")
|
||||||
|
|
||||||
|
if should_set(AISuggestionField.DOCUMENT_TYPE):
|
||||||
|
names = suggestions.get("document_types", [])
|
||||||
|
document_type = resolve_object(
|
||||||
|
DocumentType,
|
||||||
|
names,
|
||||||
|
match_document_types_by_name(names, owner),
|
||||||
|
create_missing=create_missing,
|
||||||
|
owner=owner,
|
||||||
|
)
|
||||||
|
if document_type:
|
||||||
|
document.document_type = document_type
|
||||||
|
updated_fields.append("document_type")
|
||||||
|
|
||||||
|
if should_set(AISuggestionField.STORAGE_PATH):
|
||||||
|
names = suggestions.get("storage_paths", [])
|
||||||
|
storage_path = resolve_object(
|
||||||
|
StoragePath,
|
||||||
|
names,
|
||||||
|
match_storage_paths_by_name(names, owner),
|
||||||
|
create_missing=create_missing,
|
||||||
|
owner=owner,
|
||||||
|
)
|
||||||
|
if storage_path:
|
||||||
|
document.storage_path = storage_path
|
||||||
|
updated_fields.append("storage_path")
|
||||||
|
|
||||||
|
if should_set(AISuggestionField.CREATED):
|
||||||
|
created = resolve_date(suggestions.get("dates", []))
|
||||||
|
if created:
|
||||||
|
document.created = created
|
||||||
|
updated_fields.append("created")
|
||||||
|
|
||||||
|
if updated_fields:
|
||||||
|
# save fields and update modified
|
||||||
|
document.save(update_fields=[*updated_fields, "modified"])
|
||||||
|
|
||||||
|
if AISuggestionField.TAGS in selected:
|
||||||
|
names = suggestions.get("tags", [])
|
||||||
|
tags = resolve_tags(
|
||||||
|
names,
|
||||||
|
match_tags_by_name(names, owner),
|
||||||
|
create_missing=create_missing,
|
||||||
|
owner=owner,
|
||||||
|
)
|
||||||
|
if tags:
|
||||||
|
# Suggested tags are always added, so overwrite_existing
|
||||||
|
# does not really apply here
|
||||||
|
document.add_nested_tags(tags)
|
||||||
|
updated_fields.append("tags")
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"Applied AI suggestions %s to document %s",
|
||||||
|
updated_fields or "(none)",
|
||||||
|
document.pk,
|
||||||
|
extra={"group": logging_group},
|
||||||
|
)
|
||||||
|
|
||||||
|
return updated_fields
|
||||||
@@ -105,8 +105,7 @@ def apply_assignment_to_document(
|
|||||||
field=field,
|
field=field,
|
||||||
document=document,
|
document=document,
|
||||||
).first()
|
).first()
|
||||||
# empty string is indistinguishable from no value in the UI
|
if instance and args[value_field_name] is not None:
|
||||||
if instance and args[value_field_name] not in (None, ""):
|
|
||||||
setattr(instance, value_field_name, args[value_field_name])
|
setattr(instance, value_field_name, args[value_field_name])
|
||||||
instance.save()
|
instance.save()
|
||||||
elif not instance:
|
elif not instance:
|
||||||
|
|||||||
+17
-3
@@ -339,16 +339,30 @@ def check_deprecated_v2_ocr_env_vars(
|
|||||||
|
|
||||||
@register()
|
@register()
|
||||||
def check_remote_parser_configured(app_configs: Any, **kwargs: Any) -> list[Error]:
|
def check_remote_parser_configured(app_configs: Any, **kwargs: Any) -> list[Error]:
|
||||||
|
# Import here because checks.py runs before the app registry is ready
|
||||||
|
from paperless.models import RemoteOCRMode
|
||||||
|
|
||||||
|
errors = []
|
||||||
|
|
||||||
if settings.REMOTE_OCR_ENGINE == "azureai" and not (
|
if settings.REMOTE_OCR_ENGINE == "azureai" and not (
|
||||||
settings.REMOTE_OCR_ENDPOINT and settings.REMOTE_OCR_API_KEY
|
settings.REMOTE_OCR_ENDPOINT and settings.REMOTE_OCR_API_KEY
|
||||||
):
|
):
|
||||||
return [
|
errors.append(
|
||||||
Error(
|
Error(
|
||||||
"Azure AI remote parser requires endpoint and API key to be configured.",
|
"Azure AI remote parser requires endpoint and API key to be configured.",
|
||||||
),
|
),
|
||||||
]
|
)
|
||||||
|
|
||||||
return []
|
valid_modes = {mode.value for mode in RemoteOCRMode}
|
||||||
|
if settings.REMOTE_OCR_MODE not in valid_modes:
|
||||||
|
errors.append(
|
||||||
|
Error(
|
||||||
|
f"PAPERLESS_REMOTE_OCR_MODE is set to {settings.REMOTE_OCR_MODE!r}, "
|
||||||
|
f"expected one of {sorted(valid_modes)}.",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
return errors
|
||||||
|
|
||||||
|
|
||||||
def get_tesseract_langs():
|
def get_tesseract_langs():
|
||||||
|
|||||||
@@ -9,6 +9,7 @@ from paperless.models import CleanChoices
|
|||||||
from paperless.models import ColorConvertChoices
|
from paperless.models import ColorConvertChoices
|
||||||
from paperless.models import ModeChoices
|
from paperless.models import ModeChoices
|
||||||
from paperless.models import OutputTypeChoices
|
from paperless.models import OutputTypeChoices
|
||||||
|
from paperless.models import RemoteOCRMode
|
||||||
|
|
||||||
|
|
||||||
@dataclasses.dataclass
|
@dataclasses.dataclass
|
||||||
@@ -185,6 +186,45 @@ class GeneralConfig(BaseConfig):
|
|||||||
self.app_logo = app_config.app_logo.url if app_config.app_logo else None
|
self.app_logo = app_config.app_logo.url if app_config.app_logo else None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclasses.dataclass
|
||||||
|
class RemoteOCRConfig(BaseConfig):
|
||||||
|
"""
|
||||||
|
Settings for the remote (cloud) OCR parser
|
||||||
|
"""
|
||||||
|
|
||||||
|
remote_ocr_engine: str | None = dataclasses.field(init=False)
|
||||||
|
remote_ocr_api_key: str | None = dataclasses.field(init=False)
|
||||||
|
remote_ocr_endpoint: str | None = dataclasses.field(init=False)
|
||||||
|
remote_ocr_mode: RemoteOCRMode = dataclasses.field(init=False)
|
||||||
|
|
||||||
|
def __post_init__(self) -> None:
|
||||||
|
app_config = self._get_config_instance()
|
||||||
|
|
||||||
|
self.remote_ocr_engine = (
|
||||||
|
app_config.remote_ocr_engine or settings.REMOTE_OCR_ENGINE
|
||||||
|
)
|
||||||
|
self.remote_ocr_api_key = (
|
||||||
|
app_config.remote_ocr_api_key or settings.REMOTE_OCR_API_KEY
|
||||||
|
)
|
||||||
|
self.remote_ocr_endpoint = (
|
||||||
|
app_config.remote_ocr_endpoint or settings.REMOTE_OCR_ENDPOINT
|
||||||
|
)
|
||||||
|
self.remote_ocr_mode = app_config.remote_ocr_mode or RemoteOCRMode(
|
||||||
|
settings.REMOTE_OCR_MODE,
|
||||||
|
)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def remote_ocr_by_default(self) -> bool:
|
||||||
|
"""
|
||||||
|
Whether every supported document goes to the remote engine.
|
||||||
|
|
||||||
|
When False the remote engine is used only for documents that
|
||||||
|
explicitly asked for it, i.e. a workflow matched during consumption or
|
||||||
|
the user ticked the box when reprocessing.
|
||||||
|
"""
|
||||||
|
return self.remote_ocr_mode == RemoteOCRMode.ALWAYS
|
||||||
|
|
||||||
|
|
||||||
@dataclasses.dataclass
|
@dataclasses.dataclass
|
||||||
class AIConfig(BaseConfig):
|
class AIConfig(BaseConfig):
|
||||||
"""
|
"""
|
||||||
|
|||||||
@@ -0,0 +1,44 @@
|
|||||||
|
# Generated by Django 5.2.16 on 2026-08-10 14:37
|
||||||
|
|
||||||
|
from django.db import migrations
|
||||||
|
from django.db import models
|
||||||
|
|
||||||
|
|
||||||
|
class Migration(migrations.Migration):
|
||||||
|
dependencies = [
|
||||||
|
("paperless", "0013_applicationconfiguration_llm_request_timeout"),
|
||||||
|
]
|
||||||
|
|
||||||
|
operations = [
|
||||||
|
migrations.AddField(
|
||||||
|
model_name="applicationconfiguration",
|
||||||
|
name="remote_ocr_api_key",
|
||||||
|
field=models.CharField(
|
||||||
|
blank=True,
|
||||||
|
max_length=1024,
|
||||||
|
null=True,
|
||||||
|
verbose_name="Sets the remote OCR API key",
|
||||||
|
),
|
||||||
|
),
|
||||||
|
migrations.AddField(
|
||||||
|
model_name="applicationconfiguration",
|
||||||
|
name="remote_ocr_endpoint",
|
||||||
|
field=models.CharField(
|
||||||
|
blank=True,
|
||||||
|
max_length=256,
|
||||||
|
null=True,
|
||||||
|
verbose_name="Sets the remote OCR endpoint",
|
||||||
|
),
|
||||||
|
),
|
||||||
|
migrations.AddField(
|
||||||
|
model_name="applicationconfiguration",
|
||||||
|
name="remote_ocr_engine",
|
||||||
|
field=models.CharField(
|
||||||
|
blank=True,
|
||||||
|
choices=[("azureai", "Azure AI Document Intelligence")],
|
||||||
|
max_length=32,
|
||||||
|
null=True,
|
||||||
|
verbose_name="Sets the remote OCR engine",
|
||||||
|
),
|
||||||
|
),
|
||||||
|
]
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
# Generated by Django 5.2.16 on 2026-08-10 15:43
|
||||||
|
|
||||||
|
from django.db import migrations
|
||||||
|
from django.db import models
|
||||||
|
|
||||||
|
|
||||||
|
class Migration(migrations.Migration):
|
||||||
|
dependencies = [
|
||||||
|
("paperless", "0014_applicationconfiguration_remote_ocr_api_key_and_more"),
|
||||||
|
]
|
||||||
|
|
||||||
|
operations = [
|
||||||
|
migrations.AddField(
|
||||||
|
model_name="applicationconfiguration",
|
||||||
|
name="remote_ocr_mode",
|
||||||
|
field=models.CharField(
|
||||||
|
blank=True,
|
||||||
|
choices=[
|
||||||
|
("always", "All supported documents"),
|
||||||
|
("workflow_only", "Only when a workflow enables it"),
|
||||||
|
],
|
||||||
|
max_length=32,
|
||||||
|
null=True,
|
||||||
|
verbose_name="Sets which documents are sent to the remote OCR engine",
|
||||||
|
),
|
||||||
|
),
|
||||||
|
]
|
||||||
@@ -74,6 +74,23 @@ class ColorConvertChoices(models.TextChoices):
|
|||||||
CMYK = ("CMYK", _("CMYK"))
|
CMYK = ("CMYK", _("CMYK"))
|
||||||
|
|
||||||
|
|
||||||
|
class RemoteOCREngine(models.TextChoices):
|
||||||
|
"""
|
||||||
|
Matches to PAPERLESS_REMOTE_OCR_ENGINE
|
||||||
|
"""
|
||||||
|
|
||||||
|
AZURE_AI = ("azureai", _("Azure AI Document Intelligence"))
|
||||||
|
|
||||||
|
|
||||||
|
class RemoteOCRMode(models.TextChoices):
|
||||||
|
"""
|
||||||
|
Matches to PAPERLESS_REMOTE_OCR_MODE
|
||||||
|
"""
|
||||||
|
|
||||||
|
ALWAYS = ("always", _("All supported documents"))
|
||||||
|
WORKFLOW_ONLY = ("workflow_only", _("Only when a workflow enables it"))
|
||||||
|
|
||||||
|
|
||||||
class LLMEmbeddingBackend(models.TextChoices):
|
class LLMEmbeddingBackend(models.TextChoices):
|
||||||
OPENAI_LIKE = ("openai-like", _("OpenAI-compatible"))
|
OPENAI_LIKE = ("openai-like", _("OpenAI-compatible"))
|
||||||
HUGGINGFACE = ("huggingface", _("Huggingface"))
|
HUGGINGFACE = ("huggingface", _("Huggingface"))
|
||||||
@@ -286,6 +303,44 @@ class ApplicationConfiguration(AbstractSingletonModel):
|
|||||||
null=True,
|
null=True,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
"""
|
||||||
|
Settings for the remote OCR parser
|
||||||
|
"""
|
||||||
|
|
||||||
|
# PAPERLESS_REMOTE_OCR_ENGINE
|
||||||
|
remote_ocr_engine = models.CharField(
|
||||||
|
verbose_name=_("Sets the remote OCR engine"),
|
||||||
|
blank=True,
|
||||||
|
null=True,
|
||||||
|
max_length=32,
|
||||||
|
choices=RemoteOCREngine.choices,
|
||||||
|
)
|
||||||
|
|
||||||
|
# PAPERLESS_REMOTE_OCR_API_KEY
|
||||||
|
remote_ocr_api_key = models.CharField(
|
||||||
|
verbose_name=_("Sets the remote OCR API key"),
|
||||||
|
blank=True,
|
||||||
|
null=True,
|
||||||
|
max_length=1024,
|
||||||
|
)
|
||||||
|
|
||||||
|
# PAPERLESS_REMOTE_OCR_ENDPOINT
|
||||||
|
remote_ocr_endpoint = models.CharField(
|
||||||
|
verbose_name=_("Sets the remote OCR endpoint"),
|
||||||
|
blank=True,
|
||||||
|
null=True,
|
||||||
|
max_length=256,
|
||||||
|
)
|
||||||
|
|
||||||
|
# PAPERLESS_REMOTE_OCR_MODE
|
||||||
|
remote_ocr_mode = models.CharField(
|
||||||
|
verbose_name=_("Sets which documents are sent to the remote OCR engine"),
|
||||||
|
blank=True,
|
||||||
|
null=True,
|
||||||
|
max_length=32,
|
||||||
|
choices=RemoteOCRMode.choices,
|
||||||
|
)
|
||||||
|
|
||||||
"""
|
"""
|
||||||
AI related settings
|
AI related settings
|
||||||
"""
|
"""
|
||||||
|
|||||||
@@ -134,6 +134,11 @@ class ParserProtocol(Protocol):
|
|||||||
Author or organisation name.
|
Author or organisation name.
|
||||||
url : str
|
url : str
|
||||||
URL for documentation, source code, or issue tracker.
|
URL for documentation, source code, or issue tracker.
|
||||||
|
|
||||||
|
Parsers that send document content to a remote service should additionally
|
||||||
|
set ``uses_remote_service = True`` so the registry can exclude them when
|
||||||
|
remote processing has not been requested for a document. The attribute is
|
||||||
|
optional so a parser that omits it is treated as fully local.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
# ------------------------------------------------------------------
|
# ------------------------------------------------------------------
|
||||||
@@ -145,6 +150,10 @@ class ParserProtocol(Protocol):
|
|||||||
author: str
|
author: str
|
||||||
url: str
|
url: str
|
||||||
|
|
||||||
|
# NOTE: uses_remote_service is not declared here, the registry reads it
|
||||||
|
# with getattr(cls, ..., False) for backwards-compatibility with existing
|
||||||
|
# parsers
|
||||||
|
|
||||||
# ------------------------------------------------------------------
|
# ------------------------------------------------------------------
|
||||||
# Class methods
|
# Class methods
|
||||||
# ------------------------------------------------------------------
|
# ------------------------------------------------------------------
|
||||||
|
|||||||
@@ -334,6 +334,8 @@ class ParserRegistry:
|
|||||||
mime_type: str,
|
mime_type: str,
|
||||||
filename: str,
|
filename: str,
|
||||||
path: Path | None = None,
|
path: Path | None = None,
|
||||||
|
*,
|
||||||
|
allow_remote: bool = True,
|
||||||
) -> type[ParserProtocol] | None:
|
) -> type[ParserProtocol] | None:
|
||||||
"""Return the best parser class for the given file, or None.
|
"""Return the best parser class for the given file, or None.
|
||||||
|
|
||||||
@@ -359,6 +361,11 @@ class ParserRegistry:
|
|||||||
path:
|
path:
|
||||||
Optional filesystem path to the file. Forwarded to each
|
Optional filesystem path to the file. Forwarded to each
|
||||||
parser's score method.
|
parser's score method.
|
||||||
|
allow_remote:
|
||||||
|
When False, parsers that declare ``uses_remote_service = True``
|
||||||
|
are excluded from consideration, so a document is never sent to
|
||||||
|
a remote service. Parsers that do not declare the attribute
|
||||||
|
are treated as local and are always considered.
|
||||||
|
|
||||||
Returns
|
Returns
|
||||||
-------
|
-------
|
||||||
@@ -374,6 +381,13 @@ class ParserRegistry:
|
|||||||
if mime_type not in parser_class.supported_mime_types():
|
if mime_type not in parser_class.supported_mime_types():
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
if not allow_remote and getattr(
|
||||||
|
parser_class,
|
||||||
|
"uses_remote_service",
|
||||||
|
False,
|
||||||
|
):
|
||||||
|
continue
|
||||||
|
|
||||||
score = parser_class.score(mime_type, filename, path)
|
score = parser_class.score(mime_type, filename, path)
|
||||||
if score is None:
|
if score is None:
|
||||||
continue
|
continue
|
||||||
|
|||||||
@@ -3,9 +3,7 @@ Built-in remote-OCR document parser.
|
|||||||
|
|
||||||
Handles documents by sending them to a configured remote OCR engine
|
Handles documents by sending them to a configured remote OCR engine
|
||||||
(currently Azure AI Vision / Document Intelligence) and retrieving both
|
(currently Azure AI Vision / Document Intelligence) and retrieving both
|
||||||
the extracted text and a searchable PDF with an embedded text layer. For
|
the extracted text and a searchable PDF with an embedded text layer.
|
||||||
born-digital PDFs that need no archive copy, the remote call is skipped
|
|
||||||
entirely in favor of locally-extracted text (see ``RemoteDocumentParser.parse``).
|
|
||||||
|
|
||||||
When no engine is configured, ``score()`` returns ``None`` so the parser
|
When no engine is configured, ``score()`` returns ``None`` so the parser
|
||||||
is effectively invisible to the registry — the tesseract parser handles
|
is effectively invisible to the registry — the tesseract parser handles
|
||||||
@@ -24,8 +22,6 @@ from typing import Self
|
|||||||
from django.conf import settings
|
from django.conf import settings
|
||||||
|
|
||||||
from documents.parsers import ParseError
|
from documents.parsers import ParseError
|
||||||
from paperless.parsers.utils import extract_pdf_text
|
|
||||||
from paperless.parsers.utils import post_process_text
|
|
||||||
from paperless.version import __full_version_str__
|
from paperless.version import __full_version_str__
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
@@ -61,6 +57,18 @@ class RemoteEngineConfig:
|
|||||||
self.api_key = api_key
|
self.api_key = api_key
|
||||||
self.endpoint = endpoint
|
self.endpoint = endpoint
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def from_app_config(cls) -> Self:
|
||||||
|
"""Build the config from the app config, falling back to the env."""
|
||||||
|
from paperless.config import RemoteOCRConfig
|
||||||
|
|
||||||
|
app_config = RemoteOCRConfig()
|
||||||
|
return cls(
|
||||||
|
engine=app_config.remote_ocr_engine,
|
||||||
|
api_key=app_config.remote_ocr_api_key,
|
||||||
|
endpoint=app_config.remote_ocr_endpoint,
|
||||||
|
)
|
||||||
|
|
||||||
def engine_is_valid(self) -> bool:
|
def engine_is_valid(self) -> bool:
|
||||||
"""Return True when the engine is known and fully configured."""
|
"""Return True when the engine is known and fully configured."""
|
||||||
return (
|
return (
|
||||||
@@ -74,11 +82,8 @@ class RemoteDocumentParser:
|
|||||||
"""Parse documents via a remote OCR API (currently Azure AI Vision).
|
"""Parse documents via a remote OCR API (currently Azure AI Vision).
|
||||||
|
|
||||||
This parser sends documents to a remote engine that returns both
|
This parser sends documents to a remote engine that returns both
|
||||||
extracted text and a searchable PDF with an embedded text layer,
|
extracted text and a searchable PDF with an embedded text layer.
|
||||||
except when ``parse()`` is called with ``produce_archive=False`` for
|
It does not depend on Tesseract or ocrmypdf.
|
||||||
a PDF, in which case the remote call is skipped and only locally
|
|
||||||
extracted text is returned (no archive). It does not depend on
|
|
||||||
Tesseract or ocrmypdf.
|
|
||||||
|
|
||||||
Class attributes
|
Class attributes
|
||||||
----------------
|
----------------
|
||||||
@@ -90,6 +95,9 @@ class RemoteDocumentParser:
|
|||||||
Maintainer name.
|
Maintainer name.
|
||||||
url : str
|
url : str
|
||||||
Issue tracker / source URL.
|
Issue tracker / source URL.
|
||||||
|
uses_remote_service : bool
|
||||||
|
Content is sent to a remote service, True so that the registry
|
||||||
|
can skip this parser if remote processing was not requested.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
name: str = "Paperless-ngx Remote OCR Parser"
|
name: str = "Paperless-ngx Remote OCR Parser"
|
||||||
@@ -97,6 +105,8 @@ class RemoteDocumentParser:
|
|||||||
author: str = "Paperless-ngx Contributors"
|
author: str = "Paperless-ngx Contributors"
|
||||||
url: str = "https://github.com/paperless-ngx/paperless-ngx"
|
url: str = "https://github.com/paperless-ngx/paperless-ngx"
|
||||||
|
|
||||||
|
uses_remote_service: bool = True
|
||||||
|
|
||||||
# ------------------------------------------------------------------
|
# ------------------------------------------------------------------
|
||||||
# Class methods
|
# Class methods
|
||||||
# ------------------------------------------------------------------
|
# ------------------------------------------------------------------
|
||||||
@@ -145,11 +155,7 @@ class RemoteDocumentParser:
|
|||||||
20 when the remote engine is configured and the MIME type is
|
20 when the remote engine is configured and the MIME type is
|
||||||
supported, otherwise None.
|
supported, otherwise None.
|
||||||
"""
|
"""
|
||||||
config = RemoteEngineConfig(
|
config = RemoteEngineConfig.from_app_config()
|
||||||
engine=settings.REMOTE_OCR_ENGINE,
|
|
||||||
api_key=settings.REMOTE_OCR_API_KEY,
|
|
||||||
endpoint=settings.REMOTE_OCR_ENDPOINT,
|
|
||||||
)
|
|
||||||
if not config.engine_is_valid():
|
if not config.engine_is_valid():
|
||||||
return None
|
return None
|
||||||
if mime_type not in _SUPPORTED_MIME_TYPES:
|
if mime_type not in _SUPPORTED_MIME_TYPES:
|
||||||
@@ -167,11 +173,8 @@ class RemoteDocumentParser:
|
|||||||
Returns
|
Returns
|
||||||
-------
|
-------
|
||||||
bool
|
bool
|
||||||
Always True — the remote engine is capable of returning a PDF
|
Always True — the remote engine always returns a PDF with an
|
||||||
with an embedded text layer to serve as the archive copy.
|
embedded text layer that serves as the archive copy.
|
||||||
Whether it actually does so for a given document depends on
|
|
||||||
``produce_archive`` passed to :meth:`parse` (see there for when
|
|
||||||
the remote engine call, and thus archive generation, is skipped).
|
|
||||||
"""
|
"""
|
||||||
return True
|
return True
|
||||||
|
|
||||||
@@ -228,12 +231,6 @@ class RemoteDocumentParser:
|
|||||||
) -> None:
|
) -> None:
|
||||||
"""Send the document to the remote engine and store results.
|
"""Send the document to the remote engine and store results.
|
||||||
|
|
||||||
When *produce_archive* is False for a PDF, the caller (via
|
|
||||||
``documents.consumer.should_produce_archive``) has already determined
|
|
||||||
that the document is born-digital and needs no archive — skip the
|
|
||||||
remote engine entirely rather than re-OCRing it and creating a
|
|
||||||
duplicate text layer.
|
|
||||||
|
|
||||||
Parameters
|
Parameters
|
||||||
----------
|
----------
|
||||||
document_path:
|
document_path:
|
||||||
@@ -241,14 +238,10 @@ class RemoteDocumentParser:
|
|||||||
mime_type:
|
mime_type:
|
||||||
Detected MIME type of the document.
|
Detected MIME type of the document.
|
||||||
produce_archive:
|
produce_archive:
|
||||||
Whether an archive copy is wanted. For PDFs, False skips the
|
Ignored — the remote engine always returns a searchable PDF,
|
||||||
remote engine and uses locally-extracted text instead.
|
which is stored as the archive copy regardless of this flag.
|
||||||
"""
|
"""
|
||||||
config = RemoteEngineConfig(
|
config = RemoteEngineConfig.from_app_config()
|
||||||
engine=settings.REMOTE_OCR_ENGINE,
|
|
||||||
api_key=settings.REMOTE_OCR_API_KEY,
|
|
||||||
endpoint=settings.REMOTE_OCR_ENDPOINT,
|
|
||||||
)
|
|
||||||
|
|
||||||
if not config.engine_is_valid():
|
if not config.engine_is_valid():
|
||||||
logger.warning(
|
logger.warning(
|
||||||
@@ -257,16 +250,6 @@ class RemoteDocumentParser:
|
|||||||
self._text = ""
|
self._text = ""
|
||||||
return
|
return
|
||||||
|
|
||||||
if not produce_archive and mime_type == "application/pdf":
|
|
||||||
logger.debug(
|
|
||||||
"Remote OCR: skipped — no archive requested, "
|
|
||||||
"using locally-extracted text",
|
|
||||||
)
|
|
||||||
self._text = (
|
|
||||||
post_process_text(extract_pdf_text(document_path, log=logger)) or ""
|
|
||||||
)
|
|
||||||
return
|
|
||||||
|
|
||||||
if config.engine == "azureai":
|
if config.engine == "azureai":
|
||||||
self._text = self._azure_ai_vision_parse(document_path, config)
|
self._text = self._azure_ai_vision_parse(document_path, config)
|
||||||
|
|
||||||
|
|||||||
@@ -219,6 +219,13 @@ class ApplicationConfigurationSerializer(
|
|||||||
allow_null=True,
|
allow_null=True,
|
||||||
max_length=1024,
|
max_length=1024,
|
||||||
)
|
)
|
||||||
|
remote_ocr_api_key = ObfuscatedPasswordField(
|
||||||
|
required=False,
|
||||||
|
allow_null=True,
|
||||||
|
max_length=1024,
|
||||||
|
)
|
||||||
|
|
||||||
|
OBFUSCATED_FIELDS = ("llm_api_key", "remote_ocr_api_key")
|
||||||
|
|
||||||
def run_validation(self, data):
|
def run_validation(self, data):
|
||||||
# Empty strings treated as None to avoid unexpected behavior
|
# Empty strings treated as None to avoid unexpected behavior
|
||||||
@@ -230,11 +237,13 @@ class ApplicationConfigurationSerializer(
|
|||||||
data["language"] = None
|
data["language"] = None
|
||||||
if "llm_output_language" in data and data["llm_output_language"] == "":
|
if "llm_output_language" in data and data["llm_output_language"] == "":
|
||||||
data["llm_output_language"] = None
|
data["llm_output_language"] = None
|
||||||
if "llm_api_key" in data and data["llm_api_key"] is not None:
|
for field in self.OBFUSCATED_FIELDS:
|
||||||
if data["llm_api_key"] == "":
|
if field in data and data[field] is not None:
|
||||||
data["llm_api_key"] = None
|
if data[field] == "":
|
||||||
elif len(data["llm_api_key"].replace("*", "")) == 0:
|
data[field] = None
|
||||||
del data["llm_api_key"]
|
# Not a real value, don't overwrite the stored one
|
||||||
|
elif len(data[field].replace("*", "")) == 0:
|
||||||
|
del data[field]
|
||||||
return super().run_validation(data)
|
return super().run_validation(data)
|
||||||
|
|
||||||
def update(self, instance, validated_data):
|
def update(self, instance, validated_data):
|
||||||
|
|||||||
@@ -1197,6 +1197,7 @@ WEBHOOKS_ALLOW_INTERNAL_REQUESTS = get_bool_from_env(
|
|||||||
REMOTE_OCR_ENGINE = os.getenv("PAPERLESS_REMOTE_OCR_ENGINE")
|
REMOTE_OCR_ENGINE = os.getenv("PAPERLESS_REMOTE_OCR_ENGINE")
|
||||||
REMOTE_OCR_API_KEY = os.getenv("PAPERLESS_REMOTE_OCR_API_KEY")
|
REMOTE_OCR_API_KEY = os.getenv("PAPERLESS_REMOTE_OCR_API_KEY")
|
||||||
REMOTE_OCR_ENDPOINT = os.getenv("PAPERLESS_REMOTE_OCR_ENDPOINT")
|
REMOTE_OCR_ENDPOINT = os.getenv("PAPERLESS_REMOTE_OCR_ENDPOINT")
|
||||||
|
REMOTE_OCR_MODE = os.getenv("PAPERLESS_REMOTE_OCR_MODE", "always")
|
||||||
|
|
||||||
################################################################################
|
################################################################################
|
||||||
# AI Settings #
|
# AI Settings #
|
||||||
|
|||||||
@@ -21,6 +21,7 @@ from unittest.mock import Mock
|
|||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from documents.parsers import ParseError
|
from documents.parsers import ParseError
|
||||||
|
from paperless.models import ApplicationConfiguration
|
||||||
from paperless.parsers import ParserContext
|
from paperless.parsers import ParserContext
|
||||||
from paperless.parsers import ParserProtocol
|
from paperless.parsers import ParserProtocol
|
||||||
from paperless.parsers.remote import RemoteDocumentParser
|
from paperless.parsers.remote import RemoteDocumentParser
|
||||||
@@ -33,6 +34,10 @@ if TYPE_CHECKING:
|
|||||||
from pytest_mock import MockerFixture
|
from pytest_mock import MockerFixture
|
||||||
|
|
||||||
|
|
||||||
|
# Remote ocr config from ApplicationConfiguration needs DB access
|
||||||
|
pytestmark = pytest.mark.django_db
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Module-local fixtures
|
# Module-local fixtures
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -227,6 +232,18 @@ class TestRemoteParserScore:
|
|||||||
score = RemoteDocumentParser.score("application/pdf", "doc.pdf")
|
score = RemoteDocumentParser.score("application/pdf", "doc.pdf")
|
||||||
assert score is not None and score > 10
|
assert score is not None and score > 10
|
||||||
|
|
||||||
|
@pytest.mark.usefixtures("no_engine_settings")
|
||||||
|
def test_score_uses_app_config_when_env_unset(self) -> None:
|
||||||
|
"""The app config alone is enough to activate the parser."""
|
||||||
|
config = ApplicationConfiguration.objects.first()
|
||||||
|
assert config is not None
|
||||||
|
config.remote_ocr_engine = "azureai"
|
||||||
|
config.remote_ocr_api_key = "app-config-key"
|
||||||
|
config.remote_ocr_endpoint = "https://config.cognitiveservices.azure.com"
|
||||||
|
config.save()
|
||||||
|
|
||||||
|
assert RemoteDocumentParser.score("application/pdf", "doc.pdf") == 20
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Properties
|
# Properties
|
||||||
@@ -337,117 +354,6 @@ class TestRemoteParserParse:
|
|||||||
assert remote_parser.get_date() is None
|
assert remote_parser.get_date() is None
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# parse() — produce_archive=False skips the remote engine (PDFs only)
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
class TestRemoteParserSkipsWhenNoArchiveWanted:
|
|
||||||
"""When the caller has already decided no archive is needed for a PDF
|
|
||||||
(documents.consumer.should_produce_archive), the remote engine call is
|
|
||||||
skipped entirely in favor of locally-extracted text.
|
|
||||||
"""
|
|
||||||
|
|
||||||
def test_pdf_skips_azure_when_no_archive_requested(
|
|
||||||
self,
|
|
||||||
remote_parser: RemoteDocumentParser,
|
|
||||||
simple_digital_pdf_file: Path,
|
|
||||||
azure_client: Mock,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN: produce_archive=False for a PDF
|
|
||||||
WHEN: parse() is called
|
|
||||||
THEN: Azure is never invoked, no archive is produced, and text
|
|
||||||
comes from local pdftotext extraction
|
|
||||||
"""
|
|
||||||
remote_parser.parse(
|
|
||||||
simple_digital_pdf_file,
|
|
||||||
"application/pdf",
|
|
||||||
produce_archive=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
azure_client.begin_analyze_document.assert_not_called()
|
|
||||||
assert remote_parser.get_archive_path() is None
|
|
||||||
assert remote_parser.get_text() != ""
|
|
||||||
|
|
||||||
def test_pdf_no_archive_requested_text_matches_local_extraction(
|
|
||||||
self,
|
|
||||||
remote_parser: RemoteDocumentParser,
|
|
||||||
simple_digital_pdf_file: Path,
|
|
||||||
azure_client: Mock,
|
|
||||||
mocker: MockerFixture,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN: produce_archive=False for a PDF
|
|
||||||
WHEN: parse() is called
|
|
||||||
THEN: the returned text is exactly the locally-extracted text,
|
|
||||||
not anything from the (unused) Azure mock
|
|
||||||
"""
|
|
||||||
mocker.patch(
|
|
||||||
"paperless.parsers.remote.extract_pdf_text",
|
|
||||||
return_value="Local digital text.",
|
|
||||||
)
|
|
||||||
|
|
||||||
remote_parser.parse(
|
|
||||||
simple_digital_pdf_file,
|
|
||||||
"application/pdf",
|
|
||||||
produce_archive=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
assert remote_parser.get_text() == "Local digital text."
|
|
||||||
|
|
||||||
def test_pdf_no_archive_requested_closes_no_client(
|
|
||||||
self,
|
|
||||||
remote_parser: RemoteDocumentParser,
|
|
||||||
simple_digital_pdf_file: Path,
|
|
||||||
azure_client: Mock,
|
|
||||||
) -> None:
|
|
||||||
remote_parser.parse(
|
|
||||||
simple_digital_pdf_file,
|
|
||||||
"application/pdf",
|
|
||||||
produce_archive=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
azure_client.close.assert_not_called()
|
|
||||||
|
|
||||||
def test_non_pdf_still_calls_azure_when_no_archive_requested(
|
|
||||||
self,
|
|
||||||
remote_parser: RemoteDocumentParser,
|
|
||||||
simple_digital_pdf_file: Path,
|
|
||||||
azure_client: Mock,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
Images have no local-text fallback, so produce_archive=False does
|
|
||||||
not skip the remote engine for non-PDF MIME types.
|
|
||||||
"""
|
|
||||||
remote_parser.parse(
|
|
||||||
simple_digital_pdf_file,
|
|
||||||
"image/png",
|
|
||||||
produce_archive=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
azure_client.begin_analyze_document.assert_called_once()
|
|
||||||
assert remote_parser.get_text() == _DEFAULT_TEXT
|
|
||||||
|
|
||||||
@pytest.mark.usefixtures("no_engine_settings")
|
|
||||||
def test_unconfigured_engine_takes_precedence_over_skip(
|
|
||||||
self,
|
|
||||||
remote_parser: RemoteDocumentParser,
|
|
||||||
simple_digital_pdf_file: Path,
|
|
||||||
) -> None:
|
|
||||||
"""An unconfigured engine still short-circuits before the
|
|
||||||
produce_archive check, returning empty text as before.
|
|
||||||
"""
|
|
||||||
remote_parser.parse(
|
|
||||||
simple_digital_pdf_file,
|
|
||||||
"application/pdf",
|
|
||||||
produce_archive=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
assert remote_parser.get_text() == ""
|
|
||||||
assert remote_parser.get_archive_path() is None
|
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# parse() — Azure failure path
|
# parse() — Azure failure path
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|||||||
@@ -1277,6 +1277,8 @@ class TestParserFileTypes:
|
|||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
# Remote ocr config from ApplicationConfiguration needs DB access
|
||||||
|
@pytest.mark.django_db
|
||||||
class TestRasterisedDocumentParserRegistry:
|
class TestRasterisedDocumentParserRegistry:
|
||||||
def test_registered_in_defaults(self) -> None:
|
def test_registered_in_defaults(self) -> None:
|
||||||
from paperless.parsers.registry import ParserRegistry
|
from paperless.parsers.registry import ParserRegistry
|
||||||
|
|||||||
@@ -655,6 +655,23 @@ class TestRemoteParserChecks:
|
|||||||
in msg.msg
|
in msg.msg
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def test_valid_mode(self, settings: SettingsWrapper) -> None:
|
||||||
|
settings.REMOTE_OCR_ENGINE = None
|
||||||
|
settings.REMOTE_OCR_MODE = "workflow_only"
|
||||||
|
|
||||||
|
msgs = check_remote_parser_configured(None)
|
||||||
|
|
||||||
|
assert len(msgs) == 0
|
||||||
|
|
||||||
|
def test_invalid_mode(self, settings: SettingsWrapper) -> None:
|
||||||
|
settings.REMOTE_OCR_ENGINE = None
|
||||||
|
settings.REMOTE_OCR_MODE = "sometimes"
|
||||||
|
|
||||||
|
msgs = check_remote_parser_configured(None)
|
||||||
|
|
||||||
|
assert len(msgs) == 1
|
||||||
|
assert "PAPERLESS_REMOTE_OCR_MODE is set to 'sometimes'" in msgs[0].msg
|
||||||
|
|
||||||
|
|
||||||
class TestTesseractChecks:
|
class TestTesseractChecks:
|
||||||
def test_default_language(self) -> None:
|
def test_default_language(self) -> None:
|
||||||
|
|||||||
@@ -468,6 +468,124 @@ class TestParserRegistryGetParserForFile:
|
|||||||
assert result is AcceptingBuiltin
|
assert result is AcceptingBuiltin
|
||||||
|
|
||||||
|
|
||||||
|
class TestParserRegistryRemoteParsers:
|
||||||
|
"""Verify the allow_remote filter in ParserRegistry.get_parser_for_file()."""
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _remote_parser_cls() -> type:
|
||||||
|
class RemoteParser:
|
||||||
|
name = "remote"
|
||||||
|
version = "1.0"
|
||||||
|
author = "A"
|
||||||
|
url = "https://example.com/remote"
|
||||||
|
uses_remote_service = True
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def supported_mime_types(cls):
|
||||||
|
return {"text/plain": ".txt"}
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def score(cls, mime_type, filename, path=None):
|
||||||
|
return 20
|
||||||
|
|
||||||
|
return RemoteParser
|
||||||
|
|
||||||
|
def test_remote_parser_wins_when_remote_allowed(
|
||||||
|
self,
|
||||||
|
dummy_parser_cls: type,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN: A remote parser scoring 20 and a local parser scoring 10.
|
||||||
|
WHEN: get_parser_for_file() is called with allow_remote=True.
|
||||||
|
THEN: The remote parser is returned.
|
||||||
|
"""
|
||||||
|
remote_parser_cls = self._remote_parser_cls()
|
||||||
|
registry = ParserRegistry()
|
||||||
|
registry.register_builtin(dummy_parser_cls)
|
||||||
|
registry.register_builtin(remote_parser_cls)
|
||||||
|
|
||||||
|
result = registry.get_parser_for_file(
|
||||||
|
"text/plain",
|
||||||
|
"readme.txt",
|
||||||
|
allow_remote=True,
|
||||||
|
)
|
||||||
|
assert result is remote_parser_cls
|
||||||
|
|
||||||
|
def test_remote_parser_skipped_when_remote_not_allowed(
|
||||||
|
self,
|
||||||
|
dummy_parser_cls: type,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN: A remote parser scoring 20 and a local parser scoring 10.
|
||||||
|
WHEN: get_parser_for_file() is called with allow_remote=False.
|
||||||
|
THEN: The local parser is returned despite its lower score.
|
||||||
|
"""
|
||||||
|
registry = ParserRegistry()
|
||||||
|
registry.register_builtin(dummy_parser_cls)
|
||||||
|
registry.register_builtin(self._remote_parser_cls())
|
||||||
|
|
||||||
|
result = registry.get_parser_for_file(
|
||||||
|
"text/plain",
|
||||||
|
"readme.txt",
|
||||||
|
allow_remote=False,
|
||||||
|
)
|
||||||
|
assert result is dummy_parser_cls
|
||||||
|
|
||||||
|
def test_no_parser_when_only_remote_available_and_not_allowed(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN: A registry whose only candidate declares uses_remote_service.
|
||||||
|
WHEN: get_parser_for_file() is called with allow_remote=False.
|
||||||
|
THEN: None is returned — the remote parser is never used as a
|
||||||
|
fallback when remote processing was not requested.
|
||||||
|
"""
|
||||||
|
registry = ParserRegistry()
|
||||||
|
registry.register_builtin(self._remote_parser_cls())
|
||||||
|
|
||||||
|
result = registry.get_parser_for_file(
|
||||||
|
"text/plain",
|
||||||
|
"readme.txt",
|
||||||
|
allow_remote=False,
|
||||||
|
)
|
||||||
|
assert result is None
|
||||||
|
|
||||||
|
def test_parser_without_attribute_treated_as_local(
|
||||||
|
self,
|
||||||
|
dummy_parser_cls: type,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN: A third-party parser predating uses_remote_service, so it does
|
||||||
|
not declare the attribute at all.
|
||||||
|
WHEN: get_parser_for_file() is called with allow_remote=False.
|
||||||
|
THEN: It is still considered, i.e. treated as fully local, rather
|
||||||
|
than raising AttributeError.
|
||||||
|
"""
|
||||||
|
assert not hasattr(dummy_parser_cls, "uses_remote_service")
|
||||||
|
|
||||||
|
registry = ParserRegistry()
|
||||||
|
registry.register_builtin(dummy_parser_cls)
|
||||||
|
|
||||||
|
result = registry.get_parser_for_file(
|
||||||
|
"text/plain",
|
||||||
|
"readme.txt",
|
||||||
|
allow_remote=False,
|
||||||
|
)
|
||||||
|
assert result is dummy_parser_cls
|
||||||
|
|
||||||
|
def test_remote_allowed_by_default(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN: A registry containing only a remote parser.
|
||||||
|
WHEN: get_parser_for_file() is called without allow_remote.
|
||||||
|
THEN: The remote parser is returned — callers that do not opt in to
|
||||||
|
the filter keep the previous behaviour.
|
||||||
|
"""
|
||||||
|
remote_parser_cls = self._remote_parser_cls()
|
||||||
|
registry = ParserRegistry()
|
||||||
|
registry.register_builtin(remote_parser_cls)
|
||||||
|
|
||||||
|
result = registry.get_parser_for_file("text/plain", "readme.txt")
|
||||||
|
assert result is remote_parser_cls
|
||||||
|
|
||||||
|
|
||||||
class TestDiscover:
|
class TestDiscover:
|
||||||
"""Verify entrypoint discovery in ParserRegistry.discover()."""
|
"""Verify entrypoint discovery in ParserRegistry.discover()."""
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,113 @@
|
|||||||
|
"""Tests for RemoteOCRConfig precedence between app config and Django settings."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from django.test import override_settings
|
||||||
|
|
||||||
|
from paperless.config import RemoteOCRConfig
|
||||||
|
from paperless.models import RemoteOCRMode
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from unittest.mock import MagicMock
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture()
|
||||||
|
def null_app_config(mocker) -> MagicMock:
|
||||||
|
"""Mock ApplicationConfiguration with all fields None → falls back to Django settings."""
|
||||||
|
return mocker.MagicMock(
|
||||||
|
remote_ocr_engine=None,
|
||||||
|
remote_ocr_api_key=None,
|
||||||
|
remote_ocr_endpoint=None,
|
||||||
|
remote_ocr_mode=None,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture()
|
||||||
|
def make_remote_ocr_config(mocker):
|
||||||
|
def _make(app_config, **django_settings_overrides):
|
||||||
|
mocker.patch(
|
||||||
|
"paperless.config.BaseConfig._get_config_instance",
|
||||||
|
return_value=app_config,
|
||||||
|
)
|
||||||
|
with override_settings(**django_settings_overrides):
|
||||||
|
return RemoteOCRConfig()
|
||||||
|
|
||||||
|
return _make
|
||||||
|
|
||||||
|
|
||||||
|
class TestRemoteOCRConfig:
|
||||||
|
def test_falls_back_to_settings(
|
||||||
|
self,
|
||||||
|
make_remote_ocr_config,
|
||||||
|
null_app_config,
|
||||||
|
) -> None:
|
||||||
|
cfg = make_remote_ocr_config(
|
||||||
|
null_app_config,
|
||||||
|
REMOTE_OCR_ENGINE="azureai",
|
||||||
|
REMOTE_OCR_API_KEY="env-key",
|
||||||
|
REMOTE_OCR_ENDPOINT="https://env.cognitiveservices.azure.com",
|
||||||
|
REMOTE_OCR_MODE=RemoteOCRMode.WORKFLOW_ONLY,
|
||||||
|
)
|
||||||
|
assert cfg.remote_ocr_engine == "azureai"
|
||||||
|
assert cfg.remote_ocr_api_key == "env-key"
|
||||||
|
assert cfg.remote_ocr_endpoint == "https://env.cognitiveservices.azure.com"
|
||||||
|
assert cfg.remote_ocr_mode == RemoteOCRMode.WORKFLOW_ONLY
|
||||||
|
|
||||||
|
def test_app_config_takes_precedence(
|
||||||
|
self,
|
||||||
|
make_remote_ocr_config,
|
||||||
|
mocker,
|
||||||
|
) -> None:
|
||||||
|
app_config = mocker.MagicMock(
|
||||||
|
remote_ocr_engine="azureai",
|
||||||
|
remote_ocr_api_key="db-key",
|
||||||
|
remote_ocr_endpoint="https://db.cognitiveservices.azure.com",
|
||||||
|
remote_ocr_mode=RemoteOCRMode.WORKFLOW_ONLY,
|
||||||
|
)
|
||||||
|
cfg = make_remote_ocr_config(
|
||||||
|
app_config,
|
||||||
|
REMOTE_OCR_ENGINE=None,
|
||||||
|
REMOTE_OCR_API_KEY="env-key",
|
||||||
|
REMOTE_OCR_ENDPOINT="https://env.cognitiveservices.azure.com",
|
||||||
|
REMOTE_OCR_MODE=RemoteOCRMode.ALWAYS,
|
||||||
|
)
|
||||||
|
assert cfg.remote_ocr_engine == "azureai"
|
||||||
|
assert cfg.remote_ocr_api_key == "db-key"
|
||||||
|
assert cfg.remote_ocr_endpoint == "https://db.cognitiveservices.azure.com"
|
||||||
|
assert cfg.remote_ocr_mode == RemoteOCRMode.WORKFLOW_ONLY
|
||||||
|
|
||||||
|
def test_unset_everywhere(
|
||||||
|
self,
|
||||||
|
make_remote_ocr_config,
|
||||||
|
null_app_config,
|
||||||
|
) -> None:
|
||||||
|
cfg = make_remote_ocr_config(
|
||||||
|
null_app_config,
|
||||||
|
REMOTE_OCR_ENGINE=None,
|
||||||
|
REMOTE_OCR_API_KEY=None,
|
||||||
|
REMOTE_OCR_ENDPOINT=None,
|
||||||
|
)
|
||||||
|
assert cfg.remote_ocr_engine is None
|
||||||
|
assert cfg.remote_ocr_api_key is None
|
||||||
|
assert cfg.remote_ocr_endpoint is None
|
||||||
|
|
||||||
|
|
||||||
|
class TestRemoteOCRByDefault:
|
||||||
|
def test_always_mode(self, make_remote_ocr_config, null_app_config) -> None:
|
||||||
|
cfg = make_remote_ocr_config(
|
||||||
|
null_app_config,
|
||||||
|
REMOTE_OCR_MODE=RemoteOCRMode.ALWAYS,
|
||||||
|
)
|
||||||
|
|
||||||
|
assert cfg.remote_ocr_by_default is True
|
||||||
|
|
||||||
|
def test_workflow_only_mode(self, make_remote_ocr_config, null_app_config) -> None:
|
||||||
|
cfg = make_remote_ocr_config(
|
||||||
|
null_app_config,
|
||||||
|
REMOTE_OCR_MODE=RemoteOCRMode.WORKFLOW_ONLY,
|
||||||
|
)
|
||||||
|
|
||||||
|
assert cfg.remote_ocr_by_default is False
|
||||||
@@ -23,6 +23,22 @@ def get_language_name(language_code: str) -> str:
|
|||||||
return language_code
|
return language_code
|
||||||
|
|
||||||
|
|
||||||
|
def get_llm_output_language(ai_config: AIConfig, user: User | None) -> str | None:
|
||||||
|
"""
|
||||||
|
Language to localize LLM output into: the configured language, falling back
|
||||||
|
to the user's own UI language when unset.
|
||||||
|
"""
|
||||||
|
output_language = ai_config.llm_output_language
|
||||||
|
if (
|
||||||
|
not output_language
|
||||||
|
and user is not None
|
||||||
|
and hasattr(user, "ui_settings")
|
||||||
|
and isinstance(user.ui_settings.settings, dict)
|
||||||
|
):
|
||||||
|
output_language = user.ui_settings.settings.get("language")
|
||||||
|
return output_language
|
||||||
|
|
||||||
|
|
||||||
def build_prompt_without_rag(
|
def build_prompt_without_rag(
|
||||||
document: Document,
|
document: Document,
|
||||||
config: AIConfig,
|
config: AIConfig,
|
||||||
|
|||||||
@@ -2,8 +2,6 @@ import json
|
|||||||
import logging
|
import logging
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
from django.db.models import QuerySet
|
|
||||||
|
|
||||||
from documents.models import Document
|
from documents.models import Document
|
||||||
from paperless.config import AIConfig
|
from paperless.config import AIConfig
|
||||||
from paperless_ai.client import AIClient
|
from paperless_ai.client import AIClient
|
||||||
@@ -84,21 +82,10 @@ def _build_document_reference(
|
|||||||
|
|
||||||
|
|
||||||
def _get_document_references(
|
def _get_document_references(
|
||||||
documents: QuerySet[Document],
|
documents: list[Document],
|
||||||
top_nodes: list,
|
top_nodes: list,
|
||||||
) -> list[dict[str, int | str]]:
|
) -> list[dict[str, int | str]]:
|
||||||
candidate_ids: set[int] = set()
|
allowed_documents = {doc.pk: doc for doc in documents}
|
||||||
for node in top_nodes:
|
|
||||||
try:
|
|
||||||
candidate_ids.add(int(node.metadata["document_id"]))
|
|
||||||
except (KeyError, TypeError, ValueError): # pragma: no cover
|
|
||||||
continue
|
|
||||||
|
|
||||||
if not candidate_ids:
|
|
||||||
return []
|
|
||||||
|
|
||||||
allowed_documents = {doc.pk: doc for doc in documents.filter(pk__in=candidate_ids)}
|
|
||||||
|
|
||||||
references: list[dict[str, int | str]] = []
|
references: list[dict[str, int | str]] = []
|
||||||
seen_document_ids: set[int] = set()
|
seen_document_ids: set[int] = set()
|
||||||
|
|
||||||
@@ -132,7 +119,7 @@ def _format_chat_metadata_trailer(references: list[dict[str, int | str]]) -> str
|
|||||||
|
|
||||||
def stream_chat_with_documents(
|
def stream_chat_with_documents(
|
||||||
query_str: str,
|
query_str: str,
|
||||||
documents: QuerySet[Document],
|
documents: list[Document],
|
||||||
output_language: str | None = None,
|
output_language: str | None = None,
|
||||||
):
|
):
|
||||||
try:
|
try:
|
||||||
@@ -148,10 +135,10 @@ def stream_chat_with_documents(
|
|||||||
|
|
||||||
def _stream_chat_with_documents(
|
def _stream_chat_with_documents(
|
||||||
query_str: str,
|
query_str: str,
|
||||||
documents: QuerySet[Document],
|
documents: list[Document],
|
||||||
output_language: str | None = None,
|
output_language: str | None = None,
|
||||||
):
|
):
|
||||||
if not documents.exists():
|
if not documents:
|
||||||
yield CHAT_NO_CONTENT_MESSAGE
|
yield CHAT_NO_CONTENT_MESSAGE
|
||||||
return
|
return
|
||||||
|
|
||||||
@@ -161,9 +148,7 @@ def _stream_chat_with_documents(
|
|||||||
from llama_index.core.retrievers import VectorIndexRetriever
|
from llama_index.core.retrievers import VectorIndexRetriever
|
||||||
|
|
||||||
config = AIConfig()
|
config = AIConfig()
|
||||||
filters = _document_id_filters(
|
filters = _document_id_filters(str(doc.pk) for doc in documents)
|
||||||
str(pk) for pk in documents.values_list("pk", flat=True)
|
|
||||||
)
|
|
||||||
|
|
||||||
# Hold the shared read lock for the whole operation: the query engine
|
# Hold the shared read lock for the whole operation: the query engine
|
||||||
# retrieves from the vector store again during synthesis, so the connection
|
# retrieves from the vector store again during synthesis, so the connection
|
||||||
|
|||||||
@@ -8,45 +8,48 @@ from documents.models import Correspondent
|
|||||||
from documents.models import DocumentType
|
from documents.models import DocumentType
|
||||||
from documents.models import StoragePath
|
from documents.models import StoragePath
|
||||||
from documents.models import Tag
|
from documents.models import Tag
|
||||||
from documents.permissions import get_objects_for_user_owner_aware
|
from documents.permissions import permitted_object_ids
|
||||||
|
|
||||||
MATCH_THRESHOLD = 0.8
|
MATCH_THRESHOLD = 0.8
|
||||||
|
|
||||||
logger = logging.getLogger("paperless_ai.matching")
|
logger = logging.getLogger("paperless_ai.matching")
|
||||||
|
|
||||||
|
# Note: with a None user, e.g. a workflow acting on an unowned document,
|
||||||
|
# permitted_object_ids returns unowned objects only, so it won't return
|
||||||
|
# someone's private tag.
|
||||||
|
|
||||||
def match_tags_by_name(names: list[str], user: User) -> list[Tag]:
|
|
||||||
queryset = get_objects_for_user_owner_aware(
|
def match_tags_by_name(names: list[str], user: User | None) -> list[Tag]:
|
||||||
user,
|
queryset = Tag.objects.filter(id__in=permitted_object_ids(user, Tag, "view_tag"))
|
||||||
["view_tag"],
|
return _match_names_to_queryset(names, queryset, "name")
|
||||||
Tag,
|
|
||||||
|
|
||||||
|
def match_correspondents_by_name(
|
||||||
|
names: list[str],
|
||||||
|
user: User | None,
|
||||||
|
) -> list[Correspondent]:
|
||||||
|
queryset = Correspondent.objects.filter(
|
||||||
|
id__in=permitted_object_ids(user, Correspondent, "view_correspondent"),
|
||||||
)
|
)
|
||||||
return _match_names_to_queryset(names, queryset, "name")
|
return _match_names_to_queryset(names, queryset, "name")
|
||||||
|
|
||||||
|
|
||||||
def match_correspondents_by_name(names: list[str], user: User) -> list[Correspondent]:
|
def match_document_types_by_name(
|
||||||
queryset = get_objects_for_user_owner_aware(
|
names: list[str],
|
||||||
user,
|
user: User | None,
|
||||||
["view_correspondent"],
|
) -> list[DocumentType]:
|
||||||
Correspondent,
|
queryset = DocumentType.objects.filter(
|
||||||
|
id__in=permitted_object_ids(user, DocumentType, "view_documenttype"),
|
||||||
)
|
)
|
||||||
return _match_names_to_queryset(names, queryset, "name")
|
return _match_names_to_queryset(names, queryset, "name")
|
||||||
|
|
||||||
|
|
||||||
def match_document_types_by_name(names: list[str], user: User) -> list[DocumentType]:
|
def match_storage_paths_by_name(
|
||||||
queryset = get_objects_for_user_owner_aware(
|
names: list[str],
|
||||||
user,
|
user: User | None,
|
||||||
["view_documenttype"],
|
) -> list[StoragePath]:
|
||||||
DocumentType,
|
queryset = StoragePath.objects.filter(
|
||||||
)
|
id__in=permitted_object_ids(user, StoragePath, "view_storagepath"),
|
||||||
return _match_names_to_queryset(names, queryset, "name")
|
|
||||||
|
|
||||||
|
|
||||||
def match_storage_paths_by_name(names: list[str], user: User) -> list[StoragePath]:
|
|
||||||
queryset = get_objects_for_user_owner_aware(
|
|
||||||
user,
|
|
||||||
["view_storagepath"],
|
|
||||||
StoragePath,
|
|
||||||
)
|
)
|
||||||
return _match_names_to_queryset(names, queryset, "name")
|
return _match_names_to_queryset(names, queryset, "name")
|
||||||
|
|
||||||
|
|||||||
@@ -3,12 +3,10 @@ from unittest.mock import MagicMock
|
|||||||
from unittest.mock import patch
|
from unittest.mock import patch
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
from django.db.models.signals import post_init
|
|
||||||
from llama_index.core import settings as llama_settings
|
from llama_index.core import settings as llama_settings
|
||||||
from llama_index.core.embeddings.mock_embed_model import MockEmbedding
|
from llama_index.core.embeddings.mock_embed_model import MockEmbedding
|
||||||
from llama_index.core.schema import TextNode
|
from llama_index.core.schema import TextNode
|
||||||
|
|
||||||
from documents.models import Document
|
|
||||||
from documents.tests.factories import DocumentFactory
|
from documents.tests.factories import DocumentFactory
|
||||||
from paperless_ai import chat
|
from paperless_ai import chat
|
||||||
from paperless_ai import indexing
|
from paperless_ai import indexing
|
||||||
@@ -38,6 +36,16 @@ def patch_embed_nodes():
|
|||||||
yield mock_embed_nodes
|
yield mock_embed_nodes
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def mock_document():
|
||||||
|
doc = MagicMock()
|
||||||
|
doc.pk = 1
|
||||||
|
doc.title = "Test Document"
|
||||||
|
doc.filename = "test_file.pdf"
|
||||||
|
doc.content = "This is the document content."
|
||||||
|
return doc
|
||||||
|
|
||||||
|
|
||||||
def assert_chat_output(
|
def assert_chat_output(
|
||||||
output: list[str],
|
output: list[str],
|
||||||
*,
|
*,
|
||||||
@@ -53,13 +61,6 @@ def assert_chat_output(
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
def _fake_documents_queryset(pks: list[int]) -> MagicMock:
|
|
||||||
qs = MagicMock()
|
|
||||||
qs.exists.return_value = bool(pks)
|
|
||||||
qs.values_list.return_value = pks
|
|
||||||
return qs
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
@pytest.mark.parametrize(
|
||||||
("output_language", "expected_language_line"),
|
("output_language", "expected_language_line"),
|
||||||
[
|
[
|
||||||
@@ -106,10 +107,9 @@ def test_build_refine_prompt(
|
|||||||
|
|
||||||
@pytest.mark.django_db
|
@pytest.mark.django_db
|
||||||
def test_stream_chat_with_one_document_retrieval(
|
def test_stream_chat_with_one_document_retrieval(
|
||||||
|
mock_document,
|
||||||
patch_embed_nodes,
|
patch_embed_nodes,
|
||||||
) -> None:
|
) -> None:
|
||||||
document = DocumentFactory.create(title="Test Document", content="ignored")
|
|
||||||
documents = Document.objects.filter(pk=document.pk)
|
|
||||||
with (
|
with (
|
||||||
patch("paperless_ai.chat.AIClient") as mock_client_cls,
|
patch("paperless_ai.chat.AIClient") as mock_client_cls,
|
||||||
patch("paperless_ai.chat.load_or_build_index") as mock_load_index,
|
patch("paperless_ai.chat.load_or_build_index") as mock_load_index,
|
||||||
@@ -124,19 +124,22 @@ def test_stream_chat_with_one_document_retrieval(
|
|||||||
mock_client_cls.return_value = mock_client
|
mock_client_cls.return_value = mock_client
|
||||||
mock_client.llm = MagicMock()
|
mock_client.llm = MagicMock()
|
||||||
|
|
||||||
|
mock_node = TextNode(
|
||||||
|
text="This is node content.",
|
||||||
|
metadata={"document_id": str(mock_document.pk), "title": "Test Document"},
|
||||||
|
)
|
||||||
mock_index = MagicMock()
|
mock_index = MagicMock()
|
||||||
mock_index.vector_store.get_nodes.return_value = [
|
# Simulate get_nodes returning nodes (content exists)
|
||||||
TextNode(
|
mock_index.vector_store.get_nodes.return_value = [mock_node]
|
||||||
text="This is node content.",
|
|
||||||
metadata={"document_id": str(document.pk), "title": "Test Document"},
|
|
||||||
),
|
|
||||||
]
|
|
||||||
mock_load_index.return_value = mock_index
|
mock_load_index.return_value = mock_index
|
||||||
|
|
||||||
mock_retriever_instance = MagicMock()
|
mock_retriever_instance = MagicMock()
|
||||||
mock_retriever_instance.retrieve.return_value = [
|
mock_retriever_instance.retrieve.return_value = [
|
||||||
MagicMock(
|
MagicMock(
|
||||||
metadata={"document_id": str(document.pk), "title": "Test Document"},
|
metadata={
|
||||||
|
"document_id": str(mock_document.pk),
|
||||||
|
"title": "Test Document",
|
||||||
|
},
|
||||||
),
|
),
|
||||||
]
|
]
|
||||||
|
|
||||||
@@ -150,7 +153,7 @@ def test_stream_chat_with_one_document_retrieval(
|
|||||||
"llama_index.core.retrievers.VectorIndexRetriever",
|
"llama_index.core.retrievers.VectorIndexRetriever",
|
||||||
return_value=mock_retriever_instance,
|
return_value=mock_retriever_instance,
|
||||||
):
|
):
|
||||||
output = list(stream_chat_with_documents("What is this?", documents))
|
output = list(stream_chat_with_documents("What is this?", [mock_document]))
|
||||||
|
|
||||||
mock_query_engine.query.assert_called_once_with("What is this?")
|
mock_query_engine.query.assert_called_once_with("What is this?")
|
||||||
synthesizer_kwargs = mock_get_response_synthesizer.call_args.kwargs
|
synthesizer_kwargs = mock_get_response_synthesizer.call_args.kwargs
|
||||||
@@ -163,16 +166,13 @@ def test_stream_chat_with_one_document_retrieval(
|
|||||||
output,
|
output,
|
||||||
expected_chunks=["chunk1", "chunk2"],
|
expected_chunks=["chunk1", "chunk2"],
|
||||||
expected_references=[
|
expected_references=[
|
||||||
{"id": document.pk, "title": "Test Document"},
|
{"id": mock_document.pk, "title": "Test Document"},
|
||||||
],
|
],
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
@pytest.mark.django_db
|
||||||
def test_stream_chat_with_multiple_documents_retrieval(patch_embed_nodes) -> None:
|
def test_stream_chat_with_multiple_documents_retrieval(patch_embed_nodes) -> None:
|
||||||
doc1 = DocumentFactory.create(title="Document 1", content="ignored")
|
|
||||||
doc2 = DocumentFactory.create(title="Document 2", content="ignored")
|
|
||||||
documents = Document.objects.filter(pk__in=[doc1.pk, doc2.pk])
|
|
||||||
with (
|
with (
|
||||||
patch("paperless_ai.chat.AIClient") as mock_client_cls,
|
patch("paperless_ai.chat.AIClient") as mock_client_cls,
|
||||||
patch("paperless_ai.chat.load_or_build_index") as mock_load_index,
|
patch("paperless_ai.chat.load_or_build_index") as mock_load_index,
|
||||||
@@ -184,23 +184,23 @@ def test_stream_chat_with_multiple_documents_retrieval(patch_embed_nodes) -> Non
|
|||||||
mock_client_cls.return_value = mock_client
|
mock_client_cls.return_value = mock_client
|
||||||
mock_client.llm = MagicMock()
|
mock_client.llm = MagicMock()
|
||||||
|
|
||||||
|
mock_node1 = TextNode(
|
||||||
|
text="Content for doc 1.",
|
||||||
|
metadata={"document_id": "1", "title": "Document 1"},
|
||||||
|
)
|
||||||
|
mock_node2 = TextNode(
|
||||||
|
text="Content for doc 2.",
|
||||||
|
metadata={"document_id": "2", "title": "Document 2"},
|
||||||
|
)
|
||||||
mock_index = MagicMock()
|
mock_index = MagicMock()
|
||||||
mock_index.vector_store.get_nodes.return_value = [
|
# Simulate get_nodes returning nodes (content exists)
|
||||||
TextNode(
|
mock_index.vector_store.get_nodes.return_value = [mock_node1, mock_node2]
|
||||||
text="Content for doc 1.",
|
|
||||||
metadata={"document_id": str(doc1.pk), "title": "Document 1"},
|
|
||||||
),
|
|
||||||
TextNode(
|
|
||||||
text="Content for doc 2.",
|
|
||||||
metadata={"document_id": str(doc2.pk), "title": "Document 2"},
|
|
||||||
),
|
|
||||||
]
|
|
||||||
mock_load_index.return_value = mock_index
|
mock_load_index.return_value = mock_index
|
||||||
|
|
||||||
mock_retriever_instance = MagicMock()
|
mock_retriever_instance = MagicMock()
|
||||||
mock_retriever_instance.retrieve.return_value = [
|
mock_retriever_instance.retrieve.return_value = [
|
||||||
MagicMock(metadata={"document_id": str(doc1.pk), "title": "Document 1"}),
|
MagicMock(metadata={"document_id": "1", "title": "Document 1"}),
|
||||||
MagicMock(metadata={"document_id": str(doc2.pk), "title": "Document 2"}),
|
MagicMock(metadata={"document_id": "2", "title": "Document 2"}),
|
||||||
]
|
]
|
||||||
|
|
||||||
mock_response_stream = MagicMock()
|
mock_response_stream = MagicMock()
|
||||||
@@ -210,11 +210,14 @@ def test_stream_chat_with_multiple_documents_retrieval(patch_embed_nodes) -> Non
|
|||||||
mock_query_engine_cls.return_value = mock_query_engine
|
mock_query_engine_cls.return_value = mock_query_engine
|
||||||
mock_query_engine.query.return_value = mock_response_stream
|
mock_query_engine.query.return_value = mock_response_stream
|
||||||
|
|
||||||
|
doc1 = MagicMock(pk=1, title="Document 1", filename="doc1.pdf")
|
||||||
|
doc2 = MagicMock(pk=2, title="Document 2", filename="doc2.pdf")
|
||||||
|
|
||||||
with patch(
|
with patch(
|
||||||
"llama_index.core.retrievers.VectorIndexRetriever",
|
"llama_index.core.retrievers.VectorIndexRetriever",
|
||||||
return_value=mock_retriever_instance,
|
return_value=mock_retriever_instance,
|
||||||
):
|
):
|
||||||
output = list(stream_chat_with_documents("What's up?", documents))
|
output = list(stream_chat_with_documents("What's up?", [doc1, doc2]))
|
||||||
|
|
||||||
mock_query_engine.query.assert_called_once_with("What's up?")
|
mock_query_engine.query.assert_called_once_with("What's up?")
|
||||||
patch_embed_nodes.assert_not_called()
|
patch_embed_nodes.assert_not_called()
|
||||||
@@ -222,15 +225,15 @@ def test_stream_chat_with_multiple_documents_retrieval(patch_embed_nodes) -> Non
|
|||||||
output,
|
output,
|
||||||
expected_chunks=["chunk1", "chunk2"],
|
expected_chunks=["chunk1", "chunk2"],
|
||||||
expected_references=[
|
expected_references=[
|
||||||
{"id": doc1.pk, "title": "Document 1"},
|
{"id": 1, "title": "Document 1"},
|
||||||
{"id": doc2.pk, "title": "Document 2"},
|
{"id": 2, "title": "Document 2"},
|
||||||
],
|
],
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def test_stream_chat_empty_document_list() -> None:
|
def test_stream_chat_empty_document_list() -> None:
|
||||||
with patch("paperless_ai.chat.load_or_build_index") as mock_load_index:
|
with patch("paperless_ai.chat.load_or_build_index") as mock_load_index:
|
||||||
output = list(stream_chat_with_documents("Any info?", Document.objects.none()))
|
output = list(stream_chat_with_documents("Any info?", []))
|
||||||
mock_load_index.assert_not_called()
|
mock_load_index.assert_not_called()
|
||||||
assert output == ["Sorry, I couldn't find any content to answer your question."]
|
assert output == ["Sorry, I couldn't find any content to answer your question."]
|
||||||
|
|
||||||
@@ -250,9 +253,7 @@ def test_stream_chat_no_matching_nodes() -> None:
|
|||||||
mock_index.vector_store.get_nodes.return_value = []
|
mock_index.vector_store.get_nodes.return_value = []
|
||||||
mock_load_index.return_value = mock_index
|
mock_load_index.return_value = mock_index
|
||||||
|
|
||||||
output = list(
|
output = list(stream_chat_with_documents("Any info?", [MagicMock(pk=1)]))
|
||||||
stream_chat_with_documents("Any info?", _fake_documents_queryset([1])),
|
|
||||||
)
|
|
||||||
|
|
||||||
assert output == ["Sorry, I couldn't find any content to answer your question."]
|
assert output == ["Sorry, I couldn't find any content to answer your question."]
|
||||||
|
|
||||||
@@ -281,9 +282,7 @@ def test_stream_chat_unexpected_failure_returns_generic_error(caplog) -> None:
|
|||||||
)
|
)
|
||||||
mock_retriever_cls.return_value = mock_retriever
|
mock_retriever_cls.return_value = mock_retriever
|
||||||
|
|
||||||
output = list(
|
output = list(stream_chat_with_documents("Any info?", [MagicMock(pk=1)]))
|
||||||
stream_chat_with_documents("Any info?", _fake_documents_queryset([1])),
|
|
||||||
)
|
|
||||||
|
|
||||||
assert output == [CHAT_ERROR_MESSAGE]
|
assert output == [CHAT_ERROR_MESSAGE]
|
||||||
assert "Failed to stream document chat response" in caplog.text
|
assert "Failed to stream document chat response" in caplog.text
|
||||||
@@ -299,12 +298,7 @@ class TestStreamChatRetrieval:
|
|||||||
) -> None:
|
) -> None:
|
||||||
doc = DocumentFactory.create(content="hello world")
|
doc = DocumentFactory.create(content="hello world")
|
||||||
# Nothing indexed for this document yet.
|
# Nothing indexed for this document yet.
|
||||||
out = list(
|
out = list(chat.stream_chat_with_documents("question?", [doc]))
|
||||||
chat.stream_chat_with_documents(
|
|
||||||
"question?",
|
|
||||||
Document.objects.filter(pk=doc.pk),
|
|
||||||
),
|
|
||||||
)
|
|
||||||
assert chat.CHAT_NO_CONTENT_MESSAGE in out
|
assert chat.CHAT_NO_CONTENT_MESSAGE in out
|
||||||
|
|
||||||
def test_chat_filter_contains_only_requested_document_ids(
|
def test_chat_filter_contains_only_requested_document_ids(
|
||||||
@@ -338,12 +332,7 @@ class TestStreamChatRetrieval:
|
|||||||
side_effect=capture_retriever,
|
side_effect=capture_retriever,
|
||||||
)
|
)
|
||||||
|
|
||||||
list(
|
list(chat.stream_chat_with_documents("question?", [included]))
|
||||||
chat.stream_chat_with_documents(
|
|
||||||
"question?",
|
|
||||||
Document.objects.filter(pk=included.pk),
|
|
||||||
),
|
|
||||||
)
|
|
||||||
|
|
||||||
assert captured_filters, "VectorIndexRetriever was never constructed"
|
assert captured_filters, "VectorIndexRetriever was never constructed"
|
||||||
filt = captured_filters[0]
|
filt = captured_filters[0]
|
||||||
@@ -351,47 +340,3 @@ class TestStreamChatRetrieval:
|
|||||||
filter_values = filt.filters[0].value
|
filter_values = filt.filters[0].value
|
||||||
assert str(included.pk) in filter_values
|
assert str(included.pk) in filter_values
|
||||||
assert str(excluded.pk) not in filter_values
|
assert str(excluded.pk) not in filter_values
|
||||||
|
|
||||||
@pytest.mark.django_db
|
|
||||||
def test_get_document_references_only_queries_referenced_documents(
|
|
||||||
self,
|
|
||||||
django_assert_num_queries,
|
|
||||||
) -> None:
|
|
||||||
"""Building references must not hydrate every document the caller is
|
|
||||||
permitted to see -- only the (<= CHAT_RETRIEVER_TOP_K) documents that
|
|
||||||
the retriever actually returned nodes for.
|
|
||||||
"""
|
|
||||||
referenced = DocumentFactory.create(title="Referenced Document")
|
|
||||||
# Many more documents are "accessible" but never referenced by a node.
|
|
||||||
DocumentFactory.create_batch(200)
|
|
||||||
|
|
||||||
documents = Document.objects.all()
|
|
||||||
top_nodes = [
|
|
||||||
MagicMock(
|
|
||||||
metadata={
|
|
||||||
"document_id": str(referenced.pk),
|
|
||||||
"title": "Referenced Document",
|
|
||||||
},
|
|
||||||
),
|
|
||||||
]
|
|
||||||
|
|
||||||
hydrated_count = 0
|
|
||||||
|
|
||||||
def _count_hydration(sender, instance, **kwargs):
|
|
||||||
nonlocal hydrated_count
|
|
||||||
hydrated_count += 1
|
|
||||||
|
|
||||||
post_init.connect(_count_hydration, sender=Document)
|
|
||||||
try:
|
|
||||||
# One query: `documents.filter(pk__in=candidate_ids)` for the single
|
|
||||||
# referenced id. No query should scale with the 200 unreferenced documents.
|
|
||||||
with django_assert_num_queries(1):
|
|
||||||
references = chat._get_document_references(documents, top_nodes)
|
|
||||||
finally:
|
|
||||||
post_init.disconnect(_count_hydration, sender=Document)
|
|
||||||
|
|
||||||
# The bug this guards against: the old code hydrated all 201 accessible
|
|
||||||
# documents via `{doc.pk: doc for doc in documents}` before filtering by
|
|
||||||
# top_nodes. Only the referenced document should ever be constructed.
|
|
||||||
assert hydrated_count == 1
|
|
||||||
assert references == [{"id": referenced.pk, "title": "Referenced Document"}]
|
|
||||||
|
|||||||
@@ -1,5 +1,3 @@
|
|||||||
from unittest.mock import patch
|
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
from django.test import TestCase
|
from django.test import TestCase
|
||||||
|
|
||||||
@@ -32,33 +30,25 @@ class TestAIMatching(TestCase):
|
|||||||
self.storage_path1 = StoragePath.objects.create(name="Test Storage Path 1")
|
self.storage_path1 = StoragePath.objects.create(name="Test Storage Path 1")
|
||||||
self.storage_path2 = StoragePath.objects.create(name="Test Storage Path 2")
|
self.storage_path2 = StoragePath.objects.create(name="Test Storage Path 2")
|
||||||
|
|
||||||
@patch("paperless_ai.matching.get_objects_for_user_owner_aware")
|
def test_match_tags_by_name(self) -> None:
|
||||||
def test_match_tags_by_name(self, mock_get_objects) -> None:
|
|
||||||
mock_get_objects.return_value = Tag.objects.all()
|
|
||||||
names = ["Test Tag 1", "Nonexistent Tag"]
|
names = ["Test Tag 1", "Nonexistent Tag"]
|
||||||
result = match_tags_by_name(names, user=None)
|
result = match_tags_by_name(names, user=None)
|
||||||
self.assertEqual(len(result), 1)
|
self.assertEqual(len(result), 1)
|
||||||
self.assertEqual(result[0].name, "Test Tag 1")
|
self.assertEqual(result[0].name, "Test Tag 1")
|
||||||
|
|
||||||
@patch("paperless_ai.matching.get_objects_for_user_owner_aware")
|
def test_match_correspondents_by_name(self) -> None:
|
||||||
def test_match_correspondents_by_name(self, mock_get_objects) -> None:
|
|
||||||
mock_get_objects.return_value = Correspondent.objects.all()
|
|
||||||
names = ["Test Correspondent 1", "Nonexistent Correspondent"]
|
names = ["Test Correspondent 1", "Nonexistent Correspondent"]
|
||||||
result = match_correspondents_by_name(names, user=None)
|
result = match_correspondents_by_name(names, user=None)
|
||||||
self.assertEqual(len(result), 1)
|
self.assertEqual(len(result), 1)
|
||||||
self.assertEqual(result[0].name, "Test Correspondent 1")
|
self.assertEqual(result[0].name, "Test Correspondent 1")
|
||||||
|
|
||||||
@patch("paperless_ai.matching.get_objects_for_user_owner_aware")
|
def test_match_document_types_by_name(self) -> None:
|
||||||
def test_match_document_types_by_name(self, mock_get_objects) -> None:
|
|
||||||
mock_get_objects.return_value = DocumentType.objects.all()
|
|
||||||
names = ["Test Document Type 1", "Nonexistent Document Type"]
|
names = ["Test Document Type 1", "Nonexistent Document Type"]
|
||||||
result = match_document_types_by_name(names, user=None)
|
result = match_document_types_by_name(names, user=None)
|
||||||
self.assertEqual(len(result), 1)
|
self.assertEqual(len(result), 1)
|
||||||
self.assertEqual(result[0].name, "Test Document Type 1")
|
self.assertEqual(result[0].name, "Test Document Type 1")
|
||||||
|
|
||||||
@patch("paperless_ai.matching.get_objects_for_user_owner_aware")
|
def test_match_storage_paths_by_name(self) -> None:
|
||||||
def test_match_storage_paths_by_name(self, mock_get_objects) -> None:
|
|
||||||
mock_get_objects.return_value = StoragePath.objects.all()
|
|
||||||
names = ["Test Storage Path 1", "Nonexistent Storage Path"]
|
names = ["Test Storage Path 1", "Nonexistent Storage Path"]
|
||||||
result = match_storage_paths_by_name(names, user=None)
|
result = match_storage_paths_by_name(names, user=None)
|
||||||
self.assertEqual(len(result), 1)
|
self.assertEqual(len(result), 1)
|
||||||
@@ -70,16 +60,12 @@ class TestAIMatching(TestCase):
|
|||||||
unmatched_names = extract_unmatched_names(llm_names, matched_objects)
|
unmatched_names = extract_unmatched_names(llm_names, matched_objects)
|
||||||
self.assertEqual(unmatched_names, ["Nonexistent Tag"])
|
self.assertEqual(unmatched_names, ["Nonexistent Tag"])
|
||||||
|
|
||||||
@patch("paperless_ai.matching.get_objects_for_user_owner_aware")
|
def test_match_tags_by_name_with_empty_names(self) -> None:
|
||||||
def test_match_tags_by_name_with_empty_names(self, mock_get_objects) -> None:
|
|
||||||
mock_get_objects.return_value = Tag.objects.all()
|
|
||||||
names = [None, "", " "]
|
names = [None, "", " "]
|
||||||
result = match_tags_by_name(names, user=None)
|
result = match_tags_by_name(names, user=None)
|
||||||
self.assertEqual(result, [])
|
self.assertEqual(result, [])
|
||||||
|
|
||||||
@patch("paperless_ai.matching.get_objects_for_user_owner_aware")
|
def test_match_tags_with_fuzzy_matching(self) -> None:
|
||||||
def test_match_tags_with_fuzzy_matching(self, mock_get_objects) -> None:
|
|
||||||
mock_get_objects.return_value = Tag.objects.all()
|
|
||||||
names = ["Test Taag 1", "Teest Tag 2"]
|
names = ["Test Taag 1", "Teest Tag 2"]
|
||||||
result = match_tags_by_name(names, user=None)
|
result = match_tags_by_name(names, user=None)
|
||||||
self.assertEqual(len(result), 2)
|
self.assertEqual(len(result), 2)
|
||||||
|
|||||||
Reference in New Issue
Block a user