mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-08-13 06:13:20 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f839db6ecf |
@@ -2048,18 +2048,6 @@ password. All of these options come from their similarly-named [Django settings]
|
||||
|
||||
Defaults to None.
|
||||
|
||||
#### [`PAPERLESS_REMOTE_OCR_MODE=<str>`](#PAPERLESS_REMOTE_OCR_MODE) {#PAPERLESS_REMOTE_OCR_MODE}
|
||||
|
||||
: Which documents are sent to the remote OCR engine.
|
||||
|
||||
- `always`: every document of a supported file type is sent to the remote
|
||||
engine, bypassing the local OCR engine.
|
||||
- `workflow_only`: documents are processed locally unless a workflow
|
||||
explicitly enables remote OCR for them, letting you use the remote engine
|
||||
selectively.
|
||||
|
||||
Defaults to "always".
|
||||
|
||||
## AI {#ai}
|
||||
|
||||
#### [`PAPERLESS_AI_ENABLED=<bool>`](#PAPERLESS_AI_ENABLED) {#PAPERLESS_AI_ENABLED}
|
||||
|
||||
@@ -456,20 +456,6 @@ def score(
|
||||
return 10
|
||||
```
|
||||
|
||||
**Remote services**
|
||||
|
||||
If your parser sends document content to a remote service, declare it:
|
||||
|
||||
```python
|
||||
class MyCustomParser:
|
||||
uses_remote_service = True
|
||||
```
|
||||
|
||||
Paperless-ngx excludes such parsers when the document being consumed has not
|
||||
been marked for remote processing, so users can keep remote OCR off by default
|
||||
and enable it selectively with a workflow. Parsers that do not declare the
|
||||
attribute are treated as fully local and are always considered.
|
||||
|
||||
**Archive and rendition flags**
|
||||
|
||||
```python
|
||||
|
||||
+1
-6
@@ -1086,16 +1086,11 @@ Paperless-ngx supports performing OCR on documents using remote services. At the
|
||||
[Microsoft's Azure "Document Intelligence" service](https://azure.microsoft.com/en-us/products/ai-services/ai-document-intelligence).
|
||||
This is of course a paid service (with a free tier) which requires an Azure account and subscription. Azure AI is not affiliated with
|
||||
Paperless-ngx in any way. When enabled, Paperless-ngx will automatically send appropriate documents to Azure for OCR processing, bypassing
|
||||
the local OCR engine. See the [configuration](configuration.md#PAPERLESS_REMOTE_OCR_ENGINE) options for more details. These
|
||||
settings can be supplied as environment variables or via **Application Configuration**.
|
||||
the local OCR engine. See the [configuration](configuration.md#PAPERLESS_REMOTE_OCR_ENGINE) options for more details.
|
||||
|
||||
Additionally, when using a commercial service with this feature, consider both potential costs as well as any associated file size
|
||||
or page limitations (e.g. with a free tier).
|
||||
|
||||
By default, every document of a supported file type is sent to the remote engine. To use it more selectively, set the
|
||||
[remote OCR mode](configuration.md#PAPERLESS_REMOTE_OCR_MODE) to `workflow_only`. Documents are then processed locally
|
||||
unless a workflow explicitly enables remote OCR for them, so you can limit the remote engine to particular documents.
|
||||
|
||||
## Architecture
|
||||
|
||||
Paperless-ngx consists of the following components:
|
||||
|
||||
@@ -14,48 +14,43 @@
|
||||
<a ngbNavLink>{{category}}</a>
|
||||
<ng-template ngbNavContent>
|
||||
<div class="p-3">
|
||||
@for (section of getCategorySections(category); track section) {
|
||||
@if (section) {
|
||||
<h5 class="mt-4 mb-3">{{section}}</h5>
|
||||
}
|
||||
<div class="row row-cols-1 row-cols-md-2 row-cols-lg-3 g-2">
|
||||
@for (option of getCategoryOptions(category, section); track option.key) {
|
||||
<div class="col">
|
||||
<div class="card bg-light">
|
||||
<div class="card-body">
|
||||
<div class="card-title d-flex align-items-center">
|
||||
<h6 class="mb-0">
|
||||
{{option.title}}
|
||||
</h6>
|
||||
<a class="btn btn-sm btn-link" title="Read the documentation about this setting" i18n-title [href]="getDocsUrl(option.config_key)" target="_blank" referrerpolicy="no-referrer">
|
||||
<i-bs name="info-circle"></i-bs>
|
||||
</a>
|
||||
@if (isSet(option.key)) {
|
||||
<button type="button" class="btn btn-sm btn-link text-danger ms-auto pe-0" title="Reset" i18n-title (click)="resetOption(option.key)">
|
||||
<i-bs class="me-1" name="x"></i-bs><ng-container i18n>Reset</ng-container>
|
||||
</button>
|
||||
}
|
||||
</div>
|
||||
<div class="mb-n3">
|
||||
@switch (option.type) {
|
||||
@case (ConfigOptionType.Select) { <pngx-input-select [formControlName]="option.key" [error]="errors[option.key]" [items]="option.choices" [allowNull]="true"></pngx-input-select> }
|
||||
@case (ConfigOptionType.Number) { <pngx-input-number [formControlName]="option.key" [error]="errors[option.key]" [showAdd]="false"></pngx-input-number> }
|
||||
@case (ConfigOptionType.Boolean) { <pngx-input-switch [formControlName]="option.key" [error]="errors[option.key]" [showUnsetNote]="true" [horizontal]="true" title="Enable" i18n-title></pngx-input-switch> }
|
||||
@case (ConfigOptionType.String) { <pngx-input-text [formControlName]="option.key" [error]="errors[option.key]"></pngx-input-text> }
|
||||
@case (ConfigOptionType.JSON) { <pngx-input-text [formControlName]="option.key" [error]="errors[option.key]"></pngx-input-text> }
|
||||
@case (ConfigOptionType.File) { <pngx-input-file [formControlName]="option.key" (upload)="uploadFile($event, option.key)" [error]="errors[option.key]"></pngx-input-file> }
|
||||
@case (ConfigOptionType.Password) { <pngx-input-password [formControlName]="option.key" [error]="errors[option.key]"></pngx-input-password> }
|
||||
}
|
||||
</div>
|
||||
@if (option.note) {
|
||||
<div class="form-text fst-italic">{{option.note}}</div>
|
||||
<div class="row row-cols-1 row-cols-md-2 row-cols-lg-3 g-2">
|
||||
@for (option of getCategoryOptions(category); track option.key) {
|
||||
<div class="col">
|
||||
<div class="card bg-light">
|
||||
<div class="card-body">
|
||||
<div class="card-title d-flex align-items-center">
|
||||
<h6 class="mb-0">
|
||||
{{option.title}}
|
||||
</h6>
|
||||
<a class="btn btn-sm btn-link" title="Read the documentation about this setting" i18n-title [href]="getDocsUrl(option.config_key)" target="_blank" referrerpolicy="no-referrer">
|
||||
<i-bs name="info-circle"></i-bs>
|
||||
</a>
|
||||
@if (isSet(option.key)) {
|
||||
<button type="button" class="btn btn-sm btn-link text-danger ms-auto pe-0" title="Reset" i18n-title (click)="resetOption(option.key)">
|
||||
<i-bs class="me-1" name="x"></i-bs><ng-container i18n>Reset</ng-container>
|
||||
</button>
|
||||
}
|
||||
</div>
|
||||
<div class="mb-n3">
|
||||
@switch (option.type) {
|
||||
@case (ConfigOptionType.Select) { <pngx-input-select [formControlName]="option.key" [error]="errors[option.key]" [items]="option.choices" [allowNull]="true"></pngx-input-select> }
|
||||
@case (ConfigOptionType.Number) { <pngx-input-number [formControlName]="option.key" [error]="errors[option.key]" [showAdd]="false"></pngx-input-number> }
|
||||
@case (ConfigOptionType.Boolean) { <pngx-input-switch [formControlName]="option.key" [error]="errors[option.key]" [showUnsetNote]="true" [horizontal]="true" title="Enable" i18n-title></pngx-input-switch> }
|
||||
@case (ConfigOptionType.String) { <pngx-input-text [formControlName]="option.key" [error]="errors[option.key]"></pngx-input-text> }
|
||||
@case (ConfigOptionType.JSON) { <pngx-input-text [formControlName]="option.key" [error]="errors[option.key]"></pngx-input-text> }
|
||||
@case (ConfigOptionType.File) { <pngx-input-file [formControlName]="option.key" (upload)="uploadFile($event, option.key)" [error]="errors[option.key]"></pngx-input-file> }
|
||||
@case (ConfigOptionType.Password) { <pngx-input-password [formControlName]="option.key" [error]="errors[option.key]"></pngx-input-password> }
|
||||
}
|
||||
</div>
|
||||
@if (option.note) {
|
||||
<div class="form-text fst-italic">{{option.note}}</div>
|
||||
}
|
||||
</div>
|
||||
</div>
|
||||
}
|
||||
</div>
|
||||
}
|
||||
</div>
|
||||
}
|
||||
</div>
|
||||
</div>
|
||||
</ng-template>
|
||||
</li>
|
||||
|
||||
@@ -8,11 +8,7 @@ import { NgbModule } from '@ng-bootstrap/ng-bootstrap'
|
||||
import { NgSelectModule } from '@ng-select/ng-select'
|
||||
import { NgxBootstrapIconsModule, allIcons } from 'ngx-bootstrap-icons'
|
||||
import { of, throwError } from 'rxjs'
|
||||
import {
|
||||
ConfigCategory,
|
||||
ConfigSection,
|
||||
OutputTypeConfig,
|
||||
} from 'src/app/data/paperless-config'
|
||||
import { OutputTypeConfig } from 'src/app/data/paperless-config'
|
||||
import { ConfigService } from 'src/app/services/config.service'
|
||||
import { SettingsService } from 'src/app/services/settings.service'
|
||||
import { ToastService } from 'src/app/services/toast.service'
|
||||
@@ -162,24 +158,4 @@ describe('ConfigComponent', () => {
|
||||
component.resetOption('barcodes_enabled')
|
||||
expect(component.configForm.get('barcodes_enabled').value).toBeNull()
|
||||
})
|
||||
|
||||
it('should group options into sections within a category, or not', () => {
|
||||
const sections = component.getCategorySections(ConfigCategory.OCR)
|
||||
expect(sections).toEqual([null, ConfigSection.RemoteOCR])
|
||||
expect(
|
||||
component
|
||||
.getCategoryOptions(ConfigCategory.OCR)
|
||||
.map((option) => option.key)
|
||||
).toContain('output_type')
|
||||
expect(
|
||||
component
|
||||
.getCategoryOptions(ConfigCategory.OCR, ConfigSection.RemoteOCR)
|
||||
.map((option) => option.key)
|
||||
).toEqual([
|
||||
'remote_ocr_engine',
|
||||
'remote_ocr_api_key',
|
||||
'remote_ocr_endpoint',
|
||||
'remote_ocr_mode',
|
||||
])
|
||||
})
|
||||
})
|
||||
|
||||
@@ -74,20 +74,8 @@ export class ConfigComponent
|
||||
return Object.values(ConfigCategory)
|
||||
}
|
||||
|
||||
getCategorySections(category: string): string[] {
|
||||
return [
|
||||
...new Set(
|
||||
PaperlessConfigOptions.filter((o) => o.category === category).map(
|
||||
(o) => o.section ?? null // null means no section
|
||||
)
|
||||
),
|
||||
]
|
||||
}
|
||||
|
||||
getCategoryOptions(category: string, section: string = null): ConfigOption[] {
|
||||
return PaperlessConfigOptions.filter(
|
||||
(o) => o.category === category && (o.section ?? null) === section
|
||||
)
|
||||
getCategoryOptions(category: string): ConfigOption[] {
|
||||
return PaperlessConfigOptions.filter((o) => o.category === category)
|
||||
}
|
||||
|
||||
initialConfig: PaperlessConfig
|
||||
|
||||
-9
@@ -17,10 +17,6 @@ const permissions = [
|
||||
'view_document',
|
||||
'change_document',
|
||||
'delete_document',
|
||||
'add_sharelinkbundle',
|
||||
'view_sharelinkbundle',
|
||||
'change_sharelinkbundle',
|
||||
'delete_sharelinkbundle',
|
||||
'change_tag',
|
||||
'view_documenttype',
|
||||
]
|
||||
@@ -79,7 +75,6 @@ describe('PermissionsSelectComponent', () => {
|
||||
component.ngOnInit()
|
||||
component.writeValue(permissions)
|
||||
expect(component.typesWithAllActions).toContain('Document')
|
||||
expect(component.typesWithAllActions).toContain('ShareLinkBundle')
|
||||
})
|
||||
|
||||
it('should update checkboxes on permissions set', () => {
|
||||
@@ -90,10 +85,6 @@ describe('PermissionsSelectComponent', () => {
|
||||
expect(input1.nativeElement.checked).toBeTruthy()
|
||||
const input2 = fixture.debugElement.query(By.css('input#Tag_Change'))
|
||||
expect(input2.nativeElement.checked).toBeTruthy()
|
||||
const bundleInput = fixture.debugElement.query(
|
||||
By.css('input#ShareLinkBundle_Add')
|
||||
)
|
||||
expect(bundleInput.nativeElement.checked).toBeTruthy()
|
||||
})
|
||||
|
||||
it('disable checkboxes when permissions are inherited', () => {
|
||||
|
||||
@@ -54,10 +54,6 @@ export const ConfigCategory = {
|
||||
AI: $localize`AI Settings`,
|
||||
}
|
||||
|
||||
export const ConfigSection = {
|
||||
RemoteOCR: $localize`Remote OCR`,
|
||||
}
|
||||
|
||||
export const LLMEmbeddingBackendConfig = {
|
||||
OPENAI_LIKE: 'openai-like',
|
||||
HUGGINGFACE: 'huggingface',
|
||||
@@ -69,15 +65,6 @@ export const LLMBackendConfig = {
|
||||
OLLAMA: 'ollama',
|
||||
}
|
||||
|
||||
export const RemoteOCREngineConfig = {
|
||||
AZURE_AI: 'azureai',
|
||||
}
|
||||
|
||||
export const RemoteOCRModeConfig = {
|
||||
ALWAYS: 'always',
|
||||
WORKFLOW_ONLY: 'workflow_only',
|
||||
}
|
||||
|
||||
export interface ConfigOption {
|
||||
key: string
|
||||
title: string
|
||||
@@ -85,7 +72,6 @@ export interface ConfigOption {
|
||||
choices?: Array<{ id: string; name: string }>
|
||||
config_key?: string
|
||||
category: string
|
||||
section?: string
|
||||
note?: string
|
||||
}
|
||||
|
||||
@@ -195,43 +181,6 @@ export const PaperlessConfigOptions: ConfigOption[] = [
|
||||
config_key: 'PAPERLESS_OCR_USER_ARGS',
|
||||
category: ConfigCategory.OCR,
|
||||
},
|
||||
{
|
||||
key: 'remote_ocr_engine',
|
||||
title: $localize`Remote OCR Engine`,
|
||||
type: ConfigOptionType.Select,
|
||||
choices: mapToItems(RemoteOCREngineConfig),
|
||||
config_key: 'PAPERLESS_REMOTE_OCR_ENGINE',
|
||||
category: ConfigCategory.OCR,
|
||||
section: ConfigSection.RemoteOCR,
|
||||
note: $localize`Enabling remote OCR sends documents to a third-party service for processing. Consider the privacy implications as well as potential costs before enabling.`,
|
||||
},
|
||||
{
|
||||
key: 'remote_ocr_api_key',
|
||||
title: $localize`Remote OCR API Key`,
|
||||
type: ConfigOptionType.Password,
|
||||
config_key: 'PAPERLESS_REMOTE_OCR_API_KEY',
|
||||
category: ConfigCategory.OCR,
|
||||
section: ConfigSection.RemoteOCR,
|
||||
},
|
||||
{
|
||||
key: 'remote_ocr_endpoint',
|
||||
title: $localize`Remote OCR Endpoint`,
|
||||
type: ConfigOptionType.String,
|
||||
config_key: 'PAPERLESS_REMOTE_OCR_ENDPOINT',
|
||||
category: ConfigCategory.OCR,
|
||||
section: ConfigSection.RemoteOCR,
|
||||
note: $localize`Required when using the Azure AI engine.`,
|
||||
},
|
||||
{
|
||||
key: 'remote_ocr_mode',
|
||||
title: $localize`Remote OCR Mode`,
|
||||
type: ConfigOptionType.Select,
|
||||
choices: mapToItems(RemoteOCRModeConfig),
|
||||
config_key: 'PAPERLESS_REMOTE_OCR_MODE',
|
||||
category: ConfigCategory.OCR,
|
||||
section: ConfigSection.RemoteOCR,
|
||||
note: $localize`Which documents are sent to the remote engine. Use 'workflow_only' to keep remote OCR off unless a workflow enables it for a document.`,
|
||||
},
|
||||
{
|
||||
key: 'app_logo',
|
||||
title: $localize`Application Logo`,
|
||||
@@ -449,10 +398,6 @@ export interface PaperlessConfig extends ObjectWithId {
|
||||
barcode_enable_tag: boolean
|
||||
barcode_tag_mapping: object
|
||||
barcode_tag_split: boolean
|
||||
remote_ocr_engine: string
|
||||
remote_ocr_api_key: string
|
||||
remote_ocr_endpoint: string
|
||||
remote_ocr_mode: string
|
||||
ai_enabled: boolean
|
||||
llm_embedding_backend: string
|
||||
llm_embedding_model: string
|
||||
|
||||
@@ -120,12 +120,6 @@ describe('PermissionsService', () => {
|
||||
actionKey: 'View', // PermissionAction.View
|
||||
typeKey: 'SystemMonitoring', // PermissionType.SystemMonitoring
|
||||
})
|
||||
expect(permissionsService.getPermissionKeys('add_sharelinkbundle')).toEqual(
|
||||
{
|
||||
actionKey: 'Add', // PermissionAction.Add
|
||||
typeKey: 'ShareLinkBundle', // PermissionType.ShareLinkBundle
|
||||
}
|
||||
)
|
||||
})
|
||||
|
||||
it('correctly checks explicit global permissions', () => {
|
||||
@@ -275,10 +269,6 @@ describe('PermissionsService', () => {
|
||||
'view_sharelink',
|
||||
'change_sharelink',
|
||||
'delete_sharelink',
|
||||
'add_sharelinkbundle',
|
||||
'view_sharelinkbundle',
|
||||
'change_sharelinkbundle',
|
||||
'delete_sharelinkbundle',
|
||||
'add_workflow',
|
||||
'view_workflow',
|
||||
'change_workflow',
|
||||
|
||||
@@ -26,7 +26,6 @@ export enum PermissionType {
|
||||
User = '%s_user',
|
||||
Group = '%s_group',
|
||||
ShareLink = '%s_sharelink',
|
||||
ShareLinkBundle = '%s_sharelinkbundle',
|
||||
CustomField = '%s_customfield',
|
||||
Workflow = '%s_workflow',
|
||||
ProcessedMail = '%s_processedmail',
|
||||
|
||||
@@ -53,7 +53,6 @@ from documents.utils import copy_basic_file_stats
|
||||
from documents.utils import copy_file_with_basic_stats
|
||||
from documents.utils import run_subprocess
|
||||
from paperless.config import OcrConfig
|
||||
from paperless.config import RemoteOCRConfig
|
||||
from paperless.models import ArchiveFileGenerationChoices
|
||||
from paperless.parsers import ParserContext
|
||||
from paperless.parsers import ParserProtocol
|
||||
@@ -452,19 +451,12 @@ class ConsumerPlugin(
|
||||
except Exception as e:
|
||||
self.log.error(f"Error attempting to clean PDF: {e}")
|
||||
|
||||
# Workflows have already run at this point, so the metadata knows
|
||||
# whether this document was singled out for remote OCR
|
||||
allow_remote = (
|
||||
self.metadata.remote_ocr or RemoteOCRConfig().remote_ocr_by_default
|
||||
)
|
||||
|
||||
# Based on the mime type, get the parser for that type
|
||||
parser_class: type[ParserProtocol] | None = (
|
||||
get_parser_registry().get_parser_for_file(
|
||||
mime_type,
|
||||
self.filename,
|
||||
self.working_copy,
|
||||
allow_remote=allow_remote,
|
||||
)
|
||||
)
|
||||
if not parser_class:
|
||||
|
||||
@@ -34,7 +34,6 @@ class DocumentMetadataOverrides:
|
||||
skip_asn_if_exists: bool = False
|
||||
version_label: str | None = None
|
||||
actor_id: int | None = None
|
||||
remote_ocr: bool = False
|
||||
|
||||
def update(self, other: "DocumentMetadataOverrides") -> "DocumentMetadataOverrides":
|
||||
"""
|
||||
@@ -58,8 +57,6 @@ class DocumentMetadataOverrides:
|
||||
self.actor_id = other.actor_id
|
||||
if other.skip_asn_if_exists:
|
||||
self.skip_asn_if_exists = True
|
||||
if other.remote_ocr:
|
||||
self.remote_ocr = True
|
||||
if other.version_label is not None:
|
||||
self.version_label = other.version_label
|
||||
|
||||
|
||||
+1
-10
@@ -66,7 +66,6 @@ from documents.utils import compute_checksum
|
||||
from documents.utils import identity
|
||||
from documents.workflows.utils import get_workflows_for_trigger
|
||||
from paperless.config import AIConfig
|
||||
from paperless.config import RemoteOCRConfig
|
||||
from paperless.logging import consume_task_id
|
||||
from paperless.parsers import ParserContext
|
||||
from paperless.parsers.registry import get_parser_registry
|
||||
@@ -338,17 +337,10 @@ def bulk_update_documents(document_ids) -> None:
|
||||
|
||||
|
||||
@shared_task
|
||||
def update_document_content_maybe_archive_file(
|
||||
document_id,
|
||||
*,
|
||||
remote_ocr: bool = False,
|
||||
) -> None:
|
||||
def update_document_content_maybe_archive_file(document_id) -> None:
|
||||
"""
|
||||
Re-creates OCR content and thumbnail for a document, and archive file if
|
||||
it exists.
|
||||
|
||||
Remote OCR is used only when the engine is configured to handle everything
|
||||
or if explicitly asked for via ``remote_ocr``.
|
||||
"""
|
||||
document = Document.objects.get(id=document_id)
|
||||
|
||||
@@ -358,7 +350,6 @@ def update_document_content_maybe_archive_file(
|
||||
mime_type,
|
||||
document.original_filename or "",
|
||||
document.source_path,
|
||||
allow_remote=remote_ocr or RemoteOCRConfig().remote_ocr_by_default,
|
||||
)
|
||||
|
||||
if not parser_class:
|
||||
|
||||
@@ -72,10 +72,6 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
|
||||
"barcode_enable_tag": None,
|
||||
"barcode_tag_mapping": None,
|
||||
"barcode_tag_split": None,
|
||||
"remote_ocr_engine": None,
|
||||
"remote_ocr_api_key": None,
|
||||
"remote_ocr_endpoint": None,
|
||||
"remote_ocr_mode": None,
|
||||
"ai_enabled": False,
|
||||
"llm_embedding_backend": None,
|
||||
"llm_embedding_model": None,
|
||||
@@ -874,49 +870,6 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
|
||||
config.refresh_from_db()
|
||||
self.assertEqual(config.llm_api_key, None)
|
||||
|
||||
def test_update_remote_ocr_api_key(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
- Existing config with remote_ocr_api_key specified
|
||||
WHEN:
|
||||
- API to update remote_ocr_api_key is called with all *s
|
||||
- API to update remote_ocr_api_key is called with empty string
|
||||
THEN:
|
||||
- remote_ocr_api_key is unchanged
|
||||
- remote_ocr_api_key is set to None
|
||||
"""
|
||||
config = ApplicationConfiguration.objects.first()
|
||||
assert config is not None
|
||||
config.remote_ocr_api_key = "1234567890"
|
||||
config.save()
|
||||
|
||||
# Test with all *
|
||||
response = self.client.patch(
|
||||
f"{self.ENDPOINT}1/",
|
||||
json.dumps(
|
||||
{
|
||||
"remote_ocr_api_key": "*" * 32,
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
config.refresh_from_db()
|
||||
self.assertEqual(config.remote_ocr_api_key, "1234567890")
|
||||
# Test with empty string
|
||||
response = self.client.patch(
|
||||
f"{self.ENDPOINT}1/",
|
||||
json.dumps(
|
||||
{
|
||||
"remote_ocr_api_key": "",
|
||||
},
|
||||
),
|
||||
content_type="application/json",
|
||||
)
|
||||
self.assertEqual(response.status_code, status.HTTP_200_OK)
|
||||
config.refresh_from_db()
|
||||
self.assertEqual(config.remote_ocr_api_key, None)
|
||||
|
||||
def test_enable_ai_index_triggers_update(self) -> None:
|
||||
"""
|
||||
GIVEN:
|
||||
|
||||
@@ -1559,72 +1559,6 @@ class PostConsumeTestCase(DirectoriesMixin, GetConsumerMixin, TestCase):
|
||||
consumer.run_post_consume_script(doc)
|
||||
|
||||
|
||||
class TestConsumerRemoteOCR(
|
||||
DirectoriesMixin,
|
||||
FileSystemAssertsMixin,
|
||||
GetConsumerMixin,
|
||||
TestCase,
|
||||
):
|
||||
"""
|
||||
The consumer resolves the remote OCR mode and the per-document request from
|
||||
workflows into the allow_remote flag it hands to the parser registry.
|
||||
"""
|
||||
|
||||
def setUp(self) -> None:
|
||||
super().setUp()
|
||||
|
||||
patcher = mock.patch("documents.consumer.get_parser_registry")
|
||||
self.mock_registry = patcher.start()
|
||||
self.mock_registry.return_value.get_parser_for_file.return_value = DummyParser
|
||||
self.addCleanup(patcher.stop)
|
||||
|
||||
def _consume(self, *, overrides: DocumentMetadataOverrides | None = None) -> bool:
|
||||
src = (
|
||||
Path(__file__).parent
|
||||
/ "samples"
|
||||
/ "documents"
|
||||
/ "originals"
|
||||
/ "0000001.pdf"
|
||||
)
|
||||
dst = self.dirs.scratch_dir / "sample.pdf"
|
||||
shutil.copy(src, dst)
|
||||
|
||||
with self.get_consumer(dst, overrides=overrides) as consumer:
|
||||
consumer.run()
|
||||
|
||||
_, kwargs = self.mock_registry.return_value.get_parser_for_file.call_args
|
||||
return kwargs["allow_remote"]
|
||||
|
||||
@override_settings(REMOTE_OCR_MODE="always")
|
||||
def test_always_mode_allows_remote(self) -> None:
|
||||
"""
|
||||
GIVEN: Remote OCR mode is 'always'.
|
||||
WHEN: A document is consumed without any workflow asking for it.
|
||||
THEN: The registry is allowed to pick the remote parser.
|
||||
"""
|
||||
self.assertTrue(self._consume())
|
||||
|
||||
@override_settings(REMOTE_OCR_MODE="workflow_only")
|
||||
def test_workflow_only_mode_denies_remote_by_default(self) -> None:
|
||||
"""
|
||||
GIVEN: Remote OCR mode is 'workflow_only'.
|
||||
WHEN: A document is consumed and nothing asked for remote OCR.
|
||||
THEN: The remote parser is excluded.
|
||||
"""
|
||||
self.assertFalse(self._consume())
|
||||
|
||||
@override_settings(REMOTE_OCR_MODE="workflow_only")
|
||||
def test_workflow_only_mode_allows_remote_when_requested(self) -> None:
|
||||
"""
|
||||
GIVEN: Remote OCR mode is 'workflow_only'.
|
||||
WHEN: A workflow set remote_ocr on the metadata overrides.
|
||||
THEN: The registry is allowed to pick the remote parser.
|
||||
"""
|
||||
self.assertTrue(
|
||||
self._consume(overrides=DocumentMetadataOverrides(remote_ocr=True)),
|
||||
)
|
||||
|
||||
|
||||
class TestMetadataOverrides(TestCase):
|
||||
def test_update_skip_asn_if_exists(self) -> None:
|
||||
base = DocumentMetadataOverrides()
|
||||
@@ -1632,20 +1566,6 @@ class TestMetadataOverrides(TestCase):
|
||||
base.update(incoming)
|
||||
self.assertTrue(base.skip_asn_if_exists)
|
||||
|
||||
def test_update_remote_ocr(self) -> None:
|
||||
base = DocumentMetadataOverrides()
|
||||
base.update(DocumentMetadataOverrides(remote_ocr=True))
|
||||
self.assertTrue(base.remote_ocr)
|
||||
|
||||
def test_update_remote_ocr_is_not_unset(self) -> None:
|
||||
"""
|
||||
A later workflow that says nothing must not undo an earlier one that
|
||||
asked for remote OCR.
|
||||
"""
|
||||
base = DocumentMetadataOverrides(remote_ocr=True)
|
||||
base.update(DocumentMetadataOverrides())
|
||||
self.assertTrue(base.remote_ocr)
|
||||
|
||||
def test_update_actor_and_version_label(self) -> None:
|
||||
base = DocumentMetadataOverrides(
|
||||
actor_id=1,
|
||||
|
||||
@@ -287,45 +287,6 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
|
||||
self.assertNotEqual(Document.objects.get(pk=doc.pk).content, "test")
|
||||
|
||||
|
||||
class TestUpdateContentRemoteOCR(DirectoriesMixin, TestCase):
|
||||
"""
|
||||
Consumption workflows do not run on reprocess, so the remote parser is
|
||||
used only in 'always' mode or when the caller explicitly asks for it.
|
||||
"""
|
||||
|
||||
def setUp(self) -> None:
|
||||
super().setUp()
|
||||
|
||||
patcher = mock.patch("documents.tasks.get_parser_registry")
|
||||
self.mock_registry = patcher.start()
|
||||
self.mock_registry.return_value.get_parser_for_file.return_value = None
|
||||
self.addCleanup(patcher.stop)
|
||||
|
||||
self.doc = Document.objects.create(
|
||||
title="test",
|
||||
content="my document",
|
||||
checksum="wow",
|
||||
mime_type="application/pdf",
|
||||
)
|
||||
|
||||
def _allow_remote(self, **kwargs) -> bool:
|
||||
tasks.update_document_content_maybe_archive_file(self.doc.pk, **kwargs)
|
||||
_, call_kwargs = self.mock_registry.return_value.get_parser_for_file.call_args
|
||||
return call_kwargs["allow_remote"]
|
||||
|
||||
@override_settings(REMOTE_OCR_MODE="always")
|
||||
def test_always_mode_allows_remote(self) -> None:
|
||||
self.assertTrue(self._allow_remote())
|
||||
|
||||
@override_settings(REMOTE_OCR_MODE="workflow_only")
|
||||
def test_workflow_only_mode_denies_remote_by_default(self) -> None:
|
||||
self.assertFalse(self._allow_remote())
|
||||
|
||||
@override_settings(REMOTE_OCR_MODE="workflow_only")
|
||||
def test_workflow_only_mode_allows_remote_when_requested(self) -> None:
|
||||
self.assertTrue(self._allow_remote(remote_ocr=True))
|
||||
|
||||
|
||||
class TestAIIndex(DirectoriesMixin, TestCase):
|
||||
@override_settings(
|
||||
AI_ENABLED=True,
|
||||
|
||||
@@ -338,16 +338,13 @@ def check_deprecated_v2_ocr_env_vars(
|
||||
|
||||
|
||||
@register()
|
||||
def check_remote_ocr_mode(app_configs: Any, **kwargs: Any) -> list[Error]:
|
||||
# Import here because checks.py runs before the app registry is ready
|
||||
from paperless.models import RemoteOCRMode
|
||||
|
||||
valid_modes = {mode.value for mode in RemoteOCRMode}
|
||||
if settings.REMOTE_OCR_MODE not in valid_modes:
|
||||
def check_remote_parser_configured(app_configs: Any, **kwargs: Any) -> list[Error]:
|
||||
if settings.REMOTE_OCR_ENGINE == "azureai" and not (
|
||||
settings.REMOTE_OCR_ENDPOINT and settings.REMOTE_OCR_API_KEY
|
||||
):
|
||||
return [
|
||||
Error(
|
||||
f"PAPERLESS_REMOTE_OCR_MODE is set to {settings.REMOTE_OCR_MODE!r}, "
|
||||
f"expected one of {sorted(valid_modes)}.",
|
||||
"Azure AI remote parser requires endpoint and API key to be configured.",
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
@@ -9,7 +9,6 @@ from paperless.models import CleanChoices
|
||||
from paperless.models import ColorConvertChoices
|
||||
from paperless.models import ModeChoices
|
||||
from paperless.models import OutputTypeChoices
|
||||
from paperless.models import RemoteOCRMode
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
@@ -186,45 +185,6 @@ class GeneralConfig(BaseConfig):
|
||||
self.app_logo = app_config.app_logo.url if app_config.app_logo else None
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
class RemoteOCRConfig(BaseConfig):
|
||||
"""
|
||||
Settings for the remote (cloud) OCR parser
|
||||
"""
|
||||
|
||||
remote_ocr_engine: str | None = dataclasses.field(init=False)
|
||||
remote_ocr_api_key: str | None = dataclasses.field(init=False)
|
||||
remote_ocr_endpoint: str | None = dataclasses.field(init=False)
|
||||
remote_ocr_mode: RemoteOCRMode = dataclasses.field(init=False)
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
app_config = self._get_config_instance()
|
||||
|
||||
self.remote_ocr_engine = (
|
||||
app_config.remote_ocr_engine or settings.REMOTE_OCR_ENGINE
|
||||
)
|
||||
self.remote_ocr_api_key = (
|
||||
app_config.remote_ocr_api_key or settings.REMOTE_OCR_API_KEY
|
||||
)
|
||||
self.remote_ocr_endpoint = (
|
||||
app_config.remote_ocr_endpoint or settings.REMOTE_OCR_ENDPOINT
|
||||
)
|
||||
self.remote_ocr_mode = app_config.remote_ocr_mode or RemoteOCRMode(
|
||||
settings.REMOTE_OCR_MODE,
|
||||
)
|
||||
|
||||
@property
|
||||
def remote_ocr_by_default(self) -> bool:
|
||||
"""
|
||||
Whether every supported document goes to the remote engine.
|
||||
|
||||
When False the remote engine is used only for documents that
|
||||
explicitly asked for it, i.e. a workflow matched during consumption or
|
||||
the user ticked the box when reprocessing.
|
||||
"""
|
||||
return self.remote_ocr_mode == RemoteOCRMode.ALWAYS
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
class AIConfig(BaseConfig):
|
||||
"""
|
||||
|
||||
@@ -1,44 +0,0 @@
|
||||
# Generated by Django 5.2.16 on 2026-08-10 14:37
|
||||
|
||||
from django.db import migrations
|
||||
from django.db import models
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
dependencies = [
|
||||
("paperless", "0013_applicationconfiguration_llm_request_timeout"),
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.AddField(
|
||||
model_name="applicationconfiguration",
|
||||
name="remote_ocr_api_key",
|
||||
field=models.CharField(
|
||||
blank=True,
|
||||
max_length=1024,
|
||||
null=True,
|
||||
verbose_name="Sets the remote OCR API key",
|
||||
),
|
||||
),
|
||||
migrations.AddField(
|
||||
model_name="applicationconfiguration",
|
||||
name="remote_ocr_endpoint",
|
||||
field=models.CharField(
|
||||
blank=True,
|
||||
max_length=256,
|
||||
null=True,
|
||||
verbose_name="Sets the remote OCR endpoint",
|
||||
),
|
||||
),
|
||||
migrations.AddField(
|
||||
model_name="applicationconfiguration",
|
||||
name="remote_ocr_engine",
|
||||
field=models.CharField(
|
||||
blank=True,
|
||||
choices=[("azureai", "Azure AI Document Intelligence")],
|
||||
max_length=32,
|
||||
null=True,
|
||||
verbose_name="Sets the remote OCR engine",
|
||||
),
|
||||
),
|
||||
]
|
||||
@@ -1,27 +0,0 @@
|
||||
# Generated by Django 5.2.16 on 2026-08-10 15:43
|
||||
|
||||
from django.db import migrations
|
||||
from django.db import models
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
dependencies = [
|
||||
("paperless", "0014_applicationconfiguration_remote_ocr_api_key_and_more"),
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.AddField(
|
||||
model_name="applicationconfiguration",
|
||||
name="remote_ocr_mode",
|
||||
field=models.CharField(
|
||||
blank=True,
|
||||
choices=[
|
||||
("always", "All supported documents"),
|
||||
("workflow_only", "Only when a workflow enables it"),
|
||||
],
|
||||
max_length=32,
|
||||
null=True,
|
||||
verbose_name="Sets which documents are sent to the remote OCR engine",
|
||||
),
|
||||
),
|
||||
]
|
||||
@@ -74,23 +74,6 @@ class ColorConvertChoices(models.TextChoices):
|
||||
CMYK = ("CMYK", _("CMYK"))
|
||||
|
||||
|
||||
class RemoteOCREngine(models.TextChoices):
|
||||
"""
|
||||
Matches to PAPERLESS_REMOTE_OCR_ENGINE
|
||||
"""
|
||||
|
||||
AZURE_AI = ("azureai", _("Azure AI Document Intelligence"))
|
||||
|
||||
|
||||
class RemoteOCRMode(models.TextChoices):
|
||||
"""
|
||||
Matches to PAPERLESS_REMOTE_OCR_MODE
|
||||
"""
|
||||
|
||||
ALWAYS = ("always", _("All supported documents"))
|
||||
WORKFLOW_ONLY = ("workflow_only", _("Only when a workflow enables it"))
|
||||
|
||||
|
||||
class LLMEmbeddingBackend(models.TextChoices):
|
||||
OPENAI_LIKE = ("openai-like", _("OpenAI-compatible"))
|
||||
HUGGINGFACE = ("huggingface", _("Huggingface"))
|
||||
@@ -303,44 +286,6 @@ class ApplicationConfiguration(AbstractSingletonModel):
|
||||
null=True,
|
||||
)
|
||||
|
||||
"""
|
||||
Settings for the remote OCR parser
|
||||
"""
|
||||
|
||||
# PAPERLESS_REMOTE_OCR_ENGINE
|
||||
remote_ocr_engine = models.CharField(
|
||||
verbose_name=_("Sets the remote OCR engine"),
|
||||
blank=True,
|
||||
null=True,
|
||||
max_length=32,
|
||||
choices=RemoteOCREngine.choices,
|
||||
)
|
||||
|
||||
# PAPERLESS_REMOTE_OCR_API_KEY
|
||||
remote_ocr_api_key = models.CharField(
|
||||
verbose_name=_("Sets the remote OCR API key"),
|
||||
blank=True,
|
||||
null=True,
|
||||
max_length=1024,
|
||||
)
|
||||
|
||||
# PAPERLESS_REMOTE_OCR_ENDPOINT
|
||||
remote_ocr_endpoint = models.CharField(
|
||||
verbose_name=_("Sets the remote OCR endpoint"),
|
||||
blank=True,
|
||||
null=True,
|
||||
max_length=256,
|
||||
)
|
||||
|
||||
# PAPERLESS_REMOTE_OCR_MODE
|
||||
remote_ocr_mode = models.CharField(
|
||||
verbose_name=_("Sets which documents are sent to the remote OCR engine"),
|
||||
blank=True,
|
||||
null=True,
|
||||
max_length=32,
|
||||
choices=RemoteOCRMode.choices,
|
||||
)
|
||||
|
||||
"""
|
||||
AI related settings
|
||||
"""
|
||||
|
||||
@@ -134,11 +134,6 @@ class ParserProtocol(Protocol):
|
||||
Author or organisation name.
|
||||
url : str
|
||||
URL for documentation, source code, or issue tracker.
|
||||
|
||||
Parsers that send document content to a remote service should additionally
|
||||
set ``uses_remote_service = True`` so the registry can exclude them when
|
||||
remote processing has not been requested for a document. The attribute is
|
||||
optional so a parser that omits it is treated as fully local.
|
||||
"""
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
@@ -150,10 +145,6 @@ class ParserProtocol(Protocol):
|
||||
author: str
|
||||
url: str
|
||||
|
||||
# NOTE: uses_remote_service is not declared here, the registry reads it
|
||||
# with getattr(cls, ..., False) for backwards-compatibility with existing
|
||||
# parsers
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Class methods
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
@@ -334,8 +334,6 @@ class ParserRegistry:
|
||||
mime_type: str,
|
||||
filename: str,
|
||||
path: Path | None = None,
|
||||
*,
|
||||
allow_remote: bool = True,
|
||||
) -> type[ParserProtocol] | None:
|
||||
"""Return the best parser class for the given file, or None.
|
||||
|
||||
@@ -361,11 +359,6 @@ class ParserRegistry:
|
||||
path:
|
||||
Optional filesystem path to the file. Forwarded to each
|
||||
parser's score method.
|
||||
allow_remote:
|
||||
When False, parsers that declare ``uses_remote_service = True``
|
||||
are excluded from consideration, so a document is never sent to
|
||||
a remote service. Parsers that do not declare the attribute
|
||||
are treated as local and are always considered.
|
||||
|
||||
Returns
|
||||
-------
|
||||
@@ -381,13 +374,6 @@ class ParserRegistry:
|
||||
if mime_type not in parser_class.supported_mime_types():
|
||||
continue
|
||||
|
||||
if not allow_remote and getattr(
|
||||
parser_class,
|
||||
"uses_remote_service",
|
||||
False,
|
||||
):
|
||||
continue
|
||||
|
||||
score = parser_class.score(mime_type, filename, path)
|
||||
if score is None:
|
||||
continue
|
||||
|
||||
@@ -61,18 +61,6 @@ class RemoteEngineConfig:
|
||||
self.api_key = api_key
|
||||
self.endpoint = endpoint
|
||||
|
||||
@classmethod
|
||||
def from_app_config(cls) -> Self:
|
||||
"""Build the config from the app config, falling back to the env."""
|
||||
from paperless.config import RemoteOCRConfig
|
||||
|
||||
app_config = RemoteOCRConfig()
|
||||
return cls(
|
||||
engine=app_config.remote_ocr_engine,
|
||||
api_key=app_config.remote_ocr_api_key,
|
||||
endpoint=app_config.remote_ocr_endpoint,
|
||||
)
|
||||
|
||||
def engine_is_valid(self) -> bool:
|
||||
"""Return True when the engine is known and fully configured."""
|
||||
return (
|
||||
@@ -102,9 +90,6 @@ class RemoteDocumentParser:
|
||||
Maintainer name.
|
||||
url : str
|
||||
Issue tracker / source URL.
|
||||
uses_remote_service : bool
|
||||
Content is sent to a remote service, True so that the registry
|
||||
can skip this parser if remote processing was not requested.
|
||||
"""
|
||||
|
||||
name: str = "Paperless-ngx Remote OCR Parser"
|
||||
@@ -112,8 +97,6 @@ class RemoteDocumentParser:
|
||||
author: str = "Paperless-ngx Contributors"
|
||||
url: str = "https://github.com/paperless-ngx/paperless-ngx"
|
||||
|
||||
uses_remote_service: bool = True
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Class methods
|
||||
# ------------------------------------------------------------------
|
||||
@@ -162,7 +145,11 @@ class RemoteDocumentParser:
|
||||
20 when the remote engine is configured and the MIME type is
|
||||
supported, otherwise None.
|
||||
"""
|
||||
config = RemoteEngineConfig.from_app_config()
|
||||
config = RemoteEngineConfig(
|
||||
engine=settings.REMOTE_OCR_ENGINE,
|
||||
api_key=settings.REMOTE_OCR_API_KEY,
|
||||
endpoint=settings.REMOTE_OCR_ENDPOINT,
|
||||
)
|
||||
if not config.engine_is_valid():
|
||||
return None
|
||||
if mime_type not in _SUPPORTED_MIME_TYPES:
|
||||
@@ -257,7 +244,11 @@ class RemoteDocumentParser:
|
||||
Whether an archive copy is wanted. For PDFs, False skips the
|
||||
remote engine and uses locally-extracted text instead.
|
||||
"""
|
||||
config = RemoteEngineConfig.from_app_config()
|
||||
config = RemoteEngineConfig(
|
||||
engine=settings.REMOTE_OCR_ENGINE,
|
||||
api_key=settings.REMOTE_OCR_API_KEY,
|
||||
endpoint=settings.REMOTE_OCR_ENDPOINT,
|
||||
)
|
||||
|
||||
if not config.engine_is_valid():
|
||||
logger.warning(
|
||||
|
||||
@@ -219,13 +219,6 @@ class ApplicationConfigurationSerializer(
|
||||
allow_null=True,
|
||||
max_length=1024,
|
||||
)
|
||||
remote_ocr_api_key = ObfuscatedPasswordField(
|
||||
required=False,
|
||||
allow_null=True,
|
||||
max_length=1024,
|
||||
)
|
||||
|
||||
OBFUSCATED_FIELDS = ("llm_api_key", "remote_ocr_api_key")
|
||||
|
||||
def run_validation(self, data):
|
||||
# Empty strings treated as None to avoid unexpected behavior
|
||||
@@ -237,13 +230,11 @@ class ApplicationConfigurationSerializer(
|
||||
data["language"] = None
|
||||
if "llm_output_language" in data and data["llm_output_language"] == "":
|
||||
data["llm_output_language"] = None
|
||||
for field in self.OBFUSCATED_FIELDS:
|
||||
if field in data and data[field] is not None:
|
||||
if data[field] == "":
|
||||
data[field] = None
|
||||
# Not a real value, don't overwrite the stored one
|
||||
elif len(data[field].replace("*", "")) == 0:
|
||||
del data[field]
|
||||
if "llm_api_key" in data and data["llm_api_key"] is not None:
|
||||
if data["llm_api_key"] == "":
|
||||
data["llm_api_key"] = None
|
||||
elif len(data["llm_api_key"].replace("*", "")) == 0:
|
||||
del data["llm_api_key"]
|
||||
return super().run_validation(data)
|
||||
|
||||
def update(self, instance, validated_data):
|
||||
|
||||
@@ -1197,7 +1197,6 @@ WEBHOOKS_ALLOW_INTERNAL_REQUESTS = get_bool_from_env(
|
||||
REMOTE_OCR_ENGINE = os.getenv("PAPERLESS_REMOTE_OCR_ENGINE")
|
||||
REMOTE_OCR_API_KEY = os.getenv("PAPERLESS_REMOTE_OCR_API_KEY")
|
||||
REMOTE_OCR_ENDPOINT = os.getenv("PAPERLESS_REMOTE_OCR_ENDPOINT")
|
||||
REMOTE_OCR_MODE = os.getenv("PAPERLESS_REMOTE_OCR_MODE", "always")
|
||||
|
||||
################################################################################
|
||||
# AI Settings #
|
||||
|
||||
@@ -21,7 +21,6 @@ from unittest.mock import Mock
|
||||
import pytest
|
||||
|
||||
from documents.parsers import ParseError
|
||||
from paperless.models import ApplicationConfiguration
|
||||
from paperless.parsers import ParserContext
|
||||
from paperless.parsers import ParserProtocol
|
||||
from paperless.parsers.remote import RemoteDocumentParser
|
||||
@@ -34,10 +33,6 @@ if TYPE_CHECKING:
|
||||
from pytest_mock import MockerFixture
|
||||
|
||||
|
||||
# Remote ocr config from ApplicationConfiguration needs DB access
|
||||
pytestmark = pytest.mark.django_db
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Module-local fixtures
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -232,18 +227,6 @@ class TestRemoteParserScore:
|
||||
score = RemoteDocumentParser.score("application/pdf", "doc.pdf")
|
||||
assert score is not None and score > 10
|
||||
|
||||
@pytest.mark.usefixtures("no_engine_settings")
|
||||
def test_score_uses_app_config_when_env_unset(self) -> None:
|
||||
"""The app config alone is enough to activate the parser."""
|
||||
config = ApplicationConfiguration.objects.first()
|
||||
assert config is not None
|
||||
config.remote_ocr_engine = "azureai"
|
||||
config.remote_ocr_api_key = "app-config-key"
|
||||
config.remote_ocr_endpoint = "https://config.cognitiveservices.azure.com"
|
||||
config.save()
|
||||
|
||||
assert RemoteDocumentParser.score("application/pdf", "doc.pdf") == 20
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Properties
|
||||
|
||||
@@ -1277,8 +1277,6 @@ class TestParserFileTypes:
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
# Remote ocr config from ApplicationConfiguration needs DB access
|
||||
@pytest.mark.django_db
|
||||
class TestRasterisedDocumentParserRegistry:
|
||||
def test_registered_in_defaults(self) -> None:
|
||||
from paperless.parsers.registry import ParserRegistry
|
||||
|
||||
@@ -15,7 +15,7 @@ from paperless.checks import audit_log_check
|
||||
from paperless.checks import binaries_check
|
||||
from paperless.checks import check_default_language_available
|
||||
from paperless.checks import check_deprecated_db_settings
|
||||
from paperless.checks import check_remote_ocr_mode
|
||||
from paperless.checks import check_remote_parser_configured
|
||||
from paperless.checks import check_v3_minimum_upgrade_version
|
||||
from paperless.checks import debug_mode_check
|
||||
from paperless.checks import paths_check
|
||||
@@ -631,21 +631,29 @@ class TestV3MinimumUpgradeVersionCheck:
|
||||
assert check_v3_minimum_upgrade_version(None) == []
|
||||
|
||||
|
||||
class TestRemoteOCRModeCheck:
|
||||
def test_valid_mode(self, settings: SettingsWrapper) -> None:
|
||||
settings.REMOTE_OCR_MODE = "workflow_only"
|
||||
|
||||
msgs = check_remote_ocr_mode(None)
|
||||
class TestRemoteParserChecks:
|
||||
def test_no_engine(self, settings: SettingsWrapper) -> None:
|
||||
settings.REMOTE_OCR_ENGINE = None
|
||||
msgs = check_remote_parser_configured(None)
|
||||
|
||||
assert len(msgs) == 0
|
||||
|
||||
def test_invalid_mode(self, settings: SettingsWrapper) -> None:
|
||||
settings.REMOTE_OCR_MODE = "sometimes"
|
||||
def test_azure_no_endpoint(self, settings: SettingsWrapper) -> None:
|
||||
|
||||
msgs = check_remote_ocr_mode(None)
|
||||
settings.REMOTE_OCR_ENGINE = "azureai"
|
||||
settings.REMOTE_OCR_API_KEY = "somekey"
|
||||
settings.REMOTE_OCR_ENDPOINT = None
|
||||
|
||||
msgs = check_remote_parser_configured(None)
|
||||
|
||||
assert len(msgs) == 1
|
||||
assert "PAPERLESS_REMOTE_OCR_MODE is set to 'sometimes'" in msgs[0].msg
|
||||
|
||||
msg = msgs[0]
|
||||
|
||||
assert (
|
||||
"Azure AI remote parser requires endpoint and API key to be configured."
|
||||
in msg.msg
|
||||
)
|
||||
|
||||
|
||||
class TestTesseractChecks:
|
||||
|
||||
@@ -468,124 +468,6 @@ class TestParserRegistryGetParserForFile:
|
||||
assert result is AcceptingBuiltin
|
||||
|
||||
|
||||
class TestParserRegistryRemoteParsers:
|
||||
"""Verify the allow_remote filter in ParserRegistry.get_parser_for_file()."""
|
||||
|
||||
@staticmethod
|
||||
def _remote_parser_cls() -> type:
|
||||
class RemoteParser:
|
||||
name = "remote"
|
||||
version = "1.0"
|
||||
author = "A"
|
||||
url = "https://example.com/remote"
|
||||
uses_remote_service = True
|
||||
|
||||
@classmethod
|
||||
def supported_mime_types(cls):
|
||||
return {"text/plain": ".txt"}
|
||||
|
||||
@classmethod
|
||||
def score(cls, mime_type, filename, path=None):
|
||||
return 20
|
||||
|
||||
return RemoteParser
|
||||
|
||||
def test_remote_parser_wins_when_remote_allowed(
|
||||
self,
|
||||
dummy_parser_cls: type,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN: A remote parser scoring 20 and a local parser scoring 10.
|
||||
WHEN: get_parser_for_file() is called with allow_remote=True.
|
||||
THEN: The remote parser is returned.
|
||||
"""
|
||||
remote_parser_cls = self._remote_parser_cls()
|
||||
registry = ParserRegistry()
|
||||
registry.register_builtin(dummy_parser_cls)
|
||||
registry.register_builtin(remote_parser_cls)
|
||||
|
||||
result = registry.get_parser_for_file(
|
||||
"text/plain",
|
||||
"readme.txt",
|
||||
allow_remote=True,
|
||||
)
|
||||
assert result is remote_parser_cls
|
||||
|
||||
def test_remote_parser_skipped_when_remote_not_allowed(
|
||||
self,
|
||||
dummy_parser_cls: type,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN: A remote parser scoring 20 and a local parser scoring 10.
|
||||
WHEN: get_parser_for_file() is called with allow_remote=False.
|
||||
THEN: The local parser is returned despite its lower score.
|
||||
"""
|
||||
registry = ParserRegistry()
|
||||
registry.register_builtin(dummy_parser_cls)
|
||||
registry.register_builtin(self._remote_parser_cls())
|
||||
|
||||
result = registry.get_parser_for_file(
|
||||
"text/plain",
|
||||
"readme.txt",
|
||||
allow_remote=False,
|
||||
)
|
||||
assert result is dummy_parser_cls
|
||||
|
||||
def test_no_parser_when_only_remote_available_and_not_allowed(self) -> None:
|
||||
"""
|
||||
GIVEN: A registry whose only candidate declares uses_remote_service.
|
||||
WHEN: get_parser_for_file() is called with allow_remote=False.
|
||||
THEN: None is returned — the remote parser is never used as a
|
||||
fallback when remote processing was not requested.
|
||||
"""
|
||||
registry = ParserRegistry()
|
||||
registry.register_builtin(self._remote_parser_cls())
|
||||
|
||||
result = registry.get_parser_for_file(
|
||||
"text/plain",
|
||||
"readme.txt",
|
||||
allow_remote=False,
|
||||
)
|
||||
assert result is None
|
||||
|
||||
def test_parser_without_attribute_treated_as_local(
|
||||
self,
|
||||
dummy_parser_cls: type,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN: A third-party parser predating uses_remote_service, so it does
|
||||
not declare the attribute at all.
|
||||
WHEN: get_parser_for_file() is called with allow_remote=False.
|
||||
THEN: It is still considered, i.e. treated as fully local, rather
|
||||
than raising AttributeError.
|
||||
"""
|
||||
assert not hasattr(dummy_parser_cls, "uses_remote_service")
|
||||
|
||||
registry = ParserRegistry()
|
||||
registry.register_builtin(dummy_parser_cls)
|
||||
|
||||
result = registry.get_parser_for_file(
|
||||
"text/plain",
|
||||
"readme.txt",
|
||||
allow_remote=False,
|
||||
)
|
||||
assert result is dummy_parser_cls
|
||||
|
||||
def test_remote_allowed_by_default(self) -> None:
|
||||
"""
|
||||
GIVEN: A registry containing only a remote parser.
|
||||
WHEN: get_parser_for_file() is called without allow_remote.
|
||||
THEN: The remote parser is returned — callers that do not opt in to
|
||||
the filter keep the previous behaviour.
|
||||
"""
|
||||
remote_parser_cls = self._remote_parser_cls()
|
||||
registry = ParserRegistry()
|
||||
registry.register_builtin(remote_parser_cls)
|
||||
|
||||
result = registry.get_parser_for_file("text/plain", "readme.txt")
|
||||
assert result is remote_parser_cls
|
||||
|
||||
|
||||
class TestDiscover:
|
||||
"""Verify entrypoint discovery in ParserRegistry.discover()."""
|
||||
|
||||
|
||||
@@ -1,113 +0,0 @@
|
||||
"""Tests for RemoteOCRConfig precedence between app config and Django settings."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
from django.test import override_settings
|
||||
|
||||
from paperless.config import RemoteOCRConfig
|
||||
from paperless.models import RemoteOCRMode
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def null_app_config(mocker) -> MagicMock:
|
||||
"""Mock ApplicationConfiguration with all fields None → falls back to Django settings."""
|
||||
return mocker.MagicMock(
|
||||
remote_ocr_engine=None,
|
||||
remote_ocr_api_key=None,
|
||||
remote_ocr_endpoint=None,
|
||||
remote_ocr_mode=None,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def make_remote_ocr_config(mocker):
|
||||
def _make(app_config, **django_settings_overrides):
|
||||
mocker.patch(
|
||||
"paperless.config.BaseConfig._get_config_instance",
|
||||
return_value=app_config,
|
||||
)
|
||||
with override_settings(**django_settings_overrides):
|
||||
return RemoteOCRConfig()
|
||||
|
||||
return _make
|
||||
|
||||
|
||||
class TestRemoteOCRConfig:
|
||||
def test_falls_back_to_settings(
|
||||
self,
|
||||
make_remote_ocr_config,
|
||||
null_app_config,
|
||||
) -> None:
|
||||
cfg = make_remote_ocr_config(
|
||||
null_app_config,
|
||||
REMOTE_OCR_ENGINE="azureai",
|
||||
REMOTE_OCR_API_KEY="env-key",
|
||||
REMOTE_OCR_ENDPOINT="https://env.cognitiveservices.azure.com",
|
||||
REMOTE_OCR_MODE=RemoteOCRMode.WORKFLOW_ONLY,
|
||||
)
|
||||
assert cfg.remote_ocr_engine == "azureai"
|
||||
assert cfg.remote_ocr_api_key == "env-key"
|
||||
assert cfg.remote_ocr_endpoint == "https://env.cognitiveservices.azure.com"
|
||||
assert cfg.remote_ocr_mode == RemoteOCRMode.WORKFLOW_ONLY
|
||||
|
||||
def test_app_config_takes_precedence(
|
||||
self,
|
||||
make_remote_ocr_config,
|
||||
mocker,
|
||||
) -> None:
|
||||
app_config = mocker.MagicMock(
|
||||
remote_ocr_engine="azureai",
|
||||
remote_ocr_api_key="db-key",
|
||||
remote_ocr_endpoint="https://db.cognitiveservices.azure.com",
|
||||
remote_ocr_mode=RemoteOCRMode.WORKFLOW_ONLY,
|
||||
)
|
||||
cfg = make_remote_ocr_config(
|
||||
app_config,
|
||||
REMOTE_OCR_ENGINE=None,
|
||||
REMOTE_OCR_API_KEY="env-key",
|
||||
REMOTE_OCR_ENDPOINT="https://env.cognitiveservices.azure.com",
|
||||
REMOTE_OCR_MODE=RemoteOCRMode.ALWAYS,
|
||||
)
|
||||
assert cfg.remote_ocr_engine == "azureai"
|
||||
assert cfg.remote_ocr_api_key == "db-key"
|
||||
assert cfg.remote_ocr_endpoint == "https://db.cognitiveservices.azure.com"
|
||||
assert cfg.remote_ocr_mode == RemoteOCRMode.WORKFLOW_ONLY
|
||||
|
||||
def test_unset_everywhere(
|
||||
self,
|
||||
make_remote_ocr_config,
|
||||
null_app_config,
|
||||
) -> None:
|
||||
cfg = make_remote_ocr_config(
|
||||
null_app_config,
|
||||
REMOTE_OCR_ENGINE=None,
|
||||
REMOTE_OCR_API_KEY=None,
|
||||
REMOTE_OCR_ENDPOINT=None,
|
||||
)
|
||||
assert cfg.remote_ocr_engine is None
|
||||
assert cfg.remote_ocr_api_key is None
|
||||
assert cfg.remote_ocr_endpoint is None
|
||||
|
||||
|
||||
class TestRemoteOCRByDefault:
|
||||
def test_always_mode(self, make_remote_ocr_config, null_app_config) -> None:
|
||||
cfg = make_remote_ocr_config(
|
||||
null_app_config,
|
||||
REMOTE_OCR_MODE=RemoteOCRMode.ALWAYS,
|
||||
)
|
||||
|
||||
assert cfg.remote_ocr_by_default is True
|
||||
|
||||
def test_workflow_only_mode(self, make_remote_ocr_config, null_app_config) -> None:
|
||||
cfg = make_remote_ocr_config(
|
||||
null_app_config,
|
||||
REMOTE_OCR_MODE=RemoteOCRMode.WORKFLOW_ONLY,
|
||||
)
|
||||
|
||||
assert cfg.remote_ocr_by_default is False
|
||||
@@ -24,6 +24,7 @@ from paperless_ai.embedding import get_configured_model_name
|
||||
from paperless_ai.embedding import get_embedding_model
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from django.db.models import QuerySet
|
||||
from llama_index.core.schema import BaseNode
|
||||
|
||||
from paperless_ai.vector_store import PaperlessSqliteVecVectorStore
|
||||
@@ -252,6 +253,20 @@ def _safe_related_name(document: Document, field: str) -> str | None:
|
||||
return related.name if related else None
|
||||
|
||||
|
||||
def _document_index_queryset() -> "QuerySet[Document]":
|
||||
"""Document queryset with every relation build_document_node() /
|
||||
build_llm_index_text() touches -- correspondent, document_type,
|
||||
storage_path, tags, notes, custom_fields__field -- pre-loaded, so
|
||||
indexing one document costs a fixed handful of queries regardless of
|
||||
its tag or custom field count, instead of one query per related object.
|
||||
"""
|
||||
return Document.objects.select_related(
|
||||
"correspondent",
|
||||
"document_type",
|
||||
"storage_path",
|
||||
).prefetch_related("tags", "notes", "custom_fields__field")
|
||||
|
||||
|
||||
def build_document_node(
|
||||
document: Document,
|
||||
*,
|
||||
@@ -425,11 +440,7 @@ def update_llm_index(
|
||||
"Skipping LLM index update: migration check deferred; "
|
||||
"will retry next run."
|
||||
)
|
||||
documents = Document.objects.select_related(
|
||||
"correspondent",
|
||||
"document_type",
|
||||
"storage_path",
|
||||
).prefetch_related("tags", "notes", "custom_fields__field")
|
||||
documents = _document_index_queryset()
|
||||
no_documents = not documents.exists()
|
||||
|
||||
# Fast exit before touching config: nothing to index and no existing index.
|
||||
@@ -494,6 +505,21 @@ def update_llm_index(
|
||||
def llm_index_add_or_update_document(document: Document):
|
||||
"""Add or atomically replace a document's chunks in the index."""
|
||||
config = AIConfig()
|
||||
document_id = document.id
|
||||
# Re-fetch with the same select_related/prefetch_related shape as the
|
||||
# bulk path (update_llm_index()) uses: the caller's ``document`` instance
|
||||
# (e.g. straight off a signal) has none of that loaded, and
|
||||
# build_document_node()/build_llm_index_text() touch
|
||||
# correspondent/document_type/storage_path/tags/notes/custom_fields__field
|
||||
# -- without prefetching, that's one query per related object, including
|
||||
# one per custom field instance.
|
||||
document = _document_index_queryset().filter(pk=document_id).first()
|
||||
if document is None:
|
||||
logger.info(
|
||||
"Skipping LLM index update for document %s: it no longer exists.",
|
||||
document_id,
|
||||
)
|
||||
return
|
||||
new_nodes = build_document_node(
|
||||
document,
|
||||
chunk_size=config.llm_embedding_chunk_size,
|
||||
|
||||
@@ -735,8 +735,71 @@ class TestLlmIndexAddOrUpdateDocumentEmptyContent:
|
||||
)
|
||||
mock_load = mocker.patch("paperless_ai.indexing.load_or_build_index")
|
||||
|
||||
doc = MagicMock(spec=Document)
|
||||
doc.id = 42
|
||||
doc = DocumentFactory.create()
|
||||
# Must not raise
|
||||
indexing.llm_index_add_or_update_document(doc)
|
||||
|
||||
mock_load.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.django_db
|
||||
class TestLlmIndexAddOrUpdateDocumentPrefetch:
|
||||
"""llm_index_add_or_update_document must prefetch the relations
|
||||
build_document_node()/build_llm_index_text() touch, not re-query per
|
||||
tag/note/custom field.
|
||||
"""
|
||||
|
||||
def test_query_count_does_not_scale_with_custom_field_count(
|
||||
self,
|
||||
temp_llm_index_dir: Path,
|
||||
mock_embed_model: FakeEmbedding,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN a document with several custom fields, a note, and a tag
|
||||
WHEN it is incrementally indexed via llm_index_add_or_update_document
|
||||
THEN the query count stays flat instead of growing with the number
|
||||
of custom fields -- a regression here would add one query per
|
||||
custom field instance (instance.field.name unprefetched), see
|
||||
build_llm_index_text().
|
||||
"""
|
||||
doc = DocumentFactory.create()
|
||||
Note.objects.create(document=doc, note="a note")
|
||||
for i in range(5):
|
||||
field = CustomField.objects.create(
|
||||
name=f"Field {i}",
|
||||
data_type=CustomField.FieldDataType.STRING,
|
||||
)
|
||||
CustomFieldInstance.objects.create(
|
||||
document=doc,
|
||||
field=field,
|
||||
value_text="value",
|
||||
)
|
||||
|
||||
with CaptureQueriesContext(connection) as ctx:
|
||||
indexing.llm_index_add_or_update_document(doc)
|
||||
|
||||
# Flat regardless of custom field count -- an unprefetched
|
||||
# custom_fields__field would add one query per instance (5 here) on
|
||||
# top of this budget.
|
||||
assert len(ctx.captured_queries) <= 10
|
||||
|
||||
def test_skips_write_when_document_no_longer_exists(
|
||||
self,
|
||||
temp_llm_index_dir: Path,
|
||||
mock_embed_model: FakeEmbedding,
|
||||
mocker: pytest_mock.MockerFixture,
|
||||
) -> None:
|
||||
"""
|
||||
GIVEN a document that has been deleted since the caller looked it up
|
||||
(e.g. a race between a signal firing and its async task running)
|
||||
WHEN llm_index_add_or_update_document is called with that stale instance
|
||||
THEN it skips the write instead of raising Document.DoesNotExist
|
||||
"""
|
||||
doc = DocumentFactory.create()
|
||||
doc_id = doc.pk
|
||||
Document.objects.filter(pk=doc_id).delete()
|
||||
mock_load = mocker.patch("paperless_ai.indexing.load_or_build_index")
|
||||
|
||||
# Must not raise
|
||||
indexing.llm_index_add_or_update_document(doc)
|
||||
|
||||
@@ -793,8 +856,7 @@ class TestLlmIndexLocking:
|
||||
return_value=[mock_node],
|
||||
)
|
||||
|
||||
doc = MagicMock(spec=Document)
|
||||
doc.id = 1
|
||||
doc = DocumentFactory.create()
|
||||
indexing.llm_index_add_or_update_document(doc)
|
||||
|
||||
mock_store.upsert_document.assert_called_once()
|
||||
@@ -825,8 +887,7 @@ class TestLlmIndexLocking:
|
||||
return_value=[mock_node],
|
||||
)
|
||||
|
||||
doc = MagicMock(spec=Document)
|
||||
doc.id = 1
|
||||
doc = DocumentFactory.create()
|
||||
indexing.llm_index_add_or_update_document(doc)
|
||||
|
||||
mock_store.upsert_document.assert_not_called()
|
||||
@@ -863,8 +924,7 @@ class TestLlmIndexLocking:
|
||||
return_value=[mock_node],
|
||||
)
|
||||
|
||||
doc = MagicMock(spec=Document)
|
||||
doc.id = 1
|
||||
doc = DocumentFactory.create()
|
||||
indexing.llm_index_add_or_update_document(doc)
|
||||
|
||||
mock_store.upsert_document.assert_not_called()
|
||||
|
||||
Reference in New Issue
Block a user