Files
paperless-ngx/src/documents/data_models.py
T
jurassicparkicecreamandshamoon 62cc31fbdb Feature: store barcode contents, list and search them (#14276)
* Feature: store barcode contents, list and search them

New setting PAPERLESS_CONSUMER_STORE_BARCODE_VALUES (off by default)
stores all barcodes found during consumption with the document: page,
type and content. They are listed on the metadata tab with a copy
button, returned by the documents API and searchable with barcodes:
in the advanced search. Versions keep their own barcodes, reprocessing
reads them again. Refs #9898

* Tests: cover the remaining barcode branches

Covers unchanged and failing barcode reads on reprocessing, unsupported files and DocumentBarcode.__str__, and uses toHaveLength in the barcode list spec as suggested by SonarCloud.

* Address review: keep barcodes when they can't be read, format choices

- Reprocessing and new versions keep the stored barcodes when the file
  can't be scanned or the scan fails, and replace them atomically.
- Format is a TextChoices of the zxing-cpp formats, with a test.
- Shared scan code, latest_version helper, TypedDict, serializer reuse.
- Barcodes in the split manifest, export/import tests, pytest-style tests.

* Barcode tests: TIFF reprocess, format check both ways, fixtures

- Reprocessing a TIFF with TIFF support off keeps the stored barcodes.
- The format test also fails when zxing-cpp drops a format.
- Fixtures in place of the sample dir mixin, plugin-level disabled test.
- Format labels aren't translated, OpenAPI enum named BarcodeFormatEnum.

* Review: module-level zxing reader, shorter barcode docs

- read_barcodes_zxing is a module-level function used by scan_pdf.
- Drop the trivial __str__ test, mark it no cover.
- Shorten the barcode docs and remove the duplicate in configuration.md.

* Fix header

* Use utility class

* Return barcodes from the metadata endpoint only

Drop the barcodes field and its prefetches from the document serializer,
as agreed in the review. Also remove the now empty component stylesheet.

---------

Co-authored-by: shamoon <4887959+shamoon@users.noreply.github.com>
2026-10-05 16:25:19 +00:00

225 lines
7.4 KiB
Python

import dataclasses
import datetime
from enum import IntEnum
from pathlib import Path
from typing import TypedDict
import magic
from guardian.shortcuts import get_groups_with_perms
from guardian.shortcuts import get_users_with_perms
class StoredBarcode(TypedDict):
"""
A detected barcode as it is stored with a document
"""
page: int # 1-indexed
value: str
format: str # a DocumentBarcode.Format value
@dataclasses.dataclass
class DocumentMetadataOverrides:
"""
Manages overrides for document fields which normally would
be set from content or matching. All fields default to None,
meaning no override is happening
"""
filename: str | None = None
title: str | None = None
correspondent_id: int | None = None
document_type_id: int | None = None
tag_ids: list[int] | None = None
storage_path_id: int | None = None
created: datetime.date | None = None
asn: int | None = None
owner_id: int | None = None
view_users: list[int] | None = None
view_groups: list[int] | None = None
change_users: list[int] | None = None
change_groups: list[int] | None = None
custom_fields: dict | None = None
skip_asn_if_exists: bool = False
version_label: str | None = None
actor_id: int | None = None
remote_ocr: bool = False
barcodes: list[StoredBarcode] | None = None
def update(self, other: "DocumentMetadataOverrides") -> "DocumentMetadataOverrides":
"""
Merges two DocumentMetadataOverrides objects such that object B's overrides
are applied to object A or merged if multiple are accepted.
The update is an in-place modification of self
"""
# only if empty
if other.title is not None:
self.title = other.title
if other.correspondent_id is not None:
self.correspondent_id = other.correspondent_id
if other.document_type_id is not None:
self.document_type_id = other.document_type_id
if other.storage_path_id is not None:
self.storage_path_id = other.storage_path_id
if other.owner_id is not None:
self.owner_id = other.owner_id
if other.actor_id is not None:
self.actor_id = other.actor_id
if other.skip_asn_if_exists:
self.skip_asn_if_exists = True
if other.remote_ocr:
self.remote_ocr = True
if other.version_label is not None:
self.version_label = other.version_label
# merge
if self.tag_ids is None:
self.tag_ids = other.tag_ids
elif other.tag_ids is not None:
self.tag_ids.extend(other.tag_ids)
self.tag_ids = list(set(self.tag_ids))
if self.view_users is None:
self.view_users = other.view_users
elif other.view_users is not None:
self.view_users.extend(other.view_users)
self.view_users = list(set(self.view_users))
if self.view_groups is None:
self.view_groups = other.view_groups
elif other.view_groups is not None:
self.view_groups.extend(other.view_groups)
self.view_groups = list(set(self.view_groups))
if self.change_users is None:
self.change_users = other.change_users
elif other.change_users is not None:
self.change_users.extend(other.change_users)
self.change_users = list(set(self.change_users))
if self.change_groups is None:
self.change_groups = other.change_groups
elif other.change_groups is not None:
self.change_groups.extend(other.change_groups)
self.change_groups = list(set(self.change_groups))
if self.custom_fields is None:
self.custom_fields = other.custom_fields
elif other.custom_fields is not None:
self.custom_fields.update(other.custom_fields)
return self
@staticmethod
def from_document(doc) -> "DocumentMetadataOverrides":
"""
Fills in the overrides from a document object
"""
overrides = DocumentMetadataOverrides()
overrides.title = doc.title
overrides.correspondent_id = doc.correspondent.id if doc.correspondent else None
overrides.document_type_id = doc.document_type.id if doc.document_type else None
overrides.storage_path_id = doc.storage_path.id if doc.storage_path else None
overrides.owner_id = doc.owner.id if doc.owner else None
overrides.tag_ids = list(doc.tags.values_list("id", flat=True))
overrides.created = doc.created
overrides.view_users = list(
get_users_with_perms(
doc,
only_with_perms_in=["view_document"],
).values_list("id", flat=True),
)
overrides.change_users = list(
get_users_with_perms(
doc,
only_with_perms_in=["change_document"],
).values_list("id", flat=True),
)
overrides.custom_fields = {
custom_field.field.id: custom_field.value
for custom_field in doc.custom_fields.all()
}
groups_with_perms = get_groups_with_perms(
doc,
attach_perms=True,
)
overrides.view_groups = [
group.id
for group in groups_with_perms
if "view_document" in groups_with_perms[group]
]
overrides.change_groups = [
group.id
for group in groups_with_perms
if "change_document" in groups_with_perms[group]
]
return overrides
class DocumentSource(IntEnum):
"""
The source of an incoming document. May have other uses in the future
"""
ConsumeFolder = 1
ApiUpload = 2
MailFetch = 3
WebUI = 4
@dataclasses.dataclass
class ConsumableDocument:
"""
Encapsulates an incoming document, either from consume folder, API upload
or mail fetching and certain useful operations on it.
"""
source: DocumentSource
original_file: Path
root_document_id: int | None = None
original_path: Path | None = None
mailrule_id: int | None = None
mime_type: str = dataclasses.field(init=False, default=None)
def __post_init__(self) -> None:
"""
After a dataclass is initialized, this is called to finalize some data
1. Make sure the original path is an absolute, fully qualified path
2. Get the mime type of the file
"""
# Always fully qualify the path first thing
# Just in case, convert to a path if it's a str
self.original_file = Path(self.original_file).resolve()
# Get the file type once at init
# Note this function isn't called when the object is unpickled
self.mime_type = magic.from_file(self.original_file, mime=True)
class ConsumeFileDuplicateResult(TypedDict):
"""Returned by consume_file when the file is rejected as a duplicate."""
duplicate_of: int
duplicate_in_trash: bool
class ConsumeFileSuccessResult(TypedDict):
"""Returned by consume_file when the document is created successfully."""
document_id: int
class ConsumeFileStoppedResult(TypedDict):
"""Returned by consume_file when a plugin raises StopConsumeTaskError.
Examples: barcode split dispatched child tasks, double-sided scan waiting
for the second half, workflow deleted the document during consumption.
"""
reason: str