diff --git a/docs/advanced_usage.md b/docs/advanced_usage.md index 90dc50c65..59cfa5a0b 100644 --- a/docs/advanced_usage.md +++ b/docs/advanced_usage.md @@ -1010,6 +1010,19 @@ documents to both separate and categorize them in a single operation. **Example:** A 6-page scan with TAG:invoice on page 3 and TAG:receipt on page 5 will create three documents: pages 1-2 (no tags), pages 3-4 (tagged "invoice"), and pages 5-6 (tagged "receipt"). +### Barcode Contents {#barcode-contents} + +By default, Paperless only uses barcodes for splitting, ASNs and tags. With +[`PAPERLESS_CONSUMER_STORE_BARCODE_VALUES`](configuration.md#PAPERLESS_CONSUMER_STORE_BARCODE_VALUES) +enabled, it stores the content of every barcode with the document, e.g. payment codes or QR codes. + +- Barcodes are listed on the **Metadata** tab with page, type and content, and can be copied. +- The API returns them in the `barcodes` field of `/api/documents/{id}/metadata/`. +- They can be [searched](usage.md#searching-barcodes), e.g. `barcodes:DE89370400440532013000`. +- Only the first [`PAPERLESS_CONSUMER_BARCODE_MAX_PAGES`](configuration.md#PAPERLESS_CONSUMER_BARCODE_MAX_PAGES) + pages are scanned. Reprocessing reads the barcodes of existing documents. +- Each version keeps its own barcodes, and the newest version's are shown and searched. + ## Automatic collation of double-sided documents {#collate} !!! note diff --git a/docs/configuration.md b/docs/configuration.md index 2f94c41aa..3257a815c 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -1796,6 +1796,13 @@ assigns or creates tags if a properly formatted barcode is detected. Defaults to false. +#### [`PAPERLESS_CONSUMER_STORE_BARCODE_VALUES=`](#PAPERLESS_CONSUMER_STORE_BARCODE_VALUES) {#PAPERLESS_CONSUMER_STORE_BARCODE_VALUES} + +: Stores the content of every barcode found during consumption, see +[Barcode Contents](advanced_usage.md#barcode-contents). + + Defaults to false. + ## Audit Trail #### [`PAPERLESS_AUDIT_LOG_ENABLED=`](#PAPERLESS_AUDIT_LOG_ENABLED) {#PAPERLESS_AUDIT_LOG_ENABLED} diff --git a/docs/usage.md b/docs/usage.md index 12cdaa921..a5bd40140 100644 --- a/docs/usage.md +++ b/docs/usage.md @@ -1061,6 +1061,20 @@ notes.user:alice notes.note:insurance The bare `notes:` prefix is shorthand for `notes.note:`. +#### Searching barcodes + +If [barcode contents are stored](advanced_usage.md#barcode-contents), they can be searched by +content or type, but only with a field name: + +``` +barcodes.value:DE89370400440532013000 +barcodes.format:qrcode +barcodes:wifi barcodes:guest +``` + +`barcodes:` is shorthand for `barcodes.value:`. Separators are stripped, so each part of e.g. +`WIFI:S:Guest;P:secret;;` can be searched on its own. + All of these can be combined. Syntax not described here may not work as expected, and an unknown field name is searched as ordinary text. !!! note diff --git a/src-ui/src/app/components/document-detail/document-barcodes/document-barcodes.component.html b/src-ui/src/app/components/document-detail/document-barcodes/document-barcodes.component.html new file mode 100644 index 000000000..b3e67736f --- /dev/null +++ b/src-ui/src/app/components/document-detail/document-barcodes/document-barcodes.component.html @@ -0,0 +1,46 @@ + + + + + + + + + + + @for (barcode of barcodes(); track $index) { + + + + + + + } + +
PageTypeContent
{{ barcode.page }}{{ barcode.format }} + @if (isLink(barcode.value)) { + {{ barcode.value }} + } @else { + {{ barcode.value }} + } + + +
diff --git a/src-ui/src/app/components/document-detail/document-barcodes/document-barcodes.component.spec.ts b/src-ui/src/app/components/document-detail/document-barcodes/document-barcodes.component.spec.ts new file mode 100644 index 000000000..6b40b7bc7 --- /dev/null +++ b/src-ui/src/app/components/document-detail/document-barcodes/document-barcodes.component.spec.ts @@ -0,0 +1,73 @@ +import { Clipboard } from '@angular/cdk/clipboard' +import { ComponentFixture, TestBed } from '@angular/core/testing' +import { NgxBootstrapIconsModule, allIcons } from 'ngx-bootstrap-icons' +import { DocumentBarcodesComponent } from './document-barcodes.component' + +const barcodes = [ + { page: 1, value: 'ASN00123', format: 'Code128' }, + { page: 2, value: 'https://example.com/invoice/4711', format: 'QRCode' }, + { page: 2, value: 'javascript:alert(1)', format: 'QRCode' }, +] + +describe('DocumentBarcodesComponent', () => { + let component: DocumentBarcodesComponent + let fixture: ComponentFixture + let clipboard: Clipboard + + beforeEach(async () => { + TestBed.configureTestingModule({ + imports: [ + DocumentBarcodesComponent, + NgxBootstrapIconsModule.pick(allIcons), + ], + }).compileComponents() + + fixture = TestBed.createComponent(DocumentBarcodesComponent) + component = fixture.componentInstance + clipboard = TestBed.inject(Clipboard) + fixture.componentRef.setInput('barcodes', barcodes) + fixture.detectChanges() + }) + + it('should display all barcodes', () => { + const rows = fixture.nativeElement.querySelectorAll('tbody tr') + expect(rows).toHaveLength(3) + expect(rows[0].textContent).toContain('ASN00123') + expect(rows[0].textContent).toContain('Code128') + }) + + it('should only link http(s) values', () => { + const links = fixture.nativeElement.querySelectorAll('tbody a') + expect(links).toHaveLength(1) + expect(links[0].getAttribute('href')).toEqual( + 'https://example.com/invoice/4711' + ) + expect(links[0].getAttribute('target')).toEqual('_blank') + }) + + it('should copy a value and show feedback', () => { + jest.useFakeTimers() + const copySpy = jest.spyOn(clipboard, 'copy').mockReturnValue(true) + const buttons = fixture.nativeElement.querySelectorAll('tbody button') + buttons[0].click() + fixture.detectChanges() + expect(copySpy).toHaveBeenCalledWith('ASN00123') + expect(component.copiedIndex()).toEqual(0) + expect(buttons[0].querySelector('i-bs').getAttribute('name')).toEqual( + 'clipboard-check' + ) + jest.advanceTimersByTime(3000) + fixture.detectChanges() + expect(component.copiedIndex()).toBeNull() + expect(buttons[0].querySelector('i-bs').getAttribute('name')).toEqual( + 'clipboard' + ) + jest.useRealTimers() + }) + + it('should not show feedback if copying failed', () => { + jest.spyOn(clipboard, 'copy').mockReturnValue(false) + component.copy(1) + expect(component.copiedIndex()).toBeNull() + }) +}) diff --git a/src-ui/src/app/components/document-detail/document-barcodes/document-barcodes.component.ts b/src-ui/src/app/components/document-detail/document-barcodes/document-barcodes.component.ts new file mode 100644 index 000000000..e246db0c3 --- /dev/null +++ b/src-ui/src/app/components/document-detail/document-barcodes/document-barcodes.component.ts @@ -0,0 +1,38 @@ +import { Clipboard } from '@angular/cdk/clipboard' +import { Component, inject, input, OnDestroy, signal } from '@angular/core' +import { NgxBootstrapIconsModule } from 'ngx-bootstrap-icons' +import { DocumentBarcode } from 'src/app/data/document-barcode' + +@Component({ + selector: 'pngx-document-barcodes', + templateUrl: './document-barcodes.component.html', + imports: [NgxBootstrapIconsModule], +}) +export class DocumentBarcodesComponent implements OnDestroy { + private readonly clipboard = inject(Clipboard) + + readonly barcodes = input([]) + + readonly copiedIndex = signal(null) + private copyTimeout: ReturnType + + public isLink(value: string): boolean { + try { + const url = new URL(value.trim()) + return ['http:', 'https:'].includes(url.protocol) && !!url.host + } catch { + return false + } + } + + public copy(index: number) { + if (!this.clipboard.copy(this.barcodes()[index].value)) return + this.copiedIndex.set(index) + clearTimeout(this.copyTimeout) + this.copyTimeout = setTimeout(() => this.copiedIndex.set(null), 3000) + } + + ngOnDestroy(): void { + clearTimeout(this.copyTimeout) + } +} diff --git a/src-ui/src/app/components/document-detail/document-detail.component.html b/src-ui/src/app/components/document-detail/document-detail.component.html index 64d0776aa..c0b7b6321 100644 --- a/src-ui/src/app/components/document-detail/document-detail.component.html +++ b/src-ui/src/app/components/document-detail/document-detail.component.html @@ -354,6 +354,10 @@ } + @if (metadata()?.barcodes?.length > 0) { +
Barcodes
+ + } @if (metadata()?.original_metadata?.length > 0) { } diff --git a/src-ui/src/app/components/document-detail/document-detail.component.ts b/src-ui/src/app/components/document-detail/document-detail.component.ts index 5b12ca98b..24680d35b 100644 --- a/src-ui/src/app/components/document-detail/document-detail.component.ts +++ b/src-ui/src/app/components/document-detail/document-detail.component.ts @@ -135,6 +135,7 @@ import { ShareLinksDialogComponent } from '../common/share-links-dialog/share-li import { SuggestionsDropdownComponent } from '../common/suggestions-dropdown/suggestions-dropdown.component' import { DocumentNotesComponent } from '../document-notes/document-notes.component' import { ComponentWithPermissions } from '../with-permissions/with-permissions.component' +import { DocumentBarcodesComponent } from './document-barcodes/document-barcodes.component' import { DocumentHistoryComponent } from './document-history/document-history.component' import { DocumentVersionDropdownComponent } from './document-version-dropdown/document-version-dropdown.component' import { MetadataCollapseComponent } from './metadata-collapse/metadata-collapse.component' @@ -177,6 +178,7 @@ interface IncomingDocumentUpdate { DateComponent, DocumentLinkComponent, MetadataCollapseComponent, + DocumentBarcodesComponent, PermissionsFormComponent, SelectComponent, TagsComponent, diff --git a/src-ui/src/app/data/document-barcode.ts b/src-ui/src/app/data/document-barcode.ts new file mode 100644 index 000000000..5ec480229 --- /dev/null +++ b/src-ui/src/app/data/document-barcode.ts @@ -0,0 +1,7 @@ +export interface DocumentBarcode { + page: number + + value: string + + format: string +} diff --git a/src-ui/src/app/data/document-metadata.ts b/src-ui/src/app/data/document-metadata.ts index ea353e2ac..3aa635747 100644 --- a/src-ui/src/app/data/document-metadata.ts +++ b/src-ui/src/app/data/document-metadata.ts @@ -1,3 +1,5 @@ +import { DocumentBarcode } from './document-barcode' + export interface DocumentMetadata { original_checksum?: string @@ -12,4 +14,6 @@ export interface DocumentMetadata { has_archive_version?: boolean lang?: string + + barcodes?: DocumentBarcode[] } diff --git a/src-ui/src/app/data/paperless-config.ts b/src-ui/src/app/data/paperless-config.ts index 2568358bd..99ca53821 100644 --- a/src-ui/src/app/data/paperless-config.ts +++ b/src-ui/src/app/data/paperless-config.ts @@ -330,6 +330,13 @@ export const PaperlessConfigOptions: ConfigOption[] = [ config_key: 'PAPERLESS_CONSUMER_TAG_BARCODE_SPLIT', category: ConfigCategory.Barcode, }, + { + key: 'barcode_store_values', + title: $localize`Store Barcode Contents`, + type: ConfigOptionType.Boolean, + config_key: 'PAPERLESS_CONSUMER_STORE_BARCODE_VALUES', + category: ConfigCategory.Barcode, + }, { key: 'ai_enabled', title: $localize`AI Enabled`, @@ -458,6 +465,7 @@ export interface PaperlessConfig extends ObjectWithId { barcode_enable_tag: boolean barcode_tag_mapping: object barcode_tag_split: boolean + barcode_store_values: boolean remote_ocr_engine: string remote_ocr_api_key: string remote_ocr_endpoint: string diff --git a/src/documents/barcodes.py b/src/documents/barcodes.py index 2bb96f1ea..324d65d61 100644 --- a/src/documents/barcodes.py +++ b/src/documents/barcodes.py @@ -18,6 +18,7 @@ from documents.converters import convert_from_tiff_to_pdf from documents.data_models import ConsumableDocument from documents.data_models import DocumentMetadataOverrides from documents.data_models import DocumentSource +from documents.data_models import StoredBarcode from documents.models import Document from documents.models import PaperlessTask from documents.models import Tag @@ -47,6 +48,7 @@ class Barcode: page: int value: str settings: BarcodeConfig + format: str = "" @property def is_separator(self) -> bool: @@ -78,6 +80,12 @@ class Barcode: return True return False + def stored(self) -> StoredBarcode: + """ + The barcode as it is stored with a document, page 1-indexed + """ + return {"page": self.page + 1, "value": self.value, "format": self.format} + class BarcodePlugin(ConsumeTaskPlugin): NAME: str = "BarcodePlugin" @@ -89,16 +97,12 @@ class BarcodePlugin(ConsumeTaskPlugin): - ASN from barcode detection is enabled or - Barcode support is enabled and the mime type is supported """ - if self.settings.barcode_enable_tiff_support: - supported_mimes: set[str] = {"application/pdf", "image/tiff"} - else: - supported_mimes = {"application/pdf"} - return ( self.settings.barcode_enable_asn or self.settings.barcodes_enabled or self.settings.barcode_enable_tag - ) and self.input_doc.mime_type in supported_mimes + or self.settings.barcode_store_values + ) and self.input_doc.mime_type in scannable_mime_types(self.settings) def get_settings(self) -> BarcodeConfig: """ @@ -244,6 +248,10 @@ class BarcodePlugin(ConsumeTaskPlugin): if self.settings.barcode_enable_asn and (located_asn := self.asn) is not None: self._apply_detected_asn(located_asn) + # After splitting too, so each split document keeps its own barcodes + if self.settings.barcode_store_values: + self.metadata.barcodes = [x.stored() for x in self.barcodes] or None + def cleanup(self) -> None: self.temp_dir.cleanup() @@ -262,22 +270,6 @@ class BarcodePlugin(ConsumeTaskPlugin): ) self._tiff_conversion_done = True - @staticmethod - def read_barcodes_zxing(image: Image.Image) -> list[str]: - barcodes = [] - - import zxingcpp - - detected_barcodes = zxingcpp.read_barcodes(image) - for barcode in detected_barcodes: - if barcode.text: - barcodes.append(barcode.text) - logger.debug( - f"Barcode of type {barcode.format} found: {barcode.text}", - ) - - return barcodes - def detect(self) -> None: """ Scan all pages of the PDF as images, updating barcodes and the pages @@ -291,60 +283,12 @@ class BarcodePlugin(ConsumeTaskPlugin): self.convert_from_tiff_to_pdf() try: - # Read number of pages from pdf - with Pdf.open(self.pdf_file) as pdf: - num_of_pages = len(pdf.pages) - logger.debug(f"PDF has {num_of_pages} pages") - - # Get limit from configuration - barcode_max_pages: int = ( - num_of_pages - if self.settings.barcode_max_pages == 0 - else self.settings.barcode_max_pages + self.barcodes = scan_pdf( + self.pdf_file, + self.settings, + Path(self.temp_dir.name), ) - if barcode_max_pages < num_of_pages: # pragma: no cover - logger.debug( - f"Barcodes detection will be limited to the first {barcode_max_pages} pages", - ) - - # Loop al page - for current_page_number in range(min(num_of_pages, barcode_max_pages)): - logger.debug(f"Processing page {current_page_number}") - - # Convert page to image - page = convert_from_path( - self.pdf_file, - dpi=self.settings.barcode_dpi, - output_folder=self.temp_dir.name, - first_page=current_page_number + 1, - last_page=current_page_number + 1, - )[0] - - # Remember filename, since it is lost by upscaling - page_filepath = Path(page.filename) - logger.debug(f"Image is at {page_filepath}") - - # Upscale image if configured - factor = self.settings.barcode_upscale - if factor > 1.0: - logger.debug( - f"Upscaling image by {factor} for better barcode detection", - ) - x, y = page.size - page = page.resize( - (round(x * factor), (round(y * factor))), - ) - - # Detect barcodes - for barcode_value in self.read_barcodes_zxing(page): - self.barcodes.append( - Barcode(current_page_number, barcode_value, self.settings), - ) - - # Delete temporary image file - page_filepath.unlink() - # Password protected files can't be checked # This is the exception raised for those except PasswordError as e: @@ -534,3 +478,111 @@ class BarcodePlugin(ConsumeTaskPlugin): document_paths.append(savepath) return document_paths + + +def scannable_mime_types(settings: BarcodeConfig) -> set[str]: + """ + The file types the barcode scan supports with the current settings + """ + if settings.barcode_enable_tiff_support: + return {"application/pdf", "image/tiff"} + return {"application/pdf"} + + +def read_barcodes_zxing(image: Image.Image) -> list[tuple[str, str]]: + """ + Returns the text and format (zxing enum name) of each barcode found in + the image + """ + barcodes = [] + + import zxingcpp + + detected_barcodes = zxingcpp.read_barcodes(image) + for barcode in detected_barcodes: + if barcode.text: + barcodes.append((barcode.text, barcode.format.name)) + logger.debug( + f"Barcode of type {barcode.format} found: {barcode.text}", + ) + + return barcodes + + +def scan_pdf(pdf_path: Path, settings: BarcodeConfig, work_dir: Path) -> list[Barcode]: + """ + Scans the pages of a PDF as images for barcodes. Errors are not caught, + so callers can tell a failed scan from one that found nothing. + """ + barcodes: list[Barcode] = [] + + with Pdf.open(pdf_path) as pdf: + num_of_pages = len(pdf.pages) + logger.debug(f"PDF has {num_of_pages} pages") + + # Get limit from configuration + barcode_max_pages: int = ( + num_of_pages if settings.barcode_max_pages == 0 else settings.barcode_max_pages + ) + + if barcode_max_pages < num_of_pages: # pragma: no cover + logger.debug( + f"Barcodes detection will be limited to the first {barcode_max_pages} pages", + ) + + for current_page_number in range(min(num_of_pages, barcode_max_pages)): + logger.debug(f"Processing page {current_page_number}") + + # Convert page to image + page = convert_from_path( + pdf_path, + dpi=settings.barcode_dpi, + output_folder=work_dir, + first_page=current_page_number + 1, + last_page=current_page_number + 1, + )[0] + + # Remember filename, since it is lost by upscaling + page_filepath = Path(page.filename) + logger.debug(f"Image is at {page_filepath}") + + # Upscale image if configured + factor = settings.barcode_upscale + if factor > 1.0: + logger.debug( + f"Upscaling image by {factor} for better barcode detection", + ) + x, y = page.size + page = page.resize( + (round(x * factor), (round(y * factor))), + ) + + for barcode_value, barcode_format in read_barcodes_zxing(page): + barcodes.append( + Barcode(current_page_number, barcode_value, settings, barcode_format), + ) + + # Delete temporary image file + page_filepath.unlink() + + return barcodes + + +def read_barcode_values( + path: Path, + mime_type: str, + settings: BarcodeConfig, + work_dir: Path, +) -> list[StoredBarcode] | None: + """ + Reads the barcodes of a file outside of the consumption plugins: for new + versions, which skip the barcode plugin, and when reprocessing. + + Returns None if the file can't be scanned with the current settings. + Errors while scanning are raised. + """ + if mime_type not in scannable_mime_types(settings): + return None + if mime_type == "image/tiff": + path = convert_from_tiff_to_pdf(path, work_dir) + return [x.stored() for x in scan_pdf(path, settings, work_dir)] diff --git a/src/documents/consumer.py b/src/documents/consumer.py index 4f321d88e..252ee6d1f 100644 --- a/src/documents/consumer.py +++ b/src/documents/consumer.py @@ -18,6 +18,7 @@ from django.utils import timezone from filelock import FileLock from rest_framework.reverse import reverse +from documents.barcodes import read_barcode_values from documents.classifier import load_classifier from documents.data_models import ConsumableDocument from documents.data_models import ConsumeFileSuccessResult @@ -31,6 +32,7 @@ from documents.models import Correspondent from documents.models import CustomField from documents.models import CustomFieldInstance from documents.models import Document +from documents.models import DocumentBarcode from documents.models import DocumentType from documents.models import StoragePath from documents.models import Tag @@ -53,6 +55,7 @@ from documents.utils import compute_checksum from documents.utils import copy_basic_file_stats from documents.utils import copy_file_with_basic_stats from documents.utils import run_subprocess +from paperless.config import BarcodeConfig from paperless.config import OcrConfig from paperless.config import RemoteOCRConfig from paperless.models import ArchiveFileGenerationChoices @@ -504,6 +507,13 @@ class ConsumerPlugin( f"Parser: {document_parser.name} v{document_parser.version}", ) + # New versions skip the barcode plugin, so read their barcodes here + if ( + self.input_doc.root_document_id is not None + and self.metadata.barcodes is None + ): + self._read_version_barcodes(mime_type, Path(tmpdir)) + # Parse the document. This may take some time. text = None @@ -631,6 +641,8 @@ class ConsumerPlugin( else: original_document.save() + self._store_barcodes(original_document) + # Adding a version changes the effective document, so update root modified Document.objects.filter(pk=root_doc.pk).update( modified=timezone.now(), @@ -962,6 +974,32 @@ class ConsumerPlugin( } CustomFieldInstance.objects.create(**args) # adds to document + self._store_barcodes(document) + + def _read_version_barcodes(self, mime_type: str, work_dir: Path) -> None: + barcode_settings = BarcodeConfig() + if not barcode_settings.barcode_store_values: + return + try: + self.metadata.barcodes = ( + read_barcode_values( + self.working_copy, + mime_type, + barcode_settings, + work_dir, + ) + or None + ) + except Exception as e: + self.log.warning(f"Could not read barcodes of {self.filename}: {e}") + + def _store_barcodes(self, document: Document) -> None: + if self.metadata.barcodes: + DocumentBarcode.objects.bulk_create( + DocumentBarcode(document=document, **barcode) + for barcode in self.metadata.barcodes + ) + def _write(self, source, target) -> None: with ( Path(source).open("rb") as read_file, diff --git a/src/documents/data_models.py b/src/documents/data_models.py index 230af0684..e8df1a022 100644 --- a/src/documents/data_models.py +++ b/src/documents/data_models.py @@ -9,6 +9,16 @@ from guardian.shortcuts import get_groups_with_perms from guardian.shortcuts import get_users_with_perms +class StoredBarcode(TypedDict): + """ + A detected barcode as it is stored with a document + """ + + page: int # 1-indexed + value: str + format: str # a DocumentBarcode.Format value + + @dataclasses.dataclass class DocumentMetadataOverrides: """ @@ -35,6 +45,7 @@ class DocumentMetadataOverrides: version_label: str | None = None actor_id: int | None = None remote_ocr: bool = False + barcodes: list[StoredBarcode] | None = None def update(self, other: "DocumentMetadataOverrides") -> "DocumentMetadataOverrides": """ diff --git a/src/documents/management/commands/document_exporter.py b/src/documents/management/commands/document_exporter.py index 320605235..25f4cb717 100644 --- a/src/documents/management/commands/document_exporter.py +++ b/src/documents/management/commands/document_exporter.py @@ -45,6 +45,7 @@ from documents.models import Correspondent from documents.models import CustomField from documents.models import CustomFieldInstance from documents.models import Document +from documents.models import DocumentBarcode from documents.models import DocumentType from documents.models import Note from documents.models import SavedView @@ -353,6 +354,7 @@ class Command(CryptMixin, PaperlessCommand): "workflows": Workflow.objects.all(), "custom_fields": CustomField.objects.all(), "custom_field_instances": CustomFieldInstance.global_objects.all(), + "document_barcodes": DocumentBarcode.objects.all(), "app_configs": ApplicationConfiguration.objects.all(), "notes": Note.global_objects.all(), "documents": Document.global_objects.order_by("id").all(), @@ -411,6 +413,7 @@ class Command(CryptMixin, PaperlessCommand): elif self.split_manifest and key in ( "notes", "custom_field_instances", + "document_barcodes", ): # Written per-document in _write_split_manifest pass @@ -651,6 +654,12 @@ class Command(CryptMixin, PaperlessCommand): CustomFieldInstance.global_objects.filter(document=document), ), ) + content.extend( + serializers.serialize( + "python", + DocumentBarcode.objects.filter(document=document), + ), + ) manifest_name = base_name.with_name(f"{base_name.stem}-manifest.json") if self.use_folder_prefix: manifest_name = Path("json") / manifest_name diff --git a/src/documents/management/commands/document_index.py b/src/documents/management/commands/document_index.py index 10558c211..45dd98824 100644 --- a/src/documents/management/commands/document_index.py +++ b/src/documents/management/commands/document_index.py @@ -77,6 +77,8 @@ class Command(PaperlessCommand): "notes__user", "custom_fields__field", "versions", + "barcodes", + "versions__barcodes", ) total = documents.count() rebuild_kwargs = {} diff --git a/src/documents/migrations/0027_documentbarcode.py b/src/documents/migrations/0027_documentbarcode.py new file mode 100644 index 000000000..d38ee0b0e --- /dev/null +++ b/src/documents/migrations/0027_documentbarcode.py @@ -0,0 +1,102 @@ +# Generated by Django 5.2.16 on 2026-09-30 23:01 + +import django.db.models.deletion +from django.db import migrations +from django.db import models + + +class Migration(migrations.Migration): + dependencies = [ + ("documents", "0026_alter_document_archive_checksum_and_more"), + ] + + operations = [ + migrations.CreateModel( + name="DocumentBarcode", + fields=[ + ( + "id", + models.AutoField( + auto_created=True, + primary_key=True, + serialize=False, + verbose_name="ID", + ), + ), + ( + "page", + models.PositiveIntegerField( + help_text="Page of the original file, starting at 1", + verbose_name="page", + ), + ), + ("value", models.TextField(verbose_name="value")), + ( + "format", + models.CharField( + choices=[ + ("Codabar", "Codabar"), + ("Code39", "Code 39"), + ("Code39Std", "Code 39 Standard"), + ("Code39Ext", "Code 39 Extended"), + ("Code32", "Code 32"), + ("PZN", "Pharmazentralnummer"), + ("Code93", "Code 93"), + ("Code128", "Code 128"), + ("ITF", "ITF"), + ("ITF14", "ITF-14"), + ("DataBar", "DataBar"), + ("DataBarOmni", "DataBar Omni"), + ("DataBarStk", "DataBar Stacked"), + ("DataBarStkOmni", "DataBar Stacked Omni"), + ("DataBarLtd", "DataBar Limited"), + ("DataBarExp", "DataBar Expanded"), + ("DataBarExpStk", "DataBar Expanded Stacked"), + ("EANUPC", "EAN/UPC"), + ("EAN13", "EAN-13"), + ("EAN8", "EAN-8"), + ("EAN5", "EAN-5"), + ("EAN2", "EAN-2"), + ("ISBN", "ISBN"), + ("UPCA", "UPC-A"), + ("UPCE", "UPC-E"), + ("Telepen", "Telepen"), + ("TelepenAlpha", "Telepen Alpha"), + ("TelepenNumeric", "Telepen Numeric"), + ("OtherBarcode", "Other barcode"), + ("DXFilmEdge", "DX Film Edge"), + ("PDF417", "PDF417"), + ("CompactPDF417", "Compact PDF417"), + ("MicroPDF417", "MicroPDF417"), + ("Aztec", "Aztec"), + ("AztecCode", "Aztec Code"), + ("AztecRune", "Aztec Rune"), + ("QRCode", "QR Code"), + ("QRCodeModel1", "QR Code Model 1"), + ("QRCodeModel2", "QR Code Model 2"), + ("MicroQRCode", "Micro QR Code"), + ("RMQRCode", "rMQR Code"), + ("DataMatrix", "Data Matrix"), + ("MaxiCode", "MaxiCode"), + ], + max_length=32, + verbose_name="format", + ), + ), + ( + "document", + models.ForeignKey( + on_delete=django.db.models.deletion.CASCADE, + related_name="barcodes", + to="documents.document", + verbose_name="document", + ), + ), + ], + options={ + "verbose_name": "document barcode", + "verbose_name_plural": "document barcodes", + "ordering": ("page", "id"), + }, + ), + ] diff --git a/src/documents/models.py b/src/documents/models.py index 76b0192aa..ff8632b12 100644 --- a/src/documents/models.py +++ b/src/documents/models.py @@ -366,6 +366,16 @@ class Document(SoftDeleteModel, ModelWithOwner): # type: ignore[django-manager- res += f" {self.title}" return res + def get_effective_barcodes(self) -> list["DocumentBarcode"]: + """ + Returns the stored barcodes for the document, like + get_effective_content(): for root documents those of the latest + version when there is one, as that is the file users see. + """ + from documents.versioning import latest_version + + return list(latest_version(self).barcodes.all()) + def get_effective_content(self) -> str | None: """ Returns the effective content for the document. @@ -969,6 +979,87 @@ class Note(SoftDeleteModel): return self.note +class DocumentBarcode(models.Model): + """ + A barcode found in a document during consumption, kept so its content + can be shown and copied + """ + + document = models.ForeignKey( + Document, + related_name="barcodes", + on_delete=models.CASCADE, + verbose_name=_("document"), + ) + + page = models.PositiveIntegerField( + _("page"), + help_text=_("Page of the original file, starting at 1"), + ) + + value = models.TextField(_("value")) + + class Format(models.TextChoices): + """ + The concrete barcode formats of zxing-cpp, keyed on the enum name. + The labels are symbology names and aren't translated. + """ + + CODABAR = "Codabar", "Codabar" + CODE39 = "Code39", "Code 39" + CODE39_STD = "Code39Std", "Code 39 Standard" + CODE39_EXT = "Code39Ext", "Code 39 Extended" + CODE32 = "Code32", "Code 32" + PZN = "PZN", "Pharmazentralnummer" + CODE93 = "Code93", "Code 93" + CODE128 = "Code128", "Code 128" + ITF = "ITF", "ITF" + ITF14 = "ITF14", "ITF-14" + DATA_BAR = "DataBar", "DataBar" + DATA_BAR_OMNI = "DataBarOmni", "DataBar Omni" + DATA_BAR_STK = "DataBarStk", "DataBar Stacked" + DATA_BAR_STK_OMNI = "DataBarStkOmni", "DataBar Stacked Omni" + DATA_BAR_LTD = "DataBarLtd", "DataBar Limited" + DATA_BAR_EXP = "DataBarExp", "DataBar Expanded" + DATA_BAR_EXP_STK = "DataBarExpStk", "DataBar Expanded Stacked" + EANUPC = "EANUPC", "EAN/UPC" + EAN13 = "EAN13", "EAN-13" + EAN8 = "EAN8", "EAN-8" + EAN5 = "EAN5", "EAN-5" + EAN2 = "EAN2", "EAN-2" + ISBN = "ISBN", "ISBN" + UPCA = "UPCA", "UPC-A" + UPCE = "UPCE", "UPC-E" + TELEPEN = "Telepen", "Telepen" + TELEPEN_ALPHA = "TelepenAlpha", "Telepen Alpha" + TELEPEN_NUMERIC = "TelepenNumeric", "Telepen Numeric" + OTHER_BARCODE = "OtherBarcode", "Other barcode" + DX_FILM_EDGE = "DXFilmEdge", "DX Film Edge" + PDF417 = "PDF417", "PDF417" + COMPACT_PDF417 = "CompactPDF417", "Compact PDF417" + MICRO_PDF417 = "MicroPDF417", "MicroPDF417" + AZTEC = "Aztec", "Aztec" + AZTEC_CODE = "AztecCode", "Aztec Code" + AZTEC_RUNE = "AztecRune", "Aztec Rune" + QR_CODE = "QRCode", "QR Code" + QR_CODE_MODEL1 = "QRCodeModel1", "QR Code Model 1" + QR_CODE_MODEL2 = "QRCodeModel2", "QR Code Model 2" + MICRO_QR_CODE = "MicroQRCode", "Micro QR Code" + RMQR_CODE = "RMQRCode", "rMQR Code" + DATA_MATRIX = "DataMatrix", "Data Matrix" + MAXI_CODE = "MaxiCode", "MaxiCode" + + format = models.CharField(_("format"), max_length=32, choices=Format.choices) + + class Meta: + ordering = ("page", "id") + verbose_name = _("document barcode") + verbose_name_plural = _("document barcodes") + + def __str__(self) -> str: # pragma: no cover + return self.value + + class ShareLink(SoftDeleteModel): class FileVersion(models.TextChoices): ARCHIVE = ("archive", _("Archive")) diff --git a/src/documents/search/_backend.py b/src/documents/search/_backend.py index c04ccbfd4..996d332e8 100644 --- a/src/documents/search/_backend.py +++ b/src/documents/search/_backend.py @@ -311,7 +311,13 @@ class WriteBatch: queryset = annotate_effective_content( Document.objects.filter(pk__in=ids) .select_related("correspondent", "document_type", "storage_path", "owner") - .prefetch_related("tags", "notes__user", "custom_fields__field"), + .prefetch_related( + "tags", + "notes__user", + "custom_fields__field", + "barcodes", + "versions__barcodes", + ), ) for document, grant in _DocumentViewerStream(queryset, chunk_size=1000): self.remove(document.pk) @@ -604,6 +610,16 @@ class TantivyBackend: }, ) + # Barcodes: JSON field like custom_fields, only filled when stored + for barcode in document.get_effective_barcodes(): + doc.add_json( + "barcodes", + { + "value": normalize_search_text(barcode.value), + "format": normalize_search_text(barcode.format), + }, + ) + # Dates created_date = datetime( document.created.year, diff --git a/src/documents/search/_fields.py b/src/documents/search/_fields.py index b5194ce81..29dc99468 100644 --- a/src/documents/search/_fields.py +++ b/src/documents/search/_fields.py @@ -39,4 +39,9 @@ PUBLIC_FIELDS: tuple[FieldSpec, ...] = ( FieldKind.JSON, subpaths={"name": SubpathSpec(), "value": SubpathSpec(default=True)}, ), + FieldSpec( + "barcodes", + FieldKind.JSON, + subpaths={"value": SubpathSpec(default=True), "format": SubpathSpec()}, + ), ) diff --git a/src/documents/search/_schema.py b/src/documents/search/_schema.py index 3430356e0..2539d12fe 100644 --- a/src/documents/search/_schema.py +++ b/src/documents/search/_schema.py @@ -25,7 +25,8 @@ logger = logging.getLogger("paperless.search") # order, and the write-only correspondent/document_type/storage_path/tag id # columns dropped. tantivy compares schemas by ordered field list, so an # index built by v1 rejects every write against the v2 schema. -SCHEMA_VERSION: Final[int] = 2 +# v3 - barcodes JSON field for stored barcode contents +SCHEMA_VERSION: Final[int] = 3 class FieldDescriptor(NamedTuple): diff --git a/src/documents/serialisers.py b/src/documents/serialisers.py index 4bb1ff8c1..c2c875029 100644 --- a/src/documents/serialisers.py +++ b/src/documents/serialisers.py @@ -60,6 +60,7 @@ from documents.models import Correspondent from documents.models import CustomField from documents.models import CustomFieldInstance from documents.models import Document +from documents.models import DocumentBarcode from documents.models import DocumentType from documents.models import MatchingModel from documents.models import Note @@ -987,6 +988,12 @@ class BasicUserSerializer(serializers.ModelSerializer[User]): fields = ["id", "username", "first_name", "last_name"] +class DocumentBarcodeSerializer(serializers.ModelSerializer[DocumentBarcode]): + class Meta: + model = DocumentBarcode + fields = ["page", "value", "format"] + + class NotesSerializer(serializers.ModelSerializer[Note]): user = BasicUserSerializer(read_only=True) diff --git a/src/documents/tasks.py b/src/documents/tasks.py index 33f7d4f90..456b006d9 100644 --- a/src/documents/tasks.py +++ b/src/documents/tasks.py @@ -20,6 +20,7 @@ from filelock import FileLock from documents import sanity_checker from documents.barcodes import BarcodePlugin +from documents.barcodes import read_barcode_values from documents.bulk_download import ArchiveOnlyStrategy from documents.bulk_download import OriginalsOnlyStrategy from documents.caching import clear_document_caches @@ -36,6 +37,7 @@ from documents.data_models import ConsumeFileDuplicateResult from documents.data_models import ConsumeFileStoppedResult from documents.data_models import ConsumeFileSuccessResult from documents.data_models import DocumentMetadataOverrides +from documents.data_models import StoredBarcode from documents.double_sided import CollatePlugin from documents.file_handling import create_source_path_directory from documents.file_handling import generate_unique_filename @@ -43,6 +45,7 @@ from documents.matching import prefilter_documents_by_workflowtrigger from documents.models import Correspondent from documents.models import CustomFieldInstance from documents.models import Document +from documents.models import DocumentBarcode from documents.models import DocumentType from documents.models import PaperlessTask from documents.models import ShareLink @@ -67,6 +70,7 @@ from documents.utils import identity from documents.versioning import annotate_effective_content from documents.workflows.utils import get_workflows_for_trigger from paperless.config import AIConfig +from paperless.config import BarcodeConfig from paperless.config import RemoteOCRConfig from paperless.logging import consume_task_id from paperless.parsers import ParserContext @@ -341,6 +345,29 @@ def bulk_update_documents(document_ids) -> None: ) +def _read_barcodes_for_reprocess(document: Document) -> list[StoredBarcode] | None: + """ + Reads the barcodes of the original again, e.g. for documents consumed + before storing them was enabled. Returns None if they should be left as + they are: storing is off, the file can't be scanned with the current + settings, or the scan failed. + """ + barcode_settings = BarcodeConfig() + if not barcode_settings.barcode_store_values: + return None + try: + with TemporaryDirectory(dir=settings.SCRATCH_DIR) as tmpdir: + return read_barcode_values( + document.source_path, + document.mime_type, + barcode_settings, + Path(tmpdir), + ) + except Exception as e: + logger.warning(f"Could not read barcodes of document {document}: {e}") + return None + + @shared_task def update_document_content_maybe_archive_file( document_id, @@ -387,6 +414,8 @@ def update_document_content_maybe_archive_file( produce_archive=produce_archive, ) + barcodes = _read_barcodes_for_reprocess(document) + thumbnail = parser.get_thumbnail(document.source_path, mime_type) with transaction.atomic(): @@ -443,6 +472,17 @@ def update_document_content_maybe_archive_file( action=LogEntry.Action.UPDATE, ) + if barcodes is not None: + document.barcodes.all().delete() + DocumentBarcode.objects.bulk_create( + DocumentBarcode(document=document, **barcode) + for barcode in barcodes + ) + # metadata_etag includes modified + Document.objects.filter(pk=document.pk).update( + modified=timezone.now(), + ) + with FileLock(settings.MEDIA_LOCK): if parser.get_archive_path(): create_source_path_directory(document.archive_path) diff --git a/src/documents/tests/samples/barcodes/barcode-qr-url.pdf b/src/documents/tests/samples/barcodes/barcode-qr-url.pdf new file mode 100644 index 000000000..0a6dd5af4 Binary files /dev/null and b/src/documents/tests/samples/barcodes/barcode-qr-url.pdf differ diff --git a/src/documents/tests/search/test_barcode_search.py b/src/documents/tests/search/test_barcode_search.py new file mode 100644 index 000000000..3036ede04 --- /dev/null +++ b/src/documents/tests/search/test_barcode_search.py @@ -0,0 +1,95 @@ +"""Stored barcode contents in the search index. + +Barcodes are a JSON field like notes and custom fields: barcodes: resolves to +barcodes.value:, and a plain query without the prefix does not look at them. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import pytest + +from documents.models import DocumentBarcode +from paperless_testing.factories import DocumentBarcodeFactory +from paperless_testing.factories import DocumentFactory + +if TYPE_CHECKING: + from collections.abc import Callable + + from documents.models import Document + from documents.search._backend import TantivyBackend + +pytestmark = [pytest.mark.search, pytest.mark.django_db] + + +class TestBarcodeSearch: + @pytest.fixture + def with_barcodes(self, backend: TantivyBackend) -> Document: + document = DocumentFactory(title="Letter", content="x") + DocumentBarcodeFactory( + document=document, + value="WIFI:T:WPA;S:Guest-WLAN;P:crocodile123;;", + ) + DocumentBarcodeFactory( + document=document, + page=2, + value="DE89370400440532013000", + format=DocumentBarcode.Format.CODE128, + ) + backend.add_or_update(document) + return document + + def test_bare_barcodes_prefix_searches_values( + self, + with_barcodes: Document, + index_document: Callable[..., Document], + matched_ids: Callable[[str], set[int]], + ) -> None: + """ + GIVEN: + - A document with stored barcodes, and a decoy document whose + content (not a barcode) contains the same word + WHEN: + - A bare "barcodes:" prefix query is run + THEN: + - Only the document whose barcode matches is returned, also for a + part of a barcode between separators + """ + index_document(title="Decoy", content="crocodile123 in the text") + + assert matched_ids("barcodes:crocodile123") == {with_barcodes.pk} + assert matched_ids("barcodes.value:DE89370400440532013000") == { + with_barcodes.pk, + } + + def test_barcodes_format_subpath( + self, + with_barcodes: Document, + matched_ids: Callable[[str], set[int]], + ) -> None: + """ + GIVEN: + - A document with a QR code and a Code 128 barcode + WHEN: + - The format subpath is queried + THEN: + - The document is found by its barcode formats + """ + assert matched_ids("barcodes.format:qrcode") == {with_barcodes.pk} + assert matched_ids("barcodes.format:aztec") == set() + + def test_plain_query_ignores_barcodes( + self, + with_barcodes: Document, + matched_ids: Callable[[str], set[int]], + ) -> None: + """ + GIVEN: + - A document with a barcode value not found in its text + WHEN: + - The value is searched without a field prefix + THEN: + - Nothing is found, as with notes and custom fields + """ + assert matched_ids("crocodile123") == set() diff --git a/src/documents/tests/search/test_json_subpath_completeness.py b/src/documents/tests/search/test_json_subpath_completeness.py index 54f2ca8fb..0422dc192 100644 --- a/src/documents/tests/search/test_json_subpath_completeness.py +++ b/src/documents/tests/search/test_json_subpath_completeness.py @@ -8,7 +8,7 @@ queryable-but-always-empty -- syntactically valid, silently matching nothing -- with no test failure anywhere. This indexes one real document carrying values for every JSON field -(a Note, a CustomFieldInstance) and inspects the document's own stored +(a Note, a CustomFieldInstance, a DocumentBarcode) and inspects the document's own stored JSON payload, rather than running field-specific queries: that way a future JSON field's subpaths are covered automatically, without a new per-subpath query having to be added by hand each time. @@ -27,6 +27,7 @@ from documents.models import CustomFieldInstance from documents.models import Document from documents.models import Note from documents.search._fields import PUBLIC_FIELDS +from paperless_testing.factories import DocumentBarcodeFactory from paperless_testing.factories import UserFactory if TYPE_CHECKING: @@ -42,11 +43,12 @@ class TestJsonSubpathsAreWrittenAtIndexTime: ) -> None: """ GIVEN: - - A document with a Note and a CustomFieldInstance attached + - A document with a Note, a CustomFieldInstance and a + DocumentBarcode attached WHEN: - The document is indexed via TantivyBackend.add_or_update THEN: - - Every subpath PUBLIC_FIELDS declares for notes/custom_fields + - Every subpath PUBLIC_FIELDS declares for notes/custom_fields/barcodes is present as a key in the document's stored JSON payload """ user = UserFactory(username="completeness-user") @@ -65,6 +67,7 @@ class TestJsonSubpathsAreWrittenAtIndexTime: field=field, value_text="a value", ) + DocumentBarcodeFactory(document=doc, value="a barcode") backend.add_or_update(doc) index = backend._index diff --git a/src/documents/tests/search/test_schema_fingerprint.py b/src/documents/tests/search/test_schema_fingerprint.py index 21d3e765f..972381d4a 100644 --- a/src/documents/tests/search/test_schema_fingerprint.py +++ b/src/documents/tests/search/test_schema_fingerprint.py @@ -175,6 +175,14 @@ PINNED_DESCRIPTORS: tuple[FieldDescriptor, ...] = ( fast=False, tokenizer="paperless_text", ), + FieldDescriptor( + "barcodes", + "json", + stored=True, + indexed=True, + fast=False, + tokenizer="paperless_text", + ), FieldDescriptor( "title_sort", "text", diff --git a/src/documents/tests/test_api_app_config.py b/src/documents/tests/test_api_app_config.py index 4566d668d..f617e646a 100644 --- a/src/documents/tests/test_api_app_config.py +++ b/src/documents/tests/test_api_app_config.py @@ -74,6 +74,7 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase): "barcode_enable_tag": None, "barcode_tag_mapping": None, "barcode_tag_split": None, + "barcode_store_values": None, "remote_ocr_engine": None, "remote_ocr_api_key": None, "remote_ocr_endpoint": None, diff --git a/src/documents/tests/test_barcodes.py b/src/documents/tests/test_barcodes.py index c21f75fbc..521b87a6d 100644 --- a/src/documents/tests/test_barcodes.py +++ b/src/documents/tests/test_barcodes.py @@ -1,29 +1,46 @@ +from __future__ import annotations + import shutil -from collections.abc import Generator from contextlib import contextmanager from pathlib import Path +from typing import TYPE_CHECKING import pytest +import zxingcpp from django.conf import settings from django.test import TestCase from django.test import override_settings +from rest_framework import status from documents import tasks from documents.barcodes import BarcodePlugin +from documents.barcodes import read_barcode_values from documents.consumer import ConsumerError from documents.data_models import ConsumableDocument from documents.data_models import DocumentMetadataOverrides from documents.data_models import DocumentSource from documents.models import Document +from documents.models import DocumentBarcode from documents.models import Tag from documents.plugins.base import StopConsumeTaskError from documents.tests.utils import ConsumeTaskMixin from documents.tests.utils import SampleDirMixin +from paperless.config import BarcodeConfig from paperless.models import ApplicationConfiguration from paperless_testing.assertions import FileSystemAssertsMixin from paperless_testing.dirs import DirectoriesMixin from paperless_testing.fakes.progress import FakeProgressManager +if TYPE_CHECKING: + from collections.abc import Callable + from collections.abc import Generator + + from pytest_django.fixtures import Settings + from pytest_mock import MockerFixture + from rest_framework.test import APIClient + + from paperless_testing.dirs import PaperlessDirs + class GetReaderPluginMixin: @contextmanager @@ -1127,3 +1144,301 @@ class TestTagBarcode(DirectoriesMixin, SampleDirMixin, GetReaderPluginMixin, Tes document_list = reader.separate_pages(separator_pages) self.assertEqual(len(document_list), 3) + + +SAMPLE_VALUES = [ + {"page": 1, "value": "javascript:alert(1)", "format": "QRCode"}, + {"page": 2, "value": "https://example.com/invoice/4711", "format": "QRCode"}, +] + + +@pytest.fixture +def samples_dir() -> Path: + return Path(__file__).parent / "samples" + + +@pytest.fixture +def barcode_samples_dir(samples_dir: Path) -> Path: + return samples_dir / "barcodes" + + +@pytest.fixture +def store_barcodes(settings: Settings) -> None: + settings.CONSUMER_STORE_BARCODE_VALUES = True + + +@pytest.fixture +def barcode_reader( + paperless_dirs: PaperlessDirs, +) -> Generator[Callable[[Path], BarcodePlugin], None, None]: + readers: list[BarcodePlugin] = [] + + def make(path: Path) -> BarcodePlugin: + reader = BarcodePlugin( + ConsumableDocument(DocumentSource.ConsumeFolder, original_file=path), + DocumentMetadataOverrides(), + FakeProgressManager(path.name, None), + paperless_dirs.scratch_dir, + "task-id", + ) + reader.setup() + readers.append(reader) + return reader + + yield make + for reader in readers: + reader.cleanup() + + +@pytest.fixture +def consume_sample( + paperless_dirs: PaperlessDirs, + barcode_samples_dir: Path, + fake_progress_manager: type[FakeProgressManager], + settings: Settings, +) -> Callable[..., Document]: + settings.CELERY_TASK_ALWAYS_EAGER = True + settings.OCR_MODE = "auto" + + def consume(name: str, *, root_document_id: int | None = None) -> Document: + dst = paperless_dirs.scratch_dir / name + shutil.copy(barcode_samples_dir / name, dst) + tasks.consume_file( + ConsumableDocument( + source=DocumentSource.ApiUpload + if root_document_id + else DocumentSource.ConsumeFolder, + original_file=dst, + root_document_id=root_document_id, + ), + None, + ) + return Document.objects.latest("id") + + return consume + + +def _stored(document: Document) -> list[dict]: + return list(document.barcodes.values("page", "value", "format")) + + +def test_formats_cover_zxing() -> None: + """ + DocumentBarcode.Format matches the concrete formats of zxing-cpp, so a + zxing-cpp update that adds or removes one fails here + """ + members = zxingcpp.BarcodeFormat.__members__ + # skip NONE and the groups like AllLinear, including their aliases + seen = {int(m) for n, m in members.items() if n == "NONE" or n.startswith("All")} + concrete = set() + for name, member in members.items(): + # aliases such as DataBarExpanded come after the name zxing reports + if int(member) not in seen: + seen.add(int(member)) + concrete.add(name) + assert set(DocumentBarcode.Format.values) == concrete + + +@pytest.mark.django_db +class TestBarcodeValues: + @pytest.mark.parametrize( + ("filename", "mime_type", "tiff_support", "expected"), + [ + pytest.param("simple.jpg", "image/jpeg", True, None, id="jpeg"), + pytest.param("simple.tiff", "image/tiff", False, None, id="tiff-off"), + pytest.param("simple.tiff", "image/tiff", True, [], id="tiff-on"), + ], + ) + def test_read_values_file_types( + self, + paperless_dirs: PaperlessDirs, + samples_dir: Path, + settings: Settings, + filename: str, + mime_type: str, + tiff_support: bool, # noqa: FBT001 + expected: list | None, + ) -> None: + """ + Files the scan doesn't support return None, so stored barcodes are + kept instead of being replaced with an empty list + """ + settings.CONSUMER_BARCODE_TIFF_SUPPORT = tiff_support + values = read_barcode_values( + samples_dir / filename, + mime_type, + BarcodeConfig(), + paperless_dirs.scratch_dir, + ) + assert values == expected + + def test_values_detected( + self, + barcode_reader: Callable[[Path], BarcodePlugin], + barcode_samples_dir: Path, + store_barcodes: None, + ) -> None: + reader = barcode_reader(barcode_samples_dir / "barcode-qr-url.pdf") + assert reader.able_to_run + reader.run() + assert reader.metadata.barcodes == SAMPLE_VALUES + + def test_values_disabled( + self, + barcode_reader: Callable[[Path], BarcodePlugin], + barcode_samples_dir: Path, + settings: Settings, + ) -> None: + settings.CONSUMER_ENABLE_ASN_BARCODE = True + reader = barcode_reader(barcode_samples_dir / "barcode-qr-url.pdf") + reader.run() + assert reader.metadata.barcodes is None + + def test_consume_and_reprocess( + self, + consume_sample: Callable[..., Document], + admin_client: APIClient, + store_barcodes: None, + ) -> None: + """ + GIVEN: + - PDF with a QR code on each of its two pages + WHEN: + - File is consumed, the values are lost, and the document is reprocessed + THEN: + - The barcodes are stored, shown in the API and searchable + - Reprocessing reads them again + """ + document = consume_sample("barcode-qr-url.pdf") + assert _stored(document) == SAMPLE_VALUES + + response = admin_client.get(f"/api/documents/{document.pk}/metadata/") + assert response.status_code == status.HTTP_200_OK + assert response.data["barcodes"] == SAMPLE_VALUES + response = admin_client.get(f"/api/documents/{document.pk}/") + assert "barcodes" not in response.data + response = admin_client.get("/api/documents/?query=barcodes:invoice") + assert [x["id"] for x in response.data["results"]] == [document.pk] + response = admin_client.get("/api/documents/?query=invoice") + assert response.data["results"] == [] + + document.barcodes.all().delete() + modified = Document.objects.get(pk=document.pk).modified + tasks.update_document_content_maybe_archive_file(document.pk) + + assert _stored(document) == SAMPLE_VALUES + assert Document.objects.get(pk=document.pk).modified > modified + + @pytest.mark.parametrize( + "read_values", + [ + pytest.param({"side_effect": RuntimeError("broken")}, id="scan-fails"), + pytest.param({"return_value": None}, id="not-scannable"), + ], + ) + def test_reprocess_keeps_values( + self, + consume_sample: Callable[..., Document], + mocker: MockerFixture, + store_barcodes: None, + read_values: dict, + ) -> None: + """ + GIVEN: + - A document with stored barcodes + WHEN: + - It is reprocessed, but the barcodes can't be read + THEN: + - The stored barcodes are kept + """ + document = consume_sample("barcode-qr-url.pdf") + mocker.patch("documents.tasks.read_barcode_values", **read_values) + + tasks.update_document_content_maybe_archive_file(document.pk) + + assert _stored(document) == SAMPLE_VALUES + + def test_reprocess_tiff_support_off_keeps_values( + self, + consume_sample: Callable[..., Document], + settings: Settings, + store_barcodes: None, + ) -> None: + """ + GIVEN: + - A TIFF document with barcodes stored while TIFF support was on + WHEN: + - TIFF support is turned off and the document is reprocessed + THEN: + - The stored barcodes are kept + """ + settings.CONSUMER_BARCODE_TIFF_SUPPORT = True + document = consume_sample("patch-code-t-middle.tiff") + stored = _stored(document) + assert stored + + settings.CONSUMER_BARCODE_TIFF_SUPPORT = False + tasks.update_document_content_maybe_archive_file(document.pk) + + assert _stored(document) == stored + + def test_reprocess_values_disabled( + self, + consume_sample: Callable[..., Document], + settings: Settings, + store_barcodes: None, + ) -> None: + document = consume_sample("barcode-qr-url.pdf") + settings.CONSUMER_STORE_BARCODE_VALUES = False + + assert tasks._read_barcodes_for_reprocess(document) is None + + def test_consume_version_stores_own_values( + self, + consume_sample: Callable[..., Document], + admin_client: APIClient, + store_barcodes: None, + ) -> None: + """ + GIVEN: + - A document with stored barcodes + WHEN: + - A new version with a different barcode is consumed, like after + rotating or removing pages + THEN: + - The version keeps its own barcodes, the original ones are kept + - The metadata and the search use those of the newest version + """ + root = consume_sample("barcode-qr-url.pdf") + version = consume_sample("barcode-128-custom.pdf", root_document_id=root.pk) + assert version.root_document == root + assert _stored(version) == [ + {"page": 1, "value": "CUSTOM BARCODE", "format": "Code128"}, + ] + assert root.barcodes.count() == 2 + assert [x.value for x in root.get_effective_barcodes()] == ["CUSTOM BARCODE"] + + response = admin_client.get(f"/api/documents/{root.pk}/metadata/") + assert [x["value"] for x in response.data["barcodes"]] == ["CUSTOM BARCODE"] + response = admin_client.get('/api/documents/?query=barcodes:"custom barcode"') + assert [x["id"] for x in response.data["results"]] == [root.pk] + response = admin_client.get("/api/documents/?query=barcodes:invoice") + assert response.data["results"] == [] + + def test_consume_version_scan_fails( + self, + consume_sample: Callable[..., Document], + mocker: MockerFixture, + store_barcodes: None, + ) -> None: + """ + A failed scan of a new version is logged and doesn't stop consumption + """ + root = consume_sample("barcode-qr-url.pdf") + mocker.patch( + "documents.consumer.read_barcode_values", + side_effect=RuntimeError("broken"), + ) + version = consume_sample("barcode-128-custom.pdf", root_document_id=root.pk) + assert version.root_document == root + assert not version.barcodes.exists() diff --git a/src/documents/tests/test_management_exporter.py b/src/documents/tests/test_management_exporter.py index 137744569..facea8ca2 100644 --- a/src/documents/tests/test_management_exporter.py +++ b/src/documents/tests/test_management_exporter.py @@ -32,6 +32,7 @@ from documents.models import Correspondent from documents.models import CustomField from documents.models import CustomFieldInstance from documents.models import Document +from documents.models import DocumentBarcode from documents.models import DocumentType from documents.models import Note from documents.models import ShareLink @@ -50,6 +51,7 @@ from paperless_mail.models import MailAccount from paperless_testing.assertions import FileSystemAssertsMixin from paperless_testing.dirs import DirectoriesMixin from paperless_testing.dirs import paperless_environment +from paperless_testing.factories import DocumentBarcodeFactory from paperless_testing.permissions import grant_object @@ -856,6 +858,37 @@ class TestExportImport( self.assertEqual(Document.objects.count(), 4) self.assertEqual(CustomFieldInstance.objects.count(), 1) + def _export_import_barcodes(self, *, split_manifest: bool) -> None: + shutil.rmtree(Path(self.dirs.media_dir) / "documents") + shutil.copytree( + Path(__file__).parent / "samples" / "documents", + Path(self.dirs.media_dir) / "documents", + ) + DocumentBarcodeFactory(document=self.d1, value="https://example.com") + DocumentBarcodeFactory(document=self.d2, page=2, value="DE8937") + + self._do_export(split_manifest=split_manifest) + + with paperless_environment(): + Document.objects.all().delete() + self.assertEqual(DocumentBarcode.objects.count(), 0) + call_command( + "document_importer", + "--no-progress-bar", + self.target, + skip_checks=True, + ) + self.assertEqual( + set(DocumentBarcode.objects.values_list("document", "page", "value")), + {(self.d1.pk, 1, "https://example.com"), (self.d2.pk, 2, "DE8937")}, + ) + + def test_export_import_barcodes(self) -> None: + self._export_import_barcodes(split_manifest=False) + + def test_export_import_barcodes_split_manifest(self) -> None: + self._export_import_barcodes(split_manifest=True) + def test_folder_prefix(self) -> None: """ GIVEN: diff --git a/src/documents/versioning.py b/src/documents/versioning.py index b1f0862f0..a3c005587 100644 --- a/src/documents/versioning.py +++ b/src/documents/versioning.py @@ -166,6 +166,25 @@ def get_latest_version_for_root( return latest or root_doc +def latest_version(document: Document) -> Document: + """ + The newest version of a root document, or the document itself if it has + no versions or is a version. Reads a prefetched "versions" lookup when + there is one and queries otherwise. + """ + if document.root_document_id is not None or document.pk is None: + return document + prefetched_cache = getattr(document, "_prefetched_objects_cache", None) + prefetched_versions = ( + prefetched_cache.get("versions") if isinstance(prefetched_cache, dict) else None + ) + if prefetched_versions is None: + return get_latest_version_for_root(document) + if not prefetched_versions: + return document + return sort_versions_newest_first(list(prefetched_versions))[0] + + def resolve_requested_version_for_root( root_doc: Document, request: Request, diff --git a/src/documents/views.py b/src/documents/views.py index 3bbd1c0df..515fd33d9 100644 --- a/src/documents/views.py +++ b/src/documents/views.py @@ -193,6 +193,7 @@ from documents.serialisers import BulkEditSerializer from documents.serialisers import CorrespondentSerializer from documents.serialisers import CustomFieldSerializer from documents.serialisers import DeleteDocumentsSerializer +from documents.serialisers import DocumentBarcodeSerializer from documents.serialisers import DocumentSelectionSerializer from documents.serialisers import DocumentSerializer from documents.serialisers import DocumentTypeSerializer @@ -845,6 +846,7 @@ class EmailDocumentDetailSchema(EmailSerializer): required=False, ), "lang": serializers.CharField(), + "barcodes": DocumentBarcodeSerializer(many=True), }, ), HTTPStatus.BAD_REQUEST: None, @@ -1526,6 +1528,7 @@ class DocumentViewSet( "original_filename": doc.original_filename, "archive_size": archive_filesize, "archive_metadata": archive_metadata, + "barcodes": DocumentBarcodeSerializer(doc.barcodes.all(), many=True).data, } lang = "en" diff --git a/src/paperless/config.py b/src/paperless/config.py index 116b18dde..4627a2c93 100644 --- a/src/paperless/config.py +++ b/src/paperless/config.py @@ -129,6 +129,7 @@ class BarcodeConfig(BaseConfig): barcode_enable_tag: bool = dataclasses.field(init=False) barcode_tag_mapping: dict[str, str] = dataclasses.field(init=False) barcode_tag_split: bool = dataclasses.field(init=False) + barcode_store_values: bool = dataclasses.field(init=False) def __post_init__(self) -> None: app_config = self._get_config_instance() @@ -179,6 +180,11 @@ class BarcodeConfig(BaseConfig): if app_config.barcode_tag_split is not None else settings.CONSUMER_TAG_BARCODE_SPLIT ) + self.barcode_store_values = ( + app_config.barcode_store_values + if app_config.barcode_store_values is not None + else settings.CONSUMER_STORE_BARCODE_VALUES + ) @dataclasses.dataclass diff --git a/src/paperless/migrations/0018_applicationconfiguration_barcode_store_values.py b/src/paperless/migrations/0018_applicationconfiguration_barcode_store_values.py new file mode 100644 index 000000000..01c96be85 --- /dev/null +++ b/src/paperless/migrations/0018_applicationconfiguration_barcode_store_values.py @@ -0,0 +1,21 @@ +# Generated by Django 5.2.16 on 2026-09-26 14:37 + +from django.db import migrations +from django.db import models + + +class Migration(migrations.Migration): + dependencies = [ + ("paperless", "0017_applicationconfiguration_llm_embedding_api_key"), + ] + + operations = [ + migrations.AddField( + model_name="applicationconfiguration", + name="barcode_store_values", + field=models.BooleanField( + null=True, + verbose_name="Stores the values of detected barcodes", + ), + ), + ] diff --git a/src/paperless/models.py b/src/paperless/models.py index 480b8629f..5dd502dfb 100644 --- a/src/paperless/models.py +++ b/src/paperless/models.py @@ -303,6 +303,12 @@ class ApplicationConfiguration(AbstractSingletonModel): null=True, ) + # PAPERLESS_CONSUMER_STORE_BARCODE_VALUES + barcode_store_values = models.BooleanField( + verbose_name=_("Stores the values of detected barcodes"), + null=True, + ) + """ Settings for the remote OCR parser """ diff --git a/src/paperless/settings/__init__.py b/src/paperless/settings/__init__.py index 3cdbb7b1f..7ca3bb66e 100644 --- a/src/paperless/settings/__init__.py +++ b/src/paperless/settings/__init__.py @@ -229,6 +229,7 @@ SPECTACULAR_SETTINGS = { }, "ENUM_NAME_OVERRIDES": { "MatchingAlgorithm": "documents.models.MatchingModel.MATCHING_ALGORITHMS", + "BarcodeFormatEnum": "documents.models.DocumentBarcode.Format", }, "SCHEMA_PATH_PREFIX_INSERT": FORCE_SCRIPT_NAME or "", } @@ -906,6 +907,10 @@ CONSUMER_TAG_BARCODE_SPLIT: Final[bool] = get_bool_from_env( "PAPERLESS_CONSUMER_TAG_BARCODE_SPLIT", ) +CONSUMER_STORE_BARCODE_VALUES: Final[bool] = get_bool_from_env( + "PAPERLESS_CONSUMER_STORE_BARCODE_VALUES", +) + CONSUMER_ENABLE_COLLATE_DOUBLE_SIDED: Final[bool] = get_bool_from_env( "PAPERLESS_CONSUMER_ENABLE_COLLATE_DOUBLE_SIDED", ) diff --git a/src/paperless_testing/factories.py b/src/paperless_testing/factories.py index 9ea6c8b39..eaff2c79e 100644 --- a/src/paperless_testing/factories.py +++ b/src/paperless_testing/factories.py @@ -9,6 +9,7 @@ from django.contrib.auth.models import User from documents.models import Correspondent from documents.models import Document +from documents.models import DocumentBarcode from documents.models import DocumentType from documents.models import MatchingModel from documents.models import PaperlessTask @@ -69,6 +70,16 @@ class DocumentFactory(TypedModelFactory[Document]): storage_path = None +class DocumentBarcodeFactory(TypedModelFactory[DocumentBarcode]): + class Meta: + model = DocumentBarcode + + document = factory.SubFactory(DocumentFactory) + page = 1 + value = factory.Faker("uri") + format = DocumentBarcode.Format.QR_CODE + + class UserFactory(TypedModelFactory[User]): class Meta: model = User