Feature: store barcode contents, list and search them (#14276)

* Feature: store barcode contents, list and search them

New setting PAPERLESS_CONSUMER_STORE_BARCODE_VALUES (off by default)
stores all barcodes found during consumption with the document: page,
type and content. They are listed on the metadata tab with a copy
button, returned by the documents API and searchable with barcodes:
in the advanced search. Versions keep their own barcodes, reprocessing
reads them again. Refs #9898

* Tests: cover the remaining barcode branches

Covers unchanged and failing barcode reads on reprocessing, unsupported files and DocumentBarcode.__str__, and uses toHaveLength in the barcode list spec as suggested by SonarCloud.

* Address review: keep barcodes when they can't be read, format choices

- Reprocessing and new versions keep the stored barcodes when the file
  can't be scanned or the scan fails, and replace them atomically.
- Format is a TextChoices of the zxing-cpp formats, with a test.
- Shared scan code, latest_version helper, TypedDict, serializer reuse.
- Barcodes in the split manifest, export/import tests, pytest-style tests.

* Barcode tests: TIFF reprocess, format check both ways, fixtures

- Reprocessing a TIFF with TIFF support off keeps the stored barcodes.
- The format test also fails when zxing-cpp drops a format.
- Fixtures in place of the sample dir mixin, plugin-level disabled test.
- Format labels aren't translated, OpenAPI enum named BarcodeFormatEnum.

* Review: module-level zxing reader, shorter barcode docs

- read_barcodes_zxing is a module-level function used by scan_pdf.
- Drop the trivial __str__ test, mark it no cover.
- Shorten the barcode docs and remove the duplicate in configuration.md.

* Fix header

* Use utility class

* Return barcodes from the metadata endpoint only

Drop the barcodes field and its prefetches from the document serializer,
as agreed in the review. Also remove the now empty component stylesheet.

---------

Co-authored-by: shamoon <4887959+shamoon@users.noreply.github.com>
This commit is contained in:
jurassicparkicecreamandshamoon authored and GitHub committed 2026-10-05 16:25:19 +00:00
1 parent 312684aef3
commit 62cc31fbdb
37 files changed
+1196 -80

No files matched your search

@@ -0,0 +1,95 @@
"""Stored barcode contents in the search index.
Barcodes are a JSON field like notes and custom fields: barcodes: resolves to
barcodes.value:, and a plain query without the prefix does not look at them.
"""
from __future__ import annotations
from typing import TYPE_CHECKING
import pytest
from documents.models import DocumentBarcode
from paperless_testing.factories import DocumentBarcodeFactory
from paperless_testing.factories import DocumentFactory
if TYPE_CHECKING:
from collections.abc import Callable
from documents.models import Document
from documents.search._backend import TantivyBackend
pytestmark = [pytest.mark.search, pytest.mark.django_db]
class TestBarcodeSearch:
@pytest.fixture
def with_barcodes(self, backend: TantivyBackend) -> Document:
document = DocumentFactory(title="Letter", content="x")
DocumentBarcodeFactory(
document=document,
value="WIFI:T:WPA;S:Guest-WLAN;P:crocodile123;;",
)
DocumentBarcodeFactory(
document=document,
page=2,
value="DE89370400440532013000",
format=DocumentBarcode.Format.CODE128,
)
backend.add_or_update(document)
return document
def test_bare_barcodes_prefix_searches_values(
self,
with_barcodes: Document,
index_document: Callable[..., Document],
matched_ids: Callable[[str], set[int]],
) -> None:
"""
GIVEN:
- A document with stored barcodes, and a decoy document whose
content (not a barcode) contains the same word
WHEN:
- A bare "barcodes:" prefix query is run
THEN:
- Only the document whose barcode matches is returned, also for a
part of a barcode between separators
"""
index_document(title="Decoy", content="crocodile123 in the text")
assert matched_ids("barcodes:crocodile123") == {with_barcodes.pk}
assert matched_ids("barcodes.value:DE89370400440532013000") == {
with_barcodes.pk,
}
def test_barcodes_format_subpath(
self,
with_barcodes: Document,
matched_ids: Callable[[str], set[int]],
) -> None:
"""
GIVEN:
- A document with a QR code and a Code 128 barcode
WHEN:
- The format subpath is queried
THEN:
- The document is found by its barcode formats
"""
assert matched_ids("barcodes.format:qrcode") == {with_barcodes.pk}
assert matched_ids("barcodes.format:aztec") == set()
def test_plain_query_ignores_barcodes(
self,
with_barcodes: Document,
matched_ids: Callable[[str], set[int]],
) -> None:
"""
GIVEN:
- A document with a barcode value not found in its text
WHEN:
- The value is searched without a field prefix
THEN:
- Nothing is found, as with notes and custom fields
"""
assert matched_ids("crocodile123") == set()
@@ -8,7 +8,7 @@ queryable-but-always-empty -- syntactically valid, silently matching
nothing -- with no test failure anywhere.
This indexes one real document carrying values for every JSON field
(a Note, a CustomFieldInstance) and inspects the document's own stored
(a Note, a CustomFieldInstance, a DocumentBarcode) and inspects the document's own stored
JSON payload, rather than running field-specific queries: that way a
future JSON field's subpaths are covered automatically, without a new
per-subpath query having to be added by hand each time.
@@ -27,6 +27,7 @@ from documents.models import CustomFieldInstance
from documents.models import Document
from documents.models import Note
from documents.search._fields import PUBLIC_FIELDS
from paperless_testing.factories import DocumentBarcodeFactory
from paperless_testing.factories import UserFactory
if TYPE_CHECKING:
@@ -42,11 +43,12 @@ class TestJsonSubpathsAreWrittenAtIndexTime:
) -> None:
"""
GIVEN:
- A document with a Note and a CustomFieldInstance attached
- A document with a Note, a CustomFieldInstance and a
DocumentBarcode attached
WHEN:
- The document is indexed via TantivyBackend.add_or_update
THEN:
- Every subpath PUBLIC_FIELDS declares for notes/custom_fields
- Every subpath PUBLIC_FIELDS declares for notes/custom_fields/barcodes
is present as a key in the document's stored JSON payload
"""
user = UserFactory(username="completeness-user")
@@ -65,6 +67,7 @@ class TestJsonSubpathsAreWrittenAtIndexTime:
field=field,
value_text="a value",
)
DocumentBarcodeFactory(document=doc, value="a barcode")
backend.add_or_update(doc)
index = backend._index
@@ -175,6 +175,14 @@ PINNED_DESCRIPTORS: tuple[FieldDescriptor, ...] = (
fast=False,
tokenizer="paperless_text",
),
FieldDescriptor(
"barcodes",
"json",
stored=True,
indexed=True,
fast=False,
tokenizer="paperless_text",
),
FieldDescriptor(
"title_sort",
"text",
@@ -74,6 +74,7 @@ class TestApiAppConfig(DirectoriesMixin, APITestCase):
"barcode_enable_tag": None,
"barcode_tag_mapping": None,
"barcode_tag_split": None,
"barcode_store_values": None,
"remote_ocr_engine": None,
"remote_ocr_api_key": None,
"remote_ocr_endpoint": None,
+316 -1
View File
@@ -1,29 +1,46 @@
from __future__ import annotations
import shutil
from collections.abc import Generator
from contextlib import contextmanager
from pathlib import Path
from typing import TYPE_CHECKING
import pytest
import zxingcpp
from django.conf import settings
from django.test import TestCase
from django.test import override_settings
from rest_framework import status
from documents import tasks
from documents.barcodes import BarcodePlugin
from documents.barcodes import read_barcode_values
from documents.consumer import ConsumerError
from documents.data_models import ConsumableDocument
from documents.data_models import DocumentMetadataOverrides
from documents.data_models import DocumentSource
from documents.models import Document
from documents.models import DocumentBarcode
from documents.models import Tag
from documents.plugins.base import StopConsumeTaskError
from documents.tests.utils import ConsumeTaskMixin
from documents.tests.utils import SampleDirMixin
from paperless.config import BarcodeConfig
from paperless.models import ApplicationConfiguration
from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.fakes.progress import FakeProgressManager
if TYPE_CHECKING:
from collections.abc import Callable
from collections.abc import Generator
from pytest_django.fixtures import Settings
from pytest_mock import MockerFixture
from rest_framework.test import APIClient
from paperless_testing.dirs import PaperlessDirs
class GetReaderPluginMixin:
@contextmanager
@@ -1127,3 +1144,301 @@ class TestTagBarcode(DirectoriesMixin, SampleDirMixin, GetReaderPluginMixin, Tes
document_list = reader.separate_pages(separator_pages)
self.assertEqual(len(document_list), 3)
SAMPLE_VALUES = [
{"page": 1, "value": "javascript:alert(1)", "format": "QRCode"},
{"page": 2, "value": "https://example.com/invoice/4711", "format": "QRCode"},
]
@pytest.fixture
def samples_dir() -> Path:
return Path(__file__).parent / "samples"
@pytest.fixture
def barcode_samples_dir(samples_dir: Path) -> Path:
return samples_dir / "barcodes"
@pytest.fixture
def store_barcodes(settings: Settings) -> None:
settings.CONSUMER_STORE_BARCODE_VALUES = True
@pytest.fixture
def barcode_reader(
paperless_dirs: PaperlessDirs,
) -> Generator[Callable[[Path], BarcodePlugin], None, None]:
readers: list[BarcodePlugin] = []
def make(path: Path) -> BarcodePlugin:
reader = BarcodePlugin(
ConsumableDocument(DocumentSource.ConsumeFolder, original_file=path),
DocumentMetadataOverrides(),
FakeProgressManager(path.name, None),
paperless_dirs.scratch_dir,
"task-id",
)
reader.setup()
readers.append(reader)
return reader
yield make
for reader in readers:
reader.cleanup()
@pytest.fixture
def consume_sample(
paperless_dirs: PaperlessDirs,
barcode_samples_dir: Path,
fake_progress_manager: type[FakeProgressManager],
settings: Settings,
) -> Callable[..., Document]:
settings.CELERY_TASK_ALWAYS_EAGER = True
settings.OCR_MODE = "auto"
def consume(name: str, *, root_document_id: int | None = None) -> Document:
dst = paperless_dirs.scratch_dir / name
shutil.copy(barcode_samples_dir / name, dst)
tasks.consume_file(
ConsumableDocument(
source=DocumentSource.ApiUpload
if root_document_id
else DocumentSource.ConsumeFolder,
original_file=dst,
root_document_id=root_document_id,
),
None,
)
return Document.objects.latest("id")
return consume
def _stored(document: Document) -> list[dict]:
return list(document.barcodes.values("page", "value", "format"))
def test_formats_cover_zxing() -> None:
"""
DocumentBarcode.Format matches the concrete formats of zxing-cpp, so a
zxing-cpp update that adds or removes one fails here
"""
members = zxingcpp.BarcodeFormat.__members__
# skip NONE and the groups like AllLinear, including their aliases
seen = {int(m) for n, m in members.items() if n == "NONE" or n.startswith("All")}
concrete = set()
for name, member in members.items():
# aliases such as DataBarExpanded come after the name zxing reports
if int(member) not in seen:
seen.add(int(member))
concrete.add(name)
assert set(DocumentBarcode.Format.values) == concrete
@pytest.mark.django_db
class TestBarcodeValues:
@pytest.mark.parametrize(
("filename", "mime_type", "tiff_support", "expected"),
[
pytest.param("simple.jpg", "image/jpeg", True, None, id="jpeg"),
pytest.param("simple.tiff", "image/tiff", False, None, id="tiff-off"),
pytest.param("simple.tiff", "image/tiff", True, [], id="tiff-on"),
],
)
def test_read_values_file_types(
self,
paperless_dirs: PaperlessDirs,
samples_dir: Path,
settings: Settings,
filename: str,
mime_type: str,
tiff_support: bool, # noqa: FBT001
expected: list | None,
) -> None:
"""
Files the scan doesn't support return None, so stored barcodes are
kept instead of being replaced with an empty list
"""
settings.CONSUMER_BARCODE_TIFF_SUPPORT = tiff_support
values = read_barcode_values(
samples_dir / filename,
mime_type,
BarcodeConfig(),
paperless_dirs.scratch_dir,
)
assert values == expected
def test_values_detected(
self,
barcode_reader: Callable[[Path], BarcodePlugin],
barcode_samples_dir: Path,
store_barcodes: None,
) -> None:
reader = barcode_reader(barcode_samples_dir / "barcode-qr-url.pdf")
assert reader.able_to_run
reader.run()
assert reader.metadata.barcodes == SAMPLE_VALUES
def test_values_disabled(
self,
barcode_reader: Callable[[Path], BarcodePlugin],
barcode_samples_dir: Path,
settings: Settings,
) -> None:
settings.CONSUMER_ENABLE_ASN_BARCODE = True
reader = barcode_reader(barcode_samples_dir / "barcode-qr-url.pdf")
reader.run()
assert reader.metadata.barcodes is None
def test_consume_and_reprocess(
self,
consume_sample: Callable[..., Document],
admin_client: APIClient,
store_barcodes: None,
) -> None:
"""
GIVEN:
- PDF with a QR code on each of its two pages
WHEN:
- File is consumed, the values are lost, and the document is reprocessed
THEN:
- The barcodes are stored, shown in the API and searchable
- Reprocessing reads them again
"""
document = consume_sample("barcode-qr-url.pdf")
assert _stored(document) == SAMPLE_VALUES
response = admin_client.get(f"/api/documents/{document.pk}/metadata/")
assert response.status_code == status.HTTP_200_OK
assert response.data["barcodes"] == SAMPLE_VALUES
response = admin_client.get(f"/api/documents/{document.pk}/")
assert "barcodes" not in response.data
response = admin_client.get("/api/documents/?query=barcodes:invoice")
assert [x["id"] for x in response.data["results"]] == [document.pk]
response = admin_client.get("/api/documents/?query=invoice")
assert response.data["results"] == []
document.barcodes.all().delete()
modified = Document.objects.get(pk=document.pk).modified
tasks.update_document_content_maybe_archive_file(document.pk)
assert _stored(document) == SAMPLE_VALUES
assert Document.objects.get(pk=document.pk).modified > modified
@pytest.mark.parametrize(
"read_values",
[
pytest.param({"side_effect": RuntimeError("broken")}, id="scan-fails"),
pytest.param({"return_value": None}, id="not-scannable"),
],
)
def test_reprocess_keeps_values(
self,
consume_sample: Callable[..., Document],
mocker: MockerFixture,
store_barcodes: None,
read_values: dict,
) -> None:
"""
GIVEN:
- A document with stored barcodes
WHEN:
- It is reprocessed, but the barcodes can't be read
THEN:
- The stored barcodes are kept
"""
document = consume_sample("barcode-qr-url.pdf")
mocker.patch("documents.tasks.read_barcode_values", **read_values)
tasks.update_document_content_maybe_archive_file(document.pk)
assert _stored(document) == SAMPLE_VALUES
def test_reprocess_tiff_support_off_keeps_values(
self,
consume_sample: Callable[..., Document],
settings: Settings,
store_barcodes: None,
) -> None:
"""
GIVEN:
- A TIFF document with barcodes stored while TIFF support was on
WHEN:
- TIFF support is turned off and the document is reprocessed
THEN:
- The stored barcodes are kept
"""
settings.CONSUMER_BARCODE_TIFF_SUPPORT = True
document = consume_sample("patch-code-t-middle.tiff")
stored = _stored(document)
assert stored
settings.CONSUMER_BARCODE_TIFF_SUPPORT = False
tasks.update_document_content_maybe_archive_file(document.pk)
assert _stored(document) == stored
def test_reprocess_values_disabled(
self,
consume_sample: Callable[..., Document],
settings: Settings,
store_barcodes: None,
) -> None:
document = consume_sample("barcode-qr-url.pdf")
settings.CONSUMER_STORE_BARCODE_VALUES = False
assert tasks._read_barcodes_for_reprocess(document) is None
def test_consume_version_stores_own_values(
self,
consume_sample: Callable[..., Document],
admin_client: APIClient,
store_barcodes: None,
) -> None:
"""
GIVEN:
- A document with stored barcodes
WHEN:
- A new version with a different barcode is consumed, like after
rotating or removing pages
THEN:
- The version keeps its own barcodes, the original ones are kept
- The metadata and the search use those of the newest version
"""
root = consume_sample("barcode-qr-url.pdf")
version = consume_sample("barcode-128-custom.pdf", root_document_id=root.pk)
assert version.root_document == root
assert _stored(version) == [
{"page": 1, "value": "CUSTOM BARCODE", "format": "Code128"},
]
assert root.barcodes.count() == 2
assert [x.value for x in root.get_effective_barcodes()] == ["CUSTOM BARCODE"]
response = admin_client.get(f"/api/documents/{root.pk}/metadata/")
assert [x["value"] for x in response.data["barcodes"]] == ["CUSTOM BARCODE"]
response = admin_client.get('/api/documents/?query=barcodes:"custom barcode"')
assert [x["id"] for x in response.data["results"]] == [root.pk]
response = admin_client.get("/api/documents/?query=barcodes:invoice")
assert response.data["results"] == []
def test_consume_version_scan_fails(
self,
consume_sample: Callable[..., Document],
mocker: MockerFixture,
store_barcodes: None,
) -> None:
"""
A failed scan of a new version is logged and doesn't stop consumption
"""
root = consume_sample("barcode-qr-url.pdf")
mocker.patch(
"documents.consumer.read_barcode_values",
side_effect=RuntimeError("broken"),
)
version = consume_sample("barcode-128-custom.pdf", root_document_id=root.pk)
assert version.root_document == root
assert not version.barcodes.exists()
@@ -32,6 +32,7 @@ from documents.models import Correspondent
from documents.models import CustomField
from documents.models import CustomFieldInstance
from documents.models import Document
from documents.models import DocumentBarcode
from documents.models import DocumentType
from documents.models import Note
from documents.models import ShareLink
@@ -50,6 +51,7 @@ from paperless_mail.models import MailAccount
from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.dirs import paperless_environment
from paperless_testing.factories import DocumentBarcodeFactory
from paperless_testing.permissions import grant_object
@@ -856,6 +858,37 @@ class TestExportImport(
self.assertEqual(Document.objects.count(), 4)
self.assertEqual(CustomFieldInstance.objects.count(), 1)
def _export_import_barcodes(self, *, split_manifest: bool) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
DocumentBarcodeFactory(document=self.d1, value="https://example.com")
DocumentBarcodeFactory(document=self.d2, page=2, value="DE8937")
self._do_export(split_manifest=split_manifest)
with paperless_environment():
Document.objects.all().delete()
self.assertEqual(DocumentBarcode.objects.count(), 0)
call_command(
"document_importer",
"--no-progress-bar",
self.target,
skip_checks=True,
)
self.assertEqual(
set(DocumentBarcode.objects.values_list("document", "page", "value")),
{(self.d1.pk, 1, "https://example.com"), (self.d2.pk, 2, "DE8937")},
)
def test_export_import_barcodes(self) -> None:
self._export_import_barcodes(split_manifest=False)
def test_export_import_barcodes_split_manifest(self) -> None:
self._export_import_barcodes(split_manifest=True)
def test_folder_prefix(self) -> None:
"""
GIVEN: