From 0f6bdaf5de105194ab9f22023a63695d55d3f556 Mon Sep 17 00:00:00 2001 From: Trenton H <797416+stumpylog@users.noreply.github.com> Date: Mon, 9 Mar 2026 13:54:52 -0700 Subject: [PATCH] Feature: Add parser plugin registry and ParserProtocol (Phase 1 & 2) Introduces the foundation of the entrypoint-based parser discovery system to replace the signal-based document_consumer_declaration approach. - Add ParserProtocol: runtime_checkable Protocol defining the full contract for document parsers (supported_mime_types, score, parse, context manager, result accessors) - Add ParserRegistry: lazy singleton with entrypoint discovery via importlib.metadata group 'paperless_ngx.parsers', uniform score-based selection across external and built-in parsers - Add get_parser_registry(), init_builtin_parsers(), reset_parser_registry() module-level helpers - Wire Celery worker_process_init to call init_builtin_parsers() eagerly in each worker, deferring third-party discovery to first task use - Add 28 pytest tests covering Protocol compliance, singleton lifecycle, scoring logic, entrypoint discovery, and log output Built-in parsers and consumer migration follow in Phases 3-6. Co-Authored-By: Claude Sonnet 4.6 --- src/paperless/celery.py | 16 + src/paperless/parsers/__init__.py | 341 ++++++++++++ src/paperless/parsers/registry.py | 374 ++++++++++++++ src/paperless/tests/test_registry.py | 748 +++++++++++++++++++++++++++ 4 files changed, 1479 insertions(+) create mode 100644 src/paperless/parsers/__init__.py create mode 100644 src/paperless/parsers/registry.py create mode 100644 src/paperless/tests/test_registry.py diff --git a/src/paperless/celery.py b/src/paperless/celery.py index a9a853521..5d6a01dd2 100644 --- a/src/paperless/celery.py +++ b/src/paperless/celery.py @@ -1,6 +1,7 @@ import os from celery import Celery +from celery.signals import worker_process_init # Set the default Django settings module for the 'celery' program. os.environ.setdefault("DJANGO_SETTINGS_MODULE", "paperless.settings") @@ -15,3 +16,18 @@ app.config_from_object("django.conf:settings", namespace="CELERY") # Load task modules from all registered Django apps. app.autodiscover_tasks() + + +@worker_process_init.connect +def on_worker_process_init(**kwargs) -> None: + """Register built-in parsers eagerly in each Celery worker process. + + This registers only the built-in parsers (no entrypoint discovery) so + that workers can begin consuming documents immediately. Entrypoint + discovery for third-party parsers is deferred to the first call of + ``get_parser_registry()`` inside a task, keeping ``worker_process_init`` + well within its 4-second timeout budget. + """ + from paperless.parsers.registry import init_builtin_parsers + + init_builtin_parsers() diff --git a/src/paperless/parsers/__init__.py b/src/paperless/parsers/__init__.py new file mode 100644 index 000000000..f70fb771e --- /dev/null +++ b/src/paperless/parsers/__init__.py @@ -0,0 +1,341 @@ +""" +paperless.parsers +================= + +Public interface for the Paperless-ngx parser plugin system. + +This module defines :class:`ParserProtocol` — the structural contract that +every document parser must satisfy — whether it is a built-in parser shipped +with Paperless-ngx or a third-party parser installed via a Python entrypoint. + +Phase 1/2 scope +--------------- +Only the Protocol is defined here. The transitional :class:`DocumentParser` +ABC (Phase 3) and concrete built-in parsers (Phase 3+) will be added in later +phases, so there are intentionally no imports of parser implementations here. + +Usage example (third-party parser):: + + from paperless.parsers import ParserProtocol + + class MyParser: + name = "my-parser" + version = "1.0.0" + author = "Acme Corp" + url = "https://example.com/my-parser" + + @classmethod + def supported_mime_types(cls) -> dict[str, str]: + return {"application/x-my-format": ".myf"} + + @classmethod + def score(cls, mime_type, filename, path=None): + return 10 + + # … implement remaining protocol methods … + + assert isinstance(MyParser(), ParserProtocol) +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING +from typing import Protocol +from typing import Self +from typing import runtime_checkable + +if TYPE_CHECKING: + import datetime + from pathlib import Path + +__all__ = [ + "ParserProtocol", +] + + +@runtime_checkable +class ParserProtocol(Protocol): + """Structural contract for all Paperless-ngx document parsers. + + Both built-in parsers and third-party plugins (discovered via the + ``paperless_ngx.parsers`` entrypoint group) must satisfy this Protocol. + Because it is decorated with :func:`typing.runtime_checkable`, + ``isinstance(obj, ParserProtocol)`` works at runtime based on method + presence, which is useful for validation in :meth:`ParserRegistry.discover`. + + Class-level identity attributes + -------------------------------- + Parsers are required to expose four string attributes at the **class** + level so the registry can log attribution information without + instantiating the parser: + + name : str + Human-readable parser name (e.g. ``"Tesseract OCR"``). + version : str + Semantic version string (e.g. ``"1.2.3"``). + author : str + Author or organisation name. + url : str + URL for documentation, source code, or issue tracker. + """ + + # ------------------------------------------------------------------ + # Class-level identity (checked by the registry, not Protocol methods) + # ------------------------------------------------------------------ + + name: str + version: str + author: str + url: str + + # ------------------------------------------------------------------ + # Class methods + # ------------------------------------------------------------------ + + @classmethod + def supported_mime_types(cls) -> dict[str, str]: + """Return a mapping of supported MIME types to preferred file extensions. + + The keys are MIME type strings (e.g. ``"application/pdf"``), and the + values are the preferred file extension **including** the leading dot + (e.g. ``".pdf"``). The registry uses this mapping both to decide + whether a parser is a candidate for a given file and to determine the + default extension when creating archive copies. + + Returns + ------- + dict[str, str] + ``{mime_type: extension}`` mapping — may be empty if the parser + has been temporarily disabled. + """ + ... + + @classmethod + def score( + cls, + mime_type: str, + filename: str, + path: Path | None = None, + ) -> int | None: + """Return a priority score for handling ``mime_type`` on ``filename``. + + The registry calls this method after confirming that the MIME type is + in :meth:`supported_mime_types`. Parsers may inspect ``filename`` + (and optionally the file at ``path``) to refine their confidence level. + + A higher score wins. Return ``None`` to explicitly decline handling + a file even though the MIME type is listed as supported (e.g. when the + parser detects a feature flag is disabled, or a licence has expired). + + Parameters + ---------- + mime_type: + The detected MIME type of the file to be parsed. + filename: + The original filename, including extension. + path: + Optional filesystem path to the file. Parsers that need to + inspect file content (e.g. magic-byte sniffing) may use this. + The path may be ``None`` when scoring happens before the file + is available locally. + + Returns + ------- + int | None + Priority score (higher wins), or ``None`` to decline. + """ + ... + + # ------------------------------------------------------------------ + # Properties + # ------------------------------------------------------------------ + + @property + def can_produce_archive(self) -> bool: + """Whether this parser can produce a searchable PDF archive copy. + + If ``True``, the consumption pipeline will request an archive version + when the document is processed. If ``False``, only the thumbnail and + text extraction will be performed. + """ + ... + + @property + def requires_pdf_rendition(self) -> bool: + """Whether the parser requires a pre-rendered PDF before parsing. + + Some parsers (e.g. image-based OCR engines) work on rasterised PDFs + rather than the original file. When ``True``, the pipeline will + convert the source document to PDF before calling :meth:`parse`. + """ + ... + + # ------------------------------------------------------------------ + # Core parsing interface + # ------------------------------------------------------------------ + + def parse( + self, + document_path: Path, + mime_type: str, + file_name: str | None = None, + *, + produce_archive: bool = True, + ) -> None: + """Parse ``document_path`` and populate internal state. + + After a successful call, callers retrieve results via + :meth:`get_text`, :meth:`get_date`, and :meth:`get_archive_path`. + + Parameters + ---------- + document_path: + Absolute path to the document file to parse. + mime_type: + Detected MIME type of the document. + file_name: + Original filename as provided by the user. May differ from the + stem of ``document_path`` (which is usually a UUID-based name). + produce_archive: + When ``True`` (the default) and :attr:`can_produce_archive` is + also ``True``, the parser should produce a searchable PDF at the + path returned by :meth:`get_archive_path`. Pass ``False`` when + only text extraction and thumbnail generation are required and + disk I/O should be minimised. + + Raises + ------ + documents.parsers.ParseError + If parsing fails for any reason. The consumption pipeline will + catch this and handle failure appropriately. + """ + ... + + # ------------------------------------------------------------------ + # Result accessors + # ------------------------------------------------------------------ + + def get_text(self) -> str | None: + """Return the plain-text content extracted during :meth:`parse`. + + Returns + ------- + str | None + Extracted text, or ``None`` if no text could be found. + """ + ... + + def get_date(self) -> datetime.datetime | None: + """Return the document date detected during :meth:`parse`. + + Returns + ------- + datetime.datetime | None + Detected document date, or ``None`` if no date was found. + """ + ... + + def get_archive_path(self) -> Path | None: + """Return the path to the generated archive PDF (if any). + + Returns + ------- + Path | None + Path to the searchable PDF archive, or ``None`` if no archive + was produced (e.g. because ``produce_archive=False`` was passed + to :meth:`parse`, or the parser does not support archive + production). + """ + ... + + # ------------------------------------------------------------------ + # Thumbnail and metadata + # ------------------------------------------------------------------ + + def get_thumbnail( + self, + document_path: Path, + mime_type: str, + file_name: str | None = None, + ) -> Path: + """Generate and return the path to a thumbnail image for the document. + + Unlike :meth:`parse`, this method may be called independently of + :meth:`parse`. The returned path must point to an existing WebP image + file inside the parser's temporary working directory. + + Parameters + ---------- + document_path: + Absolute path to the source document. + mime_type: + Detected MIME type of the document. + file_name: + Original filename. + + Returns + ------- + Path + Path to the generated thumbnail image (WebP format preferred). + """ + ... + + def get_page_count( + self, + document_path: Path, + mime_type: str, + ) -> int | None: + """Return the number of pages in the document, if determinable. + + Parameters + ---------- + document_path: + Absolute path to the source document. + mime_type: + Detected MIME type of the document. + + Returns + ------- + int | None + Page count, or ``None`` if the parser cannot determine it. + """ + ... + + # ------------------------------------------------------------------ + # Context manager + # ------------------------------------------------------------------ + + def __enter__(self) -> Self: + """Enter the parser context, returning the parser instance. + + Implementations should perform any resource allocation (e.g. creating + a temporary working directory) here if not done in ``__init__``. + + Returns + ------- + Self + The parser instance itself. + """ + ... + + def __exit__( + self, + exc_type: type[BaseException] | None, + exc_val: BaseException | None, + exc_tb: object, + ) -> None: + """Exit the parser context and release resources. + + Implementations must clean up all temporary files and other + resources regardless of whether an exception occurred. + + Parameters + ---------- + exc_type: + The exception class, or ``None`` if no exception was raised. + exc_val: + The exception instance, or ``None``. + exc_tb: + The traceback, or ``None``. + """ + ... diff --git a/src/paperless/parsers/registry.py b/src/paperless/parsers/registry.py new file mode 100644 index 000000000..3b2502c75 --- /dev/null +++ b/src/paperless/parsers/registry.py @@ -0,0 +1,374 @@ +""" +paperless.parsers.registry +========================== + +Singleton registry that tracks all document parsers available to +Paperless-ngx — both built-ins shipped with the application and third-party +plugins installed via Python entrypoints. + +Public surface +-------------- +:func:`get_parser_registry` + Lazy-initialise and return the shared :class:`ParserRegistry`. This is + the primary entry point for production code. + +:func:`init_builtin_parsers` + Register built-in parsers only, without entrypoint discovery. Safe to + call from Celery ``worker_process_init`` where importing all entrypoints + would be wasteful or cause side effects. + +:func:`reset_parser_registry` + Reset module-level state. **For tests only.** + +Entrypoint group +---------------- +Third-party parsers must advertise themselves under the +``paperless_ngx.parsers`` entrypoint group in their ``pyproject.toml``:: + + [project.entry-points."paperless_ngx.parsers"] + my_parser = "my_package.parsers:MyParser" + +The loaded class must expose the following attributes *at the class level* +(not just on instances) for the registry to accept it: +``name``, ``version``, ``author``, ``url``, +``supported_mime_types`` (callable), ``score`` (callable). +""" + +from __future__ import annotations + +import logging +from importlib.metadata import entry_points +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from pathlib import Path + +logger = logging.getLogger("paperless.parsers.registry") + +# --------------------------------------------------------------------------- +# Module-level singleton state +# --------------------------------------------------------------------------- + +_registry: ParserRegistry | None = None +_discovery_complete: bool = False + +# Attribute names that every registered external parser class must expose. +_REQUIRED_ATTRS: tuple[str, ...] = ( + "name", + "version", + "author", + "url", + "supported_mime_types", + "score", +) + + +# --------------------------------------------------------------------------- +# Module-level accessor functions +# --------------------------------------------------------------------------- + + +def get_parser_registry() -> ParserRegistry: + """Return the shared :class:`ParserRegistry` instance. + + On the first call this function: + + 1. Creates a new :class:`ParserRegistry`. + 2. Calls :meth:`~ParserRegistry.register_defaults` to install built-in + parsers. + 3. Calls :meth:`~ParserRegistry.discover` to load third-party plugins via + ``importlib.metadata`` entrypoints. + 4. Calls :meth:`~ParserRegistry.log_summary` to emit a startup summary. + + Subsequent calls return the same instance immediately. + + Returns + ------- + ParserRegistry + The shared registry singleton. + """ + global _registry, _discovery_complete + + if _registry is None: + _registry = ParserRegistry() + _registry.register_defaults() + + if not _discovery_complete: + _registry.discover() + _registry.log_summary() + _discovery_complete = True + + return _registry + + +def init_builtin_parsers() -> None: + """Register built-in parsers without performing entrypoint discovery. + + This function is intended for use in Celery ``worker_process_init`` + handlers and similar contexts where importing all installed entrypoints + would be wasteful, slow, or could produce undesirable side effects. + + It is safe to call this function multiple times; subsequent calls are + no-ops. Entrypoint discovery (i.e. third-party plugins) is deliberately + **not** performed. + + Returns + ------- + None + """ + global _registry + + if _registry is None: + _registry = ParserRegistry() + _registry.register_defaults() + + +def reset_parser_registry() -> None: + """Reset the module-level registry state to its initial values. + + This resets both :data:`_registry` and :data:`_discovery_complete` so + that the next call to :func:`get_parser_registry` will re-initialise + everything from scratch. + + .. warning:: + **FOR TESTS ONLY.** Do not call this function in production code. + Resetting the registry mid-request will cause all subsequent parser + lookups to go through discovery again, which is expensive and may + have unexpected side effects in multi-threaded environments. + + Returns + ------- + None + """ + global _registry, _discovery_complete + + _registry = None + _discovery_complete = False + + +# --------------------------------------------------------------------------- +# Registry class +# --------------------------------------------------------------------------- + + +class ParserRegistry: + """Registry that maps MIME types to the best available parser class. + + Parsers are partitioned into two lists: + + ``_builtins`` + Parser classes registered via :meth:`register_builtin` (populated by + :meth:`register_defaults` in Phase 3+). + + ``_external`` + Parser classes loaded from installed Python entrypoints via + :meth:`discover`. + + When resolving a parser for a file, external parsers are evaluated + alongside built-in parsers using a uniform scoring mechanism. Both lists + are iterated together; the class with the highest :meth:`~ParserProtocol.score` + wins. If an external parser wins, its attribution details are logged so + users can identify which third-party package handled their document. + """ + + def __init__(self) -> None: + self._external: list[type] = [] + self._builtins: list[type] = [] + + # ------------------------------------------------------------------ + # Registration + # ------------------------------------------------------------------ + + def register_builtin(self, parser_class: type) -> None: + """Register a built-in parser class. + + Built-in parsers are shipped with Paperless-ngx and are appended to + the ``_builtins`` list. They are never overridden by external parsers; + instead, scoring determines which parser wins for any given file. + + Parameters + ---------- + parser_class: + The parser class to register. Must satisfy + :class:`~paperless.parsers.ParserProtocol`. + """ + self._builtins.append(parser_class) + + def register_defaults(self) -> None: + """Register the built-in parsers that ship with Paperless-ngx. + + Populated in Phase 3 when built-in parsers implement the new + interface. In Phase 1/2 this is intentionally a no-op so that the + registry infrastructure can be tested in isolation without depending + on any concrete parser implementations. + """ + + # ------------------------------------------------------------------ + # Discovery + # ------------------------------------------------------------------ + + def discover(self) -> None: + """Load third-party parsers from the ``paperless_ngx.parsers`` entrypoint group. + + For each advertised entrypoint the method: + + 1. Calls ``ep.load()`` to import the class. + 2. Validates that the class exposes all required attributes. + 3. On success, appends the class to :attr:`_external` and logs an + info message. + 4. On failure (import error or missing attributes), logs an + appropriate warning/error and continues to the next entrypoint. + + Errors during discovery of a single parser do not prevent other + parsers from being loaded. + + Returns + ------- + None + """ + eps = entry_points(group="paperless_ngx.parsers") + + for ep in eps: + try: + parser_class = ep.load() + except Exception: + logger.exception( + "Failed to load parser entrypoint '%s' — skipping.", + ep.name, + ) + continue + + missing = [ + attr for attr in _REQUIRED_ATTRS if not hasattr(parser_class, attr) + ] + if missing: + logger.warning( + "Parser loaded from entrypoint '%s' is missing required " + "attributes %r — skipping.", + ep.name, + missing, + ) + continue + + self._external.append(parser_class) + logger.info( + "Loaded third-party parser '%s' v%s by %s (entrypoint: '%s').", + parser_class.name, + parser_class.version, + parser_class.author, + ep.name, + ) + + # ------------------------------------------------------------------ + # Summary logging + # ------------------------------------------------------------------ + + def log_summary(self) -> None: + """Log a startup summary of all registered parsers. + + Built-in parsers are listed first, followed by any external parsers + discovered from entrypoints. If no external parsers were found a + short informational message is logged instead of an empty list. + + Returns + ------- + None + """ + logger.info( + "Built-in parsers (%d):", + len(self._builtins), + ) + for cls in self._builtins: + logger.info( + " [built-in] %s v%s — %s", + getattr(cls, "name", repr(cls)), + getattr(cls, "version", "unknown"), + getattr(cls, "url", "built-in"), + ) + + if not self._external: + logger.info("No third-party parsers discovered.") + return + + logger.info( + "Third-party parsers (%d):", + len(self._external), + ) + for cls in self._external: + logger.info( + " [external] %s v%s by %s — report issues at %s", + getattr(cls, "name", repr(cls)), + getattr(cls, "version", "unknown"), + getattr(cls, "author", "unknown"), + getattr(cls, "url", "unknown"), + ) + + # ------------------------------------------------------------------ + # Parser resolution + # ------------------------------------------------------------------ + + def get_parser_for_file( + self, + mime_type: str, + filename: str, + path: Path | None = None, + ) -> type | None: + """Return the best parser class for the given file, or ``None``. + + All registered parsers (external first, then built-ins) are evaluated + against the file. A parser is eligible if: + + * ``mime_type`` appears in the dict returned by its + ``supported_mime_types()`` classmethod, **and** + * its ``score()`` classmethod returns a non-``None`` integer. + + The parser with the highest score wins. When two parsers return the + same score, the one that appears earlier in the evaluation order wins + (external parsers are evaluated before built-ins, giving third-party + packages a chance to override defaults at equal priority). + + When an external parser is selected, its identity is logged at + ``INFO`` level so operators can trace which package handled a document. + + Parameters + ---------- + mime_type: + The detected MIME type of the file. + filename: + The original filename, including extension. + path: + Optional filesystem path to the file. Forwarded to each + parser's ``score()`` method. + + Returns + ------- + type | None + The winning parser class, or ``None`` if no parser can handle + the file. + """ + best_score: int | None = None + best_parser: type | None = None + + # External parsers are placed first so that, at equal scores, an + # external parser wins over a built-in (first-seen policy). + for parser_class in (*self._external, *self._builtins): + if mime_type not in parser_class.supported_mime_types(): + continue + + score = parser_class.score(mime_type, filename, path) + if score is None: + continue + + if best_score is None or score > best_score: + best_score = score + best_parser = parser_class + + if best_parser is not None and best_parser in self._external: + logger.info( + "Document handled by third-party parser '%s' v%s — %s", + getattr(best_parser, "name", repr(best_parser)), + getattr(best_parser, "version", "unknown"), + getattr(best_parser, "url", "unknown"), + ) + + return best_parser diff --git a/src/paperless/tests/test_registry.py b/src/paperless/tests/test_registry.py new file mode 100644 index 000000000..8fb0c00bc --- /dev/null +++ b/src/paperless/tests/test_registry.py @@ -0,0 +1,748 @@ +""" +Tests for :mod:`paperless.parsers` (ParserProtocol) and +:mod:`paperless.parsers.registry` (ParserRegistry + module-level helpers). + +All tests use pytest-style functions/classes — no unittest.TestCase. +The ``clean_registry`` fixture ensures complete isolation between tests by +resetting the module-level singleton before and after every test. +""" + +from __future__ import annotations + +import logging +from importlib.metadata import EntryPoint +from pathlib import Path +from typing import Self +from unittest.mock import MagicMock +from unittest.mock import patch + +import pytest + +from paperless.parsers import ParserProtocol +from paperless.parsers.registry import ParserRegistry +from paperless.parsers.registry import get_parser_registry +from paperless.parsers.registry import init_builtin_parsers +from paperless.parsers.registry import reset_parser_registry + +# --------------------------------------------------------------------------- +# Shared fixtures +# --------------------------------------------------------------------------- + + +@pytest.fixture(autouse=True) +def clean_registry() -> None: + """Reset the global parser registry before and after every test. + + GIVEN: The registry module carries module-level singleton state. + WHEN: Any test is executed. + THEN: Each test starts and ends with a clean slate, preventing state + leak between tests. + """ + reset_parser_registry() + yield + reset_parser_registry() + + +@pytest.fixture() +def dummy_parser_cls() -> type: + """Return a class that fully satisfies :class:`ParserProtocol`. + + GIVEN: A need to exercise registry and Protocol logic with a minimal + but complete parser. + WHEN: A test requests this fixture. + THEN: A class with all required attributes and methods is returned. + """ + + class DummyParser: + name = "dummy-parser" + version = "0.1.0" + author = "Test Author" + url = "https://example.com/dummy-parser" + + @classmethod + def supported_mime_types(cls) -> dict[str, str]: + return {"text/plain": ".txt"} + + @classmethod + def score( + cls, + mime_type: str, + filename: str, + path: Path | None = None, + ) -> int | None: + return 10 + + @property + def can_produce_archive(self) -> bool: + return False + + @property + def requires_pdf_rendition(self) -> bool: + return False + + def parse( + self, + document_path: Path, + mime_type: str, + file_name: str | None = None, + *, + produce_archive: bool = True, + ) -> None: + pass + + def get_text(self) -> str | None: + return None + + def get_date(self) -> None: + return None + + def get_archive_path(self) -> Path | None: + return None + + def get_thumbnail( + self, + document_path: Path, + mime_type: str, + file_name: str | None = None, + ) -> Path: + return Path("/tmp/thumbnail.webp") + + def get_page_count( + self, + document_path: Path, + mime_type: str, + ) -> int | None: + return None + + def __enter__(self) -> Self: + return self + + def __exit__(self, exc_type, exc_val, exc_tb) -> None: + pass + + return DummyParser + + +# --------------------------------------------------------------------------- +# TestParserProtocol +# --------------------------------------------------------------------------- + + +class TestParserProtocol: + """Verify runtime isinstance() checks against ParserProtocol.""" + + def test_compliant_class_instance_passes_isinstance( + self, + dummy_parser_cls: type, + ) -> None: + """ + GIVEN: A class that implements every method required by ParserProtocol. + WHEN: isinstance() is called with the Protocol. + THEN: The check passes (returns True). + """ + instance = dummy_parser_cls() + assert isinstance(instance, ParserProtocol) + + def test_non_compliant_class_instance_fails_isinstance(self) -> None: + """ + GIVEN: A plain class with no parser-related methods. + WHEN: isinstance() is called with ParserProtocol. + THEN: The check fails (returns False). + """ + + class Unrelated: + pass + + assert not isinstance(Unrelated(), ParserProtocol) + + @pytest.mark.parametrize( + "missing_method", + [ + pytest.param("parse", id="missing-parse"), + pytest.param("get_text", id="missing-get_text"), + pytest.param("get_thumbnail", id="missing-get_thumbnail"), + pytest.param("__enter__", id="missing-__enter__"), + pytest.param("__exit__", id="missing-__exit__"), + ], + ) + def test_partial_compliant_fails_isinstance( + self, + dummy_parser_cls: type, + missing_method: str, + ) -> None: + """ + GIVEN: A class that satisfies ParserProtocol except for one method. + WHEN: isinstance() is called with ParserProtocol. + THEN: The check fails because the Protocol is not fully satisfied. + """ + # Create a subclass and delete the specified method to break compliance. + partial_cls = type( + "PartialParser", + (dummy_parser_cls,), + {missing_method: None}, # Replace with None — not callable + ) + assert not isinstance(partial_cls(), ParserProtocol) + + +# --------------------------------------------------------------------------- +# TestRegistrySingleton +# --------------------------------------------------------------------------- + + +class TestRegistrySingleton: + """Verify the module-level singleton lifecycle functions.""" + + def test_get_parser_registry_returns_instance(self) -> None: + """ + GIVEN: No registry has been created yet. + WHEN: get_parser_registry() is called. + THEN: A ParserRegistry instance is returned. + """ + registry = get_parser_registry() + assert isinstance(registry, ParserRegistry) + + def test_get_parser_registry_same_instance_on_repeated_calls(self) -> None: + """ + GIVEN: A registry instance was created by a prior call. + WHEN: get_parser_registry() is called a second time. + THEN: The exact same object (identity) is returned. + """ + first = get_parser_registry() + second = get_parser_registry() + assert first is second + + def test_reset_parser_registry_gives_fresh_instance(self) -> None: + """ + GIVEN: A registry instance already exists. + WHEN: reset_parser_registry() is called and then get_parser_registry() + is called again. + THEN: A new, distinct registry instance is returned. + """ + first = get_parser_registry() + reset_parser_registry() + second = get_parser_registry() + assert first is not second + + def test_init_builtin_parsers_does_not_run_discover( + self, + monkeypatch: pytest.MonkeyPatch, + ) -> None: + """ + GIVEN: discover() would raise an exception if called. + WHEN: init_builtin_parsers() is called. + THEN: No exception is raised, confirming discover() was not invoked. + """ + + def exploding_discover(self) -> None: + raise RuntimeError( + "discover() must not be called from init_builtin_parsers", + ) + + monkeypatch.setattr(ParserRegistry, "discover", exploding_discover) + + # Should complete without raising. + init_builtin_parsers() + + def test_init_builtin_parsers_idempotent(self) -> None: + """ + GIVEN: init_builtin_parsers() has already been called once. + WHEN: init_builtin_parsers() is called a second time. + THEN: No error is raised and the same registry instance is reused. + """ + init_builtin_parsers() + # Capture the registry created by the first call. + import paperless.parsers.registry as reg_module + + first_registry = reg_module._registry + + init_builtin_parsers() + + assert reg_module._registry is first_registry + + +# --------------------------------------------------------------------------- +# TestParserRegistryGetParserForFile +# --------------------------------------------------------------------------- + + +class TestParserRegistryGetParserForFile: + """Verify parser selection logic in get_parser_for_file().""" + + def test_returns_none_when_no_parsers_registered(self) -> None: + """ + GIVEN: A registry with no parsers registered. + WHEN: get_parser_for_file() is called for any MIME type. + THEN: None is returned. + """ + registry = ParserRegistry() + result = registry.get_parser_for_file("text/plain", "doc.txt") + assert result is None + + def test_returns_none_for_unsupported_mime_type( + self, + dummy_parser_cls: type, + ) -> None: + """ + GIVEN: A registry with a parser that supports only 'text/plain'. + WHEN: get_parser_for_file() is called with 'application/pdf'. + THEN: None is returned. + """ + registry = ParserRegistry() + registry.register_builtin(dummy_parser_cls) + result = registry.get_parser_for_file("application/pdf", "file.pdf") + assert result is None + + def test_returns_parser_for_supported_mime_type( + self, + dummy_parser_cls: type, + ) -> None: + """ + GIVEN: A registry with a parser registered for 'text/plain'. + WHEN: get_parser_for_file() is called with 'text/plain'. + THEN: The registered parser class is returned. + """ + registry = ParserRegistry() + registry.register_builtin(dummy_parser_cls) + result = registry.get_parser_for_file("text/plain", "readme.txt") + assert result is dummy_parser_cls + + def test_highest_score_wins(self) -> None: + """ + GIVEN: Two parsers both supporting 'text/plain' with scores 5 and 20. + WHEN: get_parser_for_file() is called for 'text/plain'. + THEN: The parser with score 20 is returned. + """ + + class LowScoreParser: + name = "low" + version = "1.0" + author = "A" + url = "https://example.com/low" + + @classmethod + def supported_mime_types(cls): + return {"text/plain": ".txt"} + + @classmethod + def score(cls, mime_type, filename, path=None): + return 5 + + class HighScoreParser: + name = "high" + version = "1.0" + author = "B" + url = "https://example.com/high" + + @classmethod + def supported_mime_types(cls): + return {"text/plain": ".txt"} + + @classmethod + def score(cls, mime_type, filename, path=None): + return 20 + + registry = ParserRegistry() + registry.register_builtin(LowScoreParser) + registry.register_builtin(HighScoreParser) + result = registry.get_parser_for_file("text/plain", "readme.txt") + assert result is HighScoreParser + + def test_parser_returning_none_score_is_skipped(self) -> None: + """ + GIVEN: A parser that returns None from score() for the given file. + WHEN: get_parser_for_file() is called. + THEN: That parser is skipped and None is returned (no other candidates). + """ + + class DecliningParser: + name = "declining" + version = "1.0" + author = "A" + url = "https://example.com" + + @classmethod + def supported_mime_types(cls): + return {"text/plain": ".txt"} + + @classmethod + def score(cls, mime_type, filename, path=None): + return None # Explicitly declines + + registry = ParserRegistry() + registry.register_builtin(DecliningParser) + result = registry.get_parser_for_file("text/plain", "readme.txt") + assert result is None + + def test_all_parsers_decline_returns_none(self) -> None: + """ + GIVEN: Multiple parsers that all return None from score(). + WHEN: get_parser_for_file() is called. + THEN: None is returned. + """ + + class AlwaysDeclines: + name = "declines" + version = "1.0" + author = "A" + url = "https://example.com" + + @classmethod + def supported_mime_types(cls): + return {"text/plain": ".txt"} + + @classmethod + def score(cls, mime_type, filename, path=None): + return None + + registry = ParserRegistry() + registry.register_builtin(AlwaysDeclines) + registry._external.append(AlwaysDeclines) + result = registry.get_parser_for_file("text/plain", "file.txt") + assert result is None + + def test_external_parser_beats_builtin_same_score(self) -> None: + """ + GIVEN: An external and a built-in parser both returning score 10. + WHEN: get_parser_for_file() is called. + THEN: The external parser wins because externals are evaluated first + and the first-seen-wins policy applies at equal scores. + """ + + class BuiltinParser: + name = "builtin" + version = "1.0" + author = "Core" + url = "https://example.com/builtin" + + @classmethod + def supported_mime_types(cls): + return {"text/plain": ".txt"} + + @classmethod + def score(cls, mime_type, filename, path=None): + return 10 + + class ExternalParser: + name = "external" + version = "2.0" + author = "Third Party" + url = "https://example.com/external" + + @classmethod + def supported_mime_types(cls): + return {"text/plain": ".txt"} + + @classmethod + def score(cls, mime_type, filename, path=None): + return 10 + + registry = ParserRegistry() + registry.register_builtin(BuiltinParser) + registry._external.append(ExternalParser) + result = registry.get_parser_for_file("text/plain", "file.txt") + assert result is ExternalParser + + def test_builtin_wins_when_external_declines(self) -> None: + """ + GIVEN: An external parser that declines (score None) and a built-in + that returns score 5. + WHEN: get_parser_for_file() is called. + THEN: The built-in parser is returned. + """ + + class DecliningExternal: + name = "declining-external" + version = "1.0" + author = "Third Party" + url = "https://example.com/declining" + + @classmethod + def supported_mime_types(cls): + return {"text/plain": ".txt"} + + @classmethod + def score(cls, mime_type, filename, path=None): + return None + + class AcceptingBuiltin: + name = "accepting-builtin" + version = "1.0" + author = "Core" + url = "https://example.com/accepting" + + @classmethod + def supported_mime_types(cls): + return {"text/plain": ".txt"} + + @classmethod + def score(cls, mime_type, filename, path=None): + return 5 + + registry = ParserRegistry() + registry.register_builtin(AcceptingBuiltin) + registry._external.append(DecliningExternal) + result = registry.get_parser_for_file("text/plain", "file.txt") + assert result is AcceptingBuiltin + + +# --------------------------------------------------------------------------- +# TestDiscover +# --------------------------------------------------------------------------- + + +class TestDiscover: + """Verify entrypoint discovery in ParserRegistry.discover().""" + + def test_discover_with_no_entrypoints(self) -> None: + """ + GIVEN: No entrypoints are registered under 'paperless_ngx.parsers'. + WHEN: discover() is called. + THEN: _external remains empty and no errors are raised. + """ + registry = ParserRegistry() + + with patch( + "paperless.parsers.registry.entry_points", + return_value=[], + ): + registry.discover() + + assert registry._external == [] + + def test_discover_adds_valid_external_parser(self) -> None: + """ + GIVEN: One valid entrypoint whose loaded class has all required attrs. + WHEN: discover() is called. + THEN: The class is appended to _external. + """ + + class ValidExternal: + name = "valid-external" + version = "3.0.0" + author = "Someone" + url = "https://example.com/valid" + + @classmethod + def supported_mime_types(cls): + return {"application/pdf": ".pdf"} + + @classmethod + def score(cls, mime_type, filename, path=None): + return 5 + + mock_ep = MagicMock(spec=EntryPoint) + mock_ep.name = "valid_external" + mock_ep.load.return_value = ValidExternal + + registry = ParserRegistry() + + with patch( + "paperless.parsers.registry.entry_points", + return_value=[mock_ep], + ): + registry.discover() + + assert ValidExternal in registry._external + + def test_discover_skips_entrypoint_with_load_error( + self, + caplog: pytest.LogCaptureFixture, + ) -> None: + """ + GIVEN: An entrypoint whose load() method raises ImportError. + WHEN: discover() is called. + THEN: The entrypoint is skipped, an error is logged, and _external + remains empty. + """ + mock_ep = MagicMock(spec=EntryPoint) + mock_ep.name = "broken_ep" + mock_ep.load.side_effect = ImportError("missing dependency") + + registry = ParserRegistry() + + with caplog.at_level(logging.ERROR, logger="paperless.parsers.registry"): + with patch( + "paperless.parsers.registry.entry_points", + return_value=[mock_ep], + ): + registry.discover() + + assert registry._external == [] + assert any( + "broken_ep" in record.message + for record in caplog.records + if record.levelno >= logging.ERROR + ) + + def test_discover_skips_entrypoint_with_missing_attrs( + self, + caplog: pytest.LogCaptureFixture, + ) -> None: + """ + GIVEN: A class loaded from an entrypoint that is missing the 'score' + attribute. + WHEN: discover() is called. + THEN: The entrypoint is skipped, a warning is logged, and _external + remains empty. + """ + + class MissingScore: + name = "missing-score" + version = "1.0" + author = "Someone" + url = "https://example.com" + + # 'score' classmethod is intentionally absent. + + @classmethod + def supported_mime_types(cls): + return {"text/plain": ".txt"} + + mock_ep = MagicMock(spec=EntryPoint) + mock_ep.name = "missing_score_ep" + mock_ep.load.return_value = MissingScore + + registry = ParserRegistry() + + with caplog.at_level(logging.WARNING, logger="paperless.parsers.registry"): + with patch( + "paperless.parsers.registry.entry_points", + return_value=[mock_ep], + ): + registry.discover() + + assert registry._external == [] + assert any( + "missing_score_ep" in record.message + for record in caplog.records + if record.levelno >= logging.WARNING + ) + + def test_discover_logs_loaded_parser_info( + self, + caplog: pytest.LogCaptureFixture, + ) -> None: + """ + GIVEN: A valid entrypoint that loads successfully. + WHEN: discover() is called. + THEN: An INFO log message is emitted containing the parser name, + version, author, and entrypoint name. + """ + + class LoggableParser: + name = "loggable" + version = "4.2.0" + author = "Log Tester" + url = "https://example.com/loggable" + + @classmethod + def supported_mime_types(cls): + return {"image/png": ".png"} + + @classmethod + def score(cls, mime_type, filename, path=None): + return 1 + + mock_ep = MagicMock(spec=EntryPoint) + mock_ep.name = "loggable_ep" + mock_ep.load.return_value = LoggableParser + + registry = ParserRegistry() + + with caplog.at_level(logging.INFO, logger="paperless.parsers.registry"): + with patch( + "paperless.parsers.registry.entry_points", + return_value=[mock_ep], + ): + registry.discover() + + info_messages = " ".join( + r.message for r in caplog.records if r.levelno == logging.INFO + ) + assert "loggable" in info_messages + assert "4.2.0" in info_messages + assert "Log Tester" in info_messages + assert "loggable_ep" in info_messages + + +# --------------------------------------------------------------------------- +# TestLogSummary +# --------------------------------------------------------------------------- + + +class TestLogSummary: + """Verify log output from ParserRegistry.log_summary().""" + + def test_log_summary_with_no_external_parsers( + self, + dummy_parser_cls: type, + caplog: pytest.LogCaptureFixture, + ) -> None: + """ + GIVEN: A registry with one built-in parser and no external parsers. + WHEN: log_summary() is called. + THEN: The built-in parser name appears in the logs. + """ + registry = ParserRegistry() + registry.register_builtin(dummy_parser_cls) + + with caplog.at_level(logging.INFO, logger="paperless.parsers.registry"): + registry.log_summary() + + all_messages = " ".join(r.message for r in caplog.records) + assert dummy_parser_cls.name in all_messages + + def test_log_summary_with_external_parsers( + self, + caplog: pytest.LogCaptureFixture, + ) -> None: + """ + GIVEN: A registry with one external parser registered. + WHEN: log_summary() is called. + THEN: The external parser name, version, author, and url appear in + the log output. + """ + + class ExtParser: + name = "ext-parser" + version = "9.9.9" + author = "Ext Corp" + url = "https://ext.example.com" + + @classmethod + def supported_mime_types(cls): + return {} + + @classmethod + def score(cls, mime_type, filename, path=None): + return None + + registry = ParserRegistry() + registry._external.append(ExtParser) + + with caplog.at_level(logging.INFO, logger="paperless.parsers.registry"): + registry.log_summary() + + all_messages = " ".join(r.message for r in caplog.records) + assert "ext-parser" in all_messages + assert "9.9.9" in all_messages + assert "Ext Corp" in all_messages + assert "https://ext.example.com" in all_messages + + def test_log_summary_logs_no_third_party_message_when_none( + self, + caplog: pytest.LogCaptureFixture, + ) -> None: + """ + GIVEN: A registry with no external parsers. + WHEN: log_summary() is called. + THEN: A message containing 'No third-party parsers discovered.' is + logged. + """ + registry = ParserRegistry() + + with caplog.at_level(logging.INFO, logger="paperless.parsers.registry"): + registry.log_summary() + + all_messages = " ".join(r.message for r in caplog.records) + assert "No third-party parsers discovered." in all_messages