mirror of
https://github.com/paperless-ngx/paperless-ngx.git
synced 2026-10-11 18:47:13 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7bdc407ed9 | ||
|
|
01e1e76f91 | ||
|
|
a41e06f196 | ||
|
|
138160cbda | ||
|
|
6d960c11d5 | ||
|
|
a35bd6e238 | ||
|
|
474c630aa4 | ||
|
|
d7a9894400 |
No files matched your search
@@ -111,7 +111,7 @@ jobs:
|
|||||||
timeout-minutes: 12
|
timeout-minutes: 12
|
||||||
uses: $/.github/actions/apt-install
|
uses: $/.github/actions/apt-install
|
||||||
with:
|
with:
|
||||||
packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils qpdf
|
packages: unpaper tesseract-ocr imagemagick ghostscript poppler-utils
|
||||||
- name: Configure ImageMagick
|
- name: Configure ImageMagick
|
||||||
run: |
|
run: |
|
||||||
sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-7/policy.xml
|
sudo cp docker/rootfs/etc/ImageMagick-6/paperless-policy.xml /etc/ImageMagick-7/policy.xml
|
||||||
|
|||||||
+18
-6
@@ -1315,17 +1315,29 @@ valid crontab(5) expression describing when to run.
|
|||||||
|
|
||||||
#### [`PAPERLESS_CONVERT_MEMORY_LIMIT=<num>`](#PAPERLESS_CONVERT_MEMORY_LIMIT) {#PAPERLESS_CONVERT_MEMORY_LIMIT}
|
#### [`PAPERLESS_CONVERT_MEMORY_LIMIT=<num>`](#PAPERLESS_CONVERT_MEMORY_LIMIT) {#PAPERLESS_CONVERT_MEMORY_LIMIT}
|
||||||
|
|
||||||
!!! warning
|
: On smaller systems, or even in the case of Very Large Documents, the
|
||||||
|
consumer may explode, complaining about how it's "unable to extend
|
||||||
|
pixel cache". In such cases, try setting this to a reasonably low
|
||||||
|
value, like 32. The default is to use whatever is necessary to do
|
||||||
|
everything without writing to disk, and units are in megabytes.
|
||||||
|
|
||||||
Deprecated and has no effect, since PDF thumbnails no longer use
|
For more information on how to use this value, you should search the
|
||||||
ImageMagick. It will be removed in a future release.
|
web for "MAGICK_MEMORY_LIMIT".
|
||||||
|
|
||||||
|
Defaults to 0, which disables the limit.
|
||||||
|
|
||||||
#### [`PAPERLESS_CONVERT_TMPDIR=<path>`](#PAPERLESS_CONVERT_TMPDIR) {#PAPERLESS_CONVERT_TMPDIR}
|
#### [`PAPERLESS_CONVERT_TMPDIR=<path>`](#PAPERLESS_CONVERT_TMPDIR) {#PAPERLESS_CONVERT_TMPDIR}
|
||||||
|
|
||||||
!!! warning
|
: Similar to the memory limit, if you've got a small system and your
|
||||||
|
OS mounts /tmp as tmpfs, you should set this to a path that's on a
|
||||||
|
physical disk, like /home/your_user/tmp or something. ImageMagick
|
||||||
|
will use this as scratch space when crunching through very large
|
||||||
|
documents.
|
||||||
|
|
||||||
Deprecated and has no effect, since PDF thumbnails no longer use
|
For more information on how to use this value, you should search the
|
||||||
ImageMagick. It will be removed in a future release.
|
web for "MAGICK_TMPDIR".
|
||||||
|
|
||||||
|
Default is none, which disables the temporary directory.
|
||||||
|
|
||||||
#### [`PAPERLESS_APPS=<string>`](#PAPERLESS_APPS) {#PAPERLESS_APPS}
|
#### [`PAPERLESS_APPS=<string>`](#PAPERLESS_APPS) {#PAPERLESS_APPS}
|
||||||
|
|
||||||
|
|||||||
+7
-4
@@ -177,12 +177,12 @@ to a positive number to enable polling and disable native filesystem notificatio
|
|||||||
- `pkg-config` for mysqlclient (python dependency)
|
- `pkg-config` for mysqlclient (python dependency)
|
||||||
- `fonts-liberation` for generating thumbnails for plain text
|
- `fonts-liberation` for generating thumbnails for plain text
|
||||||
files
|
files
|
||||||
- `imagemagick` >= 6 for image alpha handling
|
- `imagemagick` >= 6 for PDF conversion
|
||||||
- `gnupg` for decrypting GPG-encrypted email
|
- `gnupg` for decrypting GPG-encrypted email
|
||||||
- `libpq-dev` for PostgreSQL
|
- `libpq-dev` for PostgreSQL
|
||||||
- `libmagic-dev` for mime type detection
|
- `libmagic-dev` for mime type detection
|
||||||
- `mariadb-client` for MariaDB compile time
|
- `mariadb-client` for MariaDB compile time
|
||||||
- `poppler-utils` for thumbnail generation and barcode detection
|
- `poppler-utils` for barcode detection
|
||||||
|
|
||||||
Use this list for your preferred package management:
|
Use this list for your preferred package management:
|
||||||
|
|
||||||
@@ -416,8 +416,11 @@ to a positive number to enable polling and disable native filesystem notificatio
|
|||||||
You may need to change the path in the files. Example:
|
You may need to change the path in the files. Example:
|
||||||
`ExecStart=/opt/paperless/.local/bin/celery --app paperless worker --loglevel INFO`
|
`ExecStart=/opt/paperless/.local/bin/celery --app paperless worker --loglevel INFO`
|
||||||
|
|
||||||
12. Harden ImageMagick by disabling formats that Paperless-ngx does not use.
|
12. Configure ImageMagick to allow processing of PDF documents and disable
|
||||||
PDF processing is not needed and should stay disabled.
|
formats that Paperless-ngx does not use. Most distributions disable PDF
|
||||||
|
processing by default, since PDF documents can contain malware. If you
|
||||||
|
don't enable it, Paperless-ngx will fall back to Ghostscript for certain
|
||||||
|
steps such as thumbnail generation.
|
||||||
|
|
||||||
Configure the active ImageMagick policy file (commonly
|
Configure the active ImageMagick policy file (commonly
|
||||||
`/etc/ImageMagick-6/policy.xml` or `/etc/ImageMagick-7/policy.xml`) and
|
`/etc/ImageMagick-6/policy.xml` or `/etc/ImageMagick-7/policy.xml`) and
|
||||||
|
|||||||
@@ -50,6 +50,8 @@ PAPERLESS_SECRET_KEY=change-me
|
|||||||
#PAPERLESS_OCR_ROTATE_PAGES=true
|
#PAPERLESS_OCR_ROTATE_PAGES=true
|
||||||
#PAPERLESS_OCR_ROTATE_PAGES_THRESHOLD=12.0
|
#PAPERLESS_OCR_ROTATE_PAGES_THRESHOLD=12.0
|
||||||
#PAPERLESS_OCR_USER_ARGS={}
|
#PAPERLESS_OCR_USER_ARGS={}
|
||||||
|
#PAPERLESS_CONVERT_MEMORY_LIMIT=0
|
||||||
|
#PAPERLESS_CONVERT_TMPDIR=/var/tmp/paperless
|
||||||
|
|
||||||
# Software tweaks
|
# Software tweaks
|
||||||
|
|
||||||
|
|||||||
@@ -13,8 +13,14 @@ from celery import group
|
|||||||
from celery import shared_task
|
from celery import shared_task
|
||||||
from django.conf import settings
|
from django.conf import settings
|
||||||
from django.db import transaction
|
from django.db import transaction
|
||||||
|
from django.db.models import Case
|
||||||
|
from django.db.models import F
|
||||||
from django.db.models import Max
|
from django.db.models import Max
|
||||||
|
from django.db.models import OuterRef
|
||||||
from django.db.models import Q
|
from django.db.models import Q
|
||||||
|
from django.db.models import Subquery
|
||||||
|
from django.db.models import When
|
||||||
|
from django.db.models.functions import Coalesce
|
||||||
from django.utils import timezone
|
from django.utils import timezone
|
||||||
|
|
||||||
from documents.data_models import ConsumableDocument
|
from documents.data_models import ConsumableDocument
|
||||||
@@ -36,6 +42,7 @@ from documents.tasks import remove_document_from_index
|
|||||||
from documents.tasks import update_document_content_maybe_archive_file
|
from documents.tasks import update_document_content_maybe_archive_file
|
||||||
from documents.versioning import get_latest_version_for_root
|
from documents.versioning import get_latest_version_for_root
|
||||||
from documents.versioning import get_root_document
|
from documents.versioning import get_root_document
|
||||||
|
from documents.versioning import versions_newest_first
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
from collections.abc import Mapping
|
from collections.abc import Mapping
|
||||||
@@ -408,10 +415,28 @@ def reprocess(doc_ids: list[int], *, remote_ocr: bool = False) -> Literal["OK"]:
|
|||||||
|
|
||||||
Consumption workflows do not run here, so ``remote_ocr`` is how the user
|
Consumption workflows do not run here, so ``remote_ocr`` is how the user
|
||||||
asks for the remote engine when it is not configured to handle everything.
|
asks for the remote engine when it is not configured to handle everything.
|
||||||
|
|
||||||
|
A root document with versions reprocesses its latest version, which is the
|
||||||
|
file whose content, archive and thumbnail are shown for it.
|
||||||
"""
|
"""
|
||||||
for document_id in doc_ids:
|
latest_version = versions_newest_first(
|
||||||
|
Document.objects.filter(root_document=OuterRef("pk")),
|
||||||
|
).values("id")[:1]
|
||||||
|
source_ids = (
|
||||||
|
Document.objects.filter(id__in=doc_ids)
|
||||||
|
.annotate(
|
||||||
|
source_id=Case(
|
||||||
|
When(root_document__isnull=False, then=F("id")),
|
||||||
|
default=Coalesce(Subquery(latest_version), F("id")),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
.order_by()
|
||||||
|
.values_list("source_id", flat=True)
|
||||||
|
.distinct()
|
||||||
|
)
|
||||||
|
for source_id in source_ids:
|
||||||
update_document_content_maybe_archive_file.apply_async(
|
update_document_content_maybe_archive_file.apply_async(
|
||||||
kwargs={"document_id": document_id, "remote_ocr": remote_ocr},
|
kwargs={"document_id": source_id, "remote_ocr": remote_ocr},
|
||||||
headers={"trigger_source": PaperlessTask.TriggerSource.MANUAL},
|
headers={"trigger_source": PaperlessTask.TriggerSource.MANUAL},
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
+96
-159
@@ -1,8 +1,8 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import math
|
|
||||||
import mimetypes
|
import mimetypes
|
||||||
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
import subprocess
|
import subprocess
|
||||||
import tempfile
|
import tempfile
|
||||||
@@ -68,6 +68,58 @@ def get_supported_file_extensions() -> set[str]:
|
|||||||
return extensions
|
return extensions
|
||||||
|
|
||||||
|
|
||||||
|
def run_convert(
|
||||||
|
input_file,
|
||||||
|
output_file,
|
||||||
|
*,
|
||||||
|
density=None,
|
||||||
|
scale=None,
|
||||||
|
alpha=None,
|
||||||
|
strip=False,
|
||||||
|
trim=False,
|
||||||
|
type=None,
|
||||||
|
depth=None,
|
||||||
|
auto_orient=False,
|
||||||
|
use_cropbox=False,
|
||||||
|
extra=None,
|
||||||
|
logging_group=None,
|
||||||
|
) -> None:
|
||||||
|
environment = os.environ.copy()
|
||||||
|
if settings.CONVERT_MEMORY_LIMIT:
|
||||||
|
# MAGICK_MEMORY_LIMIT sets the maximum amount of RAM the pixel cache can use.
|
||||||
|
# MAGICK_MAP_LIMIT sets the maximum amount of memory-mapped I/O allowed.
|
||||||
|
#
|
||||||
|
# For large-format documents ImageMagick will hit the RAM limit and
|
||||||
|
# immediately try to "map" the remaining data. If MAGICK_MAP_LIMIT isn't
|
||||||
|
# also set, the process may trigger an OOM kill because the default
|
||||||
|
# system/policy map limit is often too restrictive for these massive bitmaps.
|
||||||
|
environment["MAGICK_MEMORY_LIMIT"] = settings.CONVERT_MEMORY_LIMIT
|
||||||
|
environment["MAGICK_MAP_LIMIT"] = settings.CONVERT_MEMORY_LIMIT
|
||||||
|
if settings.CONVERT_TMPDIR:
|
||||||
|
environment["MAGICK_TMPDIR"] = settings.CONVERT_TMPDIR
|
||||||
|
|
||||||
|
args = [settings.CONVERT_BINARY]
|
||||||
|
args += ["-density", str(density)] if density else []
|
||||||
|
args += ["-scale", str(scale)] if scale else []
|
||||||
|
args += ["-alpha", str(alpha)] if alpha else []
|
||||||
|
args += ["-strip"] if strip else []
|
||||||
|
args += ["-trim"] if trim else []
|
||||||
|
args += ["-type", str(type)] if type else []
|
||||||
|
args += ["-depth", str(depth)] if depth else []
|
||||||
|
args += ["-auto-orient"] if auto_orient else []
|
||||||
|
args += ["-define", "pdf:use-cropbox=true"] if use_cropbox else []
|
||||||
|
args += [str(input_file), str(output_file)]
|
||||||
|
|
||||||
|
logger.debug("Execute: " + " ".join(args), extra={"group": logging_group})
|
||||||
|
|
||||||
|
try:
|
||||||
|
run_subprocess(args, environment, logger)
|
||||||
|
except subprocess.CalledProcessError as e:
|
||||||
|
raise ParseError(f"Convert failed at {args}") from e
|
||||||
|
except Exception as e: # pragma: no cover
|
||||||
|
raise ParseError("Unknown error running convert") from e
|
||||||
|
|
||||||
|
|
||||||
def get_default_thumbnail() -> Path:
|
def get_default_thumbnail() -> Path:
|
||||||
"""
|
"""
|
||||||
Returns the path to a generic thumbnail
|
Returns the path to a generic thumbnail
|
||||||
@@ -75,168 +127,46 @@ def get_default_thumbnail() -> Path:
|
|||||||
return (Path(__file__).parent / "resources" / "document.webp").resolve()
|
return (Path(__file__).parent / "resources" / "document.webp").resolve()
|
||||||
|
|
||||||
|
|
||||||
_THUMBNAIL_MAX_WIDTH = 500
|
def make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group=None) -> Path:
|
||||||
_THUMBNAIL_MAX_HEIGHT = 5000
|
out_path: Path = Path(temp_dir) / "convert_gs.webp"
|
||||||
# Used only when the page geometry cannot be read
|
|
||||||
_THUMBNAIL_FALLBACK_DPI = 150
|
|
||||||
# Applied before supersampling, so tiny pages are not enlarged
|
|
||||||
_THUMBNAIL_MAX_DPI = 300
|
|
||||||
# Rendering at a multiple and downsampling keeps text crisper
|
|
||||||
_THUMBNAIL_SUPERSAMPLE = 2
|
|
||||||
|
|
||||||
|
|
||||||
def rasterize_pdf_page_to_png(
|
|
||||||
in_path: Path,
|
|
||||||
out_path: Path,
|
|
||||||
*,
|
|
||||||
dpi: int,
|
|
||||||
logging_group=None,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
Rasterizes the first page of a PDF to a PNG with pdftoppm.
|
|
||||||
"""
|
|
||||||
# -singlefile drops the page number and -png appends ".png", so pass the
|
|
||||||
# path without its suffix
|
|
||||||
args = [
|
|
||||||
"pdftoppm",
|
|
||||||
"-f",
|
|
||||||
"1",
|
|
||||||
"-l",
|
|
||||||
"1",
|
|
||||||
"-r",
|
|
||||||
str(dpi),
|
|
||||||
"-png",
|
|
||||||
"-singlefile",
|
|
||||||
"-cropbox",
|
|
||||||
str(in_path),
|
|
||||||
str(out_path.with_suffix("")),
|
|
||||||
]
|
|
||||||
|
|
||||||
logger.debug("Execute: " + " ".join(args), extra={"group": logging_group})
|
|
||||||
|
|
||||||
try:
|
|
||||||
run_subprocess(args, logger=logger)
|
|
||||||
except subprocess.CalledProcessError as e:
|
|
||||||
raise ParseError(f"pdftoppm failed at {args}") from e
|
|
||||||
except Exception as e: # pragma: no cover
|
|
||||||
raise ParseError("Unknown error running pdftoppm") from e
|
|
||||||
|
|
||||||
|
|
||||||
def encode_thumbnail_webp(
|
|
||||||
png_path: Path,
|
|
||||||
out_path: Path,
|
|
||||||
*,
|
|
||||||
supersample: int = 1,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
Flattens alpha onto white, undoes supersampling, shrinks to fit and saves as WebP.
|
|
||||||
"""
|
|
||||||
from PIL import Image
|
|
||||||
|
|
||||||
try:
|
|
||||||
with Image.open(png_path) as im:
|
|
||||||
if im.mode in ("RGBA", "LA"):
|
|
||||||
flattened = Image.new("RGB", im.size, (255, 255, 255))
|
|
||||||
flattened.paste(im, mask=im.split()[-1])
|
|
||||||
else:
|
|
||||||
flattened = im.convert("RGB")
|
|
||||||
|
|
||||||
if supersample > 1:
|
|
||||||
flattened = flattened.resize(
|
|
||||||
(
|
|
||||||
max(1, round(flattened.width / supersample)),
|
|
||||||
max(1, round(flattened.height / supersample)),
|
|
||||||
),
|
|
||||||
Image.Resampling.LANCZOS,
|
|
||||||
)
|
|
||||||
|
|
||||||
flattened.thumbnail((_THUMBNAIL_MAX_WIDTH, _THUMBNAIL_MAX_HEIGHT))
|
|
||||||
flattened.save(out_path, format="WEBP")
|
|
||||||
except (OSError, Image.DecompressionBombError) as e:
|
|
||||||
raise ParseError(f"Unable to encode thumbnail from {png_path}") from e
|
|
||||||
|
|
||||||
|
|
||||||
def _compute_thumbnail_dpi(in_path: Path, logging_group=None) -> tuple[int, int]:
|
|
||||||
"""
|
|
||||||
Returns (dpi, supersample). Unknown geometry is not supersampled, since
|
|
||||||
the render size cannot be bounded.
|
|
||||||
"""
|
|
||||||
from paperless.parsers.utils import get_pdf_first_page_size_points
|
|
||||||
|
|
||||||
size = get_pdf_first_page_size_points(in_path)
|
|
||||||
if size is None:
|
|
||||||
logger.debug(
|
|
||||||
"Could not read PDF page size, using fallback DPI",
|
|
||||||
extra={"group": logging_group},
|
|
||||||
)
|
|
||||||
return _THUMBNAIL_FALLBACK_DPI, 1
|
|
||||||
|
|
||||||
width_pts, height_pts = size
|
|
||||||
dpi_for_width = _THUMBNAIL_MAX_WIDTH * 72 / width_pts
|
|
||||||
dpi_for_height = _THUMBNAIL_MAX_HEIGHT * 72 / height_pts
|
|
||||||
# Round up so the downsampled render is never a few pixels short of the
|
|
||||||
# target; the shrink-only clamp in encode_thumbnail_webp trims the excess.
|
|
||||||
dpi = max(
|
|
||||||
1,
|
|
||||||
math.ceil(min(_THUMBNAIL_MAX_DPI, dpi_for_width, dpi_for_height)),
|
|
||||||
)
|
|
||||||
# At the 1 DPI floor the page is already oversized, so do not supersample
|
|
||||||
return dpi, 1 if dpi == 1 else _THUMBNAIL_SUPERSAMPLE
|
|
||||||
|
|
||||||
|
|
||||||
def _render_pdf_thumbnail(
|
|
||||||
in_path: Path,
|
|
||||||
png_path: Path,
|
|
||||||
out_path: Path,
|
|
||||||
logging_group=None,
|
|
||||||
) -> None:
|
|
||||||
dpi, supersample = _compute_thumbnail_dpi(in_path, logging_group=logging_group)
|
|
||||||
rasterize_pdf_page_to_png(
|
|
||||||
in_path,
|
|
||||||
png_path,
|
|
||||||
dpi=dpi * supersample,
|
|
||||||
logging_group=logging_group,
|
|
||||||
)
|
|
||||||
encode_thumbnail_webp(png_path, out_path, supersample=supersample)
|
|
||||||
|
|
||||||
|
|
||||||
def _repair_pdf_with_qpdf(in_path: Path, out_path: Path) -> None:
|
|
||||||
# qpdf exits 3 after a repair; --warning-exit-0 keeps that from failing
|
|
||||||
try:
|
|
||||||
shutil.copy(in_path, out_path)
|
|
||||||
run_subprocess(
|
|
||||||
["qpdf", "--warning-exit-0", "--replace-input", str(out_path)],
|
|
||||||
logger=logger,
|
|
||||||
)
|
|
||||||
except (subprocess.CalledProcessError, OSError) as e:
|
|
||||||
raise ParseError(f"qpdf repair failed for {in_path}") from e
|
|
||||||
|
|
||||||
|
|
||||||
def make_thumbnail_from_pdf_qpdf_fallback(
|
|
||||||
in_path: Path,
|
|
||||||
temp_dir: Path,
|
|
||||||
logging_group=None,
|
|
||||||
) -> Path:
|
|
||||||
png_path = temp_dir / "page1_repaired.png"
|
|
||||||
out_path = temp_dir / "convert_qpdf.webp"
|
|
||||||
repaired_path = temp_dir / "repaired.pdf"
|
|
||||||
|
|
||||||
|
# if convert fails, fall back to extracting
|
||||||
|
# the first PDF page as a PNG using Ghostscript
|
||||||
logger.warning(
|
logger.warning(
|
||||||
"Thumbnail generation with pdftoppm failed, attempting qpdf repair and retry.",
|
"Thumbnail generation with ImageMagick failed, falling back "
|
||||||
|
"to ghostscript. Check your /etc/ImageMagick-x/policy.xml!",
|
||||||
extra={"group": logging_group},
|
extra={"group": logging_group},
|
||||||
)
|
)
|
||||||
|
# Ghostscript doesn't handle WebP outputs
|
||||||
|
gs_out_path: Path = Path(temp_dir) / "gs_out.png"
|
||||||
|
cmd = [settings.GS_BINARY, "-q", "-sDEVICE=pngalpha", "-o", gs_out_path, in_path]
|
||||||
|
|
||||||
try:
|
try:
|
||||||
_repair_pdf_with_qpdf(in_path, repaired_path)
|
try:
|
||||||
_render_pdf_thumbnail(repaired_path, png_path, out_path, logging_group)
|
run_subprocess(cmd, logger=logger)
|
||||||
|
except subprocess.CalledProcessError as e:
|
||||||
|
raise ParseError(f"Thumbnail (gs) failed at {cmd}") from e
|
||||||
|
# then run convert on the output from gs to make WebP
|
||||||
|
run_convert(
|
||||||
|
density=300,
|
||||||
|
scale="500x5000>",
|
||||||
|
alpha="remove",
|
||||||
|
strip=True,
|
||||||
|
trim=False,
|
||||||
|
auto_orient=True,
|
||||||
|
input_file=gs_out_path,
|
||||||
|
output_file=out_path,
|
||||||
|
logging_group=logging_group,
|
||||||
|
)
|
||||||
|
|
||||||
return out_path
|
return out_path
|
||||||
|
|
||||||
except ParseError as e:
|
except ParseError as e:
|
||||||
logger.error(f"Unable to make thumbnail after qpdf repair: {e}")
|
logger.error(f"Unable to make thumbnail with Ghostscript: {e}")
|
||||||
# The caller might expect a generated thumbnail that can be moved,
|
# The caller might expect a generated thumbnail that can be moved,
|
||||||
# so we need to copy it before it gets moved.
|
# so we need to copy it before it gets moved.
|
||||||
# https://github.com/paperless-ngx/paperless-ngx/issues/3631
|
# https://github.com/paperless-ngx/paperless-ngx/issues/3631
|
||||||
default_thumbnail_path = temp_dir / "document.webp"
|
default_thumbnail_path: Path = Path(temp_dir) / "document.webp"
|
||||||
copy_file_with_basic_stats(get_default_thumbnail(), default_thumbnail_path)
|
copy_file_with_basic_stats(get_default_thumbnail(), default_thumbnail_path)
|
||||||
return default_thumbnail_path
|
return default_thumbnail_path
|
||||||
|
|
||||||
@@ -245,18 +175,25 @@ def make_thumbnail_from_pdf(in_path: Path, temp_dir: Path, logging_group=None) -
|
|||||||
"""
|
"""
|
||||||
The thumbnail of a PDF is just a 500px wide image of the first page.
|
The thumbnail of a PDF is just a 500px wide image of the first page.
|
||||||
"""
|
"""
|
||||||
png_path: Path = temp_dir / "page1.png"
|
|
||||||
out_path: Path = temp_dir / "convert.webp"
|
out_path: Path = temp_dir / "convert.webp"
|
||||||
|
|
||||||
|
# Run convert to get a decent thumbnail
|
||||||
try:
|
try:
|
||||||
_render_pdf_thumbnail(in_path, png_path, out_path, logging_group)
|
run_convert(
|
||||||
except ParseError as e:
|
density=300,
|
||||||
logger.error(f"Unable to make thumbnail with pdftoppm: {e}")
|
scale="500x5000>",
|
||||||
out_path = make_thumbnail_from_pdf_qpdf_fallback(
|
alpha="remove",
|
||||||
in_path,
|
strip=True,
|
||||||
temp_dir,
|
trim=False,
|
||||||
logging_group,
|
auto_orient=True,
|
||||||
|
use_cropbox=True,
|
||||||
|
input_file=f"{in_path}[0]",
|
||||||
|
output_file=str(out_path),
|
||||||
|
logging_group=logging_group,
|
||||||
)
|
)
|
||||||
|
except ParseError as e:
|
||||||
|
logger.error(f"Unable to make thumbnail with convert: {e}")
|
||||||
|
out_path = make_thumbnail_from_pdf_gs_fallback(in_path, temp_dir, logging_group)
|
||||||
|
|
||||||
return out_path
|
return out_path
|
||||||
|
|
||||||
|
|||||||
@@ -6,7 +6,9 @@ from django.contrib.auth.models import Permission
|
|||||||
from django.contrib.auth.models import User
|
from django.contrib.auth.models import User
|
||||||
from django.contrib.contenttypes.models import ContentType
|
from django.contrib.contenttypes.models import ContentType
|
||||||
from django.db.models import Case
|
from django.db.models import Case
|
||||||
|
from django.db.models import CharField
|
||||||
from django.db.models import Count
|
from django.db.models import Count
|
||||||
|
from django.db.models import F
|
||||||
from django.db.models import IntegerField
|
from django.db.models import IntegerField
|
||||||
from django.db.models import Model
|
from django.db.models import Model
|
||||||
from django.db.models import Q
|
from django.db.models import Q
|
||||||
@@ -14,6 +16,7 @@ from django.db.models import QuerySet
|
|||||||
from django.db.models import Value
|
from django.db.models import Value
|
||||||
from django.db.models import When
|
from django.db.models import When
|
||||||
from django.db.models.functions import Cast
|
from django.db.models.functions import Cast
|
||||||
|
from django.db.models.functions import Coalesce
|
||||||
from guardian.core import ObjectPermissionChecker
|
from guardian.core import ObjectPermissionChecker
|
||||||
from guardian.models import GroupObjectPermission
|
from guardian.models import GroupObjectPermission
|
||||||
from guardian.models import UserObjectPermission
|
from guardian.models import UserObjectPermission
|
||||||
@@ -25,6 +28,7 @@ from rest_framework.permissions import BasePermission
|
|||||||
from rest_framework.permissions import DjangoObjectPermissions
|
from rest_framework.permissions import DjangoObjectPermissions
|
||||||
|
|
||||||
from documents.models import Document
|
from documents.models import Document
|
||||||
|
from documents.versioning import get_root_document
|
||||||
|
|
||||||
|
|
||||||
class PaperlessObjectPermissions(DjangoObjectPermissions):
|
class PaperlessObjectPermissions(DjangoObjectPermissions):
|
||||||
@@ -348,6 +352,7 @@ def permitted_object_ids(
|
|||||||
perm: str,
|
perm: str,
|
||||||
*,
|
*,
|
||||||
include_deleted: bool = False,
|
include_deleted: bool = False,
|
||||||
|
parent_field: str | None = None,
|
||||||
) -> QuerySet[int]:
|
) -> QuerySet[int]:
|
||||||
"""
|
"""
|
||||||
Generic version of ``permitted_document_ids`` for any model with an
|
Generic version of ``permitted_document_ids`` for any model with an
|
||||||
@@ -356,6 +361,24 @@ def permitted_object_ids(
|
|||||||
soft-delete pattern (currently only ``Document``); for every other model
|
soft-delete pattern (currently only ``Document``); for every other model
|
||||||
it is accepted but has no effect, since those models have no soft-delete
|
it is accepted but has no effect, since those models have no soft-delete
|
||||||
concept.
|
concept.
|
||||||
|
|
||||||
|
``parent_field`` names a self-referencing foreign key whose target
|
||||||
|
authorizes the row (``Document.root_document``). A row with a parent is
|
||||||
|
visible exactly when its parent is, judged by the parent's owner and
|
||||||
|
grants, so the row's own owner and grants are ignored.
|
||||||
|
|
||||||
|
Guardian stores ``object_pk`` as a string, so the row key is cast to a
|
||||||
|
string and tested against the user's and groups' grants with a single
|
||||||
|
uncorrelated ``IN``. Postgres and SQLite build that set once. MariaDB
|
||||||
|
evaluates it as an index probe per row, which is cheap because the
|
||||||
|
lookups use guardian's unique indexes. Casting every ``object_pk`` to an
|
||||||
|
integer instead cannot use an index, and MariaDB cannot materialize it
|
||||||
|
inside the owner ``OR``, so it re-scans the user's grants for every row.
|
||||||
|
A correlated ``EXISTS`` per grant fixes MariaDB too, but Postgres and
|
||||||
|
SQLite re-run it for every row and end up slower than the original. The
|
||||||
|
user's groups are matched with an ``IN`` subquery rather than a join
|
||||||
|
through the membership table, which SQLite plans badly once the grant
|
||||||
|
tables grow.
|
||||||
"""
|
"""
|
||||||
has_soft_delete = hasattr(model, "global_objects")
|
has_soft_delete = hasattr(model, "global_objects")
|
||||||
manager = (
|
manager = (
|
||||||
@@ -363,8 +386,21 @@ def permitted_object_ids(
|
|||||||
)
|
)
|
||||||
base_qs = manager.all().only("id", "owner")
|
base_qs = manager.all().only("id", "owner")
|
||||||
|
|
||||||
|
owner_field, key_field = "owner", "pk"
|
||||||
|
if parent_field is not None:
|
||||||
|
owner_field, key_field = "authorizing_owner", "authorizing_id"
|
||||||
|
base_qs = base_qs.annotate(
|
||||||
|
authorizing_id=Coalesce(f"{parent_field}_id", "id"),
|
||||||
|
authorizing_owner=Case(
|
||||||
|
When(**{f"{parent_field}_id__isnull": True}, then=F("owner_id")),
|
||||||
|
default=F(f"{parent_field}__owner_id"),
|
||||||
|
output_field=IntegerField(),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
unowned = Q(**{f"{owner_field}__isnull": True})
|
||||||
|
|
||||||
if user is None or not getattr(user, "is_authenticated", False):
|
if user is None or not getattr(user, "is_authenticated", False):
|
||||||
return base_qs.filter(owner__isnull=True).values_list("id", flat=True)
|
return base_qs.filter(unowned).values_list("id", flat=True)
|
||||||
|
|
||||||
# Deactivated users get nothing, deactivated superusers included, so this
|
# Deactivated users get nothing, deactivated superusers included, so this
|
||||||
# has to come before the superuser shortcut. guardian's
|
# has to come before the superuser shortcut. guardian's
|
||||||
@@ -388,21 +424,26 @@ def permitted_object_ids(
|
|||||||
"permission__content_type": content_type,
|
"permission__content_type": content_type,
|
||||||
}
|
}
|
||||||
|
|
||||||
user_perm_ids = (
|
# Both grant sets are compared to the row key as strings, exactly as
|
||||||
UserObjectPermission.objects.filter(user=user, **perm_filter)
|
# guardian stores them, and are uncorrelated, so each engine can build the
|
||||||
.annotate(object_pk_int=Cast("object_pk", IntegerField()))
|
# set once instead of probing per row.
|
||||||
.values_list("object_pk_int", flat=True)
|
user_keys = UserObjectPermission.objects.filter(
|
||||||
)
|
user=user,
|
||||||
group_perm_ids = (
|
**perm_filter,
|
||||||
GroupObjectPermission.objects.filter(group__user=user, **perm_filter)
|
).values_list("object_pk", flat=True)
|
||||||
.annotate(object_pk_int=Cast("object_pk", IntegerField()))
|
group_keys = GroupObjectPermission.objects.filter(
|
||||||
.values_list("object_pk_int", flat=True)
|
group_id__in=user.groups.values("id"),
|
||||||
)
|
**perm_filter,
|
||||||
permitted_ids = user_perm_ids.union(group_perm_ids)
|
).values_list("object_pk", flat=True)
|
||||||
|
permitted_keys = user_keys.union(group_keys, all=True)
|
||||||
|
|
||||||
return base_qs.filter(
|
return (
|
||||||
Q(owner=user) | Q(owner__isnull=True) | Q(id__in=permitted_ids),
|
base_qs.annotate(permitted_key=Cast(key_field, CharField(max_length=64)))
|
||||||
).values_list("id", flat=True)
|
.filter(
|
||||||
|
Q(**{owner_field: user.pk}) | unowned | Q(permitted_key__in=permitted_keys),
|
||||||
|
)
|
||||||
|
.values_list("id", flat=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
ModelT = TypeVar("ModelT", bound=Model)
|
ModelT = TypeVar("ModelT", bound=Model)
|
||||||
@@ -470,8 +511,17 @@ def permitted_document_ids(
|
|||||||
``include_deleted=True`` for callers that need to check permission on
|
``include_deleted=True`` for callers that need to check permission on
|
||||||
soft-deleted documents (e.g. trash restore). This intentionally avoids
|
soft-deleted documents (e.g. trash restore). This intentionally avoids
|
||||||
``get_objects_for_user`` to keep the subquery small and index-friendly.
|
``get_objects_for_user`` to keep the subquery small and index-friendly.
|
||||||
|
|
||||||
|
A version is authorized by its root document, so a version's own owner and
|
||||||
|
grants never matter.
|
||||||
"""
|
"""
|
||||||
return permitted_object_ids(user, Document, perm, include_deleted=include_deleted)
|
return permitted_object_ids(
|
||||||
|
user,
|
||||||
|
Document,
|
||||||
|
perm,
|
||||||
|
include_deleted=include_deleted,
|
||||||
|
parent_field="root_document",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def get_document_count_filter_for_user(user, related_name: str = "documents"):
|
def get_document_count_filter_for_user(user, related_name: str = "documents"):
|
||||||
@@ -630,7 +680,14 @@ def has_perms_owner_aware(user, perms, obj):
|
|||||||
single-object check still has many production callers. Several callers
|
single-object check still has many production callers. Several callers
|
||||||
remain across ``documents/``, ``paperless_mail/``, and ``paperless_ai/``
|
remain across ``documents/``, ``paperless_mail/``, and ``paperless_ai/``
|
||||||
-- grep for this function name before removing it.
|
-- grep for this function name before removing it.
|
||||||
|
|
||||||
|
A document version is authorized by its root document, like in
|
||||||
|
``permitted_document_ids``, so a version's own owner and grants never
|
||||||
|
matter. Fetch the root with ``select_related("root_document__owner")`` to
|
||||||
|
avoid extra queries.
|
||||||
"""
|
"""
|
||||||
|
if isinstance(obj, Document):
|
||||||
|
obj = get_root_document(obj)
|
||||||
checker = ObjectPermissionChecker(user)
|
checker = ObjectPermissionChecker(user)
|
||||||
return obj.owner is None or obj.owner == user or checker.has_perm(perms, obj)
|
return obj.owner is None or obj.owner == user or checker.has_perm(perms, obj)
|
||||||
|
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ import threading
|
|||||||
import time
|
import time
|
||||||
from datetime import UTC
|
from datetime import UTC
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
from datetime import timedelta
|
||||||
from enum import StrEnum
|
from enum import StrEnum
|
||||||
from itertools import islice
|
from itertools import islice
|
||||||
from typing import TYPE_CHECKING
|
from typing import TYPE_CHECKING
|
||||||
@@ -31,6 +32,7 @@ from documents.search._query import parse_user_query
|
|||||||
from documents.search._schema import _write_sentinels
|
from documents.search._schema import _write_sentinels
|
||||||
from documents.search._schema import build_schema
|
from documents.search._schema import build_schema
|
||||||
from documents.search._schema import open_or_rebuild_index
|
from documents.search._schema import open_or_rebuild_index
|
||||||
|
from documents.search._schema import rebuild_in_progress
|
||||||
from documents.search._schema import wipe_index
|
from documents.search._schema import wipe_index
|
||||||
from documents.search._tokenizer import ascii_fold
|
from documents.search._tokenizer import ascii_fold
|
||||||
from documents.search._tokenizer import autocomplete_tokens
|
from documents.search._tokenizer import autocomplete_tokens
|
||||||
@@ -54,6 +56,19 @@ if TYPE_CHECKING:
|
|||||||
|
|
||||||
logger = logging.getLogger("paperless.search")
|
logger = logging.getLogger("paperless.search")
|
||||||
|
|
||||||
|
# tantivy stores dates as signed 64-bit nanoseconds since the Unix epoch, which
|
||||||
|
# covers 1677-09-21T00:12:43 to 2262-04-11T23:47:16 UTC
|
||||||
|
_INDEX_DATE_NANOS_MIN: Final[int] = -(2**63)
|
||||||
|
_INDEX_DATE_NANOS_MAX: Final[int] = 2**63 - 1
|
||||||
|
_UNIX_EPOCH: Final[datetime] = datetime(1970, 1, 1, tzinfo=UTC)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_indexable_date(value: datetime) -> bool:
|
||||||
|
"""Whether value, at whole-second precision, fits tantivy's date range."""
|
||||||
|
nanos = ((value - _UNIX_EPOCH) // timedelta(seconds=1)) * 1_000_000_000
|
||||||
|
return _INDEX_DATE_NANOS_MIN <= nanos <= _INDEX_DATE_NANOS_MAX
|
||||||
|
|
||||||
|
|
||||||
_LOCK_TIMEOUT_SECONDS: Final[float] = 10.0 # per-attempt acquire timeout
|
_LOCK_TIMEOUT_SECONDS: Final[float] = 10.0 # per-attempt acquire timeout
|
||||||
_LOCK_RETRY_ATTEMPTS: Final[int] = 4 # total attempts (1 initial + 3 retries)
|
_LOCK_RETRY_ATTEMPTS: Final[int] = 4 # total attempts (1 initial + 3 retries)
|
||||||
_LOCK_BACKOFF_BASE: Final[float] = 1.0 # seconds
|
_LOCK_BACKOFF_BASE: Final[float] = 1.0 # seconds
|
||||||
@@ -627,7 +642,15 @@ class TantivyBackend:
|
|||||||
document.created.day,
|
document.created.day,
|
||||||
tzinfo=UTC,
|
tzinfo=UTC,
|
||||||
)
|
)
|
||||||
doc.add_date("created", created_date)
|
if _is_indexable_date(created_date):
|
||||||
|
doc.add_date("created", created_date)
|
||||||
|
else:
|
||||||
|
logger.warning(
|
||||||
|
"Document %s has a created date (%s) outside the range the search "
|
||||||
|
"index can store; it will be indexed without a created date",
|
||||||
|
document.pk,
|
||||||
|
document.created,
|
||||||
|
)
|
||||||
doc.add_date("modified", document.modified)
|
doc.add_date("modified", document.modified)
|
||||||
doc.add_date("added", document.added)
|
doc.add_date("added", document.added)
|
||||||
|
|
||||||
@@ -1110,39 +1133,44 @@ class TantivyBackend:
|
|||||||
flushing a segment, deferring merge work; they do not avoid it.
|
flushing a segment, deferring merge work; they do not avoid it.
|
||||||
"""
|
"""
|
||||||
wipe_index(self._path)
|
wipe_index(self._path)
|
||||||
new_index = tantivy.Index(build_schema(), path=str(self._path))
|
# The marker covers the window where the empty index is already stamped
|
||||||
_write_sentinels(self._path)
|
# as current but not yet populated, so an interrupted rebuild is retried.
|
||||||
register_tokenizers(new_index, settings.SEARCH_LANGUAGE)
|
with rebuild_in_progress(self._path):
|
||||||
|
new_index = tantivy.Index(build_schema(), path=str(self._path))
|
||||||
|
_write_sentinels(self._path)
|
||||||
|
register_tokenizers(new_index, settings.SEARCH_LANGUAGE)
|
||||||
|
|
||||||
# Point instance at the new index so _build_tantivy_doc uses it
|
# Point instance at the new index so _build_tantivy_doc uses it
|
||||||
old_index, old_schema = self._raw_index, self._raw_schema
|
old_index, old_schema = self._raw_index, self._raw_schema
|
||||||
self._raw_index = new_index
|
self._raw_index = new_index
|
||||||
self._raw_schema = new_index.schema
|
self._raw_schema = new_index.schema
|
||||||
# Stream documents one-by-one (so the progress bar advances per
|
# Stream documents one-by-one (so the progress bar advances per
|
||||||
# document) while fetching viewer permissions one SQL query per chunk.
|
# document) while fetching viewer permissions one SQL query per
|
||||||
# The stream is Sized, so iter_wrapper can still discover the total.
|
# chunk. The stream is Sized, so iter_wrapper can still discover
|
||||||
documents_stream = _DocumentViewerStream(documents, chunk_size=1000)
|
# the total.
|
||||||
try:
|
documents_stream = _DocumentViewerStream(documents, chunk_size=1000)
|
||||||
writer = new_index.writer(heap_size=writer_heap_bytes)
|
try:
|
||||||
for document, (viewer_ids, viewer_group_ids) in iter_wrapper(
|
writer = new_index.writer(heap_size=writer_heap_bytes)
|
||||||
documents_stream,
|
for document, (viewer_ids, viewer_group_ids) in iter_wrapper(
|
||||||
):
|
documents_stream,
|
||||||
doc = self._build_tantivy_doc(
|
):
|
||||||
document,
|
doc = self._build_tantivy_doc(
|
||||||
viewer_ids=viewer_ids,
|
document,
|
||||||
viewer_group_ids=viewer_group_ids,
|
viewer_ids=viewer_ids,
|
||||||
)
|
viewer_group_ids=viewer_group_ids,
|
||||||
writer.add_document(doc)
|
)
|
||||||
writer.commit()
|
writer.add_document(doc)
|
||||||
# Wait for background merge threads to finish so all segments are
|
writer.commit()
|
||||||
# fully merged and persisted before the index is considered rebuilt.
|
# Wait for background merge threads to finish so all segments
|
||||||
writer.wait_merging_threads()
|
# are fully merged and persisted before the index is considered
|
||||||
new_index.reload()
|
# rebuilt.
|
||||||
except BaseException: # pragma: no cover
|
writer.wait_merging_threads()
|
||||||
# Restore old index on failure so the backend remains usable
|
new_index.reload()
|
||||||
self._raw_index = old_index
|
except BaseException: # pragma: no cover
|
||||||
self._raw_schema = old_schema
|
# Restore old index on failure so the backend remains usable
|
||||||
raise
|
self._raw_index = old_index
|
||||||
|
self._raw_schema = old_schema
|
||||||
|
raise
|
||||||
|
|
||||||
|
|
||||||
def chunked(iterable, size):
|
def chunked(iterable, size):
|
||||||
|
|||||||
@@ -4,6 +4,7 @@ import hashlib
|
|||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import shutil
|
import shutil
|
||||||
|
from contextlib import contextmanager
|
||||||
from typing import TYPE_CHECKING
|
from typing import TYPE_CHECKING
|
||||||
from typing import Final
|
from typing import Final
|
||||||
from typing import NamedTuple
|
from typing import NamedTuple
|
||||||
@@ -16,6 +17,7 @@ from whoosh_compat import FieldKind
|
|||||||
from documents.search._fields import PUBLIC_FIELDS
|
from documents.search._fields import PUBLIC_FIELDS
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Iterator
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
logger = logging.getLogger("paperless.search")
|
logger = logging.getLogger("paperless.search")
|
||||||
@@ -28,6 +30,11 @@ logger = logging.getLogger("paperless.search")
|
|||||||
# v3 - barcodes JSON field for stored barcode contents
|
# v3 - barcodes JSON field for stored barcode contents
|
||||||
SCHEMA_VERSION: Final[int] = 3
|
SCHEMA_VERSION: Final[int] = 3
|
||||||
|
|
||||||
|
# Present in the index directory from the moment a full rebuild starts until it
|
||||||
|
# finishes. If a rebuild is interrupted it is left behind, so the half-built
|
||||||
|
# index is not mistaken for a complete one.
|
||||||
|
REBUILD_MARKER: Final[str] = ".rebuilding"
|
||||||
|
|
||||||
|
|
||||||
class FieldDescriptor(NamedTuple):
|
class FieldDescriptor(NamedTuple):
|
||||||
"""One tantivy field, in declaration order.
|
"""One tantivy field, in declaration order.
|
||||||
@@ -255,9 +262,9 @@ def needs_rebuild(index_dir: Path) -> bool:
|
|||||||
"""
|
"""
|
||||||
Check if the search index needs rebuilding.
|
Check if the search index needs rebuilding.
|
||||||
|
|
||||||
Reads .index_settings.json to compare the stored schema version, search
|
True if a previous full rebuild never finished (the rebuild marker is still
|
||||||
language and schema fingerprint against the current configuration. Returns
|
present), or if the index's stamped settings no longer match the current
|
||||||
True if the file is missing, unparsable, or any value mismatches.
|
configuration. See _settings_mismatch().
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
index_dir: Path to the search index directory
|
index_dir: Path to the search index directory
|
||||||
@@ -265,6 +272,40 @@ def needs_rebuild(index_dir: Path) -> bool:
|
|||||||
Returns:
|
Returns:
|
||||||
True if the index needs rebuilding, False if it's up to date
|
True if the index needs rebuilding, False if it's up to date
|
||||||
"""
|
"""
|
||||||
|
if (index_dir / REBUILD_MARKER).exists():
|
||||||
|
logger.warning("Previous search index rebuild did not finish - rebuilding.")
|
||||||
|
return True
|
||||||
|
return _settings_mismatch(index_dir)
|
||||||
|
|
||||||
|
|
||||||
|
@contextmanager
|
||||||
|
def rebuild_in_progress(index_dir: Path) -> Iterator[None]:
|
||||||
|
"""
|
||||||
|
Flag the index as incomplete for the duration of a full rebuild.
|
||||||
|
|
||||||
|
The marker is cleared only if the block exits cleanly. There is deliberately
|
||||||
|
no try/finally: an exception must leave the marker behind so the next
|
||||||
|
needs_rebuild() check retries the rebuild.
|
||||||
|
"""
|
||||||
|
marker = index_dir / REBUILD_MARKER
|
||||||
|
marker.touch()
|
||||||
|
yield
|
||||||
|
marker.unlink(missing_ok=True)
|
||||||
|
|
||||||
|
|
||||||
|
def _settings_mismatch(index_dir: Path) -> bool:
|
||||||
|
"""
|
||||||
|
Check the stamped settings against the current configuration.
|
||||||
|
|
||||||
|
Reads .index_settings.json to compare the stored schema version, search
|
||||||
|
language and schema fingerprint. Returns True if the file is missing,
|
||||||
|
unparsable, or any value mismatches.
|
||||||
|
|
||||||
|
This deliberately ignores the rebuild marker: open_or_rebuild_index() uses it
|
||||||
|
so that a process opening the index while another process is mid-rebuild
|
||||||
|
(or after one died) does not wipe the partial index out from under it.
|
||||||
|
Repopulating is the job of ``document_index reindex``.
|
||||||
|
"""
|
||||||
settings_file = index_dir / ".index_settings.json"
|
settings_file = index_dir / ".index_settings.json"
|
||||||
if not settings_file.exists():
|
if not settings_file.exists():
|
||||||
return True
|
return True
|
||||||
@@ -333,7 +374,7 @@ def open_or_rebuild_index(index_dir: Path | None = None) -> tantivy.Index:
|
|||||||
index_dir = cast("Path", settings.INDEX_DIR)
|
index_dir = cast("Path", settings.INDEX_DIR)
|
||||||
if not index_dir.exists():
|
if not index_dir.exists():
|
||||||
return tantivy.Index(build_schema())
|
return tantivy.Index(build_schema())
|
||||||
if needs_rebuild(index_dir):
|
if _settings_mismatch(index_dir):
|
||||||
wipe_index(index_dir)
|
wipe_index(index_dir)
|
||||||
idx = tantivy.Index(build_schema(), path=str(index_dir))
|
idx = tantivy.Index(build_schema(), path=str(index_dir))
|
||||||
_write_sentinels(index_dir)
|
_write_sentinels(index_dir)
|
||||||
|
|||||||
@@ -490,18 +490,23 @@ def update_document_content_maybe_archive_file(
|
|||||||
shutil.move(thumbnail, document.thumbnail_path)
|
shutil.move(thumbnail, document.thumbnail_path)
|
||||||
|
|
||||||
document.refresh_from_db()
|
document.refresh_from_db()
|
||||||
|
root_document = (
|
||||||
|
document.root_document if document.root_document_id else document
|
||||||
|
)
|
||||||
logger.info(
|
logger.info(
|
||||||
f"Updating index for document {document_id} ({document.archive_checksum})",
|
f"Updating index for document {root_document.pk} ({document.archive_checksum})",
|
||||||
)
|
)
|
||||||
from documents.search import get_backend
|
from documents.search import get_backend
|
||||||
|
|
||||||
get_backend().add_or_update(document)
|
get_backend().add_or_update(root_document)
|
||||||
|
|
||||||
ai_config = AIConfig()
|
ai_config = AIConfig()
|
||||||
if ai_config.llm_index_enabled:
|
if ai_config.llm_index_enabled:
|
||||||
llm_index_add_or_update_document(document)
|
llm_index_add_or_update_document(root_document)
|
||||||
|
|
||||||
clear_document_caches(document.pk)
|
clear_document_caches(document.pk)
|
||||||
|
if root_document.pk != document.pk:
|
||||||
|
clear_document_caches(root_document.pk)
|
||||||
|
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception(
|
logger.exception(
|
||||||
|
|||||||
@@ -1,4 +1,6 @@
|
|||||||
import json
|
import json
|
||||||
|
import logging
|
||||||
|
from datetime import date
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
@@ -16,7 +18,10 @@ from documents.search._backend import TantivyBackend
|
|||||||
from documents.search._backend import WriteBatch
|
from documents.search._backend import WriteBatch
|
||||||
from documents.search._backend import get_backend
|
from documents.search._backend import get_backend
|
||||||
from documents.search._backend import reset_backend
|
from documents.search._backend import reset_backend
|
||||||
|
from documents.search._schema import REBUILD_MARKER
|
||||||
|
from documents.search._schema import needs_rebuild
|
||||||
from documents.signals.handlers import add_to_index
|
from documents.signals.handlers import add_to_index
|
||||||
|
from paperless_testing.dirs import PaperlessDirs
|
||||||
from paperless_testing.factories import CorrespondentFactory
|
from paperless_testing.factories import CorrespondentFactory
|
||||||
from paperless_testing.factories import DocumentFactory
|
from paperless_testing.factories import DocumentFactory
|
||||||
from paperless_testing.factories import DocumentTypeFactory
|
from paperless_testing.factories import DocumentTypeFactory
|
||||||
@@ -823,6 +828,53 @@ class TestRebuild:
|
|||||||
backend.rebuild(Document.objects.all(), iter_wrapper=wrapper)
|
backend.rebuild(Document.objects.all(), iter_wrapper=wrapper)
|
||||||
assert 30 in seen
|
assert 30 in seen
|
||||||
|
|
||||||
|
def test_successful_rebuild_leaves_index_up_to_date(
|
||||||
|
self,
|
||||||
|
backend: TantivyBackend,
|
||||||
|
paperless_dirs: PaperlessDirs,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A backend and one document
|
||||||
|
WHEN:
|
||||||
|
- rebuild() completes
|
||||||
|
THEN:
|
||||||
|
- needs_rebuild() is False and no rebuild marker remains
|
||||||
|
"""
|
||||||
|
DocumentFactory.create()
|
||||||
|
|
||||||
|
backend.rebuild(Document.objects.all())
|
||||||
|
|
||||||
|
assert needs_rebuild(paperless_dirs.index_dir) is False
|
||||||
|
assert not (paperless_dirs.index_dir / REBUILD_MARKER).exists()
|
||||||
|
|
||||||
|
def test_interrupted_rebuild_is_retried(
|
||||||
|
self,
|
||||||
|
backend: TantivyBackend,
|
||||||
|
paperless_dirs: PaperlessDirs,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A rebuild that dies while indexing documents (e.g. the database
|
||||||
|
connection is lost)
|
||||||
|
WHEN:
|
||||||
|
- needs_rebuild() is checked afterwards
|
||||||
|
THEN:
|
||||||
|
- It is True, even though the empty index was already stamped with
|
||||||
|
current settings, so the next start rebuilds instead of reporting
|
||||||
|
the index as up to date
|
||||||
|
"""
|
||||||
|
DocumentFactory.create()
|
||||||
|
|
||||||
|
def die(pairs):
|
||||||
|
raise RuntimeError("terminating connection due to administrator command")
|
||||||
|
yield # pragma: no cover
|
||||||
|
|
||||||
|
with pytest.raises(RuntimeError):
|
||||||
|
backend.rebuild(Document.objects.all(), iter_wrapper=die)
|
||||||
|
|
||||||
|
assert needs_rebuild(paperless_dirs.index_dir) is True
|
||||||
|
|
||||||
def test_includes_group_granted_viewers(self, backend: TantivyBackend) -> None:
|
def test_includes_group_granted_viewers(self, backend: TantivyBackend) -> None:
|
||||||
"""Rebuild must index viewer ids for group-only grants, not just direct ones.
|
"""Rebuild must index viewer ids for group-only grants, not just direct ones.
|
||||||
|
|
||||||
@@ -854,6 +906,79 @@ class TestRebuild:
|
|||||||
assert ids == [doc.pk]
|
assert ids == [doc.pk]
|
||||||
|
|
||||||
|
|
||||||
|
class TestCreatedDateOutOfRange:
|
||||||
|
"""The index stores dates as nanosecond i64 values (1677-09-22 to 2262-04-11).
|
||||||
|
|
||||||
|
A document whose created date falls outside that window must not abort
|
||||||
|
indexing: it is indexed without a created value and a warning names it.
|
||||||
|
"""
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("created", "expected_warnings"),
|
||||||
|
[
|
||||||
|
pytest.param(date(1677, 9, 22), 0, id="first-representable-day"),
|
||||||
|
pytest.param(date(2262, 4, 11), 0, id="last-representable-day"),
|
||||||
|
pytest.param(date(1677, 9, 21), 1, id="day-before-first"),
|
||||||
|
pytest.param(date(2262, 4, 12), 1, id="day-after-last"),
|
||||||
|
pytest.param(date(16, 8, 30), 1, id="two-digit-year-read-as-year-16"),
|
||||||
|
pytest.param(date(9999, 12, 31), 1, id="max-python-date"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_add_or_update_indexes_document_and_warns_when_out_of_range(
|
||||||
|
self,
|
||||||
|
backend: TantivyBackend,
|
||||||
|
caplog: pytest.LogCaptureFixture,
|
||||||
|
created: date,
|
||||||
|
expected_warnings: int,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A document with a created date at or beyond the index date limits
|
||||||
|
WHEN:
|
||||||
|
- The document is added to the index
|
||||||
|
THEN:
|
||||||
|
- The document is indexed and searchable either way
|
||||||
|
- A warning naming the document is logged only for out-of-range dates
|
||||||
|
"""
|
||||||
|
doc = DocumentFactory(created=created, content="boundarycontent")
|
||||||
|
|
||||||
|
with caplog.at_level(logging.WARNING, logger="paperless.search"):
|
||||||
|
backend.add_or_update(doc)
|
||||||
|
|
||||||
|
assert backend.search_ids("boundarycontent", user=None) == [doc.pk]
|
||||||
|
warnings = [r for r in caplog.records if r.levelno == logging.WARNING]
|
||||||
|
assert len(warnings) == expected_warnings
|
||||||
|
if expected_warnings:
|
||||||
|
assert f"Document {doc.pk}" in warnings[0].getMessage()
|
||||||
|
|
||||||
|
def test_rebuild_continues_past_out_of_range_document(
|
||||||
|
self,
|
||||||
|
backend: TantivyBackend,
|
||||||
|
caplog: pytest.LogCaptureFixture,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A document with an unrepresentable created date among valid ones
|
||||||
|
WHEN:
|
||||||
|
- The index is rebuilt
|
||||||
|
THEN:
|
||||||
|
- Rebuild completes and every document is searchable
|
||||||
|
- A warning names the offending document
|
||||||
|
"""
|
||||||
|
good = DocumentFactory(created=date(2016, 8, 30), content="rebuildcontent")
|
||||||
|
bad = DocumentFactory(created=date(16, 8, 30), content="rebuildcontent")
|
||||||
|
|
||||||
|
with caplog.at_level(logging.WARNING, logger="paperless.search"):
|
||||||
|
backend.rebuild(Document.objects.all())
|
||||||
|
|
||||||
|
assert sorted(backend.search_ids("rebuildcontent", user=None)) == sorted(
|
||||||
|
[good.pk, bad.pk],
|
||||||
|
)
|
||||||
|
warnings = [r for r in caplog.records if r.levelno == logging.WARNING]
|
||||||
|
assert len(warnings) == 1
|
||||||
|
assert f"Document {bad.pk}" in warnings[0].getMessage()
|
||||||
|
|
||||||
|
|
||||||
class TestAutocomplete:
|
class TestAutocomplete:
|
||||||
"""Test autocomplete functionality."""
|
"""Test autocomplete functionality."""
|
||||||
|
|
||||||
|
|||||||
@@ -8,6 +8,7 @@ from auditlog.models import LogEntry # type: ignore[import-untyped]
|
|||||||
from django.contrib.contenttypes.models import ContentType
|
from django.contrib.contenttypes.models import ContentType
|
||||||
from django.core.files.uploadedfile import SimpleUploadedFile
|
from django.core.files.uploadedfile import SimpleUploadedFile
|
||||||
from django.test import TestCase as DjangoTestCase
|
from django.test import TestCase as DjangoTestCase
|
||||||
|
from django.test import override_settings
|
||||||
from django.utils import timezone
|
from django.utils import timezone
|
||||||
from rest_framework import status
|
from rest_framework import status
|
||||||
from rest_framework.test import APITestCase
|
from rest_framework.test import APITestCase
|
||||||
@@ -16,13 +17,17 @@ from documents.data_models import DocumentSource
|
|||||||
from documents.filters import EffectiveContentFilter
|
from documents.filters import EffectiveContentFilter
|
||||||
from documents.filters import TitleContentFilter
|
from documents.filters import TitleContentFilter
|
||||||
from documents.models import Document
|
from documents.models import Document
|
||||||
|
from documents.models import Note
|
||||||
|
from documents.models import ShareLink
|
||||||
from documents.versioning import annotate_effective_content
|
from documents.versioning import annotate_effective_content
|
||||||
from documents.views import DocumentSelectionMixin
|
from documents.views import DocumentSelectionMixin
|
||||||
from paperless_testing.dirs import DirectoriesMixin
|
from paperless_testing.dirs import DirectoriesMixin
|
||||||
from paperless_testing.factories import DocumentFactory
|
from paperless_testing.factories import DocumentFactory
|
||||||
from paperless_testing.factories import UserFactory
|
from paperless_testing.factories import UserFactory
|
||||||
from paperless_testing.http import read_streaming_response
|
from paperless_testing.http import read_streaming_response
|
||||||
|
from paperless_testing.permissions import grant_all_global
|
||||||
from paperless_testing.permissions import grant_global
|
from paperless_testing.permissions import grant_global
|
||||||
|
from paperless_testing.permissions import grant_object
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
@@ -1043,3 +1048,152 @@ class TestBulkSelectionExcludesVersions(DjangoTestCase):
|
|||||||
)
|
)
|
||||||
|
|
||||||
self.assertEqual(selected, [root.id])
|
self.assertEqual(selected, [root.id])
|
||||||
|
|
||||||
|
|
||||||
|
class TestVersionActionPermissions(DirectoriesMixin, APITestCase):
|
||||||
|
def setUp(self):
|
||||||
|
super().setUp()
|
||||||
|
self.user = UserFactory()
|
||||||
|
grant_all_global(self.user)
|
||||||
|
self.client.force_authenticate(self.user)
|
||||||
|
self.root = DocumentFactory(owner=UserFactory())
|
||||||
|
self.version = DocumentFactory(root_document=self.root, owner=None)
|
||||||
|
|
||||||
|
@override_settings(AUDIT_LOG_ENABLED=True)
|
||||||
|
def test_actions_reject_stale_version_ownership(self):
|
||||||
|
note = Note.objects.create(document=self.version, note="Version note")
|
||||||
|
for owner in (None, self.user):
|
||||||
|
self.version.owner = owner
|
||||||
|
self.version.save(update_fields=["owner"])
|
||||||
|
for action in (
|
||||||
|
"notes",
|
||||||
|
"suggestions",
|
||||||
|
"ai_suggestions",
|
||||||
|
"history",
|
||||||
|
"share_links",
|
||||||
|
):
|
||||||
|
with self.subTest(owner=owner, action=action):
|
||||||
|
response = self.client.get(
|
||||||
|
f"/api/documents/{self.version.pk}/{action}/",
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 403)
|
||||||
|
response = self.client.post(
|
||||||
|
f"/api/documents/{self.version.pk}/notes/",
|
||||||
|
{"note": "New note"},
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 403)
|
||||||
|
response = self.client.delete(
|
||||||
|
f"/api/documents/{self.version.pk}/notes/?id={note.pk}",
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 403)
|
||||||
|
response = self.client.post(
|
||||||
|
"/api/share_links/",
|
||||||
|
{"document": self.version.pk, "file_version": "original"},
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 403)
|
||||||
|
response = self.client.post(
|
||||||
|
"/api/share_link_bundles/",
|
||||||
|
{"document_ids": [self.version.pk], "file_version": "original"},
|
||||||
|
format="json",
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 400)
|
||||||
|
response = self.client.post(
|
||||||
|
"/api/documents/email/",
|
||||||
|
{
|
||||||
|
"documents": [self.version.pk],
|
||||||
|
"addresses": "recipient@example.com",
|
||||||
|
"subject": "Version",
|
||||||
|
"message": "Version",
|
||||||
|
},
|
||||||
|
format="json",
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 403)
|
||||||
|
with (
|
||||||
|
mock.patch("documents.views.AIConfig") as ai_config,
|
||||||
|
mock.patch("documents.views.stream_chat_with_documents") as chat,
|
||||||
|
):
|
||||||
|
ai_config.return_value.ai_enabled = True
|
||||||
|
response = self.client.post(
|
||||||
|
"/api/documents/chat/",
|
||||||
|
{"q": "Version?", "document_id": self.version.pk},
|
||||||
|
format="json",
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 403)
|
||||||
|
chat.assert_not_called()
|
||||||
|
self.assertTrue(Note.objects.filter(pk=note.pk).exists())
|
||||||
|
self.assertFalse(ShareLink.objects.exists())
|
||||||
|
|
||||||
|
@mock.patch("documents.views.build_share_link_bundle.apply_async")
|
||||||
|
def test_root_permissions_allow_sharing_a_private_version(self, build_mock):
|
||||||
|
self.version.owner = UserFactory()
|
||||||
|
self.version.save(update_fields=["owner"])
|
||||||
|
grant_object(self.user, self.root, "view_document", "change_document")
|
||||||
|
note = Note.objects.create(document=self.version, note="Version note")
|
||||||
|
response = self.client.get(f"/api/documents/{self.version.pk}/notes/")
|
||||||
|
self.assertEqual(response.status_code, 200)
|
||||||
|
self.assertEqual(response.data[0]["id"], note.pk)
|
||||||
|
response = self.client.post(
|
||||||
|
"/api/share_links/",
|
||||||
|
{"document": self.version.pk, "file_version": "original"},
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 201)
|
||||||
|
self.assertEqual(ShareLink.objects.get().document_id, self.version.pk)
|
||||||
|
response = self.client.get(f"/api/documents/{self.version.pk}/share_links/")
|
||||||
|
self.assertEqual(response.status_code, 200)
|
||||||
|
self.assertEqual(len(response.data), 1)
|
||||||
|
response = self.client.post(
|
||||||
|
"/api/share_link_bundles/",
|
||||||
|
{"document_ids": [self.version.pk], "file_version": "original"},
|
||||||
|
format="json",
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 201)
|
||||||
|
build_mock.assert_called_once()
|
||||||
|
|
||||||
|
def test_root_view_permission_does_not_allow_note_changes(self):
|
||||||
|
grant_object(self.user, self.root, "view_document")
|
||||||
|
note = Note.objects.create(document=self.version, note="Version note")
|
||||||
|
response = self.client.get(f"/api/documents/{self.version.pk}/notes/")
|
||||||
|
self.assertEqual(response.status_code, 200)
|
||||||
|
response = self.client.post(
|
||||||
|
f"/api/documents/{self.version.pk}/notes/",
|
||||||
|
{"note": "New note"},
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 403)
|
||||||
|
response = self.client.delete(
|
||||||
|
f"/api/documents/{self.version.pk}/notes/?id={note.pk}",
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 403)
|
||||||
|
self.assertTrue(Note.objects.filter(pk=note.pk).exists())
|
||||||
|
|
||||||
|
@override_settings(AUDIT_LOG_ENABLED=True)
|
||||||
|
def test_history_uses_root_ownership(self):
|
||||||
|
self.root.owner = self.user
|
||||||
|
self.root.save(update_fields=["owner"])
|
||||||
|
self.version.owner = UserFactory()
|
||||||
|
self.version.save(update_fields=["owner"])
|
||||||
|
response = self.client.get(f"/api/documents/{self.version.pk}/history/")
|
||||||
|
self.assertEqual(response.status_code, 200)
|
||||||
|
|
||||||
|
def test_selection_data_rejects_stale_version_ownership(self):
|
||||||
|
for owner in (None, self.user):
|
||||||
|
self.version.owner = owner
|
||||||
|
self.version.save(update_fields=["owner"])
|
||||||
|
with self.subTest(owner=owner):
|
||||||
|
response = self.client.post(
|
||||||
|
"/api/documents/selection_data/",
|
||||||
|
{"documents": [self.version.pk]},
|
||||||
|
format="json",
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 403)
|
||||||
|
|
||||||
|
def test_selection_data_allows_private_version_of_permitted_root(self):
|
||||||
|
self.version.owner = UserFactory()
|
||||||
|
self.version.save(update_fields=["owner"])
|
||||||
|
grant_object(self.user, self.root, "view_document")
|
||||||
|
other = DocumentFactory(owner=self.user)
|
||||||
|
response = self.client.post(
|
||||||
|
"/api/documents/selection_data/",
|
||||||
|
{"documents": [self.version.pk, other.pk]},
|
||||||
|
format="json",
|
||||||
|
)
|
||||||
|
self.assertEqual(response.status_code, 200)
|
||||||
@@ -6,6 +6,7 @@ from rest_framework.test import APITestCase
|
|||||||
|
|
||||||
from documents.models import Document
|
from documents.models import Document
|
||||||
from paperless_testing.dirs import DirectoriesMixin
|
from paperless_testing.dirs import DirectoriesMixin
|
||||||
|
from paperless_testing.factories import DocumentFactory
|
||||||
from paperless_testing.factories import UserFactory
|
from paperless_testing.factories import UserFactory
|
||||||
from paperless_testing.permissions import grant_all_global
|
from paperless_testing.permissions import grant_all_global
|
||||||
|
|
||||||
@@ -279,3 +280,53 @@ class TestTrashAPI(DirectoriesMixin, APITestCase):
|
|||||||
Document.objects.filter(root_document=root).values_list("id", flat=True),
|
Document.objects.filter(root_document=root).values_list("id", flat=True),
|
||||||
[version.pk for version in versions],
|
[version.pk for version in versions],
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def test_api_trash_version_follows_root_owner(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A deleted version of user2's document, owned by nobody
|
||||||
|
- A deleted version of the user's document, owned by user2
|
||||||
|
WHEN:
|
||||||
|
- The user lists the trash and tries to restore or empty the versions
|
||||||
|
THEN:
|
||||||
|
- Only the version of the user's own document is listed
|
||||||
|
- The other version can't be restored or emptied
|
||||||
|
- The version of the user's own document can be restored
|
||||||
|
"""
|
||||||
|
user2 = UserFactory(username="user2")
|
||||||
|
other_version = DocumentFactory(
|
||||||
|
root_document=DocumentFactory(owner=user2),
|
||||||
|
version_index=1,
|
||||||
|
)
|
||||||
|
other_version.delete()
|
||||||
|
own_version = DocumentFactory(
|
||||||
|
owner=user2,
|
||||||
|
root_document=DocumentFactory(owner=self.user),
|
||||||
|
version_index=1,
|
||||||
|
)
|
||||||
|
own_version.delete()
|
||||||
|
|
||||||
|
resp = self.client.get("/api/trash/")
|
||||||
|
self.assertEqual(resp.status_code, status.HTTP_200_OK)
|
||||||
|
self.assertEqual(
|
||||||
|
[doc["id"] for doc in resp.data["results"]],
|
||||||
|
[own_version.pk],
|
||||||
|
)
|
||||||
|
|
||||||
|
for action in ("restore", "empty"):
|
||||||
|
with self.subTest(action=action):
|
||||||
|
resp = self.client.post(
|
||||||
|
"/api/trash/",
|
||||||
|
{"action": action, "documents": [other_version.pk]},
|
||||||
|
)
|
||||||
|
self.assertEqual(resp.status_code, status.HTTP_403_FORBIDDEN)
|
||||||
|
self.assertTrue(
|
||||||
|
Document.deleted_objects.filter(pk=other_version.pk).exists(),
|
||||||
|
)
|
||||||
|
|
||||||
|
resp = self.client.post(
|
||||||
|
"/api/trash/",
|
||||||
|
{"action": "restore", "documents": [own_version.pk]},
|
||||||
|
)
|
||||||
|
self.assertEqual(resp.status_code, status.HTTP_200_OK)
|
||||||
|
self.assertTrue(Document.objects.filter(pk=own_version.pk).exists())
|
||||||
@@ -4,6 +4,7 @@ from pathlib import Path
|
|||||||
from unittest import mock
|
from unittest import mock
|
||||||
|
|
||||||
import pikepdf
|
import pikepdf
|
||||||
|
import pytest
|
||||||
from django.contrib.auth.models import Group
|
from django.contrib.auth.models import Group
|
||||||
from django.contrib.auth.models import Permission
|
from django.contrib.auth.models import Permission
|
||||||
from django.contrib.auth.models import User
|
from django.contrib.auth.models import User
|
||||||
@@ -12,6 +13,7 @@ from django.test import TestCase
|
|||||||
from django.test.utils import CaptureQueriesContext
|
from django.test.utils import CaptureQueriesContext
|
||||||
from guardian.shortcuts import get_groups_with_perms
|
from guardian.shortcuts import get_groups_with_perms
|
||||||
from guardian.shortcuts import get_users_with_perms
|
from guardian.shortcuts import get_users_with_perms
|
||||||
|
from pytest_mock import MockerFixture
|
||||||
|
|
||||||
from documents import bulk_edit
|
from documents import bulk_edit
|
||||||
from documents.models import Correspondent
|
from documents.models import Correspondent
|
||||||
@@ -23,6 +25,7 @@ from documents.models import StoragePath
|
|||||||
from documents.models import Tag
|
from documents.models import Tag
|
||||||
from documents.permissions import set_permissions_for_objects
|
from documents.permissions import set_permissions_for_objects
|
||||||
from paperless_testing.dirs import DirectoriesMixin
|
from paperless_testing.dirs import DirectoriesMixin
|
||||||
|
from paperless_testing.factories import DocumentFactory
|
||||||
from paperless_testing.permissions import grant_object
|
from paperless_testing.permissions import grant_object
|
||||||
|
|
||||||
|
|
||||||
@@ -1970,18 +1973,22 @@ class TestPDFActions(DirectoriesMixin, TestCase):
|
|||||||
self.assertIn("Error removing password from document", cm.output[0])
|
self.assertIn("Error removing password from document", cm.output[0])
|
||||||
|
|
||||||
|
|
||||||
class TestBulkEditReprocess(DirectoriesMixin, TestCase):
|
@pytest.mark.django_db
|
||||||
def setUp(self) -> None:
|
class TestBulkEditReprocess:
|
||||||
super().setUp()
|
@pytest.fixture
|
||||||
|
def mock_task(self, mocker: MockerFixture) -> mock.MagicMock:
|
||||||
self.doc = Document.objects.create(
|
return mocker.patch(
|
||||||
title="test",
|
"documents.bulk_edit.update_document_content_maybe_archive_file",
|
||||||
checksum="A",
|
|
||||||
mime_type="application/pdf",
|
|
||||||
)
|
)
|
||||||
|
|
||||||
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file")
|
@staticmethod
|
||||||
def test_reprocess_defaults_to_local(self, mock_task: mock.Mock) -> None:
|
def _queued_ids(mock_task: mock.MagicMock) -> list[int]:
|
||||||
|
return [
|
||||||
|
call.kwargs["kwargs"]["document_id"]
|
||||||
|
for call in mock_task.apply_async.call_args_list
|
||||||
|
]
|
||||||
|
|
||||||
|
def test_reprocess_defaults_to_local(self, mock_task: mock.MagicMock) -> None:
|
||||||
"""
|
"""
|
||||||
GIVEN:
|
GIVEN:
|
||||||
- A reprocess request that says nothing about remote OCR
|
- A reprocess request that says nothing about remote OCR
|
||||||
@@ -1990,18 +1997,17 @@ class TestBulkEditReprocess(DirectoriesMixin, TestCase):
|
|||||||
THEN:
|
THEN:
|
||||||
- The task is queued without asking for the remote engine
|
- The task is queued without asking for the remote engine
|
||||||
"""
|
"""
|
||||||
result = bulk_edit.reprocess([self.doc.id])
|
doc = DocumentFactory()
|
||||||
|
|
||||||
|
assert bulk_edit.reprocess([doc.id]) == "OK"
|
||||||
|
|
||||||
self.assertEqual(result, "OK")
|
|
||||||
mock_task.apply_async.assert_called_once()
|
mock_task.apply_async.assert_called_once()
|
||||||
_, kwargs = mock_task.apply_async.call_args
|
assert mock_task.apply_async.call_args.kwargs["kwargs"] == {
|
||||||
self.assertEqual(
|
"document_id": doc.id,
|
||||||
kwargs["kwargs"],
|
"remote_ocr": False,
|
||||||
{"document_id": self.doc.id, "remote_ocr": False},
|
}
|
||||||
)
|
|
||||||
|
|
||||||
@mock.patch("documents.bulk_edit.update_document_content_maybe_archive_file")
|
def test_reprocess_passes_remote_ocr(self, mock_task: mock.MagicMock) -> None:
|
||||||
def test_reprocess_passes_remote_ocr(self, mock_task: mock.Mock) -> None:
|
|
||||||
"""
|
"""
|
||||||
GIVEN:
|
GIVEN:
|
||||||
- A reprocess request that explicitly asks for remote OCR
|
- A reprocess request that explicitly asks for remote OCR
|
||||||
@@ -2010,14 +2016,72 @@ class TestBulkEditReprocess(DirectoriesMixin, TestCase):
|
|||||||
THEN:
|
THEN:
|
||||||
- The request is forwarded to the task for every document
|
- The request is forwarded to the task for every document
|
||||||
"""
|
"""
|
||||||
other = Document.objects.create(
|
docs = DocumentFactory.create_batch(2)
|
||||||
title="test2",
|
|
||||||
checksum="B",
|
bulk_edit.reprocess([doc.id for doc in docs], remote_ocr=True)
|
||||||
mime_type="application/pdf",
|
|
||||||
|
assert mock_task.apply_async.call_count == 2
|
||||||
|
for call in mock_task.apply_async.call_args_list:
|
||||||
|
assert call.kwargs["kwargs"]["remote_ocr"]
|
||||||
|
|
||||||
|
def test_reprocess_root_uses_latest_version(
|
||||||
|
self,
|
||||||
|
mock_task: mock.MagicMock,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with two versions
|
||||||
|
WHEN:
|
||||||
|
- reprocess is called with the root document
|
||||||
|
THEN:
|
||||||
|
- The latest version is reprocessed, not the root's original file
|
||||||
|
"""
|
||||||
|
root = DocumentFactory()
|
||||||
|
DocumentFactory(root_document=root, version_index=1)
|
||||||
|
latest = DocumentFactory(root_document=root, version_index=2)
|
||||||
|
|
||||||
|
bulk_edit.reprocess([root.id])
|
||||||
|
|
||||||
|
assert self._queued_ids(mock_task) == [latest.id]
|
||||||
|
|
||||||
|
def test_reprocess_explicit_version(self, mock_task: mock.MagicMock) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with two versions
|
||||||
|
WHEN:
|
||||||
|
- reprocess is called with the older version
|
||||||
|
THEN:
|
||||||
|
- That version is reprocessed
|
||||||
|
"""
|
||||||
|
root = DocumentFactory()
|
||||||
|
older = DocumentFactory(root_document=root, version_index=1)
|
||||||
|
DocumentFactory(root_document=root, version_index=2)
|
||||||
|
|
||||||
|
bulk_edit.reprocess([older.id])
|
||||||
|
|
||||||
|
assert self._queued_ids(mock_task) == [older.id]
|
||||||
|
|
||||||
|
def test_reprocess_root_and_latest_version_dispatches_once(
|
||||||
|
self,
|
||||||
|
mock_task: mock.MagicMock,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with two versions, the latest created on a
|
||||||
|
different date than the root
|
||||||
|
WHEN:
|
||||||
|
- reprocess is called with both the root and its latest version
|
||||||
|
THEN:
|
||||||
|
- The latest version is reprocessed only once
|
||||||
|
"""
|
||||||
|
root = DocumentFactory(created=date(2024, 1, 1))
|
||||||
|
DocumentFactory(root_document=root, version_index=1)
|
||||||
|
latest = DocumentFactory(
|
||||||
|
root_document=root,
|
||||||
|
version_index=2,
|
||||||
|
created=date(2025, 1, 1),
|
||||||
)
|
)
|
||||||
|
|
||||||
bulk_edit.reprocess([self.doc.id, other.id], remote_ocr=True)
|
bulk_edit.reprocess([root.id, latest.id])
|
||||||
|
|
||||||
self.assertEqual(mock_task.apply_async.call_count, 2)
|
assert self._queued_ids(mock_task) == [latest.id]
|
||||||
for call in mock_task.apply_async.call_args_list:
|
|
||||||
self.assertTrue(call.kwargs["kwargs"]["remote_ocr"])
|
|
||||||
@@ -1,22 +1,11 @@
|
|||||||
import subprocess
|
|
||||||
from collections.abc import Generator
|
from collections.abc import Generator
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
import pikepdf
|
|
||||||
import pytest
|
import pytest
|
||||||
from PIL import Image
|
|
||||||
from pytest_django.fixtures import Settings
|
from pytest_django.fixtures import Settings
|
||||||
from pytest_mock import MockerFixture
|
|
||||||
|
|
||||||
from documents.parsers import ParseError
|
|
||||||
from documents.parsers import _compute_thumbnail_dpi
|
|
||||||
from documents.parsers import encode_thumbnail_webp
|
|
||||||
from documents.parsers import get_default_file_extension
|
from documents.parsers import get_default_file_extension
|
||||||
from documents.parsers import get_default_thumbnail
|
|
||||||
from documents.parsers import get_supported_file_extensions
|
from documents.parsers import get_supported_file_extensions
|
||||||
from documents.parsers import is_file_ext_supported
|
from documents.parsers import is_file_ext_supported
|
||||||
from documents.parsers import make_thumbnail_from_pdf
|
|
||||||
from documents.parsers import rasterize_pdf_page_to_png
|
|
||||||
from paperless.parsers.registry import get_parser_registry
|
from paperless.parsers.registry import get_parser_registry
|
||||||
from paperless.parsers.registry import reset_parser_registry
|
from paperless.parsers.registry import reset_parser_registry
|
||||||
from paperless.parsers.tesseract import RasterisedDocumentParser
|
from paperless.parsers.tesseract import RasterisedDocumentParser
|
||||||
@@ -136,374 +125,3 @@ class TestParserAvailability:
|
|||||||
assert is_file_ext_supported(".pdf")
|
assert is_file_ext_supported(".pdf")
|
||||||
assert not is_file_ext_supported(".hsdfh")
|
assert not is_file_ext_supported(".hsdfh")
|
||||||
assert not is_file_ext_supported("")
|
assert not is_file_ext_supported("")
|
||||||
|
|
||||||
|
|
||||||
class TestComputeThumbnailDpi:
|
|
||||||
@pytest.mark.parametrize(
|
|
||||||
("size", "expected"),
|
|
||||||
[
|
|
||||||
pytest.param((612.0, 792.0), (59, 2), id="letter-width-bound"),
|
|
||||||
pytest.param((792.0, 612.0), (46, 2), id="landscape-rounded-up"),
|
|
||||||
pytest.param((612.0, 100000.0), (4, 2), id="tall-strip-height-bound"),
|
|
||||||
pytest.param((200.0, 300.0), (180, 2), id="small-page-width-bound"),
|
|
||||||
pytest.param((72.0, 72.0), (300, 2), id="tiny-page-capped-at-300"),
|
|
||||||
pytest.param(
|
|
||||||
(1000000.0, 1000000.0),
|
|
||||||
(1, 1),
|
|
||||||
id="huge-page-minimum-one-unsupersampled",
|
|
||||||
),
|
|
||||||
pytest.param(None, (150, 1), id="unreadable-geometry-fallback"),
|
|
||||||
],
|
|
||||||
)
|
|
||||||
def test_dpi_from_page_size(
|
|
||||||
self,
|
|
||||||
mocker: MockerFixture,
|
|
||||||
tmp_path: Path,
|
|
||||||
size: tuple[float, float] | None,
|
|
||||||
expected: tuple[int, int],
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A first page of the given size, or unreadable geometry
|
|
||||||
WHEN:
|
|
||||||
- The thumbnail DPI is computed
|
|
||||||
THEN:
|
|
||||||
- The expected DPI and supersample factor are returned
|
|
||||||
"""
|
|
||||||
mocker.patch(
|
|
||||||
"paperless.parsers.utils.get_pdf_first_page_size_points",
|
|
||||||
return_value=size,
|
|
||||||
)
|
|
||||||
assert _compute_thumbnail_dpi(tmp_path / "doc.pdf") == expected
|
|
||||||
|
|
||||||
|
|
||||||
class TestMakeThumbnailFromPdf:
|
|
||||||
@pytest.fixture
|
|
||||||
def work_dir(self, tmp_path: Path) -> Path:
|
|
||||||
path = tmp_path / "work"
|
|
||||||
path.mkdir()
|
|
||||||
return path
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def _write_blank_pdf(path: Path, page_size: tuple[int, int] = (612, 792)) -> Path:
|
|
||||||
pdf = pikepdf.new()
|
|
||||||
pdf.add_blank_page(page_size=page_size)
|
|
||||||
pdf.save(path, object_stream_mode=pikepdf.ObjectStreamMode.disable)
|
|
||||||
return path
|
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
|
||||||
("size", "expected_dpi", "expected_supersample"),
|
|
||||||
[
|
|
||||||
pytest.param((612.0, 792.0), 118, 2, id="known-geometry-2x"),
|
|
||||||
pytest.param(None, 150, 1, id="unreadable-geometry-plain-fallback"),
|
|
||||||
],
|
|
||||||
)
|
|
||||||
def test_render_dpi_requested(
|
|
||||||
self,
|
|
||||||
mocker: MockerFixture,
|
|
||||||
tmp_path: Path,
|
|
||||||
work_dir: Path,
|
|
||||||
size: tuple[float, float] | None,
|
|
||||||
expected_dpi: int,
|
|
||||||
expected_supersample: int,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- Readable or unreadable page geometry
|
|
||||||
WHEN:
|
|
||||||
- A thumbnail is made
|
|
||||||
THEN:
|
|
||||||
- Rasterize and encode get the matching DPI and supersample factor
|
|
||||||
"""
|
|
||||||
mocker.patch(
|
|
||||||
"paperless.parsers.utils.get_pdf_first_page_size_points",
|
|
||||||
return_value=size,
|
|
||||||
)
|
|
||||||
rasterize = mocker.patch("documents.parsers.rasterize_pdf_page_to_png")
|
|
||||||
encode = mocker.patch("documents.parsers.encode_thumbnail_webp")
|
|
||||||
|
|
||||||
make_thumbnail_from_pdf(tmp_path / "in.pdf", work_dir)
|
|
||||||
|
|
||||||
assert rasterize.call_args.kwargs["dpi"] == expected_dpi
|
|
||||||
assert encode.call_args.kwargs["supersample"] == expected_supersample
|
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
|
||||||
("page_size", "expected_width"),
|
|
||||||
[
|
|
||||||
pytest.param((612, 792), 500, id="letter"),
|
|
||||||
pytest.param((792, 612), 500, id="landscape-letter"),
|
|
||||||
pytest.param((595, 842), 500, id="a4"),
|
|
||||||
pytest.param((200, 300), 500, id="small-page"),
|
|
||||||
pytest.param((72, 72), 300, id="tiny-page-capped"),
|
|
||||||
],
|
|
||||||
)
|
|
||||||
def test_thumbnail_width(
|
|
||||||
self,
|
|
||||||
tmp_path: Path,
|
|
||||||
work_dir: Path,
|
|
||||||
page_size: tuple[int, int],
|
|
||||||
expected_width: int,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A PDF whose first page has the given size in points
|
|
||||||
WHEN:
|
|
||||||
- A thumbnail is made from it
|
|
||||||
THEN:
|
|
||||||
- The thumbnail has the expected width
|
|
||||||
"""
|
|
||||||
pdf_path = self._write_blank_pdf(tmp_path / "in.pdf", page_size)
|
|
||||||
|
|
||||||
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
|
|
||||||
|
|
||||||
assert thumb == work_dir / "convert.webp"
|
|
||||||
with Image.open(thumb) as im:
|
|
||||||
assert im.format == "WEBP"
|
|
||||||
assert im.width == expected_width
|
|
||||||
|
|
||||||
@classmethod
|
|
||||||
def _write_pdf_without_xref(cls, path: Path) -> Path:
|
|
||||||
"""
|
|
||||||
Cuts off the xref and trailer, which pdftoppm cannot recover from but qpdf can.
|
|
||||||
"""
|
|
||||||
cls._write_blank_pdf(path)
|
|
||||||
data = path.read_bytes()
|
|
||||||
path.write_bytes(data[: data.rindex(b"\nxref")])
|
|
||||||
return path
|
|
||||||
|
|
||||||
def test_qpdf_repair_produces_real_thumbnail(
|
|
||||||
self,
|
|
||||||
tmp_path: Path,
|
|
||||||
work_dir: Path,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A PDF with its xref table and trailer cut off
|
|
||||||
WHEN:
|
|
||||||
- A thumbnail is made from it
|
|
||||||
THEN:
|
|
||||||
- The thumbnail is rendered from a qpdf repaired copy
|
|
||||||
- The original file is unchanged
|
|
||||||
"""
|
|
||||||
pdf_path = self._write_pdf_without_xref(tmp_path / "broken.pdf")
|
|
||||||
original_bytes = pdf_path.read_bytes()
|
|
||||||
|
|
||||||
with pytest.raises(ParseError):
|
|
||||||
rasterize_pdf_page_to_png(pdf_path, work_dir / "probe.png", dpi=50)
|
|
||||||
|
|
||||||
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
|
|
||||||
|
|
||||||
assert thumb == work_dir / "convert_qpdf.webp"
|
|
||||||
with Image.open(thumb) as im:
|
|
||||||
assert im.format == "WEBP"
|
|
||||||
assert im.width == 500
|
|
||||||
assert pdf_path.read_bytes() == original_bytes
|
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
|
||||||
"qpdf_error",
|
|
||||||
[
|
|
||||||
pytest.param(subprocess.CalledProcessError(2, "qpdf"), id="qpdf-fails"),
|
|
||||||
pytest.param(None, id="repaired-still-unrenderable"),
|
|
||||||
],
|
|
||||||
)
|
|
||||||
def test_double_failure_uses_default_thumbnail(
|
|
||||||
self,
|
|
||||||
mocker: MockerFixture,
|
|
||||||
tmp_path: Path,
|
|
||||||
work_dir: Path,
|
|
||||||
qpdf_error: subprocess.CalledProcessError | None,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A PDF that cannot be rendered, even after qpdf repair
|
|
||||||
WHEN:
|
|
||||||
- A thumbnail is made from it
|
|
||||||
THEN:
|
|
||||||
- A copy of the default thumbnail is returned
|
|
||||||
"""
|
|
||||||
mocker.patch(
|
|
||||||
"documents.parsers.rasterize_pdf_page_to_png",
|
|
||||||
side_effect=ParseError("Does not compute."),
|
|
||||||
)
|
|
||||||
if qpdf_error is not None:
|
|
||||||
mocker.patch("documents.parsers.run_subprocess", side_effect=qpdf_error)
|
|
||||||
pdf_path = self._write_blank_pdf(tmp_path / "in.pdf")
|
|
||||||
|
|
||||||
thumb = make_thumbnail_from_pdf(pdf_path, work_dir)
|
|
||||||
|
|
||||||
assert thumb == work_dir / "document.webp"
|
|
||||||
assert thumb.read_bytes() == get_default_thumbnail().read_bytes()
|
|
||||||
|
|
||||||
|
|
||||||
class TestRasterizePdfPageToPng:
|
|
||||||
@staticmethod
|
|
||||||
def _write_pdf(
|
|
||||||
path: Path,
|
|
||||||
*,
|
|
||||||
crop_box: tuple[float, float, float, float] | None = None,
|
|
||||||
rotate: int | None = None,
|
|
||||||
) -> Path:
|
|
||||||
pdf = pikepdf.new()
|
|
||||||
pdf.add_blank_page(page_size=(144, 72))
|
|
||||||
pdf.add_blank_page(page_size=(300, 300))
|
|
||||||
page = pdf.pages[0]
|
|
||||||
if crop_box is not None:
|
|
||||||
page.obj.CropBox = pikepdf.Array(crop_box)
|
|
||||||
if rotate is not None:
|
|
||||||
page.obj.Rotate = rotate
|
|
||||||
pdf.save(path)
|
|
||||||
return path
|
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
|
||||||
("crop_box", "rotate", "expected_size"),
|
|
||||||
[
|
|
||||||
pytest.param(None, None, (144, 72), id="plain"),
|
|
||||||
pytest.param(None, 90, (72, 144), id="rotated-90"),
|
|
||||||
pytest.param((0, 0, 72, 36), None, (72, 36), id="crop-box"),
|
|
||||||
],
|
|
||||||
)
|
|
||||||
def test_renders_first_page_to_exact_path(
|
|
||||||
self,
|
|
||||||
tmp_path: Path,
|
|
||||||
crop_box: tuple[float, float, float, float] | None,
|
|
||||||
rotate: int | None,
|
|
||||||
expected_size: tuple[int, int],
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A two page PDF, first page optionally cropped or rotated
|
|
||||||
WHEN:
|
|
||||||
- The first page is rasterized at 72 DPI
|
|
||||||
THEN:
|
|
||||||
- Only out_path is written, sized to the first page's crop and rotation
|
|
||||||
"""
|
|
||||||
pdf_path = self._write_pdf(
|
|
||||||
tmp_path / "in.pdf",
|
|
||||||
crop_box=crop_box,
|
|
||||||
rotate=rotate,
|
|
||||||
)
|
|
||||||
out_dir = tmp_path / "out"
|
|
||||||
out_dir.mkdir()
|
|
||||||
out_path = out_dir / "page1.png"
|
|
||||||
|
|
||||||
rasterize_pdf_page_to_png(pdf_path, out_path, dpi=72)
|
|
||||||
|
|
||||||
assert list(out_dir.iterdir()) == [out_path]
|
|
||||||
with Image.open(out_path) as im:
|
|
||||||
assert im.format == "PNG"
|
|
||||||
assert im.size == expected_size
|
|
||||||
|
|
||||||
def test_failure_raises_parse_error(self, tmp_path: Path) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A file that is not a PDF
|
|
||||||
WHEN:
|
|
||||||
- Rasterization is attempted
|
|
||||||
THEN:
|
|
||||||
- A ParseError is raised
|
|
||||||
"""
|
|
||||||
bad = tmp_path / "bad.pdf"
|
|
||||||
bad.write_bytes(b"not a pdf")
|
|
||||||
|
|
||||||
with pytest.raises(ParseError):
|
|
||||||
rasterize_pdf_page_to_png(bad, tmp_path / "page1.png", dpi=72)
|
|
||||||
|
|
||||||
|
|
||||||
class TestEncodeThumbnailWebp:
|
|
||||||
@pytest.mark.parametrize(
|
|
||||||
("mode", "color"),
|
|
||||||
[
|
|
||||||
pytest.param("RGBA", (0, 0, 0, 0), id="rgba"),
|
|
||||||
pytest.param("LA", (0, 0), id="la"),
|
|
||||||
],
|
|
||||||
)
|
|
||||||
def test_alpha_flattened_onto_white(
|
|
||||||
self,
|
|
||||||
tmp_path: Path,
|
|
||||||
mode: str,
|
|
||||||
color: tuple[int, ...],
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A fully transparent PNG with an alpha channel
|
|
||||||
WHEN:
|
|
||||||
- It is encoded as a thumbnail
|
|
||||||
THEN:
|
|
||||||
- The WebP output is RGB with the transparency flattened to white
|
|
||||||
"""
|
|
||||||
png_path = tmp_path / "in.png"
|
|
||||||
Image.new(mode, (20, 10), color).save(png_path)
|
|
||||||
out_path = tmp_path / "out.webp"
|
|
||||||
|
|
||||||
encode_thumbnail_webp(png_path, out_path)
|
|
||||||
|
|
||||||
with Image.open(out_path) as im:
|
|
||||||
assert im.format == "WEBP"
|
|
||||||
assert im.mode == "RGB"
|
|
||||||
assert im.size == (20, 10)
|
|
||||||
red, green, blue = im.getpixel((10, 5))
|
|
||||||
assert min(red, green, blue) >= 250
|
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
|
||||||
("in_size", "supersample", "expected_size"),
|
|
||||||
[
|
|
||||||
pytest.param((1000, 2000), 1, (500, 1000), id="too-wide-shrunk"),
|
|
||||||
pytest.param((100, 10000), 1, (50, 5000), id="too-tall-shrunk"),
|
|
||||||
pytest.param((100, 200), 1, (100, 200), id="small-not-enlarged"),
|
|
||||||
pytest.param((1000, 1400), 2, (500, 700), id="2x-halved"),
|
|
||||||
pytest.param((1001, 1401), 2, (500, 700), id="2x-odd-rounded"),
|
|
||||||
pytest.param((600, 800), 2, (300, 400), id="2x-small-not-enlarged"),
|
|
||||||
pytest.param((900, 1200), 2, (450, 600), id="2x-below-clamp"),
|
|
||||||
],
|
|
||||||
)
|
|
||||||
def test_size_clamped_and_downsampled(
|
|
||||||
self,
|
|
||||||
tmp_path: Path,
|
|
||||||
in_size: tuple[int, int],
|
|
||||||
supersample: int,
|
|
||||||
expected_size: tuple[int, int],
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A rendered image and its supersample factor
|
|
||||||
WHEN:
|
|
||||||
- It is encoded as a thumbnail with that factor
|
|
||||||
THEN:
|
|
||||||
- It is downsampled, fit within 500x5000 and never enlarged
|
|
||||||
"""
|
|
||||||
png_path = tmp_path / "in.png"
|
|
||||||
Image.new("RGB", in_size, (255, 255, 255)).save(png_path)
|
|
||||||
out_path = tmp_path / "out.webp"
|
|
||||||
|
|
||||||
encode_thumbnail_webp(png_path, out_path, supersample=supersample)
|
|
||||||
|
|
||||||
with Image.open(out_path) as im:
|
|
||||||
assert im.size == expected_size
|
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
|
||||||
"error",
|
|
||||||
[
|
|
||||||
pytest.param(OSError("broken image"), id="os-error"),
|
|
||||||
pytest.param(Image.DecompressionBombError("too large"), id="bomb"),
|
|
||||||
],
|
|
||||||
)
|
|
||||||
def test_decode_failure_raises_parse_error(
|
|
||||||
self,
|
|
||||||
tmp_path: Path,
|
|
||||||
mocker: MockerFixture,
|
|
||||||
error: Exception,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- Opening the rendered image fails
|
|
||||||
WHEN:
|
|
||||||
- It is encoded as a thumbnail
|
|
||||||
THEN:
|
|
||||||
- A ParseError is raised
|
|
||||||
"""
|
|
||||||
png_path = tmp_path / "in.png"
|
|
||||||
Image.new("RGB", (10, 10)).save(png_path)
|
|
||||||
mocker.patch("PIL.Image.open", side_effect=error)
|
|
||||||
|
|
||||||
with pytest.raises(ParseError):
|
|
||||||
encode_thumbnail_webp(png_path, tmp_path / "out.webp")
|
|
||||||
@@ -18,6 +18,7 @@ from documents.models import Correspondent
|
|||||||
from documents.models import DocumentType
|
from documents.models import DocumentType
|
||||||
from documents.models import StoragePath
|
from documents.models import StoragePath
|
||||||
from documents.models import Tag
|
from documents.models import Tag
|
||||||
|
from documents.permissions import has_perms_owner_aware
|
||||||
from documents.permissions import permitted_document_ids
|
from documents.permissions import permitted_document_ids
|
||||||
from documents.permissions import permitted_object_ids
|
from documents.permissions import permitted_object_ids
|
||||||
from documents.permissions import restrict_queryset_to_visible
|
from documents.permissions import restrict_queryset_to_visible
|
||||||
@@ -32,6 +33,8 @@ from paperless_testing.permissions import grant_global
|
|||||||
from paperless_testing.permissions import grant_object
|
from paperless_testing.permissions import grant_object
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
|
from django.contrib.auth.models import User
|
||||||
|
|
||||||
from paperless_testing.dirs import PaperlessDirs
|
from paperless_testing.dirs import PaperlessDirs
|
||||||
|
|
||||||
|
|
||||||
@@ -178,6 +181,382 @@ class TestPermittedDocumentIdsIncludeDeleted:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.django_db
|
||||||
|
class TestPermittedDocumentIdsVersions:
|
||||||
|
"""
|
||||||
|
A version is authorized by its root document: the version's own owner and
|
||||||
|
grants never matter.
|
||||||
|
"""
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("root_owner", "version_owner", "expected_visible"),
|
||||||
|
[
|
||||||
|
pytest.param(
|
||||||
|
"other",
|
||||||
|
"nobody",
|
||||||
|
False,
|
||||||
|
id="unowned-version-of-private-root",
|
||||||
|
),
|
||||||
|
pytest.param("other", "user", False, id="own-version-of-private-root"),
|
||||||
|
pytest.param("user", "other", True, id="foreign-version-of-own-root"),
|
||||||
|
pytest.param("user", "nobody", True, id="unowned-version-of-own-root"),
|
||||||
|
pytest.param("nobody", "other", True, id="private-version-of-unowned-root"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_version_follows_root_owner(
|
||||||
|
self,
|
||||||
|
root_owner: str,
|
||||||
|
version_owner: str,
|
||||||
|
*,
|
||||||
|
expected_visible: bool,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document and a version with differing owners
|
||||||
|
WHEN:
|
||||||
|
- The permitted document ids are resolved for the user
|
||||||
|
THEN:
|
||||||
|
- The version is visible exactly when its root is
|
||||||
|
"""
|
||||||
|
user = UserFactory()
|
||||||
|
owners = {"user": user, "other": UserFactory(), "nobody": None}
|
||||||
|
root = DocumentFactory(owner=owners[root_owner])
|
||||||
|
version = DocumentFactory(root_document=root, owner=owners[version_owner])
|
||||||
|
|
||||||
|
visible = set(permitted_document_ids(user))
|
||||||
|
|
||||||
|
assert (version.pk in visible) is expected_visible
|
||||||
|
assert (root.pk in visible) is expected_visible
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def grantee(user: User, kind: str) -> User | Group:
|
||||||
|
"""The user itself, or a new group the user belongs to."""
|
||||||
|
if kind == "user":
|
||||||
|
return user
|
||||||
|
group = Group.objects.create(name="shared")
|
||||||
|
user.groups.add(group)
|
||||||
|
return group
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"grantee_kind",
|
||||||
|
[pytest.param("user", id="user"), pytest.param("group", id="group")],
|
||||||
|
)
|
||||||
|
def test_grant_on_root_applies_to_version(self, grantee_kind: str) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A private root document shared with a user or one of their groups
|
||||||
|
- A version of it owned by someone else
|
||||||
|
WHEN:
|
||||||
|
- The permitted document ids are resolved for the user
|
||||||
|
THEN:
|
||||||
|
- Both the root and the version are visible
|
||||||
|
- A user without the grant sees neither
|
||||||
|
"""
|
||||||
|
user = UserFactory()
|
||||||
|
stranger = UserFactory()
|
||||||
|
root = DocumentFactory(owner=UserFactory())
|
||||||
|
version = DocumentFactory(root_document=root, owner=UserFactory())
|
||||||
|
grant_object(self.grantee(user, grantee_kind), root, "view_document")
|
||||||
|
|
||||||
|
assert_visible_document_ids(
|
||||||
|
permitted_document_ids(user),
|
||||||
|
expected_visible=[root.pk, version.pk],
|
||||||
|
expected_hidden=[],
|
||||||
|
)
|
||||||
|
assert_visible_document_ids(
|
||||||
|
permitted_document_ids(stranger),
|
||||||
|
expected_visible=[],
|
||||||
|
expected_hidden=[root.pk, version.pk],
|
||||||
|
)
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"grantee_kind",
|
||||||
|
[pytest.param("user", id="user"), pytest.param("group", id="group")],
|
||||||
|
)
|
||||||
|
def test_grant_on_version_is_ignored(self, grantee_kind: str) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A private root document
|
||||||
|
- A version with an explicit grant for the user or one of their groups
|
||||||
|
WHEN:
|
||||||
|
- The permitted document ids are resolved for the user
|
||||||
|
THEN:
|
||||||
|
- Neither the root nor the version is visible
|
||||||
|
"""
|
||||||
|
user = UserFactory()
|
||||||
|
root = DocumentFactory(owner=UserFactory())
|
||||||
|
version = DocumentFactory(root_document=root, owner=UserFactory())
|
||||||
|
grant_object(self.grantee(user, grantee_kind), version, "view_document")
|
||||||
|
|
||||||
|
assert_visible_document_ids(
|
||||||
|
permitted_document_ids(user),
|
||||||
|
expected_visible=[],
|
||||||
|
expected_hidden=[root.pk, version.pk],
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_grant_on_one_root_does_not_reach_another_roots_version(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- Two private roots, each with a version
|
||||||
|
- The user may view only the first root
|
||||||
|
WHEN:
|
||||||
|
- The permitted document ids are resolved for the user
|
||||||
|
THEN:
|
||||||
|
- Only the first root and its version are visible
|
||||||
|
"""
|
||||||
|
user = UserFactory()
|
||||||
|
first = DocumentFactory(owner=UserFactory())
|
||||||
|
first_version = DocumentFactory(root_document=first, owner=UserFactory())
|
||||||
|
second = DocumentFactory(owner=UserFactory())
|
||||||
|
second_version = DocumentFactory(root_document=second, owner=user)
|
||||||
|
grant_object(user, first, "view_document")
|
||||||
|
|
||||||
|
assert_visible_document_ids(
|
||||||
|
permitted_document_ids(user),
|
||||||
|
expected_visible=[first.pk, first_version.pk],
|
||||||
|
expected_hidden=[second.pk, second_version.pk],
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_user_in_several_groups(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A user in two groups
|
||||||
|
- Two private roots shared with one group each, and a third shared with nobody
|
||||||
|
- A version of each root
|
||||||
|
WHEN:
|
||||||
|
- The permitted document ids are resolved for the user
|
||||||
|
THEN:
|
||||||
|
- The two shared roots and their versions are visible
|
||||||
|
- The third root and its version are not
|
||||||
|
"""
|
||||||
|
user = UserFactory()
|
||||||
|
groups = [Group.objects.create(name=f"group{i}") for i in range(2)]
|
||||||
|
user.groups.add(*groups)
|
||||||
|
shared = [DocumentFactory(owner=UserFactory()) for _ in groups]
|
||||||
|
for root, group in zip(shared, groups, strict=True):
|
||||||
|
grant_object(group, root, "view_document")
|
||||||
|
unshared = DocumentFactory(owner=UserFactory())
|
||||||
|
shared_versions = [
|
||||||
|
DocumentFactory(root_document=root, owner=UserFactory()) for root in shared
|
||||||
|
]
|
||||||
|
unshared_version = DocumentFactory(root_document=unshared, owner=None)
|
||||||
|
|
||||||
|
assert_visible_document_ids(
|
||||||
|
permitted_document_ids(user),
|
||||||
|
expected_visible=[
|
||||||
|
*(root.pk for root in shared),
|
||||||
|
*(version.pk for version in shared_versions),
|
||||||
|
],
|
||||||
|
expected_hidden=[unshared.pk, unshared_version.pk],
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_permission_is_resolved_through_the_root(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A private root document where the user may view and change
|
||||||
|
WHEN:
|
||||||
|
- The permitted ids are resolved for view, change and delete
|
||||||
|
THEN:
|
||||||
|
- The version is visible for view and change only
|
||||||
|
"""
|
||||||
|
user = UserFactory()
|
||||||
|
root = DocumentFactory(owner=UserFactory())
|
||||||
|
version = DocumentFactory(root_document=root, owner=UserFactory())
|
||||||
|
grant_object(user, root, "view_document", "change_document")
|
||||||
|
|
||||||
|
assert version.pk in set(permitted_document_ids(user))
|
||||||
|
assert version.pk in set(permitted_document_ids(user, perm="change_document"))
|
||||||
|
assert version.pk in set(
|
||||||
|
permitted_document_ids(user, perm="documents.change_document"),
|
||||||
|
)
|
||||||
|
assert version.pk not in set(
|
||||||
|
permitted_document_ids(user, perm="delete_document"),
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_anonymous_sees_versions_of_unowned_roots_only(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A version owned by nobody under a private root
|
||||||
|
- A version owned by someone under an unowned root
|
||||||
|
WHEN:
|
||||||
|
- The permitted document ids are resolved for an anonymous user
|
||||||
|
THEN:
|
||||||
|
- Only the version of the unowned root is visible
|
||||||
|
"""
|
||||||
|
private_root = DocumentFactory(owner=UserFactory())
|
||||||
|
private_version = DocumentFactory(root_document=private_root, owner=None)
|
||||||
|
open_root = DocumentFactory(owner=None)
|
||||||
|
open_version = DocumentFactory(root_document=open_root, owner=UserFactory())
|
||||||
|
|
||||||
|
assert_visible_document_ids(
|
||||||
|
permitted_document_ids(AnonymousUser()),
|
||||||
|
expected_visible=[open_root.pk, open_version.pk],
|
||||||
|
expected_hidden=[private_root.pk, private_version.pk],
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_deleted_versions_follow_their_deleted_root(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A soft-deleted root document and its version, which deleting the
|
||||||
|
root soft-deletes too; the version is owned by someone else
|
||||||
|
WHEN:
|
||||||
|
- The permitted document ids are resolved with and without deleted
|
||||||
|
documents
|
||||||
|
THEN:
|
||||||
|
- Nothing is visible by default
|
||||||
|
- With deleted documents included, the version is visible to the
|
||||||
|
root's owner and not to the version's own owner
|
||||||
|
"""
|
||||||
|
owner = UserFactory()
|
||||||
|
version_owner = UserFactory()
|
||||||
|
root = DocumentFactory(owner=owner)
|
||||||
|
version = DocumentFactory(root_document=root, owner=version_owner)
|
||||||
|
root.delete()
|
||||||
|
|
||||||
|
assert not {root.pk, version.pk} & set(permitted_document_ids(owner))
|
||||||
|
assert_visible_document_ids(
|
||||||
|
permitted_document_ids(owner, include_deleted=True),
|
||||||
|
expected_visible=[root.pk, version.pk],
|
||||||
|
expected_hidden=[],
|
||||||
|
)
|
||||||
|
assert_visible_document_ids(
|
||||||
|
permitted_document_ids(version_owner, include_deleted=True),
|
||||||
|
expected_visible=[],
|
||||||
|
expected_hidden=[root.pk, version.pk],
|
||||||
|
)
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"is_superuser",
|
||||||
|
[
|
||||||
|
pytest.param(False, id="regular-user"),
|
||||||
|
pytest.param(True, id="superuser"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_inactive_user_sees_no_versions(self, *, is_superuser: bool) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- An inactive user, possibly a superuser, who owns a root and its version
|
||||||
|
WHEN:
|
||||||
|
- The permitted document ids are resolved for them
|
||||||
|
THEN:
|
||||||
|
- Nothing is visible
|
||||||
|
"""
|
||||||
|
user = UserFactory(is_active=False, is_superuser=is_superuser)
|
||||||
|
root = DocumentFactory(owner=user)
|
||||||
|
version = DocumentFactory(root_document=root, owner=user)
|
||||||
|
|
||||||
|
assert_visible_document_ids(
|
||||||
|
permitted_document_ids(user),
|
||||||
|
expected_visible=[],
|
||||||
|
expected_hidden=[root.pk, version.pk],
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_superuser_sees_all_versions(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A private root owned by someone else, with a version
|
||||||
|
WHEN:
|
||||||
|
- The permitted document ids are resolved for a superuser
|
||||||
|
THEN:
|
||||||
|
- Both the root and the version are visible
|
||||||
|
"""
|
||||||
|
superuser = UserFactory(superuser=True)
|
||||||
|
root = DocumentFactory(owner=UserFactory())
|
||||||
|
version = DocumentFactory(root_document=root, owner=UserFactory())
|
||||||
|
|
||||||
|
assert_visible_document_ids(
|
||||||
|
permitted_document_ids(superuser),
|
||||||
|
expected_visible=[root.pk, version.pk],
|
||||||
|
expected_hidden=[],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.django_db
|
||||||
|
class TestHasPermsOwnerAwareVersions:
|
||||||
|
"""
|
||||||
|
The single-object check agrees with permitted_document_ids: a version is
|
||||||
|
authorized by its root document.
|
||||||
|
"""
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("root_owner", "version_owner", "expected"),
|
||||||
|
[
|
||||||
|
pytest.param(
|
||||||
|
"other",
|
||||||
|
"nobody",
|
||||||
|
False,
|
||||||
|
id="unowned-version-of-private-root",
|
||||||
|
),
|
||||||
|
pytest.param("other", "user", False, id="own-version-of-private-root"),
|
||||||
|
pytest.param("user", "other", True, id="foreign-version-of-own-root"),
|
||||||
|
pytest.param("nobody", "other", True, id="private-version-of-unowned-root"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_version_follows_root_owner(
|
||||||
|
self,
|
||||||
|
root_owner: str,
|
||||||
|
version_owner: str,
|
||||||
|
*,
|
||||||
|
expected: bool,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document and a version with differing owners
|
||||||
|
WHEN:
|
||||||
|
- The single-object check runs for the version
|
||||||
|
THEN:
|
||||||
|
- The version is allowed exactly when its root is
|
||||||
|
"""
|
||||||
|
user = UserFactory()
|
||||||
|
owners = {"user": user, "other": UserFactory(), "nobody": None}
|
||||||
|
root = DocumentFactory(owner=owners[root_owner])
|
||||||
|
version = DocumentFactory(root_document=root, owner=owners[version_owner])
|
||||||
|
|
||||||
|
assert has_perms_owner_aware(user, "view_document", version) is expected
|
||||||
|
assert has_perms_owner_aware(user, "view_document", root) is expected
|
||||||
|
|
||||||
|
def test_grant_on_root_applies_and_grant_on_version_does_not(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A private root with a version, and a second private root with a version
|
||||||
|
- The user may change only the first root, and was granted the second
|
||||||
|
root's version directly
|
||||||
|
WHEN:
|
||||||
|
- The single-object check runs for each version
|
||||||
|
THEN:
|
||||||
|
- Only the first root's version is allowed
|
||||||
|
"""
|
||||||
|
user = UserFactory()
|
||||||
|
shared_root = DocumentFactory(owner=UserFactory())
|
||||||
|
shared_version = DocumentFactory(root_document=shared_root, owner=UserFactory())
|
||||||
|
private_root = DocumentFactory(owner=UserFactory())
|
||||||
|
private_version = DocumentFactory(
|
||||||
|
root_document=private_root,
|
||||||
|
owner=UserFactory(),
|
||||||
|
)
|
||||||
|
grant_object(user, shared_root, "change_document")
|
||||||
|
grant_object(user, private_version, "change_document")
|
||||||
|
|
||||||
|
assert has_perms_owner_aware(user, "change_document", shared_version)
|
||||||
|
assert not has_perms_owner_aware(user, "change_document", private_version)
|
||||||
|
|
||||||
|
def test_other_models_use_their_own_owner(self) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A tag owned by someone else, and one owned by the user
|
||||||
|
WHEN:
|
||||||
|
- The single-object check runs for each
|
||||||
|
THEN:
|
||||||
|
- Only the user's own tag is allowed without a grant
|
||||||
|
"""
|
||||||
|
user = UserFactory()
|
||||||
|
mine = TagFactory(owner=user)
|
||||||
|
theirs = TagFactory(owner=UserFactory())
|
||||||
|
|
||||||
|
assert has_perms_owner_aware(user, "view_tag", mine)
|
||||||
|
assert not has_perms_owner_aware(user, "view_tag", theirs)
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
@pytest.mark.django_db
|
||||||
class TestAiChatAllDocumentsPermissionBoundary:
|
class TestAiChatAllDocumentsPermissionBoundary:
|
||||||
"""
|
"""
|
||||||
|
|||||||
@@ -90,10 +90,13 @@ class ShareLinkBundleAPITests(DirectoriesMixin, APITestCase):
|
|||||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||||
self.assertIn("document_ids", response.data)
|
self.assertIn("document_ids", response.data)
|
||||||
|
|
||||||
@mock.patch("documents.views.permitted_document_ids", return_value=set())
|
def test_create_bundle_rejects_insufficient_permissions(self) -> None:
|
||||||
def test_create_bundle_rejects_insufficient_permissions(self, perms_mock) -> None:
|
requester = UserFactory(username="bundle_creator")
|
||||||
|
grant_global(requester, "add_sharelinkbundle", "view_document")
|
||||||
|
self.client.force_authenticate(requester)
|
||||||
|
document = DocumentFactory(owner=UserFactory(username="document_owner"))
|
||||||
payload = {
|
payload = {
|
||||||
"document_ids": [self.document.pk],
|
"document_ids": [self.document.pk, document.pk],
|
||||||
"file_version": ShareLink.FileVersion.ARCHIVE,
|
"file_version": ShareLink.FileVersion.ARCHIVE,
|
||||||
"expiration_days": 7,
|
"expiration_days": 7,
|
||||||
}
|
}
|
||||||
@@ -101,8 +104,8 @@ class ShareLinkBundleAPITests(DirectoriesMixin, APITestCase):
|
|||||||
response = self.client.post(self.ENDPOINT, payload, format="json")
|
response = self.client.post(self.ENDPOINT, payload, format="json")
|
||||||
|
|
||||||
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST)
|
||||||
self.assertIn("document_ids", response.data)
|
self.assertIn(str(document.pk), str(response.data["document_ids"]))
|
||||||
perms_mock.assert_called()
|
self.assertFalse(ShareLinkBundle.objects.exists())
|
||||||
|
|
||||||
@mock.patch("documents.views.build_share_link_bundle.apply_async")
|
@mock.patch("documents.views.build_share_link_bundle.apply_async")
|
||||||
def test_rebuild_bundle_resets_state(self, delay_mock) -> None:
|
def test_rebuild_bundle_resets_state(self, delay_mock) -> None:
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ from documents.sanity_checker import SanityCheckMessages
|
|||||||
from documents.tests.helpers import dummy_preprocess
|
from documents.tests.helpers import dummy_preprocess
|
||||||
from paperless_testing.assertions import FileSystemAssertsMixin
|
from paperless_testing.assertions import FileSystemAssertsMixin
|
||||||
from paperless_testing.dirs import DirectoriesMixin
|
from paperless_testing.dirs import DirectoriesMixin
|
||||||
|
from paperless_testing.factories import DocumentFactory
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
@pytest.mark.django_db
|
||||||
@@ -287,6 +288,82 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
|
|||||||
tasks.update_document_content_maybe_archive_file(doc.pk)
|
tasks.update_document_content_maybe_archive_file(doc.pk)
|
||||||
self.assertNotEqual(Document.objects.get(pk=doc.pk).content, "test")
|
self.assertNotEqual(Document.objects.get(pk=doc.pk).content, "test")
|
||||||
|
|
||||||
|
def _create_root_with_version(self) -> tuple[Document, Document]:
|
||||||
|
sample1 = self.dirs.scratch_dir / "sample.pdf"
|
||||||
|
shutil.copy(
|
||||||
|
Path(__file__).parent
|
||||||
|
/ "samples"
|
||||||
|
/ "documents"
|
||||||
|
/ "originals"
|
||||||
|
/ "0000001.pdf",
|
||||||
|
sample1,
|
||||||
|
)
|
||||||
|
root = DocumentFactory(content="root content", mime_type="application/pdf")
|
||||||
|
version = DocumentFactory(
|
||||||
|
content="my document",
|
||||||
|
filename=sample1,
|
||||||
|
mime_type="application/pdf",
|
||||||
|
root_document=root,
|
||||||
|
version_index=1,
|
||||||
|
)
|
||||||
|
return root, version
|
||||||
|
|
||||||
|
@mock.patch("documents.tasks.clear_document_caches")
|
||||||
|
@mock.patch("documents.search.get_backend")
|
||||||
|
def test_update_content_version_indexes_root(
|
||||||
|
self,
|
||||||
|
mock_get_backend: mock.Mock,
|
||||||
|
mock_clear_caches: mock.Mock,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with a version
|
||||||
|
WHEN:
|
||||||
|
- Update content task is called for the version
|
||||||
|
THEN:
|
||||||
|
- The version's content is updated
|
||||||
|
- The root document is indexed rather than the version
|
||||||
|
- Caches are cleared for both
|
||||||
|
"""
|
||||||
|
root, version = self._create_root_with_version()
|
||||||
|
|
||||||
|
tasks.update_document_content_maybe_archive_file(version.pk)
|
||||||
|
|
||||||
|
self.assertNotEqual(
|
||||||
|
Document.objects.get(pk=version.pk).content,
|
||||||
|
"my document",
|
||||||
|
)
|
||||||
|
self.assertEqual(Document.objects.get(pk=root.pk).content, "root content")
|
||||||
|
indexed = mock_get_backend.return_value.add_or_update.call_args.args[0]
|
||||||
|
self.assertEqual(indexed.pk, root.pk)
|
||||||
|
mock_clear_caches.assert_has_calls(
|
||||||
|
[mock.call(version.pk), mock.call(root.pk)],
|
||||||
|
)
|
||||||
|
|
||||||
|
@override_settings(AI_ENABLED=True, LLM_EMBEDDING_BACKEND="huggingface")
|
||||||
|
@mock.patch("documents.tasks.llm_index_add_or_update_document")
|
||||||
|
@mock.patch("documents.search.get_backend")
|
||||||
|
def test_update_content_version_updates_llm_index_for_root(
|
||||||
|
self,
|
||||||
|
mock_get_backend: mock.Mock,
|
||||||
|
mock_llm_index: mock.Mock,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A root document with a version
|
||||||
|
- The LLM index is enabled
|
||||||
|
WHEN:
|
||||||
|
- Update content task is called for the version
|
||||||
|
THEN:
|
||||||
|
- The LLM index is updated for the root document, not the version
|
||||||
|
"""
|
||||||
|
root, version = self._create_root_with_version()
|
||||||
|
|
||||||
|
tasks.update_document_content_maybe_archive_file(version.pk)
|
||||||
|
|
||||||
|
mock_llm_index.assert_called_once()
|
||||||
|
self.assertEqual(mock_llm_index.call_args.args[0].pk, root.pk)
|
||||||
|
|
||||||
|
|
||||||
class TestUpdateContentRemoteOCR(DirectoriesMixin, TestCase):
|
class TestUpdateContentRemoteOCR(DirectoriesMixin, TestCase):
|
||||||
"""
|
"""
|
||||||
|
|||||||
+78
-47
@@ -1550,7 +1550,10 @@ class DocumentViewSet(
|
|||||||
)
|
)
|
||||||
def suggestions(self, request, pk=None):
|
def suggestions(self, request, pk=None):
|
||||||
doc = get_object_or_404(
|
doc = get_object_or_404(
|
||||||
Document.objects.select_related("owner").prefetch_related("versions"),
|
Document.objects.select_related(
|
||||||
|
"owner",
|
||||||
|
"root_document__owner",
|
||||||
|
).prefetch_related("versions"),
|
||||||
pk=pk,
|
pk=pk,
|
||||||
)
|
)
|
||||||
if request.user is not None and not has_perms_owner_aware(
|
if request.user is not None and not has_perms_owner_aware(
|
||||||
@@ -1610,7 +1613,10 @@ class DocumentViewSet(
|
|||||||
@method_decorator(cache_control(no_cache=True))
|
@method_decorator(cache_control(no_cache=True))
|
||||||
def ai_suggestions(self, request, pk=None):
|
def ai_suggestions(self, request, pk=None):
|
||||||
doc = get_object_or_404(
|
doc = get_object_or_404(
|
||||||
Document.objects.select_related("owner").prefetch_related("versions"),
|
Document.objects.select_related(
|
||||||
|
"owner",
|
||||||
|
"root_document__owner",
|
||||||
|
).prefetch_related("versions"),
|
||||||
pk=pk,
|
pk=pk,
|
||||||
)
|
)
|
||||||
if request.user is not None and not has_perms_owner_aware(
|
if request.user is not None and not has_perms_owner_aware(
|
||||||
@@ -1856,9 +1862,14 @@ class DocumentViewSet(
|
|||||||
currentUser = request.user
|
currentUser = request.user
|
||||||
try:
|
try:
|
||||||
doc = (
|
doc = (
|
||||||
Document.objects.select_related("owner")
|
Document.objects.select_related("owner", "root_document__owner")
|
||||||
.prefetch_related("notes")
|
.prefetch_related("notes")
|
||||||
.only("pk", "owner__id")
|
.only(
|
||||||
|
"pk",
|
||||||
|
"owner__id",
|
||||||
|
"root_document__id",
|
||||||
|
"root_document__owner__id",
|
||||||
|
)
|
||||||
.get(pk=pk)
|
.get(pk=pk)
|
||||||
)
|
)
|
||||||
if currentUser is not None and not has_perms_owner_aware(
|
if currentUser is not None and not has_perms_owner_aware(
|
||||||
@@ -1973,7 +1984,9 @@ class DocumentViewSet(
|
|||||||
def share_links(self, request, pk=None):
|
def share_links(self, request, pk=None):
|
||||||
currentUser = request.user
|
currentUser = request.user
|
||||||
try:
|
try:
|
||||||
doc = Document.objects.select_related("owner").get(pk=pk)
|
doc = Document.objects.select_related("owner", "root_document__owner").get(
|
||||||
|
pk=pk,
|
||||||
|
)
|
||||||
if currentUser is not None and not has_perms_owner_aware(
|
if currentUser is not None and not has_perms_owner_aware(
|
||||||
currentUser,
|
currentUser,
|
||||||
"change_document",
|
"change_document",
|
||||||
@@ -2008,10 +2021,11 @@ class DocumentViewSet(
|
|||||||
if not settings.AUDIT_LOG_ENABLED:
|
if not settings.AUDIT_LOG_ENABLED:
|
||||||
return HttpResponseBadRequest("Audit log is disabled")
|
return HttpResponseBadRequest("Audit log is disabled")
|
||||||
try:
|
try:
|
||||||
doc = Document.objects.get(pk=pk)
|
doc = Document.objects.select_related("root_document__owner").get(pk=pk)
|
||||||
|
root_doc = get_root_document(doc)
|
||||||
if not request.user.has_perm("auditlog.view_logentry") or (
|
if not request.user.has_perm("auditlog.view_logentry") or (
|
||||||
doc.owner is not None
|
root_doc.owner is not None
|
||||||
and doc.owner != request.user
|
and root_doc.owner != request.user
|
||||||
and not request.user.is_superuser
|
and not request.user.is_superuser
|
||||||
):
|
):
|
||||||
return HttpResponseForbidden(
|
return HttpResponseForbidden(
|
||||||
@@ -2102,9 +2116,7 @@ class DocumentViewSet(
|
|||||||
documents = Document.objects.filter(pk__in=document_ids)
|
documents = Document.objects.filter(pk__in=document_ids)
|
||||||
if (
|
if (
|
||||||
request.user is not None
|
request.user is not None
|
||||||
and documents.exclude(
|
and documents.exclude(id__in=permitted_document_ids(request.user)).exists()
|
||||||
pk__in=permitted_document_ids(request.user),
|
|
||||||
).exists()
|
|
||||||
):
|
):
|
||||||
return HttpResponseForbidden("Insufficient permissions")
|
return HttpResponseForbidden("Insufficient permissions")
|
||||||
|
|
||||||
@@ -2430,11 +2442,17 @@ class ChatStreamingView(GenericAPIView[Any]):
|
|||||||
|
|
||||||
if doc_id:
|
if doc_id:
|
||||||
try:
|
try:
|
||||||
document = Document.objects.get(id=doc_id)
|
document = Document.objects.select_related(
|
||||||
|
"root_document__owner",
|
||||||
|
).get(id=doc_id)
|
||||||
except Document.DoesNotExist:
|
except Document.DoesNotExist:
|
||||||
return HttpResponseBadRequest("Document not found")
|
return HttpResponseBadRequest("Document not found")
|
||||||
|
|
||||||
if not has_perms_owner_aware(request.user, "view_document", document):
|
if not has_perms_owner_aware(
|
||||||
|
request.user,
|
||||||
|
"view_document",
|
||||||
|
document,
|
||||||
|
):
|
||||||
return HttpResponseForbidden("Insufficient permissions")
|
return HttpResponseForbidden("Insufficient permissions")
|
||||||
|
|
||||||
documents = Document.objects.filter(pk=document.pk)
|
documents = Document.objects.filter(pk=document.pk)
|
||||||
@@ -2990,12 +3008,8 @@ class DocumentOperationPermissionMixin(PassUserMixin, DocumentSelectionMixin):
|
|||||||
user.has_perm(
|
user.has_perm(
|
||||||
"documents.change_document",
|
"documents.change_document",
|
||||||
)
|
)
|
||||||
and not Document.global_objects.filter(
|
and not Document.global_objects.filter(pk__in=documents)
|
||||||
pk__in=[doc.pk for doc in root_docs],
|
.exclude(pk__in=permitted_document_ids(user, perm="change_document"))
|
||||||
)
|
|
||||||
.exclude(
|
|
||||||
pk__in=permitted_document_ids(user, perm="change_document"),
|
|
||||||
)
|
|
||||||
.exists()
|
.exists()
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -3605,10 +3619,11 @@ class SelectionDataView(DocumentSelectionMixin, GenericAPIView[Any]):
|
|||||||
user=request.user,
|
user=request.user,
|
||||||
validated_data=serializer.validated_data,
|
validated_data=serializer.validated_data,
|
||||||
)
|
)
|
||||||
permitted_documents = Document.objects.filter(
|
documents = Document.objects.filter(pk__in=ids)
|
||||||
id__in=permitted_document_ids(request.user),
|
if (
|
||||||
)
|
documents.count() != len(ids)
|
||||||
if permitted_documents.filter(pk__in=ids).count() != len(ids):
|
or documents.exclude(id__in=permitted_document_ids(request.user)).exists()
|
||||||
|
):
|
||||||
return HttpResponseForbidden("Insufficient permissions")
|
return HttpResponseForbidden("Insufficient permissions")
|
||||||
|
|
||||||
correspondents = Correspondent.objects.annotate(
|
correspondents = Correspondent.objects.annotate(
|
||||||
@@ -4104,21 +4119,16 @@ class BulkDownloadView(DocumentSelectionMixin, GenericAPIView[Any]):
|
|||||||
validated_data=serializer.validated_data,
|
validated_data=serializer.validated_data,
|
||||||
)
|
)
|
||||||
documents = Document.objects.filter(pk__in=ids)
|
documents = Document.objects.filter(pk__in=ids)
|
||||||
versioned_documents = []
|
|
||||||
compression = serializer.validated_data.get("compression")
|
compression = serializer.validated_data.get("compression")
|
||||||
content = serializer.validated_data.get("content")
|
content = serializer.validated_data.get("content")
|
||||||
follow_filename_format = serializer.validated_data.get("follow_formatting")
|
follow_filename_format = serializer.validated_data.get("follow_formatting")
|
||||||
|
|
||||||
permitted_ids = set(permitted_document_ids(request.user))
|
if documents.exclude(id__in=permitted_document_ids(request.user)).exists():
|
||||||
for document in documents:
|
return HttpResponseForbidden("Insufficient permissions")
|
||||||
root_doc = get_root_document(document)
|
versioned_documents = [
|
||||||
if root_doc.pk not in permitted_ids:
|
get_latest_version_for_root(get_root_document(document))
|
||||||
return HttpResponseForbidden("Insufficient permissions")
|
for document in documents
|
||||||
versioned_documents.append(
|
]
|
||||||
get_latest_version_for_root(
|
|
||||||
root_doc,
|
|
||||||
),
|
|
||||||
)
|
|
||||||
|
|
||||||
if content == "both":
|
if content == "both":
|
||||||
strategy_class = OriginalAndArchiveStrategy
|
strategy_class = OriginalAndArchiveStrategy
|
||||||
@@ -4790,19 +4800,23 @@ class ShareLinkBundleViewSet(PassUserMixin, ModelViewSet[ShareLinkBundle]):
|
|||||||
},
|
},
|
||||||
)
|
)
|
||||||
|
|
||||||
documents = list(documents_qs)
|
denied_id = (
|
||||||
permitted_ids = set(permitted_document_ids(request.user))
|
documents_qs.exclude(id__in=permitted_document_ids(request.user))
|
||||||
for document in documents:
|
.order_by("pk")
|
||||||
if document.pk not in permitted_ids:
|
.values_list("pk", flat=True)
|
||||||
raise ValidationError(
|
.first()
|
||||||
{
|
)
|
||||||
"document_ids": _(
|
if denied_id is not None:
|
||||||
"Insufficient permissions to share document %(id)s.",
|
raise ValidationError(
|
||||||
)
|
{
|
||||||
% {"id": document.pk},
|
"document_ids": _(
|
||||||
},
|
"Insufficient permissions to share document %(id)s.",
|
||||||
)
|
)
|
||||||
|
% {"id": denied_id},
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
documents = list(documents_qs)
|
||||||
document_map = {document.pk: document for document in documents}
|
document_map = {document.pk: document for document in documents}
|
||||||
ordered_documents = [document_map[doc_id] for doc_id in document_ids]
|
ordered_documents = [document_map[doc_id] for doc_id in document_ids]
|
||||||
|
|
||||||
@@ -5587,6 +5601,23 @@ class TrashView(ListModelMixin, PassUserMixin):
|
|||||||
class _TrashPermittedObjectsFilter(PermittedObjectsFilter):
|
class _TrashPermittedObjectsFilter(PermittedObjectsFilter):
|
||||||
include_granted = False
|
include_granted = False
|
||||||
|
|
||||||
|
def filter_queryset(self, request, queryset, view):
|
||||||
|
if request.user.is_superuser or not request.user.is_active:
|
||||||
|
return super().filter_queryset(request, queryset, view)
|
||||||
|
|
||||||
|
# A version belongs to whoever owns its root
|
||||||
|
def owned_or_unowned(prefix: str) -> Q:
|
||||||
|
return Q(**{f"{prefix}owner": request.user}) | Q(
|
||||||
|
**{f"{prefix}owner__isnull": True},
|
||||||
|
)
|
||||||
|
|
||||||
|
return queryset.filter(
|
||||||
|
(Q(root_document__isnull=True) & owned_or_unowned(""))
|
||||||
|
| (
|
||||||
|
Q(root_document__isnull=False) & owned_or_unowned("root_document__")
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
filter_backends = (_TrashPermittedObjectsFilter,)
|
filter_backends = (_TrashPermittedObjectsFilter,)
|
||||||
pagination_class = StandardPagination
|
pagination_class = StandardPagination
|
||||||
|
|
||||||
@@ -5617,7 +5648,7 @@ class TrashView(ListModelMixin, PassUserMixin):
|
|||||||
else self.filter_queryset(self.get_queryset()).all()
|
else self.filter_queryset(self.get_queryset()).all()
|
||||||
)
|
)
|
||||||
if docs.exclude(
|
if docs.exclude(
|
||||||
pk__in=permitted_document_ids(
|
id__in=permitted_document_ids(
|
||||||
request.user,
|
request.user,
|
||||||
perm="delete_document",
|
perm="delete_document",
|
||||||
include_deleted=True,
|
include_deleted=True,
|
||||||
|
|||||||
@@ -2,7 +2,7 @@ msgid ""
|
|||||||
msgstr ""
|
msgstr ""
|
||||||
"Project-Id-Version: paperless-ngx\n"
|
"Project-Id-Version: paperless-ngx\n"
|
||||||
"Report-Msgid-Bugs-To: \n"
|
"Report-Msgid-Bugs-To: \n"
|
||||||
"POT-Creation-Date: 2026-10-06 15:12+0000\n"
|
"POT-Creation-Date: 2026-10-08 18:39+0000\n"
|
||||||
"PO-Revision-Date: 2022-02-17 04:17\n"
|
"PO-Revision-Date: 2022-02-17 04:17\n"
|
||||||
"Last-Translator: \n"
|
"Last-Translator: \n"
|
||||||
"Language-Team: English\n"
|
"Language-Team: English\n"
|
||||||
@@ -1652,49 +1652,49 @@ msgstr ""
|
|||||||
msgid "workflow runs"
|
msgid "workflow runs"
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/serialisers.py:515 documents/serialisers.py:872
|
#: documents/serialisers.py:516 documents/serialisers.py:873
|
||||||
#: documents/serialisers.py:2902 documents/views.py:343 documents/views.py:2732
|
#: documents/serialisers.py:2903 documents/views.py:344 documents/views.py:2751
|
||||||
#: paperless_mail/serialisers.py:156
|
#: paperless_mail/serialisers.py:156
|
||||||
msgid "Insufficient permissions."
|
msgid "Insufficient permissions."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/serialisers.py:708
|
#: documents/serialisers.py:709
|
||||||
msgid "Invalid color."
|
msgid "Invalid color."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/serialisers.py:2369
|
#: documents/serialisers.py:2370
|
||||||
#, python-format
|
#, python-format
|
||||||
msgid "File type %(type)s not supported"
|
msgid "File type %(type)s not supported"
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/serialisers.py:2413
|
#: documents/serialisers.py:2414
|
||||||
#, python-format
|
#, python-format
|
||||||
msgid "Custom field id must be an integer: %(id)s"
|
msgid "Custom field id must be an integer: %(id)s"
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/serialisers.py:2420
|
#: documents/serialisers.py:2421
|
||||||
#, python-format
|
#, python-format
|
||||||
msgid "Custom field with id %(id)s does not exist"
|
msgid "Custom field with id %(id)s does not exist"
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/serialisers.py:2437 documents/serialisers.py:2447
|
#: documents/serialisers.py:2438 documents/serialisers.py:2448
|
||||||
msgid ""
|
msgid ""
|
||||||
"Custom fields must be a list of integers or an object mapping ids to values."
|
"Custom fields must be a list of integers or an object mapping ids to values."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/serialisers.py:2442
|
#: documents/serialisers.py:2443
|
||||||
msgid "Some custom fields don't exist or were specified twice."
|
msgid "Some custom fields don't exist or were specified twice."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/serialisers.py:2589
|
#: documents/serialisers.py:2590
|
||||||
msgid "Invalid variable detected."
|
msgid "Invalid variable detected."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/serialisers.py:2958
|
#: documents/serialisers.py:2959
|
||||||
msgid "Duplicate document identifiers are not allowed."
|
msgid "Duplicate document identifiers are not allowed."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/serialisers.py:2988 documents/views.py:4787
|
#: documents/serialisers.py:2989 documents/views.py:4807
|
||||||
#, python-format
|
#, python-format
|
||||||
msgid "Documents not found: %(ids)s"
|
msgid "Documents not found: %(ids)s"
|
||||||
msgstr ""
|
msgstr ""
|
||||||
@@ -1945,40 +1945,40 @@ msgstr ""
|
|||||||
msgid ", "
|
msgid ", "
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/views.py:336 documents/views.py:2729
|
#: documents/views.py:337 documents/views.py:2748
|
||||||
msgid "Invalid more_like_id"
|
msgid "Invalid more_like_id"
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/views.py:1676
|
#: documents/views.py:1683
|
||||||
msgid "Invalid AI configuration."
|
msgid "Invalid AI configuration."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/views.py:1687
|
#: documents/views.py:1694
|
||||||
msgid "AI backend request timed out."
|
msgid "AI backend request timed out."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/views.py:1699
|
#: documents/views.py:1706
|
||||||
msgid "AI backend rejected the request. Check logs for details."
|
msgid "AI backend rejected the request. Check logs for details."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/views.py:2554 documents/views.py:2870
|
#: documents/views.py:2573 documents/views.py:2889
|
||||||
msgid "Specify only one of text, title_search, query, or more_like_id."
|
msgid "Specify only one of text, title_search, query, or more_like_id."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/views.py:4800
|
#: documents/views.py:4823
|
||||||
#, python-format
|
#, python-format
|
||||||
msgid "Insufficient permissions to share document %(id)s."
|
msgid "Insufficient permissions to share document %(id)s."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/views.py:4846
|
#: documents/views.py:4870
|
||||||
msgid "Bundle is already being processed."
|
msgid "Bundle is already being processed."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/views.py:4910
|
#: documents/views.py:4934
|
||||||
msgid "The share link bundle is still being prepared. Please try again later."
|
msgid "The share link bundle is still being prepared. Please try again later."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
#: documents/views.py:4924
|
#: documents/views.py:4948
|
||||||
msgid "The share link bundle is unavailable."
|
msgid "The share link bundle is unavailable."
|
||||||
msgstr ""
|
msgstr ""
|
||||||
|
|
||||||
|
|||||||
@@ -265,52 +265,6 @@ def get_page_count_for_pdf(
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def get_pdf_first_page_size_points(
|
|
||||||
path: Path,
|
|
||||||
log: logging.Logger | None = None,
|
|
||||||
) -> tuple[float, float] | None:
|
|
||||||
"""Return the first page's (width, height) in PDF points, post-rotation.
|
|
||||||
|
|
||||||
Uses the CropBox (MediaBox if absent), which must match pdftoppm's
|
|
||||||
``-cropbox`` or the computed DPI targets the wrong box.
|
|
||||||
|
|
||||||
Swaps width and height for 90/270 rotation. ``page.rotation`` resolves
|
|
||||||
inherited and negative ``/Rotate`` values, a raw lookup does not.
|
|
||||||
|
|
||||||
Parameters
|
|
||||||
----------
|
|
||||||
path:
|
|
||||||
Absolute path to the PDF file.
|
|
||||||
log:
|
|
||||||
Logger for warnings. Falls back to the module-level logger when omitted.
|
|
||||||
|
|
||||||
Returns
|
|
||||||
-------
|
|
||||||
tuple[float, float] | None
|
|
||||||
``(width_points, height_points)``, or ``None`` if the file cannot be
|
|
||||||
opened, has no pages, or the page box is degenerate.
|
|
||||||
"""
|
|
||||||
import pikepdf
|
|
||||||
|
|
||||||
_log = log or logger
|
|
||||||
try:
|
|
||||||
with pikepdf.Pdf.open(path) as pdf:
|
|
||||||
if len(pdf.pages) == 0:
|
|
||||||
return None
|
|
||||||
page = pdf.pages[0]
|
|
||||||
llx, lly, urx, ury = (float(v) for v in page.cropbox)
|
|
||||||
width = abs(urx - llx)
|
|
||||||
height = abs(ury - lly)
|
|
||||||
if width <= 0 or height <= 0:
|
|
||||||
return None
|
|
||||||
if page.rotation in (90, 270):
|
|
||||||
width, height = height, width
|
|
||||||
return width, height
|
|
||||||
except Exception as e:
|
|
||||||
_log.warning("Could not determine PDF page size for %s: %s", path, e)
|
|
||||||
return None
|
|
||||||
|
|
||||||
|
|
||||||
def extract_pdf_metadata(
|
def extract_pdf_metadata(
|
||||||
document_path: Path,
|
document_path: Path,
|
||||||
log: logging.Logger | None = None,
|
log: logging.Logger | None = None,
|
||||||
|
|||||||
@@ -986,6 +986,8 @@ GNUPG_HOME = os.getenv("HOME", "/tmp")
|
|||||||
|
|
||||||
# Convert is part of the ImageMagick package
|
# Convert is part of the ImageMagick package
|
||||||
CONVERT_BINARY = os.getenv("PAPERLESS_CONVERT_BINARY", "convert")
|
CONVERT_BINARY = os.getenv("PAPERLESS_CONVERT_BINARY", "convert")
|
||||||
|
CONVERT_TMPDIR = os.getenv("PAPERLESS_CONVERT_TMPDIR")
|
||||||
|
CONVERT_MEMORY_LIMIT = os.getenv("PAPERLESS_CONVERT_MEMORY_LIMIT")
|
||||||
|
|
||||||
GS_BINARY = os.getenv("PAPERLESS_GS_BINARY", "gs")
|
GS_BINARY = os.getenv("PAPERLESS_GS_BINARY", "gs")
|
||||||
|
|
||||||
|
|||||||
@@ -15,10 +15,9 @@ from typing import TYPE_CHECKING
|
|||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
from ocrmypdf import SubprocessOutputError
|
from ocrmypdf import SubprocessOutputError
|
||||||
from PIL import Image
|
|
||||||
|
|
||||||
from documents.parsers import ParseError
|
from documents.parsers import ParseError
|
||||||
from documents.parsers import rasterize_pdf_page_to_png
|
from documents.parsers import run_convert
|
||||||
from paperless.models import ModeChoices
|
from paperless.models import ModeChoices
|
||||||
from paperless.parsers import ParserProtocol
|
from paperless.parsers import ParserProtocol
|
||||||
from paperless.parsers.tesseract import RasterisedDocumentParser
|
from paperless.parsers.tesseract import RasterisedDocumentParser
|
||||||
@@ -281,71 +280,24 @@ class TestGetThumbnail:
|
|||||||
)
|
)
|
||||||
assert thumb.is_file()
|
assert thumb.is_file()
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
def test_thumbnail_fallback_on_convert_error(
|
||||||
("filename", "expected_height"),
|
|
||||||
[
|
|
||||||
pytest.param("simple-digital.pdf", 647, id="portrait-letter"),
|
|
||||||
pytest.param("rotated.pdf", 386, id="landscape"),
|
|
||||||
],
|
|
||||||
)
|
|
||||||
def test_thumbnail_is_correct_format_and_size(
|
|
||||||
self,
|
|
||||||
tesseract_parser: RasterisedDocumentParser,
|
|
||||||
tesseract_samples_dir: Path,
|
|
||||||
filename: str,
|
|
||||||
expected_height: int,
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A portrait or landscape PDF
|
|
||||||
WHEN:
|
|
||||||
- A thumbnail is generated
|
|
||||||
THEN:
|
|
||||||
- A 500px wide WebP keeping the page's aspect ratio
|
|
||||||
"""
|
|
||||||
thumb = tesseract_parser.get_thumbnail(
|
|
||||||
tesseract_samples_dir / filename,
|
|
||||||
"application/pdf",
|
|
||||||
)
|
|
||||||
with Image.open(thumb) as im:
|
|
||||||
assert im.format == "WEBP"
|
|
||||||
assert im.width == 500
|
|
||||||
assert im.height == pytest.approx(expected_height, abs=2)
|
|
||||||
|
|
||||||
def test_thumbnail_fallback_on_pdftoppm_error(
|
|
||||||
self,
|
self,
|
||||||
mocker: MockerFixture,
|
mocker: MockerFixture,
|
||||||
tesseract_parser: RasterisedDocumentParser,
|
tesseract_parser: RasterisedDocumentParser,
|
||||||
tesseract_samples_dir: Path,
|
tesseract_samples_dir: Path,
|
||||||
) -> None:
|
) -> None:
|
||||||
"""
|
def _raise_on_pdf(input_file, output_file, **kwargs) -> None:
|
||||||
GIVEN:
|
if ".pdf" in str(input_file):
|
||||||
- Rasterizing the original PDF fails
|
|
||||||
WHEN:
|
|
||||||
- A thumbnail is generated
|
|
||||||
THEN:
|
|
||||||
- The PDF is repaired with qpdf and a real thumbnail is rendered
|
|
||||||
"""
|
|
||||||
original = tesseract_samples_dir / "simple-digital.pdf"
|
|
||||||
|
|
||||||
def _fail_on_original(in_path: Path, out_path: Path, **kwargs) -> None:
|
|
||||||
if in_path == original:
|
|
||||||
raise ParseError("Does not compute.")
|
raise ParseError("Does not compute.")
|
||||||
rasterize_pdf_page_to_png(in_path, out_path, **kwargs)
|
run_convert(input_file=input_file, output_file=output_file, **kwargs)
|
||||||
|
|
||||||
rasterize = mocker.patch(
|
mocker.patch("documents.parsers.run_convert", side_effect=_raise_on_pdf)
|
||||||
"documents.parsers.rasterize_pdf_page_to_png",
|
|
||||||
side_effect=_fail_on_original,
|
thumb = tesseract_parser.get_thumbnail(
|
||||||
|
tesseract_samples_dir / "simple-digital.pdf",
|
||||||
|
"application/pdf",
|
||||||
)
|
)
|
||||||
|
|
||||||
thumb = tesseract_parser.get_thumbnail(original, "application/pdf")
|
|
||||||
|
|
||||||
assert rasterize.call_count == 2
|
|
||||||
assert thumb.is_file()
|
assert thumb.is_file()
|
||||||
assert thumb.name == "convert_qpdf.webp"
|
|
||||||
with Image.open(thumb) as im:
|
|
||||||
assert im.format == "WEBP"
|
|
||||||
assert im.width == 500
|
|
||||||
|
|
||||||
def test_thumbnail_encrypted_pdf(
|
def test_thumbnail_encrypted_pdf(
|
||||||
self,
|
self,
|
||||||
|
|||||||
@@ -6,10 +6,8 @@ import codecs
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import TYPE_CHECKING
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
import pikepdf
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from paperless.parsers.utils import get_pdf_first_page_size_points
|
|
||||||
from paperless.parsers.utils import is_tagged_pdf
|
from paperless.parsers.utils import is_tagged_pdf
|
||||||
from paperless.parsers.utils import pdf_born_digital_text
|
from paperless.parsers.utils import pdf_born_digital_text
|
||||||
from paperless.parsers.utils import post_process_text
|
from paperless.parsers.utils import post_process_text
|
||||||
@@ -72,149 +70,6 @@ class TestIsTaggedPdf:
|
|||||||
assert is_tagged_pdf(bad) is False
|
assert is_tagged_pdf(bad) is False
|
||||||
|
|
||||||
|
|
||||||
class TestGetPdfFirstPageSizePoints:
|
|
||||||
@staticmethod
|
|
||||||
def _write_pdf(
|
|
||||||
path: Path,
|
|
||||||
*,
|
|
||||||
media_box: tuple[float, float, float, float] = (0, 0, 600, 800),
|
|
||||||
crop_box: tuple[float, float, float, float] | None = None,
|
|
||||||
page_rotate: int | None = None,
|
|
||||||
inherited_rotate: int | None = None,
|
|
||||||
) -> Path:
|
|
||||||
pdf = pikepdf.new()
|
|
||||||
pdf.add_blank_page(page_size=(media_box[2], media_box[3]))
|
|
||||||
page = pdf.pages[0]
|
|
||||||
page.obj.MediaBox = pikepdf.Array(media_box)
|
|
||||||
if crop_box is not None:
|
|
||||||
page.obj.CropBox = pikepdf.Array(crop_box)
|
|
||||||
if page_rotate is not None:
|
|
||||||
page.obj.Rotate = page_rotate
|
|
||||||
if inherited_rotate is not None:
|
|
||||||
pdf.Root.Pages.Rotate = inherited_rotate
|
|
||||||
pdf.save(path)
|
|
||||||
return path
|
|
||||||
|
|
||||||
def test_letter_sample(self) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A US Letter sample PDF with no CropBox and no rotation
|
|
||||||
WHEN:
|
|
||||||
- The first page size is requested
|
|
||||||
THEN:
|
|
||||||
- The MediaBox size in points is returned
|
|
||||||
"""
|
|
||||||
assert get_pdf_first_page_size_points(SAMPLES / "simple-digital.pdf") == (
|
|
||||||
612.0,
|
|
||||||
792.0,
|
|
||||||
)
|
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
|
||||||
("rotate", "expected"),
|
|
||||||
[
|
|
||||||
pytest.param(0, (600.0, 800.0), id="rotate-0"),
|
|
||||||
pytest.param(90, (800.0, 600.0), id="rotate-90"),
|
|
||||||
pytest.param(180, (600.0, 800.0), id="rotate-180"),
|
|
||||||
pytest.param(270, (800.0, 600.0), id="rotate-270"),
|
|
||||||
pytest.param(-90, (800.0, 600.0), id="rotate-negative-90"),
|
|
||||||
],
|
|
||||||
)
|
|
||||||
def test_page_rotation_swaps_dimensions(
|
|
||||||
self,
|
|
||||||
tmp_path: Path,
|
|
||||||
rotate: int,
|
|
||||||
expected: tuple[float, float],
|
|
||||||
) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A portrait PDF page with /Rotate set directly on the page
|
|
||||||
WHEN:
|
|
||||||
- The first page size is requested
|
|
||||||
THEN:
|
|
||||||
- Width and height are swapped for quarter-turn rotations only
|
|
||||||
"""
|
|
||||||
pdf_path = self._write_pdf(tmp_path / "rotated.pdf", page_rotate=rotate)
|
|
||||||
assert get_pdf_first_page_size_points(pdf_path) == expected
|
|
||||||
|
|
||||||
def test_inherited_rotation_swaps_dimensions(self, tmp_path: Path) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A page inheriting /Rotate 90 from the /Pages node
|
|
||||||
WHEN:
|
|
||||||
- The first page size is requested
|
|
||||||
THEN:
|
|
||||||
- Width and height are swapped
|
|
||||||
"""
|
|
||||||
pdf_path = self._write_pdf(tmp_path / "inherited.pdf", inherited_rotate=90)
|
|
||||||
assert get_pdf_first_page_size_points(pdf_path) == (800.0, 600.0)
|
|
||||||
|
|
||||||
def test_crop_box_preferred_over_media_box(self, tmp_path: Path) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A PDF page with a CropBox smaller than its MediaBox
|
|
||||||
WHEN:
|
|
||||||
- The first page size is requested
|
|
||||||
THEN:
|
|
||||||
- The CropBox size is returned
|
|
||||||
"""
|
|
||||||
pdf_path = self._write_pdf(
|
|
||||||
tmp_path / "cropped.pdf",
|
|
||||||
crop_box=(50, 100, 350, 500),
|
|
||||||
)
|
|
||||||
assert get_pdf_first_page_size_points(pdf_path) == (300.0, 400.0)
|
|
||||||
|
|
||||||
def test_degenerate_box_returns_none(self, tmp_path: Path) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A PDF page whose box has zero width
|
|
||||||
WHEN:
|
|
||||||
- The first page size is requested
|
|
||||||
THEN:
|
|
||||||
- None is returned
|
|
||||||
"""
|
|
||||||
pdf_path = self._write_pdf(
|
|
||||||
tmp_path / "degenerate.pdf",
|
|
||||||
media_box=(0, 0, 600, 800),
|
|
||||||
crop_box=(100, 0, 100, 800),
|
|
||||||
)
|
|
||||||
assert get_pdf_first_page_size_points(pdf_path) is None
|
|
||||||
|
|
||||||
def test_nonexistent_path_returns_none(self) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A path that does not exist
|
|
||||||
WHEN:
|
|
||||||
- The first page size is requested
|
|
||||||
THEN:
|
|
||||||
- None is returned and nothing is raised
|
|
||||||
"""
|
|
||||||
assert get_pdf_first_page_size_points(Path("/nonexistent/file.pdf")) is None
|
|
||||||
|
|
||||||
def test_corrupt_pdf_returns_none(self, tmp_path: Path) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A file that is not a PDF
|
|
||||||
WHEN:
|
|
||||||
- The first page size is requested
|
|
||||||
THEN:
|
|
||||||
- None is returned and nothing is raised
|
|
||||||
"""
|
|
||||||
bad = tmp_path / "bad.pdf"
|
|
||||||
bad.write_bytes(b"not a pdf")
|
|
||||||
assert get_pdf_first_page_size_points(bad) is None
|
|
||||||
|
|
||||||
def test_encrypted_pdf_returns_none(self) -> None:
|
|
||||||
"""
|
|
||||||
GIVEN:
|
|
||||||
- A password protected PDF
|
|
||||||
WHEN:
|
|
||||||
- The first page size is requested
|
|
||||||
THEN:
|
|
||||||
- None is returned and nothing is raised
|
|
||||||
"""
|
|
||||||
assert get_pdf_first_page_size_points(SAMPLES / "encrypted.pdf") is None
|
|
||||||
|
|
||||||
|
|
||||||
class TestPostProcessText:
|
class TestPostProcessText:
|
||||||
@pytest.mark.parametrize(
|
@pytest.mark.parametrize(
|
||||||
("source", "expected"),
|
("source", "expected"),
|
||||||
|
|||||||
@@ -4,8 +4,7 @@ from django.conf import settings
|
|||||||
from django.contrib.auth.models import User
|
from django.contrib.auth.models import User
|
||||||
|
|
||||||
from documents.models import Document
|
from documents.models import Document
|
||||||
from documents.permissions import permitted_object_ids
|
from documents.permissions import permitted_document_ids
|
||||||
from documents.permissions import restrict_queryset_to_visible
|
|
||||||
from documents.permissions import user_is_unrestricted
|
from documents.permissions import user_is_unrestricted
|
||||||
from paperless.config import AIConfig
|
from paperless.config import AIConfig
|
||||||
from paperless_ai.base_model import ClassificationSuggestions
|
from paperless_ai.base_model import ClassificationSuggestions
|
||||||
@@ -54,8 +53,8 @@ def _fulltext_similar_documents(
|
|||||||
active superuser - see user_is_unrestricted) is normalized to ``None``
|
active superuser - see user_is_unrestricted) is normalized to ``None``
|
||||||
before calling, since the backend's permission filter has no superuser
|
before calling, since the backend's permission filter has no superuser
|
||||||
short-circuit of its own. Results are re-checked with
|
short-circuit of its own. Results are re-checked with
|
||||||
restrict_queryset_to_visible() since Tantivy's indexed permission fields
|
permitted_document_ids() since Tantivy's indexed permission fields lag the
|
||||||
lag the DB via async reindexing.
|
DB via async reindexing and judge a version by its own owner.
|
||||||
"""
|
"""
|
||||||
from documents.search import get_backend
|
from documents.search import get_backend
|
||||||
|
|
||||||
@@ -69,10 +68,9 @@ def _fulltext_similar_documents(
|
|||||||
)
|
)
|
||||||
if not unrestricted:
|
if not unrestricted:
|
||||||
allowed_ids = set(
|
allowed_ids = set(
|
||||||
restrict_queryset_to_visible(
|
Document.objects.filter(
|
||||||
Document.objects.filter(pk__in=similar_ids),
|
pk__in=similar_ids,
|
||||||
user,
|
id__in=permitted_document_ids(user),
|
||||||
"view_document",
|
|
||||||
).values_list("pk", flat=True),
|
).values_list("pk", flat=True),
|
||||||
)
|
)
|
||||||
similar_ids = [doc_id for doc_id in similar_ids if doc_id in allowed_ids]
|
similar_ids = [doc_id for doc_id in similar_ids if doc_id in allowed_ids]
|
||||||
@@ -200,13 +198,13 @@ def get_taxonomy_context(
|
|||||||
# quadratic scan in the vector store at best, and past ~32,763
|
# quadratic scan in the vector store at best, and past ~32,763
|
||||||
# documents a hard sqlite3.OperationalError (SQLite's
|
# documents a hard sqlite3.OperationalError (SQLite's
|
||||||
# bound-parameter limit) at worst.
|
# bound-parameter limit) at worst.
|
||||||
# permitted_object_ids() has its own superuser shortcut that would
|
# permitted_document_ids() has its own superuser shortcut that would
|
||||||
# return every Document's id anyway, so this changes nothing about
|
# return every Document's id anyway, so this changes nothing about
|
||||||
# which documents are considered -- only how we get there.
|
# which documents are considered -- only how we get there.
|
||||||
visible_document_ids = (
|
visible_document_ids = (
|
||||||
None
|
None
|
||||||
if user_is_unrestricted(user)
|
if user_is_unrestricted(user)
|
||||||
else list(permitted_object_ids(user, Document, "view_document"))
|
else list(permitted_document_ids(user))
|
||||||
)
|
)
|
||||||
nodes = retrieve_similar_nodes(
|
nodes = retrieve_similar_nodes(
|
||||||
document,
|
document,
|
||||||
|
|||||||
@@ -552,7 +552,7 @@ class TestGetTaxonomyContextVisibility:
|
|||||||
return_value=[],
|
return_value=[],
|
||||||
)
|
)
|
||||||
mock_permitted = mocker.patch(
|
mock_permitted = mocker.patch(
|
||||||
"paperless_ai.ai_classifier.permitted_object_ids",
|
"paperless_ai.ai_classifier.permitted_document_ids",
|
||||||
)
|
)
|
||||||
user = UserFactory.create(is_superuser=True)
|
user = UserFactory.create(is_superuser=True)
|
||||||
|
|
||||||
@@ -582,7 +582,7 @@ class TestGetTaxonomyContextVisibility:
|
|||||||
return_value=[],
|
return_value=[],
|
||||||
)
|
)
|
||||||
mock_permitted = mocker.patch(
|
mock_permitted = mocker.patch(
|
||||||
"paperless_ai.ai_classifier.permitted_object_ids",
|
"paperless_ai.ai_classifier.permitted_document_ids",
|
||||||
)
|
)
|
||||||
|
|
||||||
get_taxonomy_context(document, None)
|
get_taxonomy_context(document, None)
|
||||||
@@ -611,16 +611,55 @@ class TestGetTaxonomyContextVisibility:
|
|||||||
return_value=[],
|
return_value=[],
|
||||||
)
|
)
|
||||||
mock_permitted = mocker.patch(
|
mock_permitted = mocker.patch(
|
||||||
"paperless_ai.ai_classifier.permitted_object_ids",
|
"paperless_ai.ai_classifier.permitted_document_ids",
|
||||||
return_value=[1, 2, 3],
|
return_value=[1, 2, 3],
|
||||||
)
|
)
|
||||||
user = UserFactory.create(is_superuser=False)
|
user = UserFactory.create(is_superuser=False)
|
||||||
|
|
||||||
get_taxonomy_context(document, user)
|
get_taxonomy_context(document, user)
|
||||||
|
|
||||||
mock_permitted.assert_called_once_with(user, Document, "view_document")
|
mock_permitted.assert_called_once_with(user)
|
||||||
assert mock_retrieve.call_args.kwargs["document_ids"] == [1, 2, 3]
|
assert mock_retrieve.call_args.kwargs["document_ids"] == [1, 2, 3]
|
||||||
|
|
||||||
|
@pytest.mark.django_db
|
||||||
|
@override_settings(LLM_EMBEDDING_BACKEND="huggingface")
|
||||||
|
def test_version_of_private_root_is_not_visible(
|
||||||
|
self,
|
||||||
|
mocker: pytest_mock.MockerFixture,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A private root document owned by someone else
|
||||||
|
- A version of it whose own owner is unset, as when the root
|
||||||
|
changed hands after the version was created
|
||||||
|
WHEN:
|
||||||
|
- get_taxonomy_context() is called for a non-superuser
|
||||||
|
THEN:
|
||||||
|
- Neither the root nor the version is in the visible ids passed to
|
||||||
|
retrieve_similar_nodes(), since a version follows its root
|
||||||
|
"""
|
||||||
|
owner = UserFactory.create()
|
||||||
|
viewer = UserFactory.create(is_superuser=False)
|
||||||
|
root = DocumentFactory.create(content="private", owner=owner)
|
||||||
|
version = DocumentFactory.create(
|
||||||
|
content="private",
|
||||||
|
owner=None,
|
||||||
|
root_document=root,
|
||||||
|
version_index=1,
|
||||||
|
)
|
||||||
|
source = DocumentFactory.create(content="Some content", owner=viewer)
|
||||||
|
mock_retrieve = mocker.patch(
|
||||||
|
"paperless_ai.ai_classifier.retrieve_similar_nodes",
|
||||||
|
return_value=[],
|
||||||
|
)
|
||||||
|
|
||||||
|
get_taxonomy_context(source, viewer)
|
||||||
|
|
||||||
|
visible = mock_retrieve.call_args.kwargs["document_ids"]
|
||||||
|
assert source.pk in visible
|
||||||
|
assert root.pk not in visible
|
||||||
|
assert version.pk not in visible
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
@pytest.mark.django_db
|
||||||
class TestFulltextSimilarDocuments:
|
class TestFulltextSimilarDocuments:
|
||||||
@@ -803,7 +842,7 @@ class TestFulltextSimilarDocuments:
|
|||||||
- _fulltext_similar_documents() is called with that user
|
- _fulltext_similar_documents() is called with that user
|
||||||
THEN:
|
THEN:
|
||||||
- Only the still-permitted document is returned - the DB
|
- Only the still-permitted document is returned - the DB
|
||||||
re-check via restrict_queryset_to_visible() must catch the
|
re-check via permitted_document_ids() must catch the
|
||||||
document Tantivy's stale index still thinks is visible
|
document Tantivy's stale index still thinks is visible
|
||||||
"""
|
"""
|
||||||
owner = UserFactory.create()
|
owner = UserFactory.create()
|
||||||
@@ -834,6 +873,45 @@ class TestFulltextSimilarDocuments:
|
|||||||
|
|
||||||
assert [s["document_id"] for s in result] == [permitted.pk]
|
assert [s["document_id"] for s in result] == [permitted.pk]
|
||||||
|
|
||||||
|
def test_excludes_version_of_private_root_for_regular_user(
|
||||||
|
self,
|
||||||
|
fulltext_backend: TantivyBackend,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
GIVEN:
|
||||||
|
- A regular user and a private root owned by someone else
|
||||||
|
- A version of that root with no owner of its own, which the
|
||||||
|
Tantivy index therefore treats as visible to everyone
|
||||||
|
WHEN:
|
||||||
|
- _fulltext_similar_documents() is called with that user
|
||||||
|
THEN:
|
||||||
|
- The version is not returned, since the DB re-check judges it by
|
||||||
|
its root
|
||||||
|
"""
|
||||||
|
owner = UserFactory.create()
|
||||||
|
viewer = UserFactory.create(is_superuser=False)
|
||||||
|
source = DocumentFactory.create(
|
||||||
|
content="shared content phrase",
|
||||||
|
owner=viewer,
|
||||||
|
)
|
||||||
|
root = DocumentFactory.create(
|
||||||
|
content="shared content phrase",
|
||||||
|
owner=owner,
|
||||||
|
)
|
||||||
|
version = DocumentFactory.create(
|
||||||
|
content="shared content phrase",
|
||||||
|
owner=None,
|
||||||
|
root_document=root,
|
||||||
|
version_index=1,
|
||||||
|
)
|
||||||
|
fulltext_backend.add_or_update(source)
|
||||||
|
fulltext_backend.add_or_update(root)
|
||||||
|
fulltext_backend.add_or_update(version)
|
||||||
|
|
||||||
|
result = _fulltext_similar_documents(source, user=viewer, top_k=5)
|
||||||
|
|
||||||
|
assert version.pk not in [s["document_id"] for s in result]
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.django_db
|
@pytest.mark.django_db
|
||||||
@override_settings(LLM_EMBEDDING_BACKEND="huggingface")
|
@override_settings(LLM_EMBEDDING_BACKEND="huggingface")
|
||||||
|
|||||||
Reference in new issue
Block a user