Compare commits

...
42 changed files with 348 additions and 330 deletions

No files matched your search

+2 -2
View File
@@ -256,8 +256,8 @@ isort.force-single-line = true
[tool.codespell]
ignore-words-list = "criterias,afterall,valeu,ureue,equest,ure,assertIn,Oktober,commitish,NIN,nin,reprot"
skip = """\
src-ui/src/locale/*,src-ui/pnpm-lock.yaml,src-ui/e2e/*,src/paperless_mail/tests/samples/*,src/paperless/tests/samples\
/mail/*,src/documents/tests/samples/*,*.po,*.json\
src-ui/src/locale/*,src-ui/pnpm-lock.yaml,src-ui/e2e/*,src/paperless/tests/samples/mail/*,src/documents/tests/samples\
/*,src/paperless_testing/sample_files/*,*.po,*.json\
"""
write-changes = true
+38
View File
@@ -11,6 +11,8 @@ from typing import TYPE_CHECKING
import pytest
from paperless_testing import samples
if TYPE_CHECKING:
from collections.abc import Generator
from pathlib import Path
@@ -149,3 +151,39 @@ def fake_progress_manager(
monkeypatch.setattr("documents.tasks.ProgressManager", FakeProgressManager)
return FakeProgressManager
@pytest.fixture(scope="session")
def shared_samples_dir() -> Path:
"""Directory of sample files used by more than one app's tests."""
return samples.SHARED_SAMPLES_DIR
@pytest.fixture(scope="session")
def simple_digital_pdf_file() -> Path:
"""One-page PDF with a text layer."""
return samples.SIMPLE_DIGITAL_PDF
@pytest.fixture(scope="session")
def multi_page_digital_pdf_file() -> Path:
"""Three-page PDF with a text layer."""
return samples.MULTI_PAGE_DIGITAL_PDF
@pytest.fixture(scope="session")
def with_form_pdf_file() -> Path:
"""PDF containing a fillable form."""
return samples.WITH_FORM_PDF
@pytest.fixture(scope="session")
def multi_page_images_pdf_file() -> Path:
"""Multi-page PDF of scanned images, no text layer."""
return samples.MULTI_PAGE_IMAGES_PDF
@pytest.fixture(scope="session")
def thumbnail_webp_file() -> Path:
"""Small WebP thumbnail."""
return samples.THUMBNAIL_WEBP
+6 -10
View File
@@ -12,29 +12,25 @@ if TYPE_CHECKING:
from paperless_testing.dirs import PaperlessDirs
@pytest.fixture(scope="session")
def document_samples_dir() -> Path:
"""Path to the shared test sample documents."""
return Path(__file__).parent / "samples" / "documents"
@pytest.fixture()
def sample_doc(
paperless_dirs: "PaperlessDirs",
document_samples_dir: Path,
simple_digital_pdf_file: Path,
multi_page_images_pdf_file: Path,
thumbnail_webp_file: Path,
) -> "Document":
"""Create a document with valid files and matching checksums."""
with filelock.FileLock(paperless_dirs.media_lock):
shutil.copy(
document_samples_dir / "originals" / "0000001.pdf",
simple_digital_pdf_file,
paperless_dirs.originals_dir / "0000001.pdf",
)
shutil.copy(
document_samples_dir / "archive" / "0000001.pdf",
multi_page_images_pdf_file,
paperless_dirs.archive_dir / "0000001.pdf",
)
shutil.copy(
document_samples_dir / "thumbnails" / "0000001.webp",
thumbnail_webp_file,
paperless_dirs.thumbnail_dir / "0000001.webp",
)
Binary file not shown.

Before

Width:  |  Height:  |  Size: 32 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.6 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.6 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.6 KiB

Binary file not shown.
-1
View File
@@ -1 +0,0 @@
This is a test file.
@@ -53,7 +53,7 @@ class TestBulkDownload(DirectoriesMixin, SampleDirMixin, APITestCase):
archive_checksum="D",
)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", self.doc2.source_path)
shutil.copy(self.SIMPLE_PDF, self.doc2.source_path)
shutil.copy(self.SAMPLE_DIR / "simple.png", self.doc2b.source_path)
shutil.copy(self.SAMPLE_DIR / "simple.jpg", self.doc3.source_path)
shutil.copy(self.SAMPLE_DIR / "test_with_bom.pdf", self.doc3.archive_path)
+49 -33
View File
@@ -56,6 +56,8 @@ from paperless_testing.http import read_streaming_response
from paperless_testing.permissions import grant_all_global
from paperless_testing.permissions import grant_global
from paperless_testing.permissions import grant_object
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
from paperless_testing.samples import THUMBNAIL_WEBP
class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
@@ -1852,7 +1854,11 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SimpleUploadedFile(
"simple.pdf",
SIMPLE_DIGITAL_PDF.read_bytes(),
content_type="application/pdf",
) as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f},
@@ -1880,7 +1886,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
payload = SimpleUploadedFile(
"../../outside.pdf",
(Path(__file__).parent / "samples" / "simple.pdf").read_bytes(),
SIMPLE_DIGITAL_PDF.read_bytes(),
content_type="application/pdf",
)
@@ -1909,7 +1915,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
pdf_bytes = (Path(__file__).parent / "samples" / "simple.pdf").read_bytes()
pdf_bytes = SIMPLE_DIGITAL_PDF.read_bytes()
boundary = "paperless-boundary"
payload = (
(
@@ -1985,7 +1991,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
def test_upload_insufficient_permissions(self) -> None:
self.client.force_authenticate(user=UserFactory(username="testuser2"))
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f},
@@ -1998,7 +2004,11 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SimpleUploadedFile(
"simple.pdf",
SIMPLE_DIGITAL_PDF.read_bytes(),
content_type="application/pdf",
) as f:
response = self.client.post(
"/api/documents/post_document/",
{
@@ -2031,7 +2041,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"documenst": f},
@@ -2057,7 +2067,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "title": "my custom title"},
@@ -2077,7 +2087,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
)
c = Correspondent.objects.create(name="test-corres")
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "correspondent": c.id},
@@ -2096,7 +2106,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "correspondent": 3456},
@@ -2111,7 +2121,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
)
dt = DocumentType.objects.create(name="invoice")
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "document_type": dt.id},
@@ -2130,7 +2140,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "document_type": 34578},
@@ -2145,7 +2155,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
)
sp = StoragePath.objects.create(name="invoices")
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "storage_path": sp.id},
@@ -2164,7 +2174,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "storage_path": 34578},
@@ -2180,7 +2190,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
t1 = Tag.objects.create(name="tag1")
t2 = Tag.objects.create(name="tag2")
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "tags": [t2.id, t1.id]},
@@ -2201,7 +2211,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
t1 = Tag.objects.create(name="tag1")
t2 = Tag.objects.create(name="tag2")
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "tags": [t2.id, t1.id, 734563]},
@@ -2225,7 +2235,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
0,
tzinfo=zoneinfo.ZoneInfo("America/Los_Angeles"),
)
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "created": created},
@@ -2241,7 +2251,11 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SimpleUploadedFile(
"simple.pdf",
SIMPLE_DIGITAL_PDF.read_bytes(),
content_type="application/pdf",
) as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "archive_serial_number": 500},
@@ -2268,7 +2282,11 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
data_type=CustomField.FieldDataType.STRING,
)
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SimpleUploadedFile(
"simple.pdf",
SIMPLE_DIGITAL_PDF.read_bytes(),
content_type="application/pdf",
) as f:
response = self.client.post(
"/api/documents/post_document/",
{
@@ -2323,7 +2341,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
w1.actions.add(action1)
w1.save()
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{
@@ -2364,7 +2382,11 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
data_type=CustomField.FieldDataType.INT,
)
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SimpleUploadedFile(
"simple.pdf",
SIMPLE_DIGITAL_PDF.read_bytes(),
content_type="application/pdf",
) as f:
response = self.client.post(
"/api/documents/post_document/",
{
@@ -2413,7 +2435,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
]
for payload in error_payloads:
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
data = {"document": f, **payload}
response = self.client.post(
"/api/documents/post_document/",
@@ -2471,7 +2493,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
id=str(uuid.uuid4()),
)
with (Path(__file__).parent / "samples" / "simple.pdf").open("rb") as f:
with SIMPLE_DIGITAL_PDF.open("rb") as f:
response = self.client.post(
"/api/documents/post_document/",
{"document": f, "from_webui": True},
@@ -2510,14 +2532,8 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
archive_filename="archive.pdf",
)
source_file: Path = (
Path(__file__).parent
/ "samples"
/ "documents"
/ "thumbnails"
/ "0000001.webp"
)
archive_file: Path = Path(__file__).parent / "samples" / "simple.pdf"
source_file: Path = THUMBNAIL_WEBP
archive_file: Path = SIMPLE_DIGITAL_PDF
shutil.copy(source_file, doc.source_path)
shutil.copy(archive_file, doc.archive_path)
@@ -2550,7 +2566,7 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
mime_type="application/pdf",
)
shutil.copy(Path(__file__).parent / "samples" / "simple.pdf", doc.source_path)
shutil.copy(SIMPLE_DIGITAL_PDF, doc.source_path)
response = self.client.get(f"/api/documents/{doc.pk}/metadata/")
self.assertEqual(response.status_code, status.HTTP_200_OK)
@@ -4288,8 +4304,8 @@ class TestDocumentApi(DirectoriesMixin, ConsumeTaskMixin, APITestCase):
filename="test2.pdf",
)
archive_file = Path(__file__).parent / "samples" / "simple.pdf"
source_file = Path(__file__).parent / "samples" / "simple.pdf"
archive_file = SIMPLE_DIGITAL_PDF
source_file = SIMPLE_DIGITAL_PDF
shutil.copy(archive_file, doc.archive_path)
shutil.copy(source_file, doc2.source_path)
+8 -5
View File
@@ -12,6 +12,9 @@ from documents.tests.utils import SampleDirMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.factories import UserFactory
from paperless_testing.permissions import grant_global
from paperless_testing.samples import MULTI_PAGE_DIGITAL_PDF
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
from paperless_testing.samples import WITH_FORM_PDF
class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
@@ -42,15 +45,15 @@ class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
# Copy sample files to document paths (using different files to distinguish versions)
shutil.copy(
self.SAMPLE_DIR / "documents" / "originals" / "0000001.pdf",
SIMPLE_DIGITAL_PDF,
self.doc1.archive_path,
)
shutil.copy(
self.SAMPLE_DIR / "documents" / "originals" / "0000002.pdf",
MULTI_PAGE_DIGITAL_PDF,
self.doc1.source_path,
)
shutil.copy(
self.SAMPLE_DIR / "documents" / "originals" / "0000003.pdf",
WITH_FORM_PDF,
self.doc2.source_path,
)
@@ -377,7 +380,7 @@ class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
checksum="3",
filename="test3.pdf",
)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", doc3.source_path)
shutil.copy(self.SIMPLE_PDF, doc3.source_path)
doc4 = Document.objects.create(
title="test1",
@@ -386,7 +389,7 @@ class TestEmail(DirectoriesMixin, SampleDirMixin, APITestCase):
checksum="4",
filename="test4.pdf",
)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", doc4.source_path)
shutil.copy(self.SIMPLE_PDF, doc4.source_path)
response = self.client.post(
self.ENDPOINT,
+2 -2
View File
@@ -150,7 +150,7 @@ class TestBarcode(
- No barcodes detected
- No pages to split on
"""
test_file = self.SAMPLE_DIR / "simple.pdf"
test_file = self.SIMPLE_PDF
with self.get_reader(test_file) as reader:
reader.detect()
separator_page_numbers = reader.get_separation_pages()
@@ -442,7 +442,7 @@ class TestBarcode(
THEN:
- Nothing happens
"""
test_file = self.SAMPLE_DIR / "simple.pdf"
test_file = self.SIMPLE_PDF
with self.get_reader(test_file) as reader:
try:
+9 -30
View File
@@ -24,6 +24,9 @@ from documents.models import Tag
from documents.permissions import set_permissions_for_objects
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.permissions import grant_object
from paperless_testing.samples import MULTI_PAGE_DIGITAL_PDF
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
from paperless_testing.samples import WITH_FORM_PDF
class TestBulkEdit(DirectoriesMixin, TestCase):
@@ -701,47 +704,27 @@ class TestPDFActions(DirectoriesMixin, TestCase):
super().setUp()
sample1 = self.dirs.scratch_dir / "sample.pdf"
shutil.copy(
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
SIMPLE_DIGITAL_PDF,
sample1,
)
sample1_archive = self.dirs.archive_dir / "sample_archive.pdf"
shutil.copy(
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
SIMPLE_DIGITAL_PDF,
sample1_archive,
)
sample2 = self.dirs.scratch_dir / "sample2.pdf"
shutil.copy(
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000002.pdf",
MULTI_PAGE_DIGITAL_PDF,
sample2,
)
sample2_archive = self.dirs.archive_dir / "sample2_archive.pdf"
shutil.copy(
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000002.pdf",
MULTI_PAGE_DIGITAL_PDF,
sample2_archive,
)
sample3 = self.dirs.scratch_dir / "sample3.pdf"
shutil.copy(
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000003.pdf",
WITH_FORM_PDF,
sample3,
)
self.doc1 = Document.objects.create(
@@ -775,11 +758,7 @@ class TestPDFActions(DirectoriesMixin, TestCase):
)
img_doc_archive = self.dirs.archive_dir / "sample_image.pdf"
shutil.copy(
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
SIMPLE_DIGITAL_PDF,
img_doc_archive,
)
self.img_doc = Document.objects.create(
+9 -28
View File
@@ -36,6 +36,9 @@ from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.factories import UserFactory
from paperless_testing.fakes.progress import FakeProgressManager
from paperless_testing.samples import MULTI_PAGE_DIGITAL_PDF
from paperless_testing.samples import MULTI_PAGE_IMAGES_PDF
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
class _BaseNewStyleParser:
@@ -112,9 +115,7 @@ class _BaseNewStyleParser:
class DummyParser(_BaseNewStyleParser):
_ARCHIVE_SRC = (
Path(__file__).parent / "samples" / "documents" / "archive" / "0000001.pdf"
)
_ARCHIVE_SRC = MULTI_PAGE_IMAGES_PDF
def parse(self, document_path, mime_type, *, produce_archive: bool = True) -> None:
self._text = "The Text"
@@ -200,33 +201,19 @@ class TestConsumer(
self.addCleanup(patcher.stop)
def get_test_file(self):
src = (
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf"
)
src = SIMPLE_DIGITAL_PDF
dst = self.dirs.scratch_dir / "sample.pdf"
shutil.copy(src, dst)
return dst
def get_test_file2(self):
src = (
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000002.pdf"
)
src = MULTI_PAGE_DIGITAL_PDF
dst = self.dirs.scratch_dir / "sample2.pdf"
shutil.copy(src, dst)
return dst
def get_test_archive_file(self):
src = (
Path(__file__).parent / "samples" / "documents" / "archive" / "0000001.pdf"
)
src = MULTI_PAGE_IMAGES_PDF
dst = self.dirs.scratch_dir / "sample_archive.pdf"
shutil.copy(src, dst)
return dst
@@ -1062,7 +1049,7 @@ class TestConsumer(
@mock.patch("documents.consumer.get_parser_registry")
def test_similar_filenames(self, m) -> None:
shutil.copy(
Path(__file__).parent / "samples" / "simple.pdf",
SIMPLE_DIGITAL_PDF,
settings.CONSUMPTION_DIR / "simple.pdf",
)
shutil.copy(
@@ -1603,13 +1590,7 @@ class TestConsumerRemoteOCR(
self.addCleanup(patcher.stop)
def _consume(self, *, overrides: DocumentMetadataOverrides | None = None) -> bool:
src = (
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf"
)
src = SIMPLE_DIGITAL_PDF
dst = self.dirs.scratch_dir / "sample.pdf"
shutil.copy(src, dst)
+5 -2
View File
@@ -37,12 +37,15 @@ class TestDoubleSided(
self.double_sided_dir.mkdir()
self.staging_file = self.dirs.scratch_dir / STAGING_FILE_NAME
def _sample(self, name: str) -> Path:
return self.SIMPLE_PDF if name == "simple.pdf" else self.SAMPLE_DIR / name
def consume_file(self, srcname, dstname: str | Path = "foo.pdf"):
"""
Starts the consume process and also ensures the
destination file does not exist afterwards
"""
src = self.SAMPLE_DIR / srcname
src = self._sample(srcname)
dst = self.double_sided_dir / dstname
dst.parent.mkdir(parents=True, exist_ok=True)
shutil.copy(src, dst)
@@ -57,7 +60,7 @@ class TestDoubleSided(
return msg
def create_staging_file(self, src="double-sided-odd.pdf", datetime=None) -> None:
shutil.copy(self.SAMPLE_DIR / src, self.staging_file)
shutil.copy(self._sample(src), self.staging_file)
if datetime is None:
datetime = dt.datetime.now()
os.utime(str(self.staging_file), (datetime.timestamp(),) * 2)
+2 -1
View File
@@ -22,8 +22,9 @@ from documents.models import Document
from documents.tasks import update_document_content_maybe_archive_file
from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
sample_file: Path = Path(__file__).parent / "samples" / "simple.pdf"
sample_file: Path = SIMPLE_DIGITAL_PDF
@pytest.mark.management
+20 -76
View File
@@ -51,6 +51,7 @@ from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.dirs import paperless_environment
from paperless_testing.permissions import grant_object
from paperless_testing.samples import install_document_samples
@pytest.mark.management
@@ -196,10 +197,7 @@ class TestExportImport(
def test_exporter(self, *, use_filename_format=False) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
num_permission_objects = Permission.objects.count()
@@ -303,10 +301,7 @@ class TestExportImport(
def test_exporter_with_filename_format(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
with override_settings(
FILENAME_FORMAT="{created_year}/{correspondent}/{title}",
@@ -315,10 +310,7 @@ class TestExportImport(
def test_exporter_includes_share_links_and_bundles(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
share_link = ShareLink.objects.create(
slug="share-link-slug",
@@ -417,10 +409,7 @@ class TestExportImport(
def test_update_export_changed_time(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
self._do_export()
self.assertIsFile(self.target / "manifest.json")
@@ -456,10 +445,7 @@ class TestExportImport(
def test_update_export_changed_checksum(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
self._do_export()
@@ -486,10 +472,7 @@ class TestExportImport(
def test_update_export_deleted_document(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
manifest = self._do_export()
@@ -521,10 +504,7 @@ class TestExportImport(
@override_settings(FILENAME_FORMAT="{title}/{correspondent}")
def test_update_export_changed_location(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
self._do_export(use_filename_format=True)
self.assertIsFile(self.target / "wow1" / "c.pdf")
@@ -566,10 +546,7 @@ class TestExportImport(
- Zipfile contains exported files
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
args = ["document_exporter", self.target, "--zip"]
@@ -598,10 +575,7 @@ class TestExportImport(
- Zipfile contains exported files
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
args = ["document_exporter", self.target, "--zip", "--use-filename-format"]
@@ -637,10 +611,7 @@ class TestExportImport(
- The existing file and directory in target are removed
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
# Create stuff in target directory
existing_file = self.target / "test.txt"
@@ -736,10 +707,7 @@ class TestExportImport(
- Documents can be imported again
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
manifest = self._do_export()
has_archive = False
@@ -782,10 +750,7 @@ class TestExportImport(
- Documents can be imported again
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
manifest = self._do_export()
has_thumbnail = False
@@ -830,10 +795,7 @@ class TestExportImport(
- Documents can be imported again
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
manifest = self._do_export(split_manifest=True)
has_document = False
@@ -866,10 +828,7 @@ class TestExportImport(
- Documents can be imported again
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
self._do_export(use_folder_prefix=True)
@@ -896,10 +855,7 @@ class TestExportImport(
- Documents can be imported again
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
self._do_export(use_folder_prefix=True, split_manifest=True)
@@ -926,10 +882,7 @@ class TestExportImport(
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
num_content_type_objects = ContentType.objects.count()
num_permission_objects = Permission.objects.count()
@@ -966,10 +919,7 @@ class TestExportImport(
def test_exporter_with_auditlog_disabled(self) -> None:
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
with override_settings(
AUDIT_LOG_ENABLED=False,
@@ -989,10 +939,7 @@ class TestExportImport(
survive the round-trip with deleted_at preserved
"""
shutil.rmtree(Path(self.dirs.media_dir) / "documents")
shutil.copytree(
Path(__file__).parent / "samples" / "documents",
Path(self.dirs.media_dir) / "documents",
)
install_document_samples(Path(self.dirs.media_dir) / "documents")
# d1 has self.note and self.cfi1 attached via setUp
self.d1.delete()
@@ -1036,10 +983,7 @@ class TestExportImport(
"""
shutil.rmtree(self.dirs.media_dir / "documents")
shutil.copytree(
self.SAMPLE_DIR / "documents",
self.dirs.media_dir / "documents",
)
install_document_samples(self.dirs.media_dir / "documents")
_ = self._do_export(data_only=True)
@@ -11,6 +11,7 @@ from documents.models import Document
from documents.parsers import get_default_thumbnail
from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
@pytest.mark.management
@@ -24,7 +25,7 @@ class TestMakeThumbnails(DirectoriesMixin, FileSystemAssertsMixin, TestCase):
filename="test.pdf",
)
shutil.copy(
Path(__file__).parent / "samples" / "simple.pdf",
SIMPLE_DIGITAL_PDF,
self.d1.source_path,
)
@@ -36,7 +37,7 @@ class TestMakeThumbnails(DirectoriesMixin, FileSystemAssertsMixin, TestCase):
filename="test2.pdf",
)
shutil.copy(
Path(__file__).parent / "samples" / "simple.pdf",
SIMPLE_DIGITAL_PDF,
self.d2.source_path,
)
+61
View File
@@ -0,0 +1,61 @@
import hashlib
from collections import defaultdict
from pathlib import Path
from paperless_testing.samples import SHARED_SAMPLES_DIR
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
from paperless_testing.samples import install_document_samples
EXPECTED_MEDIA_FILES = [
"originals/0000001.pdf",
"originals/0000002.pdf",
"originals/0000003.pdf",
"originals/0000004.pdf",
"originals/0000005.pdf",
"originals/0000006.pdf",
"archive/0000001.pdf",
"thumbnails/0000001.webp",
"thumbnails/0000002.webp",
"thumbnails/0000003.webp",
"thumbnails/0000004.webp",
]
def _sha(path: Path) -> str:
return hashlib.sha256(path.read_bytes()).hexdigest()
def test_install_recreates_every_historical_opaque_name(tmp_path: Path) -> None:
install_document_samples(tmp_path)
for relpath in EXPECTED_MEDIA_FILES:
assert (tmp_path / relpath).is_file(), relpath
def test_install_uses_the_shared_bytes(tmp_path: Path) -> None:
install_document_samples(tmp_path)
assert _sha(tmp_path / "originals" / "0000001.pdf") == _sha(SIMPLE_DIGITAL_PDF)
APP_SAMPLE_TREES = [
Path(__file__).parent / "samples",
Path(__file__).parents[2] / "paperless" / "tests" / "samples",
SHARED_SAMPLES_DIR,
]
def test_no_two_sample_files_have_identical_bytes() -> None:
by_hash: dict[str, list[Path]] = defaultdict(list)
for tree in APP_SAMPLE_TREES:
assert tree.is_dir(), f"Sample tree does not exist: {tree}"
for path in tree.rglob("*"):
# Skip empty placeholders, which would all hash identically
if path.is_file() and path.stat().st_size > 0:
by_hash[_sha(path)].append(path)
assert by_hash, "No sample files were found to compare"
duplicates = {h: ps for h, ps in by_hash.items() if len(ps) > 1}
lines = [" " + ", ".join(str(p) for p in ps) for ps in duplicates.values()]
message = (
"Duplicate sample files, move one to paperless_testing/sample_files:\n"
+ "\n".join(lines)
)
assert not duplicates, message
+4 -15
View File
@@ -20,6 +20,7 @@ from documents.sanity_checker import SanityCheckMessages
from documents.tests.helpers import dummy_preprocess
from paperless_testing.assertions import FileSystemAssertsMixin
from paperless_testing.dirs import DirectoriesMixin
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
@pytest.mark.django_db
@@ -228,20 +229,12 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
"""
sample1 = self.dirs.scratch_dir / "sample.pdf"
shutil.copy(
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
SIMPLE_DIGITAL_PDF,
sample1,
)
sample1_archive = self.dirs.archive_dir / "sample_archive.pdf"
shutil.copy(
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
SIMPLE_DIGITAL_PDF,
sample1_archive,
)
doc = Document.objects.create(
@@ -269,11 +262,7 @@ class TestUpdateContent(DirectoriesMixin, TestCase):
"""
sample1 = self.dirs.scratch_dir / "sample.pdf"
shutil.copy(
Path(__file__).parent
/ "samples"
/ "documents"
/ "originals"
/ "0000001.pdf",
SIMPLE_DIGITAL_PDF,
sample1,
)
doc = Document.objects.create(
+27 -27
View File
@@ -178,7 +178,7 @@ class TestWorkflows(
self.assertEqual(action.__str__(), "WorkflowAction 1")
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -290,7 +290,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -409,7 +409,7 @@ class TestWorkflows(
w2.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -478,7 +478,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -531,7 +531,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -608,7 +608,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -687,7 +687,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -765,7 +765,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -876,7 +876,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -928,7 +928,7 @@ class TestWorkflows(
generated = generate_unique_filename(doc)
destination = (settings.ORIGINALS_DIR / generated).resolve()
create_source_path_directory(destination)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", destination)
shutil.copy(self.SIMPLE_PDF, destination)
Document.objects.filter(pk=doc.pk).update(filename=generated.as_posix())
doc.refresh_from_db()
@@ -2023,7 +2023,7 @@ class TestWorkflows(
superuser = UserFactory(username="superuser", superuser=True)
self.client.force_authenticate(user=superuser)
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
tasks.consume_file(
@@ -3009,7 +3009,7 @@ class TestWorkflows(
generated = generate_unique_filename(doc)
destination = (settings.ORIGINALS_DIR / generated).resolve()
create_source_path_directory(destination)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", destination)
shutil.copy(self.SIMPLE_PDF, destination)
Document.objects.filter(pk=doc.pk).update(filename=generated.as_posix())
doc.refresh_from_db()
doc.tags.set([self.t1, self.t2])
@@ -3073,7 +3073,7 @@ class TestWorkflows(
generated = generate_unique_filename(doc)
destination = (settings.ORIGINALS_DIR / generated).resolve()
create_source_path_directory(destination)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", destination)
shutil.copy(self.SIMPLE_PDF, destination)
Document.objects.filter(pk=doc.pk).update(filename=generated.as_posix())
doc.refresh_from_db()
doc.tags.set([self.t1])
@@ -3222,7 +3222,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -3346,7 +3346,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -3496,7 +3496,7 @@ class TestWorkflows(
workflow.actions.set([assignment_action, email_action])
temp_working_copy = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "working-copy.pdf",
)
@@ -3596,7 +3596,7 @@ class TestWorkflows(
# move the file
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -3704,7 +3704,7 @@ class TestWorkflows(
generated = generate_unique_filename(doc)
destination = (settings.ORIGINALS_DIR / generated).resolve()
create_source_path_directory(destination)
shutil.copy(self.SAMPLE_DIR / "simple.pdf", destination)
shutil.copy(self.SIMPLE_PDF, destination)
Document.objects.filter(pk=doc.pk).update(filename=generated.as_posix())
run_workflows(WorkflowTrigger.WorkflowTriggerType.DOCUMENT_UPDATED, doc)
@@ -3915,7 +3915,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -4036,7 +4036,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -4387,7 +4387,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -4835,7 +4835,7 @@ class TestWorkflows(
w.save()
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
@@ -5027,7 +5027,7 @@ class TestWorkflows(
# Create a test file to be consumed
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
test_file_path = Path(test_file)
@@ -5091,7 +5091,7 @@ class TestWorkflows(
# Create a test file to be consumed
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple2.pdf",
)
test_file_path = Path(test_file)
@@ -5529,7 +5529,7 @@ class TestDateWorkflowLocalization(
)
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
tmp_path / "simple.pdf",
)
@@ -5596,7 +5596,7 @@ class TestRemoteOCRWorkflowAction(DirectoriesMixin, SampleDirMixin, APITestCase)
self._make_workflow(WorkflowTrigger.WorkflowTriggerType.CONSUMPTION)
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
overrides = DocumentMetadataOverrides()
@@ -5814,7 +5814,7 @@ class TestApplyAISuggestionsWorkflowAction(
)
test_file = shutil.copy(
self.SAMPLE_DIR / "simple.pdf",
self.SIMPLE_PDF,
self.dirs.scratch_dir / "simple.pdf",
)
+3
View File
@@ -11,6 +11,7 @@ from documents.data_models import ConsumableDocument
from documents.data_models import DocumentMetadataOverrides
from documents.data_models import DocumentSource
from paperless_testing.fakes.progress import FakeProgressManager
from paperless_testing.samples import SIMPLE_DIGITAL_PDF
class ConsumeTaskMixin:
@@ -53,6 +54,8 @@ class SampleDirMixin:
BARCODE_SAMPLE_DIR = SAMPLE_DIR / "barcodes"
SIMPLE_PDF = SIMPLE_DIGITAL_PDF
class GetConsumerMixin:
@contextmanager
-24
View File
@@ -432,30 +432,6 @@ def tesseract_samples_dir(parser_samples_dir: Path) -> Path:
return parser_samples_dir / "tesseract"
@pytest.fixture(scope="session")
def multi_page_images_pdf_file(tesseract_samples_dir: Path) -> Path:
"""Path to a multi-page PDF with images.
Returns
-------
Path
Absolute path to ``tesseract/multi-page-images.pdf``.
"""
return tesseract_samples_dir / "multi-page-images.pdf"
@pytest.fixture(scope="session")
def simple_digital_pdf_file(tesseract_samples_dir: Path) -> Path:
"""Path to a simple digital PDF sample file.
Returns
-------
Path
Absolute path to ``tesseract/simple-digital.pdf``.
"""
return tesseract_samples_dir / "simple-digital.pdf"
@pytest.fixture(scope="session")
def simple_no_dpi_png_file(tesseract_samples_dir: Path) -> Path:
"""Path to a simple PNG without DPI information.
@@ -160,11 +160,11 @@ class TestGetPageCount:
def test_single_page_pdf(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
simple_digital_pdf_file: Path,
) -> None:
assert (
tesseract_parser.get_page_count(
tesseract_samples_dir / "simple-digital.pdf",
simple_digital_pdf_file,
"application/pdf",
)
== 1
@@ -187,7 +187,7 @@ class TestGetPageCount:
self,
mocker: MockerFixture,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
simple_digital_pdf_file: Path,
caplog,
) -> None:
"""
@@ -203,7 +203,7 @@ class TestGetPageCount:
with caplog.at_level(logging.WARNING):
page_count = tesseract_parser.get_page_count(
tesseract_samples_dir / "simple-digital.pdf",
simple_digital_pdf_file,
"application/pdf",
)
assert page_count is None
@@ -272,10 +272,10 @@ class TestGetThumbnail:
def test_thumbnail_is_file(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
simple_digital_pdf_file: Path,
) -> None:
thumb = tesseract_parser.get_thumbnail(
tesseract_samples_dir / "simple-digital.pdf",
simple_digital_pdf_file,
"application/pdf",
)
assert thumb.is_file()
@@ -284,7 +284,7 @@ class TestGetThumbnail:
self,
mocker: MockerFixture,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
simple_digital_pdf_file: Path,
) -> None:
def _raise_on_pdf(input_file, output_file, **kwargs) -> None:
if ".pdf" in str(input_file):
@@ -294,7 +294,7 @@ class TestGetThumbnail:
mocker.patch("documents.parsers.run_convert", side_effect=_raise_on_pdf)
thumb = tesseract_parser.get_thumbnail(
tesseract_samples_dir / "simple-digital.pdf",
simple_digital_pdf_file,
"application/pdf",
)
assert thumb.is_file()
@@ -320,11 +320,11 @@ class TestExtractText:
def test_extract_text_from_digital_pdf(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
simple_digital_pdf_file: Path,
) -> None:
text = tesseract_parser.extract_text(
None,
tesseract_samples_dir / "simple-digital.pdf",
simple_digital_pdf_file,
)
assert text is not None
assert "This is a test document." in text.strip()
@@ -339,7 +339,7 @@ class TestParsePdf:
def test_simple_digital_creates_archive(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
multi_page_digital_pdf_file: Path,
) -> None:
"""
GIVEN:
@@ -353,7 +353,7 @@ class TestParsePdf:
- Text is extracted
"""
tesseract_parser.parse(
tesseract_samples_dir / "multi-page-digital.pdf",
multi_page_digital_pdf_file,
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -368,10 +368,10 @@ class TestParsePdf:
def test_with_form_default(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
with_form_pdf_file: Path,
) -> None:
tesseract_parser.parse(
tesseract_samples_dir / "with-form.pdf",
with_form_pdf_file,
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -384,11 +384,11 @@ class TestParsePdf:
def test_with_form_redo_no_archive_when_not_requested(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
with_form_pdf_file: Path,
) -> None:
tesseract_parser.settings.mode = ModeChoices.REDO
tesseract_parser.parse(
tesseract_samples_dir / "with-form.pdf",
with_form_pdf_file,
"application/pdf",
produce_archive=False,
)
@@ -401,11 +401,11 @@ class TestParsePdf:
def test_with_form_force(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
with_form_pdf_file: Path,
) -> None:
tesseract_parser.settings.mode = ModeChoices.FORCE
tesseract_parser.parse(
tesseract_samples_dir / "with-form.pdf",
with_form_pdf_file,
"application/pdf",
)
assert_ordered_substrings(
@@ -446,7 +446,7 @@ class TestParsePdf:
self,
mocker: MockerFixture,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
simple_digital_pdf_file: Path,
) -> None:
mocker.patch(
"ocrmypdf.ocr",
@@ -454,7 +454,7 @@ class TestParsePdf:
)
with pytest.raises(ParseError):
tesseract_parser.parse(
tesseract_samples_dir / "simple-digital.pdf",
simple_digital_pdf_file,
"application/pdf",
)
@@ -530,10 +530,10 @@ class TestParseMultiPage:
def test_multi_page_digital(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
multi_page_digital_pdf_file: Path,
) -> None:
tesseract_parser.parse(
tesseract_samples_dir / "multi-page-digital.pdf",
multi_page_digital_pdf_file,
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -557,12 +557,12 @@ class TestParseMultiPage:
self,
mode: str,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
multi_page_digital_pdf_file: Path,
) -> None:
tesseract_parser.settings.pages = 2
tesseract_parser.settings.mode = mode
tesseract_parser.parse(
tesseract_samples_dir / "multi-page-digital.pdf",
multi_page_digital_pdf_file,
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -576,11 +576,11 @@ class TestParseMultiPage:
def test_multi_page_images_skip(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
multi_page_images_pdf_file: Path,
) -> None:
tesseract_parser.settings.mode = ModeChoices.AUTO
tesseract_parser.parse(
tesseract_samples_dir / "multi-page-images.pdf",
multi_page_images_pdf_file,
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -594,7 +594,7 @@ class TestParseMultiPage:
def test_multi_page_images_redo_pages_2(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
multi_page_images_pdf_file: Path,
) -> None:
"""
GIVEN:
@@ -609,7 +609,7 @@ class TestParseMultiPage:
tesseract_parser.settings.pages = 2
tesseract_parser.settings.mode = ModeChoices.REDO
tesseract_parser.parse(
tesseract_samples_dir / "multi-page-images.pdf",
multi_page_images_pdf_file,
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -622,7 +622,7 @@ class TestParseMultiPage:
def test_multi_page_images_force_page_1(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
multi_page_images_pdf_file: Path,
) -> None:
"""
GIVEN:
@@ -637,7 +637,7 @@ class TestParseMultiPage:
tesseract_parser.settings.pages = 1
tesseract_parser.settings.mode = ModeChoices.FORCE
tesseract_parser.parse(
tesseract_samples_dir / "multi-page-images.pdf",
multi_page_images_pdf_file,
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -733,7 +733,7 @@ class TestSkipArchive:
def test_skip_noarchive_with_text_layer(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
multi_page_digital_pdf_file: Path,
) -> None:
"""
GIVEN:
@@ -747,7 +747,7 @@ class TestSkipArchive:
"""
tesseract_parser.settings.mode = ModeChoices.AUTO
tesseract_parser.parse(
tesseract_samples_dir / "multi-page-digital.pdf",
multi_page_digital_pdf_file,
"application/pdf",
produce_archive=False,
)
@@ -762,7 +762,7 @@ class TestSkipArchive:
def test_skip_noarchive_image_only_creates_archive(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
multi_page_images_pdf_file: Path,
) -> None:
"""
GIVEN:
@@ -775,7 +775,7 @@ class TestSkipArchive:
"""
tesseract_parser.settings.mode = ModeChoices.AUTO
tesseract_parser.parse(
tesseract_samples_dir / "multi-page-images.pdf",
multi_page_images_pdf_file,
"application/pdf",
)
assert tesseract_parser.archive_path is not None
@@ -787,29 +787,29 @@ class TestSkipArchive:
)
@pytest.mark.parametrize(
("produce_archive", "filename", "expect_archive"),
("produce_archive", "sample_fixture", "expect_archive"),
[
pytest.param(
True,
"multi-page-digital.pdf",
"multi_page_digital_pdf_file",
True,
id="produce-archive-with-text",
),
pytest.param(
True,
"multi-page-images.pdf",
"multi_page_images_pdf_file",
True,
id="produce-archive-no-text",
),
pytest.param(
False,
"multi-page-digital.pdf",
"multi_page_digital_pdf_file",
False,
id="no-archive-with-text-layer",
),
pytest.param(
False,
"multi-page-images.pdf",
"multi_page_images_pdf_file",
False,
id="no-archive-no-text-layer",
),
@@ -818,10 +818,10 @@ class TestSkipArchive:
def test_produce_archive_flag(
self,
produce_archive: bool, # noqa: FBT001
filename: str,
sample_fixture: str,
expect_archive: bool, # noqa: FBT001
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
request: pytest.FixtureRequest,
) -> None:
"""
GIVEN:
@@ -834,8 +834,9 @@ class TestSkipArchive:
- Text is always extracted
"""
tesseract_parser.settings.mode = ModeChoices.AUTO
sample = request.getfixturevalue(sample_fixture)
tesseract_parser.parse(
tesseract_samples_dir / filename,
sample,
"application/pdf",
produce_archive=produce_archive,
)
@@ -852,7 +853,7 @@ class TestSkipArchive:
self,
mocker: MockerFixture,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
simple_digital_pdf_file: Path,
) -> None:
"""
GIVEN:
@@ -868,7 +869,7 @@ class TestSkipArchive:
tesseract_parser.settings.mode = ModeChoices.AUTO
mock_ocr = mocker.patch("ocrmypdf.ocr")
tesseract_parser.parse(
tesseract_samples_dir / "simple-digital.pdf",
simple_digital_pdf_file,
"application/pdf",
produce_archive=False,
)
@@ -907,7 +908,7 @@ class TestSkipArchive:
def test_tagged_pdf_produces_pdfa_archive_without_ocr(
self,
tesseract_parser: RasterisedDocumentParser,
tesseract_samples_dir: Path,
simple_digital_pdf_file: Path,
) -> None:
"""
GIVEN:
@@ -922,7 +923,7 @@ class TestSkipArchive:
"""
tesseract_parser.settings.mode = ModeChoices.AUTO
tesseract_parser.parse(
tesseract_samples_dir / "simple-digital.pdf",
simple_digital_pdf_file,
"application/pdf",
produce_archive=True,
)
@@ -1,19 +0,0 @@
<html>
<head>
<meta http-equiv="content-type" content="text/html; charset=UTF-8">
</head>
<body>
<p>Some Text</p>
<p>
<img src="cid:part1.pNdUSz0s.D3NqVtPg@example.de" alt="Has to be rewritten to work..">
<img src="http://localhost:8080/assets/logo_full_white.svg" alt="This image should not be shown.">
</p>
<p>and an embedded image.<br>
</p>
<p id="changeme">Paragraph unchanged.</p>
<scRipt>
document.getElementById("changeme").innerHTML = "Paragraph changed via Java Script.";
</script>
</body>
</html>
Binary file not shown.
Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.8 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 6.9 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 32 KiB

+4 -6
View File
@@ -16,8 +16,6 @@ from paperless.parsers.utils import read_file_handle_unicode_errors
if TYPE_CHECKING:
from pytest_mock import MockerFixture
SAMPLES = Path(__file__).parent / "samples" / "tesseract"
class TestReadFileHandleUnicodeErrors:
def test_plain_utf8(self, tmp_path: Path) -> None:
@@ -55,11 +53,11 @@ class TestReadFileHandleUnicodeErrors:
class TestIsTaggedPdf:
def test_tagged_pdf_returns_true(self) -> None:
assert is_tagged_pdf(SAMPLES / "simple-digital.pdf") is True
def test_tagged_pdf_returns_true(self, simple_digital_pdf_file: Path) -> None:
assert is_tagged_pdf(simple_digital_pdf_file) is True
def test_untagged_pdf_returns_false(self) -> None:
assert is_tagged_pdf(SAMPLES / "multi-page-images.pdf") is False
def test_untagged_pdf_returns_false(self, multi_page_images_pdf_file: Path) -> None:
assert is_tagged_pdf(multi_page_images_pdf_file) is False
def test_nonexistent_path_returns_false(self) -> None:
assert is_tagged_pdf(Path("/nonexistent/file.pdf")) is False
+48
View File
@@ -0,0 +1,48 @@
"""Test sample files shared across tests.
A sample used by a single app stays in that app's ``tests/samples`` tree.
Samples needed by more than one app, or stored under more than one name, live
here exactly once.
"""
import shutil
from pathlib import Path
SHARED_SAMPLES_DIR = (Path(__file__).parent / "sample_files").resolve()
SIMPLE_DIGITAL_PDF = SHARED_SAMPLES_DIR / "simple-digital.pdf"
MULTI_PAGE_DIGITAL_PDF = SHARED_SAMPLES_DIR / "multi-page-digital.pdf"
WITH_FORM_PDF = SHARED_SAMPLES_DIR / "with-form.pdf"
MULTI_PAGE_IMAGES_PDF = SHARED_SAMPLES_DIR / "multi-page-images.pdf"
THUMBNAIL_WEBP = SHARED_SAMPLES_DIR / "thumbnail.webp"
# The fake media tree under documents/tests/samples/documents keeps only the
# files unique to it. The files that duplicate a shared sample are laid back
# down under their historical opaque names (DB rows and exporter manifests
# refer to them) by install_document_samples().
_DOCUMENT_SAMPLES_DIR = (
Path(__file__).parent.parent / "documents" / "tests" / "samples" / "documents"
).resolve()
_SHARED_AS_MEDIA = (
("originals/0000001.pdf", SIMPLE_DIGITAL_PDF),
("originals/0000002.pdf", MULTI_PAGE_DIGITAL_PDF),
("originals/0000003.pdf", WITH_FORM_PDF),
("archive/0000001.pdf", MULTI_PAGE_IMAGES_PDF),
("thumbnails/0000001.webp", THUMBNAIL_WEBP),
("thumbnails/0000002.webp", THUMBNAIL_WEBP),
("thumbnails/0000003.webp", THUMBNAIL_WEBP),
("thumbnails/0000004.webp", THUMBNAIL_WEBP),
)
def install_document_samples(dest: Path) -> None:
"""Populate ``dest`` with the full fake media tree.
Replaces ``shutil.copytree(<samples>/documents, dest)`` in tests.
"""
shutil.copytree(_DOCUMENT_SAMPLES_DIR, dest, dirs_exist_ok=True)
for relpath, source in _SHARED_AS_MEDIA:
target = dest / relpath
target.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(source, target)