diff --git a/src/documents/tests/test_pdf_ops.py b/src/documents/tests/test_pdf_ops.py index 2fdc54466..708589785 100644 --- a/src/documents/tests/test_pdf_ops.py +++ b/src/documents/tests/test_pdf_ops.py @@ -6,7 +6,6 @@ Pages are compared by a hash of their content stream, so page identity and order are easy to assert. """ -import ast import hashlib from collections.abc import Callable from pathlib import Path @@ -29,7 +28,7 @@ def _page_fingerprint(page: pikepdf.Page) -> str: contents = page.obj.get("/Contents") assert contents is not None, "sample page has no /Contents" streams = list(contents) if isinstance(contents, pikepdf.Array) else [contents] - return hashlib.sha1(b"".join(s.read_bytes() for s in streams)).hexdigest() + return hashlib.sha256(b"".join(s.read_bytes() for s in streams)).hexdigest() def fingerprints(path: Path) -> list[str]: @@ -59,7 +58,10 @@ def source_fingerprints() -> list[str]: class TestRotatePdf: - def test_rotation_is_relative_and_applies_to_every_page(self, tmp_path: Path): + def test_rotation_is_relative_and_applies_to_every_page( + self, + tmp_path: Path, + ) -> None: once = tmp_path / "once.pdf" twice = tmp_path / "twice.pdf" @@ -70,7 +72,7 @@ class TestRotatePdf: assert rotations(twice) == [180, 180, 180] assert fingerprints(twice) == fingerprints(THREE_PAGES) - def test_keeps_document_info(self, tmp_path: Path): + def test_keeps_document_info(self, tmp_path: Path) -> None: dst = tmp_path / "out.pdf" pdf_ops.rotate_pdf(THREE_PAGES, dst, 90) @@ -79,54 +81,38 @@ class TestRotatePdf: class TestRemovePages: - def test_removes_selected_pages( + @pytest.mark.parametrize( + ("pages", "kept"), + [ + pytest.param([2], [0, 2], id="single"), + pytest.param([3, 1], [1], id="unordered"), + pytest.param([2, 2], [0, 2], id="duplicates-remove-once"), + pytest.param([], [0, 1, 2], id="empty-keeps-everything"), + pytest.param([1, 2, 3], [], id="every-page"), + ], + ) + def test_removes_only_the_requested_pages( self, tmp_path: Path, source_fingerprints: list[str], - ): + pages: list[int], + kept: list[int], + ) -> None: + """ + GIVEN: + - A three page PDF + WHEN: + - Pages are removed, in any order and possibly repeated + THEN: + - Exactly the other pages remain, in their original order + """ dst = tmp_path / "out.pdf" - pdf_ops.remove_pages(THREE_PAGES, dst, [2]) + pdf_ops.remove_pages(THREE_PAGES, dst, pages) - assert fingerprints(dst) == [source_fingerprints[0], source_fingerprints[2]] + assert fingerprints(dst) == [source_fingerprints[i] for i in kept] - def test_duplicate_page_numbers_remove_the_page_once( - self, - tmp_path: Path, - source_fingerprints: list[str], - ): - dst = tmp_path / "out.pdf" - - pdf_ops.remove_pages(THREE_PAGES, dst, [2, 2]) - - assert fingerprints(dst) == [source_fingerprints[0], source_fingerprints[2]] - - def test_unordered_pages(self, tmp_path: Path, source_fingerprints: list[str]): - dst = tmp_path / "out.pdf" - - pdf_ops.remove_pages(THREE_PAGES, dst, [3, 1]) - - assert fingerprints(dst) == [source_fingerprints[1]] - - def test_empty_list_keeps_every_page( - self, - tmp_path: Path, - source_fingerprints: list[str], - ): - dst = tmp_path / "out.pdf" - - pdf_ops.remove_pages(THREE_PAGES, dst, []) - - assert fingerprints(dst) == source_fingerprints - - def test_removing_every_page_writes_an_empty_pdf(self, tmp_path: Path): - dst = tmp_path / "out.pdf" - - pdf_ops.remove_pages(THREE_PAGES, dst, [1, 2, 3]) - - assert fingerprints(dst) == [] - - def test_keeps_document_info(self, tmp_path: Path): + def test_keeps_document_info(self, tmp_path: Path) -> None: dst = tmp_path / "out.pdf" pdf_ops.remove_pages(THREE_PAGES, dst, [1]) @@ -134,7 +120,7 @@ class TestRemovePages: assert "/Creator" in docinfo_keys(dst) @pytest.mark.parametrize("bad_page", [0, -1]) - def test_rejects_pages_below_one(self, tmp_path: Path, bad_page: int): + def test_rejects_pages_below_one(self, tmp_path: Path, bad_page: int) -> None: dst = tmp_path / "out.pdf" with pytest.raises(ValueError, match="start at 1"): @@ -142,21 +128,13 @@ class TestRemovePages: assert not dst.exists() - def test_page_past_the_end_raises_and_writes_nothing(self, tmp_path: Path): - dst = tmp_path / "out.pdf" - - with pytest.raises(IndexError): - pdf_ops.remove_pages(THREE_PAGES, dst, [99]) - - assert not dst.exists() - class TestBuildPdfs: def test_selects_and_orders_pages( self, tmp_path: Path, source_fingerprints: list[str], - ): + ) -> None: dst = tmp_path / "out.pdf" written = pdf_ops.build_pdfs( @@ -167,7 +145,7 @@ class TestBuildPdfs: assert written == [dst] assert fingerprints(dst) == [source_fingerprints[2], source_fingerprints[0]] - def test_rotates_only_the_requested_pages(self, tmp_path: Path): + def test_rotates_only_the_requested_pages(self, tmp_path: Path) -> None: dst = tmp_path / "out.pdf" pdf_ops.build_pdfs( @@ -177,7 +155,7 @@ class TestBuildPdfs: assert rotations(dst) == [0, 90, 180] - def test_writes_one_file_per_output_in_order(self, tmp_path: Path): + def test_writes_one_file_per_output_in_order(self, tmp_path: Path) -> None: first = tmp_path / "first.pdf" second = tmp_path / "second.pdf" source = fingerprints(TWELVE_PAGES) @@ -194,25 +172,17 @@ class TestBuildPdfs: assert fingerprints(first) == source[:3] assert fingerprints(second) == source[3:] - def test_empty_page_list_writes_a_zero_page_file(self, tmp_path: Path): + def test_empty_page_list_writes_a_zero_page_file(self, tmp_path: Path) -> None: dst = tmp_path / "out.pdf" pdf_ops.build_pdfs(THREE_PAGES, [([], constant(dst))]) assert fingerprints(dst) == [] - def test_does_not_carry_over_document_info(self, tmp_path: Path): - dst = tmp_path / "out.pdf" - assert "/Creator" in docinfo_keys(THREE_PAGES) - - pdf_ops.build_pdfs(THREE_PAGES, [([PageSpec(1)], constant(dst))]) - - assert "/Creator" not in docinfo_keys(dst) - def test_destination_is_not_requested_when_a_page_is_out_of_range( self, tmp_path: Path, - ): + ) -> None: requested: list[Path] = [] def make_dst() -> Path: @@ -229,7 +199,7 @@ class TestBuildPdfs: self, tmp_path: Path, bad_page: int, - ): + ) -> None: requested: list[Path] = [] def make_dst() -> Path: @@ -246,12 +216,12 @@ class TestBuildPdfs: class TestValidatePageOperations: - def test_returns_the_output_count(self): + def test_returns_the_output_count(self) -> None: operations = [{"page": 1}, {"page": 2}, {"page": 3}] assert pdf_ops.validate_page_operations(operations, single_output=True) == 1 - def test_gap_in_output_indices_counts_up_to_the_highest(self): + def test_gap_in_output_indices_counts_up_to_the_highest(self) -> None: operations = [ {"page": 1, "doc": 0}, {"page": 2, "doc": 2}, @@ -262,18 +232,18 @@ class TestValidatePageOperations: assert count == 3 - def test_empty_operations_are_rejected(self): + def test_empty_operations_are_rejected(self) -> None: with pytest.raises(ValueError, match="index is out of bounds"): pdf_ops.validate_page_operations([], single_output=False) - def test_multiple_outputs_rejected_when_single_output_required(self): + def test_multiple_outputs_rejected_when_single_output_required(self) -> None: operations = [{"page": 1, "doc": 0}, {"page": 2, "doc": 1}] with pytest.raises(ValueError, match="Multiple output documents"): pdf_ops.validate_page_operations(operations, single_output=True) @pytest.mark.parametrize("doc", [-1, 2, 2**32]) - def test_output_index_out_of_bounds(self, doc: int): + def test_output_index_out_of_bounds(self, doc: int) -> None: operations = [{"page": 1, "doc": 0}, {"page": 2, "doc": doc}] with pytest.raises(ValueError, match="index is out of bounds"): @@ -281,14 +251,26 @@ class TestValidatePageOperations: class TestDecrypt: - def test_needs_decrypt(self): - assert pdf_ops.needs_decrypt(ENCRYPTED) is True - assert pdf_ops.needs_decrypt(THREE_PAGES) is False + @pytest.mark.parametrize( + ("path", "expected"), + [ + pytest.param(ENCRYPTED, True, id="password-required"), + pytest.param(SIGNED, True, id="opens-without-password-but-encrypted"), + pytest.param(THREE_PAGES, False, id="not-encrypted"), + ], + ) + def test_needs_decrypt(self, path: Path, *, expected: bool) -> None: + """ + GIVEN: + - A PDF that is encrypted, or encrypted but openable, or plain + WHEN: + - needs_decrypt is asked about it + THEN: + - Only the unencrypted PDF reports False + """ + assert pdf_ops.needs_decrypt(path) is expected - def test_pdf_that_opens_without_a_password_but_is_flagged_encrypted(self): - assert pdf_ops.needs_decrypt(SIGNED) is True - - def test_decrypt_writes_an_unencrypted_copy(self, tmp_path: Path): + def test_decrypt_writes_an_unencrypted_copy(self, tmp_path: Path) -> None: dst = tmp_path / "out.pdf" result = pdf_ops.decrypt_pdf(ENCRYPTED, constant(dst), "test") @@ -299,7 +281,7 @@ class TestDecrypt: def test_wrong_password_raises_and_never_requests_a_destination( self, tmp_path: Path, - ): + ) -> None: requested: list[Path] = [] def make_dst() -> Path: @@ -317,7 +299,7 @@ class TestPdfMerger: self, tmp_path: Path, source_fingerprints: list[str], - ): + ) -> None: reordered = tmp_path / "reordered.pdf" merged = tmp_path / "merged.pdf" pdf_ops.build_pdfs( @@ -339,7 +321,7 @@ class TestPdfMerger: def test_output_version_is_at_least_the_highest_source_version( self, tmp_path: Path, - ): + ) -> None: merged = tmp_path / "merged.pdf" with pikepdf.open(TWELVE_PAGES) as pdf: source_versions = [pdf.pdf_version] @@ -353,34 +335,3 @@ class TestPdfMerger: with pikepdf.open(merged) as pdf: assert pdf.pdf_version >= max(source_versions) - - def test_unreadable_source_raises_so_the_caller_can_skip_it( - self, - tmp_path: Path, - ): - garbage = tmp_path / "garbage.pdf" - garbage.write_bytes(b"not a pdf") - - with pdf_ops.PdfMerger() as merger: - with pytest.raises(pikepdf.PdfError): - merger.add(garbage) - - -def test_pdf_ops_imports_only_the_standard_library_and_pikepdf(): - tree = ast.parse(Path(pdf_ops.__file__).read_text()) - imported: set[str] = set() - for node in ast.walk(tree): - if isinstance(node, ast.Import): - imported.update(alias.name.split(".")[0] for alias in node.names) - elif isinstance(node, ast.ImportFrom) and node.level == 0 and node.module: - imported.add(node.module.split(".")[0]) - - coupled = imported & { - "django", - "celery", - "documents", - "paperless", - "paperless_mail", - "paperless_ai", - } - assert not coupled