Storing more stuff for stuff

This commit is contained in:
stumpylog
2026-07-22 14:17:27 -07:00
parent db14bd6b61
commit 1944208de4
3 changed files with 416 additions and 0 deletions
+73
View File
@@ -0,0 +1,73 @@
#!/usr/bin/env python
"""Inspect the LanceDB vector index: row count, schema, per-column sizes, and disk layout.
Usage (from repo root):
PAPERLESS_SECRET_KEY=x uv run python inspect_lancedb_index.py [index_path]
Default index_path: data/llm_index (relative to cwd) or the container-testing path.
"""
import sys
from pathlib import Path
import lancedb
INDEX_PATHS = [
"/tank/users/trenton/projects/container-testing/paperless-ngx/data/llm_index",
"data/llm_index",
]
path = (
sys.argv[1]
if len(sys.argv) > 1
else next(
(p for p in INDEX_PATHS if Path(p).exists()),
None,
)
)
if path is None or not Path(path).exists():
print(f"No index found. Pass path as argument or create one at: {INDEX_PATHS[0]}")
sys.exit(1)
print(f"Index path: {path}")
lance_dir = Path(path) / "documents.lance"
# Disk layout
print("\n--- Disk layout ---")
for subdir in ["data", "_versions", "_transactions", "_indices"]:
p = lance_dir / subdir
if p.exists():
size = sum(f.stat().st_size for f in p.rglob("*") if f.is_file())
print(f" {subdir:20s}: {size / 1024 / 1024:.1f} MB")
db = lancedb.connect(path)
tbl = db.open_table("documents")
total_rows = tbl.count_rows()
print("\n--- Table stats ---")
print(f" Rows: {total_rows}")
print(f" Schema: {tbl.schema}")
dim = tbl.schema.field("vector").type.list_size
print(
f"\n Vector dim: {dim}, float32 raw: {total_rows * dim * 4 / 1024 / 1024:.2f} MB",
)
print("\n--- In-memory column sizes (Arrow buffers) ---")
arrow_table = tbl.search().limit(total_rows).to_arrow()
print(f" {'TOTAL':20s}: {arrow_table.nbytes / 1024 / 1024:.2f} MB")
for i, field in enumerate(arrow_table.schema):
col = arrow_table.column(i)
print(f" {field.name:20s}: {col.nbytes / 1024 / 1024:.2f} MB")
# Sample node_content to see what's inside
print("\n--- node_content sample (first row, keys only) ---")
import json
sample = json.loads(arrow_table.column("node_content")[0].as_py())
print(f" Top-level keys: {list(sample.keys())}")
if "_node_content" in sample:
inner = json.loads(sample["_node_content"])
print(f" _node_content keys: {list(inner.keys())}")
if "metadata" in inner:
print(f" metadata keys: {list(inner['metadata'].keys())}")
+173
View File
@@ -0,0 +1,173 @@
"""
Temporary profiling utilities for comparing implementations.
Usage in a management command or shell::
from profiling import profile_block, profile_cpu, measure_memory
with profile_block("new check_sanity"):
messages = check_sanity()
with profile_block("old check_sanity"):
messages = check_sanity_old()
Drop this file when done.
"""
from __future__ import annotations
import resource
import tracemalloc
from collections.abc import Callable # noqa: TC003
from collections.abc import Generator # noqa: TC003
from contextlib import contextmanager
from time import perf_counter
from typing import Any
from django.db import connection
from django.db import reset_queries
from django.test.utils import override_settings
def _rss_kib() -> int:
"""Return current process RSS in KiB (Linux: /proc/self/status; fallback: getrusage)."""
try:
with open("/proc/self/status") as f:
for line in f:
if line.startswith("VmRSS:"):
return int(line.split()[1])
except OSError:
pass
# getrusage reports in KB on Linux, bytes on macOS
import sys
ru = resource.getrusage(resource.RUSAGE_SELF)
return ru.ru_maxrss if sys.platform != "darwin" else ru.ru_maxrss // 1024
@contextmanager
def profile_block(label: str = "block") -> Generator[None, None, None]:
"""Profile memory, wall time, and DB queries for a code block.
Prints a summary to stdout on exit. Requires no external packages.
Enables DEBUG temporarily to capture Django's query log.
Reports both Python-level (tracemalloc) and process-level (RSS) memory.
"""
rss_before = _rss_kib()
tracemalloc.start()
snapshot_before = tracemalloc.take_snapshot()
with override_settings(DEBUG=True):
reset_queries()
start = perf_counter()
yield
elapsed = perf_counter() - start
queries = list(connection.queries)
snapshot_after = tracemalloc.take_snapshot()
_, peak = tracemalloc.get_traced_memory()
tracemalloc.stop()
rss_after = _rss_kib()
# Compare snapshots for top allocations
stats = snapshot_after.compare_to(snapshot_before, "lineno")
query_time = sum(float(q["time"]) for q in queries)
mem_diff = sum(s.size_diff for s in stats)
print(f"\n{'=' * 60}") # noqa: T201
print(f" Profile: {label}") # noqa: T201
print(f"{'=' * 60}") # noqa: T201
print(f" Wall time: {elapsed:.4f}s") # noqa: T201
print(f" Queries: {len(queries)} ({query_time:.4f}s)") # noqa: T201
print(
f" RSS delta: {rss_after - rss_before:+d} KiB (before={rss_before} KiB, after={rss_after} KiB)",
)
print(f" Py mem delta: {mem_diff / 1024:.1f} KiB (tracemalloc — Python only)") # noqa: T201
print(f" Py peak: {peak / 1024:.1f} KiB") # noqa: T201
print("\n Top 5 allocations:") # noqa: T201
for stat in stats[:5]:
print(f" {stat}") # noqa: T201
print(f"{'=' * 60}\n") # noqa: T201
def profile_cpu(
fn: Callable[[], Any],
*,
label: str,
top: int = 30,
sort: str = "cumtime",
) -> tuple[Any, float]:
"""Run *fn()* under cProfile, print stats, return (result, elapsed_s).
Args:
fn: Zero-argument callable to profile.
label: Human-readable label printed in the header.
top: Number of cProfile rows to print.
sort: cProfile sort key (default: cumulative time).
Returns:
``(result, elapsed_s)`` where *result* is the return value of *fn()*.
"""
import cProfile
import io
import pstats
pr = cProfile.Profile()
t0 = perf_counter()
pr.enable()
result = fn()
pr.disable()
elapsed = perf_counter() - t0
buf = io.StringIO()
ps = pstats.Stats(pr, stream=buf).sort_stats(sort)
ps.print_stats(top)
print(f"\n{'=' * 72}") # noqa: T201
print(f" {label}") # noqa: T201
print(f" wall time: {elapsed * 1000:.1f} ms") # noqa: T201
print(f"{'=' * 72}") # noqa: T201
print(buf.getvalue()) # noqa: T201
return result, elapsed
def measure_memory(fn: Callable[[], Any], *, label: str) -> tuple[Any, float, float]:
"""Run *fn()* under tracemalloc, print allocation report.
Args:
fn: Zero-argument callable to profile.
label: Human-readable label printed in the header.
Returns:
``(result, peak_kib, delta_kib)``.
"""
tracemalloc.start()
snapshot_before = tracemalloc.take_snapshot()
t0 = perf_counter()
result = fn()
elapsed = perf_counter() - t0
snapshot_after = tracemalloc.take_snapshot()
_, peak = tracemalloc.get_traced_memory()
tracemalloc.stop()
stats = snapshot_after.compare_to(snapshot_before, "lineno")
delta_kib = sum(s.size_diff for s in stats) / 1024
print(f"\n{'=' * 72}") # noqa: T201
print(f" [memory] {label}") # noqa: T201
print(f" wall time: {elapsed * 1000:.1f} ms") # noqa: T201
print(f" memory delta: {delta_kib:+.1f} KiB") # noqa: T201
print(f" peak traced: {peak / 1024:.1f} KiB") # noqa: T201
print(f"{'=' * 72}") # noqa: T201
print(" Top allocation sites (by size_diff):") # noqa: T201
for stat in stats[:20]:
if stat.size_diff != 0:
print( # noqa: T201
f" {stat.size_diff / 1024:+8.1f} KiB {stat.traceback.format()[0]}",
)
return result, peak / 1024, delta_kib