"""Measure one first-load and cached retrieval per indexed source file.""" from contextlib import closing from datetime import datetime, timezone import json import platform import time import numpy as np from catalog import ROOT, PROBS from remote_catalog import RemoteCatalog def main(): catalog = RemoteCatalog() with closing(catalog.connect()) as conn: ids = [r[0] for r in conn.execute("SELECT min(id) FROM segments GROUP BY object_path ORDER BY object_path")] results = [] for index in ids: record = catalog.records[index] start = time.perf_counter() found = catalog.lookup(record["record_name"]) lookup_ms = (time.perf_counter() - start) * 1000 assert index in found table, cold = catalog.fetch(index) # Confirm that actual returned probability arrays agree with indexed coordinates. length = record["segment_end_bp"] - record["segment_start_bp"] for column in PROBS: values = table[column][0].values.to_numpy() assert len(values) == length assert np.isfinite(values).all() assert ((values >= 0) & (values <= 1)).all() cached, warm = catalog.fetch(index) assert warm["cache_hit"] and warm["bytes_read"] == 0 and table.equals(cached) start = time.perf_counter() frame, step = catalog.window(index, table=cached) plot_seconds = time.perf_counter() - start assert len(frame) <= 2400 result = {"accession": record["record_name"], "assembly": record["assembly_accession"], "organism": record["organism_name"], "bases": length, "lookup_ms": lookup_ms, "first_load": cold, "cached_load": warm, "plot_seconds": plot_seconds, "plot_bin_bp": step} results.append(result) print(json.dumps(result), flush=True) del table, cached, values report = {"created_at": datetime.now(timezone.utc).isoformat(), "environment": platform.system() + " / Python " + platform.python_version(), "note": "Workspace measurements; not Space latency. Empty app cache before each source's first load; upstream caches uncontrolled. Bytes count returned file ranges, excluding HTTP overhead.", "index_bytes": catalog.path.stat().st_size, "results": results} (ROOT / "data" / "benchmark.json").write_text(json.dumps(report, indent=2) + "\n") if __name__ == "__main__": main()