carbon-a-database-explorer / benchmark_remote.py
cgeorgiaw's picture
cgeorgiaw HF Staff
Add SQLite accession index and bounded on-demand bucket retrieval
f0190da verified
Raw History Blame Contribute Delete
2.48 kB
"""Measure one first-load and cached retrieval per indexed source file."""
from contextlib import closing
from datetime import datetime, timezone
import json
import platform
import time
import numpy as np
from catalog import ROOT, PROBS
from remote_catalog import RemoteCatalog
def main():
catalog = RemoteCatalog()
with closing(catalog.connect()) as conn:
ids = [r[0] for r in conn.execute("SELECT min(id) FROM segments GROUP BY object_path ORDER BY object_path")]
results = []
for index in ids:
record = catalog.records[index]
start = time.perf_counter()
found = catalog.lookup(record["record_name"])
lookup_ms = (time.perf_counter() - start) * 1000
assert index in found
table, cold = catalog.fetch(index)
# Confirm that actual returned probability arrays agree with indexed coordinates.
length = record["segment_end_bp"] - record["segment_start_bp"]
for column in PROBS:
values = table[column][0].values.to_numpy()
assert len(values) == length
assert np.isfinite(values).all()
assert ((values >= 0) & (values <= 1)).all()
cached, warm = catalog.fetch(index)
assert warm["cache_hit"] and warm["bytes_read"] == 0 and table.equals(cached)
start = time.perf_counter()
frame, step = catalog.window(index, table=cached)
plot_seconds = time.perf_counter() - start
assert len(frame) <= 2400
result = {"accession": record["record_name"], "assembly": record["assembly_accession"],
"organism": record["organism_name"], "bases": length,
"lookup_ms": lookup_ms, "first_load": cold, "cached_load": warm,
"plot_seconds": plot_seconds, "plot_bin_bp": step}
results.append(result)
print(json.dumps(result), flush=True)
del table, cached, values
report = {"created_at": datetime.now(timezone.utc).isoformat(),
"environment": platform.system() + " / Python " + platform.python_version(),
"note": "Workspace measurements; not Space latency. Empty app cache before each source's first load; upstream caches uncontrolled. Bytes count returned file ranges, excluding HTTP overhead.",
"index_bytes": catalog.path.stat().st_size, "results": results}
(ROOT / "data" / "benchmark.json").write_text(json.dumps(report, indent=2) + "\n")
if __name__ == "__main__":
main()