File size: 2,484 Bytes
f0190da
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
"""Measure one first-load and cached retrieval per indexed source file."""
from contextlib import closing
from datetime import datetime, timezone
import json
import platform
import time

import numpy as np

from catalog import ROOT, PROBS
from remote_catalog import RemoteCatalog


def main():
    catalog = RemoteCatalog()
    with closing(catalog.connect()) as conn:
        ids = [r[0] for r in conn.execute("SELECT min(id) FROM segments GROUP BY object_path ORDER BY object_path")]
    results = []
    for index in ids:
        record = catalog.records[index]
        start = time.perf_counter()
        found = catalog.lookup(record["record_name"])
        lookup_ms = (time.perf_counter() - start) * 1000
        assert index in found
        table, cold = catalog.fetch(index)
        # Confirm that actual returned probability arrays agree with indexed coordinates.
        length = record["segment_end_bp"] - record["segment_start_bp"]
        for column in PROBS:
            values = table[column][0].values.to_numpy()
            assert len(values) == length
            assert np.isfinite(values).all()
            assert ((values >= 0) & (values <= 1)).all()
        cached, warm = catalog.fetch(index)
        assert warm["cache_hit"] and warm["bytes_read"] == 0 and table.equals(cached)
        start = time.perf_counter()
        frame, step = catalog.window(index, table=cached)
        plot_seconds = time.perf_counter() - start
        assert len(frame) <= 2400
        result = {"accession": record["record_name"], "assembly": record["assembly_accession"],
                  "organism": record["organism_name"], "bases": length,
                  "lookup_ms": lookup_ms, "first_load": cold, "cached_load": warm,
                  "plot_seconds": plot_seconds, "plot_bin_bp": step}
        results.append(result)
        print(json.dumps(result), flush=True)
        del table, cached, values
    report = {"created_at": datetime.now(timezone.utc).isoformat(),
              "environment": platform.system() + " / Python " + platform.python_version(),
              "note": "Workspace measurements; not Space latency. Empty app cache before each source's first load; upstream caches uncontrolled. Bytes count returned file ranges, excluding HTTP overhead.",
              "index_bytes": catalog.path.stat().st_size, "results": results}
    (ROOT / "data" / "benchmark.json").write_text(json.dumps(report, indent=2) + "\n")


if __name__ == "__main__":
    main()