syntheogenesis / dee /core /resolution_cache.py
github-actions[bot]
Deploy c0f4b8a
6c0aef4
Raw
History Blame Contribute Delete
6.57 kB
"""Resolve TP53 once. Never pay for it again.
THE PROBLEM
-----------
Every gene-symbol and accession lookup goes out to Ensembl, UniProt or NCBI,
every time, for every user. `exon.py` has a process-local dict, but a Hugging
Face Space sleeps on idle and restarts cold, so in practice the same handful of
genes — TP53, GFP, Cas9, pUC19 — are re-fetched from scratch forever.
That costs three things: latency on the request a user is actually waiting on,
a shared outbound IP burning someone else's rate limit, and correctness — when
Ensembl is slow or throttling, a lookup that has succeeded a thousand times
fails, and the user is told their gene might not exist.
WHY THIS IS THE HONEST KIND OF "GETS SMARTER"
---------------------------------------------
It compounds with use and involves no model, no training and no inference. The
thousandth user asking for TP53 gets an instant answer *because* nine hundred
and ninety-nine people asked first. That is a real flywheel, and unlike a
learned one it cannot be wrong in a way nobody notices — the cached value is
byte-identical to what the database returned.
WHAT MAY AND MAY NOT BE CACHED
------------------------------
Only resolutions keyed by a PUBLIC identifier: (gene symbol, organism) or an
accession. Those keys are not personal, and the values are public database
records, which is why one shared cache across all users is correct rather than
a leak.
A pasted sequence is never cached. Not by content, not by hash, not as a key.
`kind == "sequence"` is refused at the door — see :func:`cacheable`. The
standing rule is that a user's own sequence never leaves the Space, and a
cross-user cache is very much leaving.
STALENESS
---------
Database records change: RefSeq versions increment, Ensembl re-annotates. So
entries carry a TTL and are re-fetched after it. The TTL is long because these
records are stable on the timescale of a design project, and a stale-by-a-week
CDS is a far smaller problem than the lookup failing outright.
"""
from __future__ import annotations
import re
import threading
import time
from collections import OrderedDict
from typing import Any, Dict, Optional
# Entries older than this are re-fetched. Sequence records are stable over
# weeks; this is about eventually noticing a re-annotation, not about
# freshness in any real-time sense.
TTL_SECONDS = 14 * 24 * 3600 # 14 days
MAX_ENTRIES = 2_000 # in-process ceiling; ~50 MB worst case
# Never cached. A pasted sequence is the user's own data and a shared cache is
# by definition cross-user.
UNCACHEABLE_KINDS = {"sequence", "empty", "unknown"}
_LOCK = threading.Lock()
_MEM: "OrderedDict[str, Dict[str, Any]]" = OrderedDict()
_STATS = {"hits": 0, "misses": 0, "stores": 0, "evictions": 0, "refused": 0}
def make_key(kind: str, identifier: str, organism: str = "") -> str:
"""Stable cache key. Case- and whitespace-insensitive.
Organism is part of the key because TP53 exists in dozens of species and
they are different sequences — a key that dropped it would serve human
TP53 to someone who asked for zebrafish, which is worse than a miss.
"""
ident = re.sub(r"\s+", "", (identifier or "")).upper()
org = re.sub(r"\s+", " ", (organism or "")).strip().lower()
return f"{(kind or '').lower()}|{ident}|{org}"
def cacheable(kind: str, result: Optional[Dict[str, Any]] = None) -> bool:
"""May a resolution of this kind be stored in a SHARED cache?
The gate that keeps user sequences out. Deliberately a whitelist-free
check on the one thing that disqualifies — a pasted sequence — rather than
a list of approved kinds, so a new public identifier type is cacheable the
day it is added instead of silently bypassing the cache forever.
"""
if (kind or "").lower() in UNCACHEABLE_KINDS:
return False
if result is not None and not result.get("ok"):
# A failure is not cached. Ensembl being down for ten seconds must not
# become "this gene does not exist" for the next fortnight.
return False
return True
def get(kind: str, identifier: str, organism: str = "") -> Optional[Dict[str, Any]]:
"""A previously resolved record, or None. Never raises."""
if not cacheable(kind):
return None
key = make_key(kind, identifier, organism)
now = time.time()
with _LOCK:
entry = _MEM.get(key)
if entry is None:
_STATS["misses"] += 1
return None
if now - entry["at"] > TTL_SECONDS:
_MEM.pop(key, None)
_STATS["misses"] += 1
return None
_MEM.move_to_end(key) # LRU: a hit is a recency signal
_STATS["hits"] += 1
# A copy, so a caller mutating the result cannot poison the cache for
# everyone else — the failure mode of a shared cache that is very hard
# to trace back later.
out = dict(entry["value"])
out["cached"] = True
out["cached_age_seconds"] = int(now - entry["at"])
return out
def put(kind: str, identifier: str, result: Dict[str, Any],
organism: str = "") -> bool:
"""Store a successful resolution. Returns whether it was stored."""
if not cacheable(kind, result):
with _LOCK:
_STATS["refused"] += 1
return False
if not (result or {}).get("sequence"):
return False
key = make_key(kind, identifier, organism)
# `cached` is a property of the read, not of the record. Storing it would
# make the first served copy claim it came from cache.
value = {k: v for k, v in result.items()
if k not in ("cached", "cached_age_seconds")}
with _LOCK:
_MEM[key] = {"value": value, "at": time.time()}
_MEM.move_to_end(key)
_STATS["stores"] += 1
while len(_MEM) > MAX_ENTRIES:
_MEM.popitem(last=False)
_STATS["evictions"] += 1
return True
def stats() -> Dict[str, Any]:
"""Hit rate and size. The number worth watching is `hit_rate`: it is the
whole claim that this compounds."""
with _LOCK:
total = _STATS["hits"] + _STATS["misses"]
return {
**_STATS,
"entries": len(_MEM),
"hit_rate": round(_STATS["hits"] / total, 3) if total else 0.0,
"ttl_days": TTL_SECONDS // 86400,
}
def clear() -> None:
"""Tests, and the admin path when a record is known to have changed."""
with _LOCK:
_MEM.clear()
for k in _STATS:
_STATS[k] = 0