"""Gradio dashboard: the live Stateless state-machine graph is the centrepiece. Everything shown is folded from the harness's EVENT lines (see server.Live) and rendered server-side on a timer, so there is no custom JavaScript to keep in sync. The graph itself is drawn from the machine's own GetInfo() export, not hard-coded. """ import io, json from html import escape from pathlib import Path import gradio as gr import numpy as np from fastapi import HTTPException from PIL import Image COLORS = {"Queued": "#8b93a1", "Inferring": "#4f8cff", "Auditing": "#38bdf8", "Enforcing": "#a78bfa", "Correct": "#34d399", "WrongButSafe": "#2dd4bf", "Overblocked": "#fbbf24", "Unsafe": "#f87171", "Misclassified": "#fb923c", "Errored": "#f472b6"} STATES = list(COLORS) # order must match server.TEST_STATES (the compact per-test state codes) GROUPS = [("activated", ["Inferring", "Auditing", "Enforcing"]), ("passed", ["Correct", "WrongButSafe"]), ("failed", ["Overblocked", "Unsafe", "Misclassified", "Errored"]), ("queued", ["Queued"])] CRUMBS = ["Pending", "ModelReady", "SessionLoaded", "WarmedUp", "Measured", "Scored"] # Hand-placed layout for the known test machine. Unknown states fall into a spare row so a machine change never hides a node. POS = {"Queued": (60, 190), "Inferring": (200, 190), "Auditing": (340, 190), "Enforcing": (480, 190), "Correct": (670, 50), "WrongButSafe": (670, 120), "Overblocked": (670, 215), "Unsafe": (670, 285), "Misclassified": (670, 350), "Errored": (830, 285)} W, H = 118, 46 TXT = "var(--body-text-color)" MUT = "var(--body-text-color-subdued)" def graph_svg(graph, counts, edges): if not graph: return f'
Waiting for the harness to report its state machine — start a run from the Run tab.
' states = graph["states"] pos, spare = dict(POS), 0 for s in states: if s["id"] not in pos and not any(x["parent"] == s["id"] for x in states): pos[s["id"]] = (70 + 120 * spare, 400); spare += 1 boxes = {} for cid in {s["parent"] for s in states if s["parent"]}: kids = [k for k in states if k["parent"] == cid and k["id"] in pos] if kids: xs, ys = [pos[k["id"]][0] for k in kids], [pos[k["id"]][1] for k in kids] boxes[cid] = (min(xs) - W / 2 - 14, min(ys) - H / 2 - 24, max(xs) + W / 2 + 14, max(ys) + H / 2 + 14) def centre(i): if i in pos: return pos[i] b = boxes[i]; return ((b[0] + b[2]) / 2, (b[1] + b[3]) / 2) def anchor(i, toward): cx, cy = centre(i) w, h = (W, H) if i in pos else (boxes[i][2] - boxes[i][0], boxes[i][3] - boxes[i][1]) dx, dy = toward[0] - cx, toward[1] - cy k = min((w / 2) / max(abs(dx), 1e-9), (h / 2) / max(abs(dy), 1e-9)) return cx + dx * k, cy + dy * k out = ['', f'' f''] for cid, b in boxes.items(): label = "activated" if cid == "Active" else cid.lower() out.append(f'' f'{label.upper()}') mx = max([1, *edges.values()]) import math for e in graph["edges"]: if (e["from"] not in pos and e["from"] not in boxes) or (e["to"] not in pos and e["to"] not in boxes): continue a, b = anchor(e["from"], centre(e["to"])), anchor(e["to"], centre(e["from"])) c = edges.get(f'{e["from"]}>{e["to"]}', 0) if not c and e["from"] in boxes: # edge declared on a superstate: count its children's transitions c = sum(edges.get(f'{k["id"]}>{e["to"]}', 0) for k in states if k["parent"] == e["from"]) mxp, myp = (a[0] + b[0]) / 2, (a[1] + b[1]) / 2 - (0 if a[0] == b[0] else 8) width = 1 + 3 * math.log2(1 + c) / math.log2(1 + mx) tip = escape(f'{e["trigger"]}{" — " + e["guard"] if e.get("guard") else ""} ({c} observed)') out.append(f'{tip}') for s in states: if s["id"] not in pos or s["id"] in boxes: continue x, y = pos[s["id"]]; col = COLORS.get(s["id"], MUT) out.append(f'' f'{s["id"]}' f'{counts.get(s["id"], 0)}') out.append("") return "".join(out) def dataset_html(v, registry): """One card per registered dataset with live progress; the dataset being processed right now is highlighted. A test's suite comes from its domain's policy pack (packDomains) or is 'classification'; the registry entry whose `provides` lists that suite is its dataset. Tests run in file order (suite by suite), so the active dataset is the one holding the furthest started test.""" plan, states = v["plan"], v["states"] if not plan: return f'
No run yet.
' packs = plan.get("packDomains", {}) by_suite = {sn: e for e in registry for sn in e["provides"]} ds_of = [] for k in plan["keys"]: dom = k.split("|")[1] suite = packs.get(dom) if isinstance(packs.get(dom), str) else ("automation" if dom in packs else "classification") ds_of.append(by_suite.get(suite, {}).get("id", "?")) info = {e["id"]: e for e in registry} tot, done = {}, {} for i, d in enumerate(ds_of): tot[d] = tot.get(d, 0) + 1 if states[i] not in (0, 1, 2, 3): # not Queued / Inferring / Auditing / Enforcing done[d] = done.get(d, 0) + 1 started = [i for i in range(len(states)) if states[i] != 0] cur = ds_of[max(started)] if started else ds_of[0] measuring = v["runState"] == "Measured" # the run machine's state while tests are being processed cards = [] for e in registry: name = e["id"] if name not in tot: continue n, dn = tot[name], done.get(name, 0) is_active = measuring and name == cur state = "processing now" if is_active else ("done" if dn == n else "waiting") border = "#4f8cff" if is_active else "var(--border-color-primary)" glow = "box-shadow:0 0 0 3px #4f8cff55;" if is_active else "" badge = {"processing now": "background:#4f8cff;color:#fff", "done": "background:#34d399;color:#fff", "waiting": f"border:1px solid {MUT};color:{MUT}"}[state] cards.append( f'
' f'
{escape(e["title"])}' f'{state}
' f'
{escape(e.get("what", ""))}
' f'
{" · ".join(f"{escape(r.strip())}" for r in e.get("repo", "").split(",") if r.strip())} · ' f'suites: {", ".join(e["provides"])} · {escape(e.get("license", ""))}
' f'
' f'
{dn}/{n} tests
') head = (f'
Now processing: {escape(info[cur]["title"])} ' f'({", ".join(info[cur]["provides"])})
') if measuring and cur in info else "" return head + f'
{"".join(cards)}
' def checks_html(listing, results): """Cedar checks: status, summary, findings and the check-specific tables.""" if not listing: return f'
Check list unavailable.
' out = [] for c in listing: r = results.get(c["id"]) if r is None: badge, col = "not run", MUT elif r["passed"]: badge, col = ("PASS" if c["hard"] else "done"), "#34d399" else: badge, col = "FAIL", "#f87171" body = "" if r: d = r.get("details") if c["id"] == "conformance" and d: body += "" + "".join(f'' for x in d) + "
{escape(x["category"])}{x["pass"]}/{x["total"]}
" if c["id"] == "properties" and d: body += "" + "".join( f'' for x in d["properties"]) + "
propertychecksviolations
{escape(x["property"])}{x["checkedCount"]}{x["violations"]}
" if c["id"] == "bench" and d: body += "" + "".join( f'' for x in d) + "
domainpoliciescallsp50 msp95 ms
{escape(x["domain"])}{x["policies"]}{x["calls"]}{x["p50Ms"]}{x["p95Ms"]}
" if c["id"] == "noise-sweep" and d: body += ("" + "".join(f"" for x in d[0]["sweep"]) + "" + "".join(f'' + "".join(f'' for x in t["sweep"]) + "" for t in d) + "
domainsuite{int(x['rate'] * 100)}% noise: unsafe / overblocked
{escape(t["domain"])}{t["suite"]}{x["unsafePct"]}% / {x["overblockedPct"]}%
") if c["id"] == "gold-coverage" and d: body += "
domain / action outcomes on gold labels" + "".join( f'' for x in d) + "
pairallowdeny
{escape(x["pair"])}{x["allow"]}{x["deny"]}
" if c["id"] == "curation-sim" and d: for st in d: body += (f"
{escape(st['set'])}
" "" + "".join( f'' f'' for x in st["rows"]) + "
scenarioadmittedhumanselflabel noise %yield %errors caught %leaks
{escape(x["scenario"])}{x["admitted"]}{x["admittedHuman"]}{x["admittedSelf"]}{x["labelNoisePct"]}{x["yieldPct"]}{x["errorsCaughtPct"]}{x["leakNoConsent"] + x["leakNotOpen"] + x["leakPii"] + x["leakAttackSelf"]}
") if c["id"] == "promotion-gate" and d: body += "" + "".join( f'' for x in d) + "
candidatechampioncomparabledecisionby
{escape(x["candidate"])}{escape(x["champion"])}{x["comparable"]}{"allow" if x["allow"] else "deny"}{escape(", ".join(x["by"]))}
" if r["findings"]: body += "
findings
" out.append( f'
' f'
{escape(c["title"])}' f'{badge}{" · gate" if c["hard"] else ""}
' f'
{escape(c["description"])}
' + (f'
{escape(r["summary"])} ({r["seconds"]}s)
' if r else "") + body + "
") return "".join(out) def status_html(v, job): rs = v["runState"] or "Pending"; i = CRUMBS.index(rs) if rs in CRUMBS else -1 chips = [] for k, c in enumerate(CRUMBS): col = "#f87171" if rs == "Failed" else ("#34d399" if k < i else "#4f8cff" if k == i else MUT) fill = "background:" + col + ";color:#fff;" if (k == i and rs != "Failed") else "" chips.append(f'{c}') if rs == "Failed": chips.append('Failed') pol = "".join(f'' f'{"✓" if p["allow"] else "✗"} Cedar {p["action"]} ' f'{escape(", ".join(p["by"]) or "no permit matched")}' for p in v["runPolicy"]) line = "no job yet" if job: p, e = job.get("progress") or {}, job.get("eta") or {} unk = f" (+ unknown: {', '.join(e['queued_unknown'])})" if e.get("queued_unknown") else "" line = (f'{job["status"]} · {job.get("stage") or "idle"} · {p.get("phase", "")} {p.get("done", "")}/{p.get("total", "")} ' f'({job.get("percent", 0)}%) · mean {p.get("meanMs", "?")} ms · elapsed {round(job["elapsed_s"])}s · ' f'ETA this model {e.get("current_s", "?")}s · queued {e.get("queued_s", 0)}s{unk} · done: {", ".join(job["done_models"]) or "none"}') if job["errors"]: line += " · errors: " + " | ".join(f"{k}: {escape(str(x))}" for k, x in job["errors"].items()) pct = (job or {}).get("percent") or 0 return (f'
{escape(v["model"] or "")}
{"".join(chips)}
' f'
' f'
{line}
{pol}
') def legend_html(counts): parts = [] for g, ss in GROUPS: inner = " ".join(f' {s} {counts.get(s, 0)}' for s in ss) parts.append(f'{g} {sum(counts.get(s, 0) for s in ss)}  {inner}') return "".join(parts) CELL, GAP, COLS = 10, 2, 60 def grid_image(states, n, flagged=b""): """One cell per test, coloured by state. Returns (PIL image, cols) — cell (i) sits at column i % COLS, row i // COLS.""" if not n: return None rows = -(-n // COLS) step = CELL + GAP img = np.zeros((rows * step, COLS * step, 3), dtype=np.uint8) img[:] = (24, 26, 33) pal = [tuple(int(COLORS[s][k:k + 2], 16) for k in (1, 3, 5)) for s in STATES] for i in range(n): y, x = (i // COLS) * step, (i % COLS) * step img[y:y + CELL, x:x + CELL] = pal[states[i]] if i < len(flagged) and flagged[i]: # oracle-flagged: a white centre dot img[y + 3:y + CELL - 3, x + 3:x + CELL - 3] = (255, 255, 255) return Image.fromarray(img) def oracle_rows(rules): """Live view of the label-free oracle rules: how often each flags, how often a flag was a real error, and whether it also fires on gold.""" rows = [] for rule, o in sorted(rules.items(), key=lambda kv: -kv[1]["flagged"]): prec = f'{100 * o["tp"] / o["flagged"]:.0f}%' if o["flagged"] else "–" rows.append([rule, o["flagged"], prec, o["gold"]]) return rows def policy_rows(hits): return [[k, h["allow"], h["deny"]] for k, h in sorted(hits.items(), key=lambda kv: -(kv[1]["allow"] + kv[1]["deny"]))] def feed_rows(recent): rows = [] for r in reversed(recent): flips = "; ".join(f'{a["a"]}: model {"allow" if a["p"] else "deny"} / gold {"allow" if a["g"] else "deny"}' for a in (r.get("acts") or []) if a["p"] != a["g"]) heads = ", ".join(f'{h["t"]}: {h["p"]} ({h["c"]}%) vs {h["g"]}' for h in (r.get("heads") or [])) rows.append([r["i"], r["to"], r["k"], heads, flips or (r.get("err") or "")]) return rows def domain_rows(v): plan, states = v["plan"], v["states"] if not plan: return [] doms, packs = {}, plan.get("packDomains", {}) for i, k in enumerate(plan["keys"]): d = k.split("|")[1] row = doms.setdefault(d, [0] * len(STATES)) row[states[i]] += 1 return [[d, packs.get(d) if isinstance(packs.get(d), str) else ("automation" if d in packs else "classification"), *doms[d]] for d in sorted(doms)] def build_ui(srv): """Build the Blocks app against the server module `srv` (LIVE, jobs, start_job, ranking...).""" registry = srv.suite_registry() defaults = srv.policy_params() try: listing = srv.checks_endpoint()["checks"] except HTTPException: listing = [] def latest_job(): return list(srv._jobs.values())[-1] if srv._jobs else None def tick(): v = srv.LIVE.view() job = latest_job() snap = srv.snapshot(job) if job else None counts = v["counts"] img = grid_image(v["states"], v["plan"]["tests"], v["flagged"]) if v["plan"] else None g = v["graphs"].get("test") return (status_html(v, snap), dataset_html(v, registry), graph_svg(g, counts, v["edges"]), legend_html(counts), img, policy_rows(v["policyHits"]), feed_rows(v["recent"]), domain_rows(v), checks_html(listing, srv.CHECKS["results"]), oracle_rows(v["oracle"])) def inspect(i): try: i = int(i) except (TypeError, ValueError): return "enter a test index" with srv.LIVE.lock: ver = srv.LIVE.verdicts.get(i) return json.dumps(ver, indent=1) if ver else f"no verdict for test {i} (yet)" def on_select(evt: gr.SelectData): x, y = evt.index i = int(y // (CELL + GAP)) * COLS + int(x // (CELL + GAP)) return i, inspect(i) def start(model, all_models, skip_done, threads, limit, key, suites, *pvals): need = srv.os.environ.get("API_KEY") if need and key != need: return "API key required (Space secret API_KEY)." overrides = {k: int(v) for k, v in zip(defaults, pvals) if v is not None and int(v) != defaults[k]} try: models = list(srv.REGISTRY) if all_models else [model] r = srv.start_runs(models, int(threads), int(limit), 20, bool(skip_done), list(suites or []), overrides) return f"started: {json.dumps(r)}" except HTTPException as e: return f"not started: {e.detail}" def start_checks(selected, key): need = srv.os.environ.get("API_KEY") if need and key != need: return "API key required (Space secret API_KEY)." try: return f"started: {json.dumps(srv.run_checks_endpoint(srv.CheckRequest(checks=list(selected or []) or None)))}" except HTTPException as e: return f"not started: {e.detail}" def ranking(): try: return srv.ranking()["markdown"] except HTTPException as e: return f"ranking unavailable: {e.detail}" with gr.Blocks(title="FindAJev live", fill_width=True) as demo: gr.Markdown("# FindAJev — live test state machines with Cedar policy enforcement\n" "Each test is a [Stateless](https://github.com/dotnet-state-machine/stateless) machine; [Cedar](https://www.cedarpolicy.com/) " "decides every action with the model's labels and again with gold labels — a difference is a guardrail failure.") with gr.Tabs(): with gr.Tab("Live"): status = gr.HTML() datasets = gr.HTML() graph = gr.HTML() legend = gr.HTML() with gr.Row(): grid = gr.Image(label="tests (click a cell to inspect)", interactive=False, buttons=[], scale=3) pol = gr.Dataframe(headers=["Cedar policy (model labels)", "allow", "deny"], interactive=False, scale=2, max_height=420) gr.Markdown("**Oracle rules** — label-free Cedar audits that flag suspicious predictions (white dot in the grid). *flagged* = how often the rule fired; " "*precision* = share of flags that were real errors (compare with the base error rate); *on gold* = times the rule also fires on the gold labels " "(a sound rule ≈ 0).") oracle = gr.Dataframe(headers=["rule", "flagged", "precision", "on gold"], interactive=False, max_height=260) with gr.Tab("Failures & inspect"): feed = gr.Dataframe(headers=["test", "state", "key", "model vs gold", "decision flips"], interactive=False, max_height=420) with gr.Row(): idx = gr.Number(label="test index", precision=0, scale=1) btn = gr.Button("Inspect", scale=1) detail = gr.Code(label="verdict (labels, confidence, every Cedar decision with policy ids)", language="json") with gr.Tab("By domain"): dom = gr.Dataframe(headers=["domain", "suite", *STATES], interactive=False, max_height=600) with gr.Tab("Ranking"): rank = gr.Markdown() gr.Button("Refresh").click(ranking, outputs=rank) with gr.Tab("Cedar checks"): gr.Markdown("Tests of the Cedar policies and of Cedar itself — independent of any model. **Gate** checks run automatically before every " "benchmark job and must pass. They use CPU, so they share the single job slot with benchmark runs.") with gr.Row(): check_pick = gr.CheckboxGroup([(c["title"], c["id"]) for c in listing], value=[c["id"] for c in listing], label="checks (discovered from the harness)") check_key = gr.Textbox(label="API key", type="password") run_checks_btn = gr.Button("Run selected checks", variant="primary") check_out = gr.Textbox(label="result", interactive=False) checks_view = gr.HTML() run_checks_btn.click(start_checks, [check_pick, check_key], check_out) with gr.Tab("Run"): gr.Markdown(f"Runs are sequential (one at a time) on **{srv.CPUS} CPU(s)**. Models: {', '.join(srv.REGISTRY)}.") with gr.Row(): model = gr.Dropdown(list(srv.REGISTRY), value=list(srv.REGISTRY)[0], label="model") threads = gr.Number(value=srv.CPUS, precision=0, label="threads") limit = gr.Number(value=0, precision=0, label="limit (0 = all)") suite_pick = gr.CheckboxGroup([(f'{e["title"]} — {", ".join(e["provides"])}', e["id"]) for e in registry], value=[e["id"] for e in registry], label="suites (from suites.json)") with gr.Accordion("Cedar policy parameters (defaults from policies/params.json; a changed value gets its own ranking)", open=False): pnums = [gr.Number(value=v, precision=0, label=k) for k, v in defaults.items()] with gr.Row(): allm = gr.Checkbox(label="run every model") skip = gr.Checkbox(label="skip models that already have a result", value=True) key = gr.Textbox(label="API key", type="password") go = gr.Button("Start", variant="primary") out = gr.Textbox(label="result", interactive=False) go.click(start, [model, allm, skip, threads, limit, key, suite_pick, *pnums], out) timer = gr.Timer(0.5) timer.tick(tick, outputs=[status, datasets, graph, legend, grid, pol, feed, dom, checks_view, oracle]) grid.select(on_select, outputs=[idx, detail]) btn.click(inspect, idx, detail) demo.load(ranking, outputs=rank) return demo