coding-router / router.json
kfallah's picture
v1.2-traffic: router.json
a2cddfe verified
Raw
History Blame Contribute Delete
6.63 kB
{
"version": "v1.2-traffic",
"kind": "trained",
"hf_repo": "experiential-labs/coding-router",
"embed_model_mlx": "encoder-mlx-4bit",
"embed_model_torch": "encoder-fp16",
"n_tasks": 207,
"n_repos": 88,
"embed_backend": "local",
"T": 0.05453708011655199,
"lam": 0.1,
"sim_floor": 0.4699,
"fallback_arm_index": 5,
"arms": [
"fable5_xhigh",
"luna_high",
"luna_low",
"luna_max",
"luna_medium",
"opus5_high",
"sol_xhigh",
"sonnet5_high",
"terra_high",
"terra_max"
],
"arm_spec": {
"fable5_xhigh": {
"model": "claude-fable-5",
"effort": "xhigh",
"provider": "anthropic",
"request_kwargs": {
"model": "claude-fable-5",
"thinking": {
"type": "adaptive"
},
"output_config": {
"effort": "xhigh"
}
}
},
"luna_high": {
"model": "gpt-5.6-luna",
"effort": "high",
"provider": "openai",
"request_kwargs": {
"model": "gpt-5.6-luna",
"reasoning": {
"effort": "high"
}
}
},
"luna_low": {
"model": "gpt-5.6-luna",
"effort": "low",
"provider": "openai",
"request_kwargs": {
"model": "gpt-5.6-luna",
"reasoning": {
"effort": "low"
}
}
},
"luna_max": {
"model": "gpt-5.6-luna",
"effort": "max",
"provider": "openai",
"request_kwargs": {
"model": "gpt-5.6-luna",
"reasoning": {
"effort": "max"
}
}
},
"luna_medium": {
"model": "gpt-5.6-luna",
"effort": "medium",
"provider": "openai",
"request_kwargs": {
"model": "gpt-5.6-luna",
"reasoning": {
"effort": "medium"
}
}
},
"opus5_high": {
"model": "claude-opus-5",
"effort": "high",
"provider": "anthropic",
"request_kwargs": {
"model": "claude-opus-5",
"thinking": {
"type": "adaptive"
},
"output_config": {
"effort": "high"
}
}
},
"sol_xhigh": {
"model": "gpt-5.6-sol",
"effort": "xhigh",
"provider": "openai",
"request_kwargs": {
"model": "gpt-5.6-sol",
"reasoning": {
"effort": "xhigh"
}
}
},
"sonnet5_high": {
"model": "claude-sonnet-5",
"effort": "high",
"provider": "anthropic",
"request_kwargs": {
"model": "claude-sonnet-5",
"thinking": {
"type": "adaptive"
},
"output_config": {
"effort": "high"
}
}
},
"terra_high": {
"model": "gpt-5.6-terra",
"effort": "high",
"provider": "openai",
"request_kwargs": {
"model": "gpt-5.6-terra",
"reasoning": {
"effort": "high"
}
}
},
"terra_max": {
"model": "gpt-5.6-terra",
"effort": "max",
"provider": "openai",
"request_kwargs": {
"model": "gpt-5.6-terra",
"reasoning": {
"effort": "max"
}
}
}
},
"provenance": "EXP-012 winning recipe `reward_lcb_b0.2`: Qwen3-Embedding-0.6B with LoRA r=16 (q/k/v/o projections) trained by exact expected reward -E_pi[graded - 3.0*cost] on the LiveCodeBench 7x76 matrix, KL(pi||pi_init) anchor beta=0.2, 150 steps, checkpoint step 50 (T=0.0545), lam=0.01 -- (step, lam) selected ONLY on the six standard seeds' inner 75/25 DeepSWE-train splits (feasibility inner_graded >= inner-best-arm - 0.01, then max mean cost ratio; feasible on 5/6 inner splits). Memory/calibration side: the full 110-task DeepSWE v1.1 evidence (41 arms x 110 tasks, 88 repos), bank embedded with the merged fp16 encoder via sentence-transformers. Sweep evidence (per-seed selection, 6 seeds): selected-holdout mean graded 0.9484 at $60.17/split -- +0.015 graded and 2.1x cheaper than the deployable always-best-train baseline (0.9336/$126.28); parity, NOT better, vs the hindsight-best static arm. Cost figures are matrix-based (June 2026 collection) and pending live re-benchmark per EXP-014 (live drift measured). | v1.1 REBUILD (2026-08-01): bank cells replaced with the EXP-015 LIVE DeepSWE matrix (1,130 trials, dense 10 arms x 113 tasks; the 3 tasks without bank texts dropped -> 10x110) because the published June matrix is quality-stale live (e.g. luna_max 0.946 published vs 0.687 live). Roster 41 -> 10 live arms; med_cost/fallback from live cells (opus5_high, live f2p 0.951); tuned encoder, bank embeddings and T unchanged; lambda re-selected on inner train-side splits only (0.005). | v1.2-traffic (2026-08-01): union bank = DeepSWE-live 110 (unchanged cells) + 65 LCB-live + 32 utility tasks, all arms measured LIVE this lane (LCB via the E2B harness on the 92-task covered subset; utility = 32 short interactive asks with deterministic verifiers on 4 pinned OSS repos). graded = harness fraction; med_cost per arm = median of live per-task costs pooled across the three sources. sim_floor = LOO-nearest p10 over the union bank. lam re-selected train-side only under a 50/30/20 utility/LCB/DeepSWE traffic weighting.",
"scope_warning": "Validated on repo-issue statements (DeepSWE), competitive-programming agent tasks (LCB) and SHORT interactive asks (utility suite: bugfix/feature/refactor/question, 40-250 chars) \u2014 short prompts now land in-distribution instead of abstaining. Abstention (escalate to strongest arm) is reserved for text unlike any bank task; always check Decision.off_distribution.",
"selection": {
"lam_grid": [
{
"lam": 0.001,
"feas": 6,
"mean_ratio": 1.081393772849211,
"mean_inner_graded": 0.7660379605122892
},
{
"lam": 0.002,
"feas": 6,
"mean_ratio": 1.1000667545641503,
"mean_inner_graded": 0.7659905447036618
},
{
"lam": 0.005,
"feas": 6,
"mean_ratio": 1.2811509404979988,
"mean_inner_graded": 0.765745203615449
},
{
"lam": 0.01,
"feas": 6,
"mean_ratio": 1.7326780876087458,
"mean_inner_graded": 0.7656267731370289
},
{
"lam": 0.02,
"feas": 6,
"mean_ratio": 2.452558818623846,
"mean_inner_graded": 0.7645313997014641
},
{
"lam": 0.05,
"feas": 6,
"mean_ratio": 3.2512255254226896,
"mean_inner_graded": 0.7644669476479534
},
{
"lam": 0.1,
"feas": 6,
"mean_ratio": 3.5954684774287116,
"mean_inner_graded": 0.7617608602098715
},
{
"lam": 0.2,
"feas": 5,
"mean_ratio": 3.9614547647054508,
"mean_inner_graded": 0.7608279933293333
},
{
"lam": 0.5,
"feas": 5,
"mean_ratio": 4.002295380963825,
"mean_inner_graded": 0.7593800766626666
},
{
"lam": 1.0,
"feas": 5,
"mean_ratio": 4.061556689358738,
"mean_inner_graded": 0.7589185461017406
}
],
"selected_lam": 0.1,
"policy": "6 seeded group-stratified splits, inner train-side only; metric weights sources 50% utility / 30% LCB / 20% DeepSWE (assumed traffic mix, documented); encoder/T frozen"
}
}