| <!doctype html> |
| <html lang="en"> |
| <head> |
| <meta charset="utf-8" /> |
| <meta name="viewport" content="width=device-width, initial-scale=1" /> |
| <title>DarwinX: Evolving Agent Harnesses Through Natural Selection</title> |
| <meta name="description" content="DarwinX treats agent self-evolution as selection over a population of harnesses with the model frozen. 84.7% on Terminal-Bench 2.1, 68.3% on held-out TerminalWorld, 93.0% audit-clean on WebArena-Infinity, 84.2% zero-shot transfer to SWE-bench Verified." /> |
| <meta property="og:type" content="article" /> |
| <meta property="og:title" content="DarwinX: Evolving Agent Harnesses Through Natural Selection" /> |
| <meta property="og:description" content="Population selection over agent harnesses with the model frozen. Terminal-Bench 2.1 frontier at 84.7%." /> |
| <meta property="og:image" content="assets/teaser.png" /> |
| <meta name="twitter:card" content="summary_large_image" /> |
| <meta name="twitter:title" content="DarwinX: Evolving Agent Harnesses Through Natural Selection" /> |
| <meta name="twitter:description" content="A frozen model need not be a fixed agent: harness selection turns evaluation compute into durable capability." /> |
| <meta name="twitter:image" content="assets/teaser.png" /> |
| <link rel="icon" href="data:image/svg+xml,<svg xmlns='http://www.w3.org/2000/svg' viewBox='0 0 32 32'><rect width='32' height='32' rx='7' fill='%23032d60'/><path d='M12.6 14.8 19.4 10.6M12.6 17.2 19.4 21.4' stroke='%23fff' stroke-width='1.7'/><circle cx='10' cy='16' r='3.3' fill='%230176d3'/><circle cx='22' cy='9' r='3.3' fill='%23fff'/><circle cx='22' cy='23' r='3.3' fill='%23fff'/></svg>" /> |
| <style> |
| :root { |
| --navy: #032d60; |
| --blue: #0176d3; |
| --ink: #17202c; |
| --muted: #5a6b7d; |
| --line: #dfe6ee; |
| --tint: #f2f7fc; |
| --bg: #ffffff; |
| --maxw: 920px; |
| --navh: 48px; |
| } |
| * { box-sizing: border-box; } |
| html { -webkit-text-size-adjust: 100%; scroll-behavior: smooth; } |
| body { |
| margin: 0; |
| background: var(--bg); |
| color: var(--ink); |
| font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", "Helvetica Neue", Arial, sans-serif; |
| font-size: 17px; |
| line-height: 1.65; |
| } |
| .wrap { max-width: var(--maxw); margin: 0 auto; padding: 0 22px; } |
| a { color: var(--blue); text-decoration: none; } |
| a:hover { text-decoration: underline; } |
| |
| |
| nav.top { |
| position: sticky; top: 0; z-index: 50; |
| background: rgba(255,255,255,0.94); |
| backdrop-filter: saturate(180%) blur(8px); |
| border-bottom: 1px solid var(--line); |
| height: var(--navh); |
| } |
| nav.top .wrap { display: flex; align-items: center; gap: 4px; height: 100%; overflow-x: auto; } |
| nav.top .brand { font-weight: 700; color: var(--navy); margin-right: 12px; white-space: nowrap; } |
| nav.top a { |
| color: var(--muted); font-size: 0.85rem; padding: 5px 9px; border-radius: 5px; white-space: nowrap; |
| } |
| nav.top a:hover { background: var(--tint); color: var(--navy); text-decoration: none; } |
| nav.top .spacer { flex: 1 1 auto; } |
| nav.top a.cta { color: #fff; background: var(--navy); font-weight: 500; } |
| nav.top a.cta:hover { background: #06407f; color: #fff; } |
| section[id] { scroll-margin-top: calc(var(--navh) + 14px); } |
| |
| |
| header { padding: 52px 0 34px; border-bottom: 1px solid var(--line); } |
| h1 { |
| font-size: 2.05rem; line-height: 1.25; letter-spacing: -0.015em; |
| color: var(--navy); margin: 0 0 20px; font-weight: 700; |
| } |
| h1 .sub { display: block; font-size: 1.32rem; font-weight: 600; margin-top: 6px; color: var(--ink); } |
| .authors { font-size: 1.02rem; margin: 0 0 8px; } |
| .authors b { font-weight: 600; } |
| sup { font-size: 0.7em; color: var(--blue); } |
| .affil, .legend { color: var(--muted); font-size: 0.9rem; margin: 4px 0 0; } |
| .legend { margin-top: 12px; font-size: 0.84rem; } |
| |
| .links { display: flex; flex-wrap: wrap; gap: 10px; margin: 26px 0 0; } |
| .btn { |
| display: inline-flex; align-items: center; padding: 9px 17px; border-radius: 7px; |
| background: var(--navy); color: #fff; font-size: 0.94rem; font-weight: 500; |
| border: 1px solid var(--navy); cursor: pointer; |
| } |
| .btn:hover { background: #06407f; text-decoration: none; } |
| .btn.ghost { background: #fff; color: var(--navy); } |
| .btn.ghost:hover { background: var(--tint); } |
| |
| |
| section { padding: 42px 0; border-bottom: 1px solid var(--line); } |
| section:last-of-type { border-bottom: 0; } |
| h2 { |
| font-size: 0.83rem; text-transform: uppercase; letter-spacing: 0.1em; |
| color: var(--muted); font-weight: 700; margin: 0 0 20px; |
| } |
| h3 { font-size: 1.16rem; color: var(--navy); margin: 0 0 4px; font-weight: 600; letter-spacing: -0.01em; } |
| h3 .tag { font-weight: 400; color: var(--muted); font-size: 0.8rem; letter-spacing: 0.02em; } |
| h4 { font-size: 1rem; color: var(--ink); margin: 0 0 5px; font-weight: 600; } |
| p { margin: 0 0 15px; } |
| p:last-child { margin-bottom: 0; } |
| |
| figure { margin: 0; } |
| figure img { width: 100%; height: auto; display: block; border: 1px solid var(--line); border-radius: 8px; background: #fff; } |
| figure a.zoom { display: block; position: relative; } |
| figure a.zoom:hover img { border-color: var(--blue); } |
| figure a.zoom::after { |
| content: "\2317"; position: absolute; right: 9px; bottom: 9px; |
| font-size: 0.85rem; line-height: 1; padding: 5px 7px; border-radius: 5px; |
| background: rgba(3,45,96,0.82); color: #fff; opacity: 0; transition: opacity .15s; |
| } |
| figure a.zoom:hover::after { opacity: 1; } |
| figcaption { color: var(--muted); font-size: 0.87rem; margin-top: 10px; line-height: 1.55; } |
| .figrow { display: grid; grid-template-columns: 1fr 1fr; gap: 20px; align-items: start; } |
| .abstract { font-size: 1.02rem; } |
| |
| |
| .cards { display: grid; grid-template-columns: repeat(2, 1fr); gap: 14px; } |
| .card { border: 1px solid var(--line); border-radius: 8px; padding: 17px 18px; background: var(--tint); } |
| .card .label { font-size: 0.79rem; color: var(--muted); text-transform: uppercase; letter-spacing: 0.06em; font-weight: 600; } |
| .card .big { font-size: 1.75rem; font-weight: 700; color: var(--navy); margin: 7px 0 2px; letter-spacing: -0.02em; } |
| .card .big .delta { font-size: 0.9rem; font-weight: 600; color: var(--blue); margin-left: 7px; letter-spacing: 0; } |
| .card .note { font-size: 0.85rem; color: var(--muted); line-height: 1.5; } |
| |
| |
| .bench { padding-top: 34px; margin-top: 34px; border-top: 1px solid var(--line); } |
| .bench:first-of-type { padding-top: 0; margin-top: 0; border-top: 0; } |
| .bench > .lede { color: var(--muted); font-size: 0.93rem; margin: 0 0 20px; } |
| .block { margin-top: 28px; } |
| |
| table { width: 100%; border-collapse: collapse; font-size: 0.94rem; } |
| caption { text-align: left; color: var(--muted); font-size: 0.87rem; padding-bottom: 10px; line-height: 1.5; } |
| th, td { text-align: left; padding: 8px 9px; border-bottom: 1px solid var(--line); vertical-align: top; } |
| th { font-size: 0.78rem; text-transform: uppercase; letter-spacing: 0.04em; color: var(--muted); border-bottom: 1.5px solid #c9d4e0; font-weight: 700; } |
| td.num, th.num { text-align: right; font-variant-numeric: tabular-nums; } |
| tr.ours td { background: var(--tint); font-weight: 600; color: var(--navy); } |
| tr.base td { color: var(--muted); font-style: italic; } |
| tr.total td { border-top: 1.5px solid #c9d4e0; font-weight: 700; } |
| td.delta { color: var(--blue); font-weight: 600; } |
| .tnote { color: var(--muted); font-size: 0.85rem; margin-top: 11px; line-height: 1.55; } |
| .scroller { overflow-x: auto; } |
| .scroller table { min-width: 640px; } |
| code { |
| font-family: ui-monospace, SFMono-Regular, "SF Mono", Menlo, monospace; |
| font-size: 0.85em; background: #eef3f8; padding: 1px 5px; border-radius: 4px; color: var(--navy); |
| white-space: nowrap; |
| } |
| |
| ul.mech { list-style: none; padding: 0; margin: 0; } |
| ul.mech li { padding: 0 0 18px 0; } |
| ul.mech li:last-child { padding-bottom: 0; } |
| ul.lim { padding-left: 20px; margin: 0; } |
| ul.lim li { padding-bottom: 12px; } |
| ul.lim li:last-child { padding-bottom: 0; } |
| ul.lim b { color: var(--navy); } |
| |
| |
| .widget { border: 1px solid var(--line); border-radius: 9px; padding: 18px; background: #fcfdfe; } |
| .widget .wtop { |
| display: flex; flex-wrap: wrap; align-items: center; gap: 8px 14px; |
| padding-bottom: 15px; margin-bottom: 15px; border-bottom: 1px solid var(--line); |
| } |
| .widget .wtop .wlabel { font-size: 0.78rem; text-transform: uppercase; letter-spacing: 0.06em; color: var(--muted); font-weight: 700; } |
| .chips { display: flex; flex-wrap: wrap; gap: 7px; } |
| .chip { |
| font: inherit; font-size: 0.85rem; padding: 5px 12px; border-radius: 20px; |
| border: 1px solid var(--line); background: #fff; color: var(--muted); cursor: pointer; |
| transition: background .12s, border-color .12s, color .12s; |
| } |
| .chip:hover { border-color: #b9c7d6; } |
| .chip[aria-pressed="true"] { background: var(--navy); border-color: var(--navy); color: #fff; } |
| .chip .cnt { opacity: 0.62; font-variant-numeric: tabular-nums; } |
| .toggle { display: inline-flex; align-items: center; gap: 7px; font-size: 0.86rem; color: var(--muted); cursor: pointer; } |
| .toggle input { accent-color: var(--navy); width: 15px; height: 15px; } |
| |
| .readout { display: flex; flex-wrap: wrap; gap: 10px; margin-bottom: 16px; } |
| .ro { |
| flex: 1 1 150px; border: 1px solid var(--line); border-radius: 7px; |
| padding: 11px 13px; background: #fff; |
| } |
| .ro .k { font-size: 0.75rem; text-transform: uppercase; letter-spacing: 0.05em; color: var(--muted); font-weight: 600; } |
| .ro .v { font-size: 1.35rem; font-weight: 700; color: var(--navy); font-variant-numeric: tabular-nums; letter-spacing: -0.01em; } |
| .ro.hi { background: var(--tint); border-color: #c9dcf0; } |
| .ro .v small { font-size: 0.62em; font-weight: 600; color: var(--muted); } |
| |
| .tgrid { display: grid; grid-template-columns: repeat(auto-fill, minmax(26px, 1fr)); gap: 5px; } |
| .tcell { |
| position: relative; aspect-ratio: 1; border-radius: 4px; background: #e6ecf2; |
| border: 1px solid transparent; cursor: default; transition: background .12s; |
| } |
| .tcell.on { background: var(--blue); } |
| .tcell.gain { box-shadow: 0 0 0 2px #e8a33d; z-index: 1; } |
| .tcell.loss { box-shadow: 0 0 0 2px #d05a4e; z-index: 1; } |
| .tcell:hover { border-color: var(--navy); } |
| |
| .wlegend { display: flex; flex-wrap: wrap; gap: 14px; margin-top: 14px; font-size: 0.8rem; color: var(--muted); } |
| .wlegend span { display: inline-flex; align-items: center; gap: 6px; } |
| .sw { width: 13px; height: 13px; border-radius: 3px; display: inline-block; } |
| |
| #tip { |
| position: fixed; z-index: 100; pointer-events: none; opacity: 0; |
| background: var(--navy); color: #fff; font-size: 0.78rem; line-height: 1.45; |
| padding: 7px 10px; border-radius: 6px; max-width: 260px; |
| transition: opacity .1s; box-shadow: 0 4px 14px rgba(3,45,96,0.25); |
| } |
| #tip b { color: #9ecbf5; font-weight: 600; } |
| |
| |
| .cbars { display: flex; flex-direction: column; gap: 9px; } |
| .cbar { display: grid; grid-template-columns: minmax(120px, 1.5fr) 3fr auto; gap: 12px; align-items: center; } |
| .cbar .cname { font-size: 0.86rem; color: var(--ink); } |
| .cbar .cname .cn { color: var(--muted); font-size: 0.78rem; } |
| .cbar .track { position: relative; height: 26px; background: #f0f4f8; border-radius: 4px; overflow: hidden; } |
| .cbar .fill { position: absolute; top: 0; height: 100%; border-radius: 4px; transition: width .35s ease, left .35s ease; } |
| .cbar .fill.b { background: #c6d3e0; } |
| .cbar .fill.e { background: var(--blue); opacity: 0.92; } |
| .cbar .cd { font-size: 0.84rem; font-weight: 700; font-variant-numeric: tabular-nums; min-width: 52px; text-align: right; } |
| .cbar .cd.up { color: var(--blue); } |
| .cbar .cd.down { color: #b3453a; } |
| |
| |
| .chart { width: 100%; height: auto; display: block; overflow: visible; } |
| .chart .ax { stroke: #ccd7e2; stroke-width: 1; } |
| .chart .gl { stroke: #eef2f6; stroke-width: 1; } |
| .chart .lbl { font-size: 10.5px; fill: var(--muted); } |
| .chart .best { fill: none; stroke: var(--blue); stroke-width: 2.2; } |
| .chart .dot { fill: #b9c7d6; } |
| .chart .dot:hover, .chart .dot.act { fill: var(--navy); } |
| .chart .hit { fill: transparent; cursor: pointer; } |
| |
| |
| .panels { display: grid; grid-template-columns: repeat(4, 1fr); gap: 12px; } |
| .panel { text-align: center; } |
| .panel .pt { font-size: 0.95rem; font-weight: 700; color: var(--navy); } |
| .panel .ps { font-size: 0.74rem; color: var(--muted); line-height: 1.4; margin-bottom: 6px; } |
| .panel svg { width: 100%; height: auto; display: block; overflow: visible; } |
| .panel .bar { transition: y .35s ease, height .35s ease; } |
| .panel .bar.b { fill: #c6d3e0; } |
| .panel .bar.p { fill: #8fb4e8; } |
| .panel .bar.e { fill: var(--blue); } |
| .panel .dl { font-size: 11.5px; font-weight: 700; fill: var(--blue); } |
| .panel .tick { font-size: 9.5px; fill: var(--muted); } |
| .panel .axl { stroke: #ccd7e2; stroke-width: 1; } |
| .panel .foot { font-size: 0.72rem; color: var(--muted); margin-top: 4px; } |
| |
| |
| .hrows { display: flex; flex-direction: column; gap: 8px; } |
| .hrow { display: grid; grid-template-columns: minmax(96px, 1.1fr) 3fr auto; gap: 12px; align-items: center; } |
| .hrow .hn { font-size: 0.86rem; } |
| .hrow .ht { position: relative; height: 24px; background: #f0f4f8; border-radius: 4px; } |
| .hrow .hf { position: absolute; left: 0; top: 0; height: 100%; border-radius: 4px; transition: width .35s ease; } |
| .hrow .hv { font-size: 0.84rem; font-weight: 700; font-variant-numeric: tabular-nums; min-width: 68px; text-align: right; color: var(--navy); } |
| .hrow.dim .hn, .hrow.dim .hv { color: var(--muted); } |
| .refline { position: absolute; top: -3px; bottom: -3px; width: 0; border-left: 2px dashed #e4572e; } |
| .reftag { position: absolute; top: -17px; transform: translateX(-50%); font-size: 0.7rem; color: #e4572e; white-space: nowrap; } |
| |
| |
| .stack { display: flex; height: 30px; border-radius: 5px; overflow: hidden; background: #f0f4f8; } |
| .stack i { display: block; height: 100%; transition: width .35s ease; } |
| .srow { margin-bottom: 16px; } |
| .srow .sh { display: flex; justify-content: space-between; font-size: 0.82rem; color: var(--muted); margin-bottom: 5px; } |
| .srow .sh b { color: var(--navy); font-size: 0.95rem; } |
| .keys { display: flex; flex-wrap: wrap; gap: 6px 14px; margin-top: 12px; font-size: 0.78rem; color: var(--muted); } |
| .keys span { display: inline-flex; align-items: center; gap: 6px; } |
| |
| |
| .dumb { display: flex; flex-direction: column; gap: 7px; } |
| .drow { display: grid; grid-template-columns: minmax(84px, 1fr) 4fr; gap: 12px; align-items: center; } |
| .drow.tot { padding-bottom: 7px; border-bottom: 1px solid var(--line); } |
| .drow.tot .dn { font-weight: 700; color: var(--navy); } |
| .drow .dn { font-size: 0.82rem; } |
| .drow .dt { position: relative; height: 30px; |
| background-image: repeating-linear-gradient(to right, #edf1f5 0 1px, transparent 1px 25%); } |
| .drow.dax { align-items: end; } |
| .drow.dax .dt { height: 16px; background: none; } |
| .drow.dax .dt span { position: absolute; bottom: 0; transform: translateX(-50%); |
| font-size: 0.7rem; color: var(--muted); } |
| .drow .seg { position: absolute; height: 3px; border-radius: 2px; } |
| .drow .pt2 { position: absolute; width: 9px; height: 9px; border-radius: 50%; transform: translate(-50%, -50%); } |
| .drow .hollow { background: #fff; } |
| |
| .nojs { display: none; } |
| .nojs img { width: 100%; height: auto; border: 1px solid var(--line); border-radius: 8px; } |
| |
| th.sortable { cursor: pointer; user-select: none; white-space: nowrap; } |
| th.sortable:hover { color: var(--navy); } |
| th.sortable::after { content: " \2195"; opacity: 0.35; font-size: 0.9em; } |
| th.sortable.asc::after { content: " \2191"; opacity: 1; color: var(--blue); } |
| th.sortable.desc::after { content: " \2193"; opacity: 1; color: var(--blue); } |
| |
| .bibhead { display: flex; align-items: center; justify-content: space-between; gap: 12px; margin-bottom: 12px; } |
| .bibhead h2 { margin: 0; } |
| #copybtn { |
| font: inherit; font-size: 0.84rem; padding: 6px 13px; border-radius: 6px; |
| border: 1px solid var(--line); background: #fff; color: var(--navy); cursor: pointer; |
| } |
| #copybtn:hover { background: var(--tint); border-color: #c9d4e0; } |
| pre { |
| background: #f7f9fb; border: 1px solid var(--line); border-radius: 8px; |
| padding: 16px; overflow-x: auto; font-size: 0.85rem; line-height: 1.55; |
| font-family: ui-monospace, SFMono-Regular, "SF Mono", Menlo, monospace; |
| color: #223; margin: 0; |
| } |
| footer { padding: 30px 0 60px; color: var(--muted); font-size: 0.88rem; } |
| |
| @media (max-width: 700px) { |
| body { font-size: 16px; } |
| h1 { font-size: 1.6rem; } |
| header { padding: 32px 0 26px; } |
| .cards, .figrow { grid-template-columns: 1fr; } |
| nav.top .hideable { display: none; } |
| .panels { grid-template-columns: repeat(2, 1fr); gap: 18px 10px; } |
| .hrow { grid-template-columns: minmax(72px, 1fr) 2fr auto; gap: 8px; } |
| .hrow .hn, .drow .dn { font-size: 0.78rem; } |
| .hrow .hv { min-width: 54px; font-size: 0.78rem; } |
| .drow { grid-template-columns: minmax(66px, 1fr) 3fr; gap: 8px; } |
| } |
| </style> |
| </head> |
| <body> |
| <noscript><style>.widget{display:none}.nojs{display:block}</style></noscript> |
|
|
| <nav class="top"> |
| <div class="wrap"> |
| <span class="brand">DarwinX</span> |
| <a href="#abstract">Abstract</a> |
| <a href="#method">Method</a> |
| <a href="#results">Results</a> |
| <a href="#benchmarks" class="hideable">Benchmarks</a> |
| <a href="#ablation" class="hideable">Ablation</a> |
| <a href="#limitations" class="hideable">Limitations</a> |
| <a href="#bibtex" class="hideable">BibTeX</a> |
| <span class="spacer"></span> |
| <a class="cta" href="https://arxiv.org/abs/2608.07545">arXiv</a> |
| </div> |
| </nav> |
|
|
| <header> |
| <div class="wrap"> |
| <h1>DarwinX<span class="sub">Evolving Agent Harnesses Through Natural Selection</span></h1> |
|
|
| <p class="authors"> |
| <b>Yifan Zhang</b><sup>◦1</sup>, |
| <b>Yutong Dai</b><sup>◦1</sup>, |
| Juntao Tan<sup>*1</sup>, |
| Luyu Yang<sup>*1</sup>, |
| Rishi Mullur<sup>1</sup>, |
| Thai Hoang<sup>1</sup>, |
| Zhiyuan Hu<sup>1</sup>, |
| James Zhu<sup>†2</sup>, |
| Phil Mui<sup>†2</sup>, |
| Silvio Savarese<sup>†1</sup>, |
| Ran Xu<sup>†1</sup>, |
| Zeyuan Chen<sup>†1</sup> |
| </p> |
| <p class="affil"><sup>1</sup>Salesforce AI Research <sup>2</sup>Salesforce Agentforce</p> |
| <p class="legend"><sup>◦</sup>First authors <sup>*</sup>Core authors <sup>†</sup>Senior authors</p> |
|
|
| <div class="links"> |
| <a class="btn" href="https://arxiv.org/abs/2608.07545">arXiv:2608.07545</a> |
| <a class="btn ghost" href="https://arxiv.org/pdf/2608.07545">PDF</a> |
| <a class="btn ghost" href="https://huggingface.co/papers/2608.07545">Hugging Face</a> |
| <a class="btn ghost" href="mailto:yifan.zhang@salesforce.com?subject=DarwinX%20code%20request">Code available on request</a> |
| </div> |
| </div> |
| </header> |
|
|
| <section> |
| <div class="wrap"> |
| <figure> |
| <a class="zoom" href="assets/full/teaser.png" target="_blank" rel="noopener"> |
| <img src="assets/teaser_schematic.png" alt="Variants are evolved from a base agent, survivors are kept and merged, pruned lineages are dropped, all with the model frozen." /> |
| </a> |
| <figcaption> |
| <b>With the base model frozen, evolving the harness alone matches or beats the strongest prior |
| agent on four benchmarks.</b> Variants survive on measured fitness (avg@<i>k</i>, no gold solutions) |
| and complementary survivors are merged. Click for the paper's full figure. |
| </figcaption> |
| </figure> |
|
|
| <div class="block"> |
| <div class="widget" id="hl-widget"> |
| <div class="wtop"> |
| <span class="wlabel">y-axis</span> |
| <div class="chips" id="hl-chips"></div> |
| <span style="flex:1"></span> |
| <label class="toggle"><input type="checkbox" id="hl-prior" checked /> best prior agent</label> |
| </div> |
| <div class="panels" id="hl-panels"></div> |
| <div class="wlegend" id="hl-legend"></div> |
| <p class="tnote" id="hl-note"></p> |
| </div> |
| <noscript class="nojs"><img src="assets/teaser.png" alt="Four-benchmark summary: base vs. evolved vs. best prior agent." /></noscript> |
| </div> |
| </div> |
| </section> |
|
|
| <section id="abstract"> |
| <div class="wrap"> |
| <h2>Abstract</h2> |
| <div class="abstract"> |
| <p> |
| An LLM agent's capability depends not only on model weights but on its <i>harness</i>: prompts, |
| tools, skills, and control flow. Self-improvement loops already edit harnesses, yet single-lineage |
| search is path-dependent and local wins often regress other tasks. We introduce <b>DarwinX</b>, which |
| treats self-evolution as <b>selection over a population of harnesses</b> with the model frozen: a |
| preserve-and-extend contract admits only variants that extend coverage without regressing, an archive |
| keeps alternative lineages for recombination, and failure-, teacher-, and self-derived evidence share |
| one edit interface. Fitness comes from each benchmark's own verifier: no gold solutions, no |
| hand-picked winners. |
| </p> |
| <p> |
| Across four benchmarks that progressively separate the evolution signal from the test, one loop adds |
| about 17 points on average: Terminal-Bench 2.1 rises +7.7 to 83.2% on a matched base and to the |
| verified frontier at <b>84.7%</b> on a stronger one; TerminalWorld's held-out split reaches |
| <b>68.3%</b>, ahead of every off-the-shelf agent; WebArena-Infinity real-task pass@1 rises from 43.5% |
| to <b>93.0%</b> audit-clean; and a Terminal-Bench 2.1 harness transfers unchanged to SWE-bench |
| Verified. What evolves is general agent competence, not benchmark-specific patches, so it survives |
| changes of task, verifier, and base model. A frozen model need not be a fixed agent: harness selection |
| turns evaluation compute into durable capability. |
| </p> |
| </div> |
| </div> |
| </section> |
|
|
| <section id="method"> |
| <div class="wrap"> |
| <h2>Method</h2> |
|
|
| <figure> |
| <a class="zoom" href="assets/full/overview.png" target="_blank" rel="noopener"> |
| <img src="assets/overview.png" alt="Overview of the DarwinX loop." loading="lazy" /> |
| </a> |
| </figure> |
|
|
| <ul class="mech" style="margin-top:26px"> |
| <li> |
| <h4>Selection over a population, not one lineage</h4> |
| <p>A run maintains an archive shaped as a tree, where each node is a harness snapshot with its edit |
| delta, per-task scores, trial evidence, and distilled lessons. Nothing is discarded: a variant that |
| loses overall may still hold the one edit that, combined with another branch's, unlocks a task |
| neither solves alone. Parallel branches target different capability clusters, so the archive grows |
| specialists with different solved-task signatures.</p> |
| </li> |
| <li> |
| <h4>A preserve-and-extend contract</h4> |
| <p>A child is admitted only when the measured evidence shows it solved something new |
| <i>without breaking what it already solved</i>, which bounds what a win is allowed to cost |
| elsewhere. When variants solve complementary tasks, their additive edits are merged, and the merged |
| child is kept only if it covers the union of its parents' wins. Fitness is the benchmark's own |
| verifier at avg@<i>k</i> — no gold solutions and no hand-picked winners.</p> |
| </li> |
| <li> |
| <h4>One edit interface for three signals</h4> |
| <p>Failure-derived diagnosis, teacher-derived demonstrations, and self-derived rollout contrast are |
| all converted into harness edits rather than model-weight updates. The base model never changes, so |
| every gain reported here is a statement about the harness alone.</p> |
| </li> |
| </ul> |
|
|
| <figure style="margin-top:30px"> |
| <a class="zoom" href="assets/full/population.png" target="_blank" rel="noopener"> |
| <img src="assets/population.png" alt="DarwinX per-generation operators: mutation loop, variant classification, and the merge operator." loading="lazy" /> |
| </a> |
| <figcaption> |
| <b>DarwinX's per-generation operators.</b> <b>Left:</b> the mutation loop and the three learning |
| signals that drive it. <b>Middle:</b> variants classified by how their solved set changes, where those |
| preserving inherited solves stay eligible for recombination while the rest contribute only distilled |
| lessons. <b>Right:</b> the merge operator and its acceptance criterion. |
| </figcaption> |
| </figure> |
| </div> |
| </section> |
|
|
| <section id="results"> |
| <div class="wrap"> |
| <h2>Results at a glance</h2> |
| <div class="cards"> |
| <div class="card"> |
| <div class="label">Terminal-Bench 2.1 · avg@5</div> |
| <div class="big">84.7%<span class="delta">+7.7 on matched base</span></div> |
| <div class="note">Verified-leaderboard frontier on a frozen GPT-5.6 Sol at <i>medium</i> effort. |
| On GPT-5.5, 75.5% → 83.2%.</div> |
| </div> |
| <div class="card"> |
| <div class="label">TerminalWorld · held-out split</div> |
| <div class="big">68.3%<span class="delta">+7.3</span></div> |
| <div class="note">28/41 held-out tasks on a frozen Opus 4.8, above every off-the-shelf agent |
| we measured.</div> |
| </div> |
| <div class="card"> |
| <div class="label">WebArena-Infinity · audit-clean pass@1</div> |
| <div class="big">93.0%<span class="delta">+49.5 pp</span></div> |
| <div class="note">1,260 real tasks, evolved on synthetic intents only. Base 43.5%; best same-model |
| baseline 86.1%.</div> |
| </div> |
| <div class="card"> |
| <div class="label">SWE-bench Verified · zero-shot transfer</div> |
| <div class="big">84.2%<span class="delta">+3.4</span></div> |
| <div class="note">421/500 official pass@1 from the Terminal-Bench 2.1 harness run |
| <i>unchanged</i>, with no SWE-V feedback.</div> |
| </div> |
| </div> |
| <p class="tnote"> |
| The four benchmarks progressively separate the evolution signal from the test: in-domain |
| (Terminal-Bench 2.1), held-out task split (TerminalWorld), synthetic → real intent |
| shift (WebArena-Infinity), and zero-shot cross-benchmark transfer (SWE-bench Verified). |
| </p> |
| </div> |
| </section> |
|
|
| <section id="benchmarks"> |
| <div class="wrap"> |
| <h2>Benchmark detail</h2> |
|
|
| |
| <div class="bench"> |
| <h3>Terminal-Bench 2.1 <span class="tag">— in-domain evolution</span></h3> |
| <p class="lede">89 tasks. Evolution and evaluation share the task distribution; the model is frozen and |
| only the harness changes.</p> |
|
|
| <table class="sortable"> |
| <caption>Terminal-Bench 2.1, avg@5 under the strict leaderboard rule (binary avg@5, |
| <i>k</i>=5, errored trials score zero). Both DarwinX rows are leaderboard submissions, reported |
| before the leaderboard's uniform reward-hacking pass.</caption> |
| <thead> |
| <tr><th>Agent</th><th>Model / effort</th><th class="num">avg@5</th></tr> |
| </thead> |
| <tbody> |
| <tr class="ours"><td>Monet (DarwinX)</td><td>GPT-5.6 Sol / medium</td><td class="num">84.7 ± 1.2</td></tr> |
| <tr><td>Claude Code</td><td>Fable 5 / xhigh</td><td class="num">83.8 ± 1.2</td></tr> |
| <tr class="ours"><td>Monet (DarwinX)</td><td>GPT-5.5 / high</td><td class="num">83.2 ± 1.2</td></tr> |
| <tr><td>Codex</td><td>GPT-5.5 / xhigh</td><td class="num">83.1 ± 1.1</td></tr> |
| <tr><td>OpenAI reference</td><td>GPT-5.6 Sol / medium</td><td class="num">81.8</td></tr> |
| <tr><td>Terminus 2</td><td>GPT-5.5 / xhigh</td><td class="num">78.0 ± 1.2</td></tr> |
| <tr class="base"><td>Monet (base)</td><td>GPT-5.5 / default</td><td class="num">75.5 ± 3.5</td></tr> |
| </tbody> |
| </table> |
| <p class="tnote"> |
| At 84.7% DarwinX matches or exceeds the verified leader (Claude Code + Fable 5, 83.8% at |
| <i>xhigh</i>) while running at a <i>lower</i> effort setting, and adds +2.9 points over OpenAI's own |
| native single-agent GPT-5.6 Sol at the same medium effort. Against a neutral harness on the |
| <i>same</i> GPT-5.5 base (Terminus 2, 78.0%), the pure harness gain is +5.2 points — so the |
| improvement is the harness, not the model or the effort setting. |
| </p> |
|
|
| <div class="block"> |
| <h4>Where evolution helps</h4> |
| <p class="tnote" style="margin-top:0;margin-bottom:14px"> |
| Per-cluster avg@5 on a frozen GPT-5.5. Sort by any column and hover a bar for exact rates. |
| </p> |
| <div class="widget" id="cl-widget"> |
| <div class="wtop"> |
| <span class="wlabel">Sort by</span> |
| <div class="chips" id="cl-chips"></div> |
| <span class="spacer" style="flex:1"></span> |
| <label class="toggle"><input type="checkbox" id="cl-base" checked /> show base</label> |
| </div> |
| <div class="cbars" id="cl-bars"></div> |
| <div class="wlegend"> |
| <span><i class="sw" style="background:#c6d3e0"></i> Monet (base)</span> |
| <span><i class="sw" style="background:var(--blue)"></i> Monet (DarwinX)</span> |
| </div> |
| </div> |
| <noscript class="nojs"><img src="assets/tb21_category_gain.png" alt="Per-cluster Terminal-Bench 2.1 gain." /></noscript> |
| <p class="tnote"> |
| Gains concentrate where a frozen base has the most headroom: <i>ML & scientific computing</i> |
| (+14.8, the largest cluster at 23 tasks) and <i>data & databases</i> (+13.8). Clusters where the |
| base is already strong barely move, and none regresses beyond the per-task noise band. Paired |
| protocol over 88 tasks, which excludes one task with no paired evolved trial; the headline 83.2% |
| uses the stricter 89-task leaderboard rule above. |
| </p> |
| </div> |
|
|
| <div class="block"> |
| <h4>The gain is the harness, not compute</h4> |
| <p class="tnote" style="margin-top:0;margin-bottom:14px"> |
| Per-task compute, median over clean attempts. |
| </p> |
| <div class="widget" id="cp-widget"> |
| <div class="wtop"> |
| <span class="wlabel">Metric</span> |
| <div class="chips" id="cp-chips"></div> |
| </div> |
| <div class="hrows" id="cp-rows"></div> |
| <div class="wlegend"> |
| <span><i class="sw" style="background:#c6d3e0"></i> Monet (base)</span> |
| <span><i class="sw" style="background:var(--blue)"></i> Monet (DarwinX)</span> |
| </div> |
| <p class="tnote" id="cp-note"></p> |
| </div> |
| <noscript class="nojs"><img src="assets/tb21_compute.png" alt="Per-task compute, newly solved vs. already solved." /></noscript> |
| </div> |
| </div> |
|
|
| |
| <div class="bench"> |
| <h3>TerminalWorld <span class="tag">— held-out task split</span></h3> |
| <p class="lede">Evolve on 94 verifier-scored training tasks, then freeze the harness and evaluate on 41 |
| disjoint held-out tasks. Single-attempt pass@1, no retries and no best-of-<i>k</i>, so no held-out |
| reward can feed back into the archive.</p> |
|
|
| <table class="sortable"> |
| <caption>TerminalWorld held-out pass@1 (41 tasks, single attempt, frozen base models, identical |
| verifier and infrastructure).</caption> |
| <thead> |
| <tr><th>Agent</th><th>Model</th><th class="num">pass@1</th></tr> |
| </thead> |
| <tbody> |
| <tr class="ours"><td>Monet (DarwinX)</td><td>Opus 4.8</td><td class="num">68.3%</td></tr> |
| <tr><td>Claude Code</td><td>Opus 4.8</td><td class="num">65.9%</td></tr> |
| <tr class="base"><td>Monet (base)</td><td>Opus 4.8</td><td class="num">61.0%</td></tr> |
| <tr><td>Terminus-2</td><td>GPT-5.5</td><td class="num">61.0%</td></tr> |
| <tr><td>Terminus-2</td><td>Opus 4.8</td><td class="num">58.5%</td></tr> |
| <tr><td>Monet (DarwinX)</td><td>GPT-5.5</td><td class="num">56.1%</td></tr> |
| <tr><td>Codex</td><td>GPT-5.5</td><td class="num">51.2%</td></tr> |
| <tr class="base"><td>Monet (base)</td><td>GPT-5.5</td><td class="num">48.8%</td></tr> |
| </tbody> |
| </table> |
|
|
| <div class="figrow block"> |
| <div> |
| <div class="widget" id="sp-widget"> |
| <div class="wtop"> |
| <span class="wlabel">x-axis</span> |
| <div class="chips" id="sp-chips"></div> |
| <span style="flex:1"></span> |
| <label class="toggle"><input type="checkbox" id="sp-counts" /> task counts</label> |
| </div> |
| <div class="hrows" id="sp-rows"></div> |
| <p class="tnote" id="sp-note"></p> |
| </div> |
| <noscript class="nojs"><img src="assets/tw_heldout_specialists.png" alt="Held-out pass@1 of four specialists and their merge." /></noscript> |
| <p class="tnote"> |
| Held-out generalization reflects <b>archive diversity</b>: four evolved specialists each solve a |
| different subset of the 41 tasks, and the merged Monet (DarwinX) exceeds every specialist and the |
| Claude Code reference (dashed). |
| </p> |
| </div> |
| <div> |
| <h4>The in-loop proxy overfits; the population absorbs it</h4> |
| <p style="font-size:0.95rem"> |
| During evolution the training-subset score saturates from 0.505 to 1.000, yet held-out pass@1 is |
| 68.3% — a <b>31.7-point gap</b> between the proxy the search maximizes and the held-out |
| truth it never sees. Crucially, the variant that best fits the proxy is <i>not</i> the best |
| generalizer: four high-scoring specialists solve 24, 25, 26, and 27 of the 41 held-out tasks on |
| overlapping but distinct subsets, and their merge reaches 28/41. Keeping a population, rather than |
| following the single best-scoring lineage, is what converts an overfit proxy into held-out gain. |
| </p> |
| </div> |
| </div> |
|
|
| <div class="block"> |
| <h4>Explore the merge: which specialists cover which held-out tasks</h4> |
| <p class="tnote" style="margin-top:0;margin-bottom:14px"> |
| Every cell is one of the 41 held-out tasks, filled when at least one selected specialist solves it. |
| Toggle specialists to watch coverage grow, and turn on the comparison to see how the harness DarwinX |
| actually produced differs from the plain union. Hover any cell for its task id. |
| Built from the per-task records of the five evaluation runs. |
| </p> |
| <div class="widget" id="tw-widget"> |
| <div class="wtop"> |
| <span class="wlabel">Specialists</span> |
| <div class="chips" id="tw-chips"></div> |
| <span class="spacer" style="flex:1"></span> |
| <label class="toggle"><input type="checkbox" id="tw-cmp" /> compare with realized merge</label> |
| </div> |
| <div class="readout" id="tw-readout"></div> |
| <div class="tgrid" id="tw-grid"></div> |
| <div class="wlegend" id="tw-legend"> |
| <span><i class="sw" style="background:var(--blue)"></i> solved by selection</span> |
| <span><i class="sw" style="background:#e6ecf2"></i> unsolved</span> |
| </div> |
| <p class="tnote" id="tw-note"></p> |
| </div> |
| </div> |
| </div> |
|
|
| |
| <div class="bench"> |
| <h3>WebArena-Infinity <span class="tag">— synthetic → real intent shift</span></h3> |
| <p class="lede">Evolution sees synthetic intents only; evaluation is the official 1,260-task real suite |
| across 10 applications. We report <i>audit-clean</i> pass@1 throughout: trajectories flagged invalid |
| by the validity audit count as failures even when the verifier passes them.</p> |
|
|
| <div class="scroller"> |
| <table class="sortable"> |
| <caption>WebArena-Infinity audit-clean pass@1 by application. Click any column header to sort. External baselines are as-reported |
| without re-auditing, which makes this comparison conservative for DarwinX. BU = Browser Use.</caption> |
| <thead> |
| <tr> |
| <th>Application</th> |
| <th class="num">Kimi</th><th class="num">Qwen</th><th class="num">Gemini+BU</th> |
| <th class="num">GPT-5.5+BU</th><th class="num">Monet (base)</th> |
| <th class="num">Monet (DarwinX)</th><th class="num">Δ</th> |
| </tr> |
| </thead> |
| <tbody> |
| <tr><td>Elation clinical records</td><td class="num">50.0</td><td class="num">54.2</td><td class="num">81.7</td><td class="num">92.5</td><td class="num">95.8</td><td class="num"><b>96.7</b></td><td class="num delta">+0.9</td></tr> |
| <tr><td>Elation prescriptions</td><td class="num">23.3</td><td class="num">41.7</td><td class="num">80.8</td><td class="num">90.8</td><td class="num">20.0</td><td class="num"><b>95.0</b></td><td class="num delta">+75.0</td></tr> |
| <tr><td>GitLab plan and track</td><td class="num">39.3</td><td class="num">37.1</td><td class="num">63.6</td><td class="num">77.9</td><td class="num">63.6</td><td class="num"><b>97.9</b></td><td class="num delta">+34.3</td></tr> |
| <tr><td>Gmail</td><td class="num">70.0</td><td class="num">56.7</td><td class="num">75.0</td><td class="num">85.0</td><td class="num">25.0</td><td class="num"><b>98.3</b></td><td class="num delta">+73.3</td></tr> |
| <tr><td>Gmail accounts and contacts</td><td class="num">40.0</td><td class="num">33.3</td><td class="num">61.7</td><td class="num">87.5</td><td class="num">21.7</td><td class="num"><b>91.7</b></td><td class="num delta">+70.0</td></tr> |
| <tr><td>Handshake career exploration</td><td class="num">50.0</td><td class="num">50.5</td><td class="num">50.5</td><td class="num">83.5</td><td class="num">36.5</td><td class="num"><b>84.0</b></td><td class="num delta">+47.5</td></tr> |
| <tr><td>Linear account settings</td><td class="num">54.2</td><td class="num">65.8</td><td class="num">73.3</td><td class="num">81.7</td><td class="num">43.3</td><td class="num"><b>94.2</b></td><td class="num delta">+50.9</td></tr> |
| <tr><td>PayPal wallet</td><td class="num">70.7</td><td class="num">71.4</td><td class="num">88.6</td><td class="num">90.0</td><td class="num">49.3</td><td class="num"><b>95.7</b></td><td class="num delta">+46.4</td></tr> |
| <tr><td>Superhuman general</td><td class="num">15.0</td><td class="num">25.8</td><td class="num">50.0</td><td class="num">80.8</td><td class="num">31.7</td><td class="num"><b>87.5</b></td><td class="num delta">+55.8</td></tr> |
| <tr><td>Xero invoicing</td><td class="num">52.5</td><td class="num">55.8</td><td class="num">80.8</td><td class="num">93.3</td><td class="num">39.2</td><td class="num"><b>96.7</b></td><td class="num delta">+57.5</td></tr> |
| <tr class="total"><td>Overall</td><td class="num">43.3</td><td class="num">48.3</td><td class="num">69.3</td><td class="num">86.1</td><td class="num">43.5</td><td class="num">93.0</td><td class="num delta">+49.5</td></tr> |
| </tbody> |
| </table> |
| </div> |
| <p class="tnote"> |
| 93.0% audit-clean beats the strongest same-model baseline (GPT-5.5 + Browser Use, 86.1%) by |
| 6.9 points and the top public agent (Gemini 3 Flash + Browser Use, 69.3%) by |
| 23.7 points. Given differing action spaces across public agents, we treat the same-model gap as |
| the primary comparison. In counts, the base solves 548/1,260 and the evolved harness 1,171/1,260. |
| </p> |
|
|
| <div class="block"> |
| <h4>Evolution as optimization</h4> |
| <p class="tnote" style="margin-top:0;margin-bottom:14px"> |
| Screening score of every evaluated variant in the WebArena-Infinity run, in archive order. Hover a |
| point for its score; the line is the best-so-far envelope. |
| </p> |
| <div class="widget" id="cv-widget"> |
| <div class="wtop"> |
| <span class="wlabel">Show</span> |
| <label class="toggle"><input type="checkbox" id="cv-all" checked /> every evaluated variant</label> |
| <label class="toggle"><input type="checkbox" id="cv-best" checked /> best so far</label> |
| </div> |
| <svg class="chart" id="cv-chart" viewBox="0 0 860 320" role="img" |
| aria-label="Screening score per evaluated variant, with the best-so-far envelope."></svg> |
| <p class="tnote" id="cv-note"></p> |
| </div> |
| <noscript class="nojs"><img src="assets/wai_evolution_curve.png" alt="Best-so-far screening score across accepted generations." /></noscript> |
| </div> |
|
|
| <figure class="block"> |
| <a class="zoom" href="assets/wai_tree.png" target="_blank" rel="noopener"> |
| <img src="assets/wai_tree.png" alt="Archive lineage tree with accepted and reverted variants, the primary lineage, and recombination edges." loading="lazy" /> |
| </a> |
| <figcaption> |
| The same run as an archive lineage tree (node size ∝ screening score): accepted (blue) and |
| reverted (grey) variants, the primary lineage (gold, base → evolved), and |
| recombination edges (dashed). The reverted nodes are the preserve-and-extend contract doing its job: |
| a variant that broke an inherited solve contributes its distilled lesson but not its edit. |
| </figcaption> |
| </figure> |
|
|
| <div class="block"> |
| <h4>Anti-cheating audit</h4> |
| <table style="margin-top:12px"> |
| <caption>Validity audit on the 1,260 real tasks. Trajectories that reach the goal through the |
| evaluation plane, privileged knowledge, or exploits are counted as failures.</caption> |
| <thead> |
| <tr><th>Measure</th><th class="num">Monet (base)</th><th class="num">Monet (DarwinX)</th><th class="num">Δ</th></tr> |
| </thead> |
| <tbody> |
| <tr class="ours"><td>Audit-clean pass@1</td><td class="num">43.5</td><td class="num">93.0</td><td class="num delta">+49.5 pp</td></tr> |
| <tr><td>Confirmed invalid</td><td class="num">23.5%</td><td class="num">1.4%</td><td class="num">−22.1 pp</td></tr> |
| <tr><td>Human review</td><td class="num">5.1%</td><td class="num">0.1%</td><td class="num">−5.0 pp</td></tr> |
| <tr><td>Blocked attempt</td><td class="num">14.3%</td><td class="num">0.1%</td><td class="num">−14.2 pp</td></tr> |
| </tbody> |
| </table> |
| <p class="tnote"> |
| Evolution does not buy the gain with shortcuts: it cuts invalid trajectories from 293 to 17, the |
| evaluation-plane, privileged-knowledge, and exploit mechanisms disappear entirely, and the residual |
| 17 are all raw-state mutations concentrated in a single application. Dropping unaudited successes |
| too still leaves 92.9%. |
| </p> |
| <div class="block"> |
| <div class="widget" id="iv-widget"> |
| <div class="wtop"> |
| <span class="wlabel">Break down by</span> |
| <div class="chips" id="iv-chips"></div> |
| <span style="flex:1"></span> |
| <label class="toggle"><input type="checkbox" id="iv-norm" /> scale each row to its own total</label> |
| </div> |
| <div id="iv-rows"></div> |
| <div class="keys" id="iv-keys"></div> |
| <p class="tnote" id="iv-note"></p> |
| </div> |
| <noscript class="nojs"><img src="assets/wai_invalid_composition.png" alt="Invalid trajectories before vs. after evolution, 293 to 17." /></noscript> |
| <p class="tnote"> |
| Invalid trajectories before vs. after evolution. On a shared scale the second row is almost |
| invisible, which is the result: 293 invalid trajectories become 17. Scale each row to its own |
| total to inspect what the remainder is made of. |
| </p> |
| </div> |
|
|
| <div class="block"> |
| <h4>What the audit costs each harness</h4> |
| <p class="tnote" style="margin-top:0;margin-bottom:14px"> |
| Each line runs from raw pass@1 to pass@1 after the validity audit. A long line means the harness |
| was leaning on trajectories the audit rejects. |
| </p> |
| <div class="widget" id="ad-widget"> |
| <div class="wtop"> |
| <span class="wlabel">Show</span> |
| <label class="toggle"><input type="checkbox" id="ad-base" checked /> Monet (base)</label> |
| <label class="toggle"><input type="checkbox" id="ad-dx" checked /> Monet (DarwinX)</label> |
| <span style="flex:1"></span> |
| <label class="toggle"><input type="checkbox" id="ad-sort" /> sort by audit loss</label> |
| </div> |
| <div class="dumb" id="ad-rows"></div> |
| <div class="wlegend"> |
| <span><i class="sw" style="background:#fff;box-shadow:inset 0 0 0 2px #9aa8b8;border-radius:50%"></i> raw pass@1</span> |
| <span><i class="sw" style="background:#8794a6;border-radius:50%"></i> base, audited</span> |
| <span><i class="sw" style="background:var(--blue);border-radius:50%"></i> DarwinX, audited</span> |
| </div> |
| <p class="tnote" id="ad-note"></p> |
| </div> |
| <noscript class="nojs"><img src="assets/wai_audit_by_app.png" alt="Raw vs. audited pass@1 per application." /></noscript> |
| </div> |
| </div> |
| </div> |
|
|
| |
| <div class="bench"> |
| <h3>SWE-bench Verified <span class="tag">— zero-shot cross-benchmark transfer</span></h3> |
| <p class="lede">The best Terminal-Bench 2.1 harness is run <i>unchanged</i> on all 500 SWE-bench |
| Verified instances. No SWE-V feedback ever enters the loop; we report it as a transfer target only and |
| run no in-domain SWE-V evolution.</p> |
|
|
| <table> |
| <caption>SWE-bench Verified official pass@1, graded by SWE-V's own test harness.</caption> |
| <thead> |
| <tr><th>Harness</th><th class="num">Solved</th><th class="num">pass@1</th></tr> |
| </thead> |
| <tbody> |
| <tr class="ours"><td>Monet (DarwinX), TB2.1-specialized, transferred unchanged</td><td class="num">421/500</td><td class="num">84.2%</td></tr> |
| <tr><td>Fix-skill reference</td><td class="num">—</td><td class="num">80.8%</td></tr> |
| </tbody> |
| </table> |
| <p class="tnote"> |
| +3.4 points over the reference without receiving any SWE-V feedback. The transferred agent preserves |
| strong repository-level competence, so what evolved on terminal tasks is general agent capability |
| rather than benchmark-specific patching. |
| </p> |
| </div> |
| </div> |
| </section> |
|
|
| <section id="ablation"> |
| <div class="wrap"> |
| <h2>Ablation: what evolution changes</h2> |
| <p> |
| Anchored on Terminal-Bench 2.1, we compare base Monet (75.5%) with the evolved best (83.2%) along |
| two axes: the <b>skill-bundle diff</b> between the two harnesses, and <b>where</b> on the benchmark the |
| gains land. This is an <i>exploratory attribution, not a per-skill causal ablation</i> — the skills |
| were co-selected, not independently randomized. |
| </p> |
|
|
| <table style="margin-top:20px"> |
| <caption>The seven skills the evolved lineage adds over base Monet. Every one belongs to a single |
| family: verification / artifact-contract. None adds domain knowledge.</caption> |
| <thead> |
| <tr><th style="width:42%">Evolved skills</th><th>Role</th></tr> |
| </thead> |
| <tbody> |
| <tr> |
| <td><code>verifier-contract</code><br /><code>contract-candidate</code></td> |
| <td>Derive the task's acceptance contract and check the solution against it before finalizing.</td> |
| </tr> |
| <tr> |
| <td><code>graded-artifact-final-check</code><br /><code>artifact-verification-loop</code></td> |
| <td>Verify the graded artifact (output file, format, and values) and iterate a fix-and-recheck loop.</td> |
| </tr> |
| <tr> |
| <td><code>real-tool-artifact</code><br /><code>tool-grounded-artifact</code></td> |
| <td>Ground outputs in real tool execution rather than asserted or simulated results.</td> |
| </tr> |
| <tr> |
| <td><code>security-contract-repair</code></td> |
| <td>Repair the solution against security and contract checks.</td> |
| </tr> |
| </tbody> |
| </table> |
|
|
| <div class="block" style="margin-top:30px"> |
| <h4>Where the gains land</h4> |
| <p> |
| Gains concentrate where a frozen base has the most headroom and where verifying against a contract |
| unblocks fragile multi-step work: <i>ML & scientific-computing</i> (+14.8, 60→75%) and |
| <i>data/database</i> (+13.8, 84→98%). Clusters where the base is already strong barely move |
| (system administration 92→98%, security 85→84%, within noise), and no cluster regresses |
| beyond the per-task noise band. <b>That asymmetry is the empirical footprint of the |
| preserve-and-extend rule</b>: extend the fragile clusters, hold the solved ones fixed. The unblocked |
| difficulty is procedural — long dependency installs, environment setup, output verification, |
| multi-step tool use — rather than knowledge-bound, which matches a verification/artifact-contract |
| bundle rather than a stronger model. |
| </p> |
| </div> |
|
|
| <div class="block"> |
| <h4>Cross-benchmark cross-checks</h4> |
| <p> |
| <b>WebArena-Infinity, the same family in a different modality.</b> The evolved browser harness adds |
| the same kind of skills (state and action contracts), and its confirmed-invalid rate falls from 23.5% |
| to 1.4% while audit-clean pass@1 rises 49.5 points, with the largest gains on state-changing |
| applications. Verification-before-finalization is thus visible under a different interaction modality |
| and reward source. |
| </p> |
| <p> |
| <b>TerminalWorld, diversity rather than a single skill.</b> Individual specialists solve 24–27 |
| held-out tasks and the merged harness solves 28, so the archive is valuable as a source of |
| complementary behaviors even when the training-subset ranking is noisy. |
| </p> |
| <p class="tnote"> |
| Together these make verification-before-finalization and contract-aware tool use a plausible shared |
| mechanism across benchmarks, offered as an explanation rather than a per-skill causal estimate. |
| </p> |
| </div> |
| </div> |
| </section> |
|
|
| <section id="limitations"> |
| <div class="wrap"> |
| <h2>Limitations</h2> |
| <ul class="lim"> |
| <li><b>Transfer is measured in one direction, and is much smaller.</b> A TB2.1-evolved harness reaches |
| 84.2% on SWE-bench Verified, ahead of the fix-skill reference but inside a narrow band around it |
| (official scores across the harnesses we compare span just 80.8–84.2%). The transferred gain is |
| far smaller than the in-domain ones, and we make no in-domain SWE-V claim.</li> |
| <li><b>Recombination's own contribution is not isolated.</b> The archive and merge machinery can |
| preserve and combine variants, but population search needs diverse wins before inheritance is useful, |
| and the contribution of recombination relative to single-lineage mutation still requires controlled |
| ablation.</li> |
| <li><b>The claim is system-level, not per-operator.</b> The experiments evaluate the complete system: |
| archive, parent selector, recombination operator, and inference effort are not independently |
| randomized, and public leaderboard rows use different models and effort settings. Matched-model deltas |
| support a harness claim; any individual operator's contribution, and the attribution to a |
| verification/contract mechanism, remain plausible rather than causal.</li> |
| <li><b>TerminalWorld is small.</b> With only 41 held-out tasks, one solve moves pass@1 by 2.4 points, |
| and the matched Opus comparison (25/41 vs. 28/41, McNemar <i>p</i>=0.45) is suggestive rather than |
| decisive.</li> |
| <li><b>The validity audit is not a formal sandbox.</b> Our WAI policy permits client-visible observation |
| and semantic application operations but rejects privileged knowledge, evaluation-plane access, |
| raw-state fabrication, database manipulation, and exploits. The static-plus-LLM audit is far stronger |
| than a keyword heuristic, yet deeply dynamic construction can require human review and a small number |
| of trajectories were unavailable, so we report both raw and conservative audit-clean scores.</li> |
| <li><b>Freezing the base model is a choice, not a ceiling.</b> It is what makes these deltas |
| attributable to the harness. Coupling harness selection with weight updates, and treating the harness |
| as an asset that outlives a base-model generation, are left to future work.</li> |
| </ul> |
| </div> |
| </section> |
|
|
| <section id="bibtex"> |
| <div class="wrap"> |
| <div class="bibhead"> |
| <h2>BibTeX</h2> |
| <button id="copybtn" type="button">Copy</button> |
| </div> |
| <pre id="bib">@article{zhang2026darwinx, |
| title = {DarwinX: Evolving Agent Harnesses Through Natural Selection}, |
| author = {Zhang, Yifan and Dai, Yutong and Tan, Juntao and Yang, Luyu and |
| Mullur, Rishi and Hoang, Thai and Hu, Zhiyuan and Zhu, James and |
| Mui, Phil and Savarese, Silvio and Xu, Ran and Chen, Zeyuan}, |
| journal = {arXiv preprint arXiv:2608.07545}, |
| year = {2026} |
| }</pre> |
| </div> |
| </section> |
|
|
| <footer> |
| <div class="wrap"> |
| Salesforce AI Research · 2026. All figures and numbers are from |
| <a href="https://arxiv.org/abs/2608.07545">arXiv:2608.07545</a>. |
| Click any figure to open it at full resolution. |
| </div> |
| </footer> |
|
|
| <div id="tip" role="tooltip"></div> |
|
|
| <script src="assets/data.js"></script> |
| <script> |
| (function () { |
| "use strict"; |
| var D = window.DARWINX; |
| var $ = function (id) { return document.getElementById(id); }; |
| |
| |
| var tip = $("tip"); |
| function showTip(html, evt) { |
| tip.innerHTML = html; |
| tip.style.opacity = "1"; |
| var r = tip.getBoundingClientRect(); |
| var x = Math.min(Math.max(8, evt.clientX - r.width / 2), window.innerWidth - r.width - 8); |
| var y = evt.clientY - r.height - 12; |
| if (y < 8) { y = evt.clientY + 16; } |
| tip.style.left = x + "px"; |
| tip.style.top = y + "px"; |
| } |
| function hideTip() { tip.style.opacity = "0"; } |
| function bindTip(el, htmlFn) { |
| el.addEventListener("mousemove", function (e) { showTip(htmlFn(), e); }); |
| el.addEventListener("mouseleave", hideTip); |
| } |
| |
| |
| (function () { |
| if (!D || !D.headline) { return; } |
| var P = D.headline; |
| var modes = [ |
| { id: "own", label: "per panel" }, |
| { id: "shared", label: "shared 0\u2013100" } |
| ]; |
| var mode = "own"; |
| |
| var chips = $("hl-chips"); |
| modes.forEach(function (m) { |
| var b = document.createElement("button"); |
| b.className = "chip"; |
| b.type = "button"; |
| b.textContent = m.label; |
| b.setAttribute("aria-pressed", m.id === mode ? "true" : "false"); |
| b.addEventListener("click", function () { |
| mode = m.id; |
| Array.prototype.forEach.call(chips.children, function (c) { |
| c.setAttribute("aria-pressed", c === b ? "true" : "false"); |
| }); |
| render(); |
| }); |
| chips.appendChild(b); |
| }); |
| |
| var NS = "http://www.w3.org/2000/svg"; |
| function mk(n, a) { |
| var e = document.createElementNS(NS, n); |
| Object.keys(a).forEach(function (k) { e.setAttribute(k, a[k]); }); |
| return e; |
| } |
| |
| function render() { |
| var showPrior = $("hl-prior").checked; |
| var box = $("hl-panels"); |
| box.innerHTML = ""; |
| |
| P.forEach(function (p) { |
| var lo = mode === "shared" ? 0 : p.lo; |
| var hi = mode === "shared" ? 100 : p.hi; |
| var W = 190, H = 168, T = 20, B = 22, L = 26; |
| var ih = H - T - B; |
| var y = function (v) { return T + ih - ((v - lo) / (hi - lo)) * ih; }; |
| |
| var series = [{ c: "b", label: p.baseLabel, v: p.base }]; |
| if (showPrior && p.prior !== null) { series.push({ c: "p", label: p.priorLabel, v: p.prior }); } |
| series.push({ c: "e", label: p.evolvedLabel, v: p.evolved }); |
| |
| var wrapEl = document.createElement("div"); |
| wrapEl.className = "panel"; |
| wrapEl.innerHTML = "<div class=\"pt\">" + p.short + "</div><div class=\"ps\">" + |
| p.metric + "<br>on " + p.model + "</div>"; |
| |
| var svg = mk("svg", { viewBox: "0 0 " + W + " " + H }); |
| [lo, hi].forEach(function (v) { |
| var t = mk("text", { class: "tick", x: L - 6, y: y(v) + 3.5, "text-anchor": "end" }); |
| t.textContent = v; |
| svg.appendChild(t); |
| }); |
| svg.appendChild(mk("line", { class: "axl", x1: L, x2: W - 6, y1: y(lo), y2: y(lo) })); |
| |
| var slot = (W - 6 - L) / series.length; |
| var bw = Math.min(34, slot * 0.62); |
| series.forEach(function (s, i) { |
| var cx = L + slot * (i + 0.5); |
| var top = y(Math.max(s.v, lo)); |
| var rect = mk("rect", { |
| class: "bar " + s.c, x: cx - bw / 2, y: top, width: bw, |
| height: Math.max(0, y(lo) - top), rx: 2 |
| }); |
| svg.appendChild(rect); |
| var hit = mk("rect", { x: cx - slot / 2, y: T, width: slot, height: ih, fill: "transparent" }); |
| hit.addEventListener("mousemove", function (e) { |
| showTip("<b>" + p.name + "</b><br>" + s.label + ": " + s.v.toFixed(1) + "%<br>" + |
| p.metric + ", " + p.scope, e); |
| }); |
| hit.addEventListener("mouseleave", hideTip); |
| svg.appendChild(hit); |
| }); |
| |
| var ev = series[series.length - 1]; |
| var dl = mk("text", { class: "dl", x: L + slot * (series.length - 0.5), |
| y: y(ev.v) - 6, "text-anchor": "middle" }); |
| dl.textContent = "+" + p.delta.toFixed(1); |
| svg.appendChild(dl); |
| |
| wrapEl.appendChild(svg); |
| var foot = document.createElement("div"); |
| foot.className = "foot"; |
| foot.textContent = p.prior === null ? "transfer only" : "vs " + p.priorLabel.split(" + ")[0]; |
| wrapEl.appendChild(foot); |
| box.appendChild(wrapEl); |
| }); |
| |
| $("hl-legend").innerHTML = |
| "<span><i class=\"sw\" style=\"background:#c6d3e0\"></i> base / reference</span>" + |
| (showPrior ? "<span><i class=\"sw\" style=\"background:#8fb4e8\"></i> best prior agent</span>" : "") + |
| "<span><i class=\"sw\" style=\"background:var(--blue)\"></i> Monet (DarwinX)</span>"; |
| |
| $("hl-note").innerHTML = mode === "own" |
| ? "Each panel is scaled to its own range, which is how the paper draws it: on a shared axis " + |
| "WebArena-Infinity's +49.5 flattens the two terminal benchmarks into slivers. Switch to the " + |
| "shared axis to see the honest relative magnitudes \u2014 the annotated deltas are the same either way." |
| : "On one 0\u2013100 axis the four regimes are directly comparable, and it is clear how much of the " + |
| "range each benchmark actually occupies. Bars within a panel share one frozen base model; the " + |
| "prior-agent bar uses a different model and effort setting, so it is context rather than a " + |
| "controlled comparison. On SWE-bench Verified the grey bar is the fix-skill reference, not an " + |
| "unevolved Monet, and no prior-agent bar is drawn."; |
| } |
| |
| $("hl-prior").addEventListener("change", render); |
| render(); |
| })(); |
| |
| |
| (function () { |
| if (!D || !D.twBars) { return; } |
| var T = D.twBars; |
| function pct(s) { return (100 * s / T.n); } |
| |
| |
| |
| var scales = [{ id: "zoom", label: "55\u201370", lo: 55, hi: 70 }, |
| { id: "full", label: "0\u2013100", lo: 0, hi: 100 }]; |
| var scale = scales[0]; |
| |
| var chips = $("sp-chips"); |
| scales.forEach(function (s) { |
| var b = document.createElement("button"); |
| b.className = "chip"; |
| b.type = "button"; |
| b.textContent = s.label; |
| b.setAttribute("aria-pressed", s === scale ? "true" : "false"); |
| b.addEventListener("click", function () { |
| scale = s; |
| Array.prototype.forEach.call(chips.children, function (c) { |
| c.setAttribute("aria-pressed", c === b ? "true" : "false"); |
| }); |
| render(); |
| }); |
| chips.appendChild(b); |
| }); |
| |
| function render() { |
| var counts = $("sp-counts").checked; |
| var pos = function (p) { |
| return Math.max(0, Math.min(100, 100 * (p - scale.lo) / (scale.hi - scale.lo))); |
| }; |
| var refPos = pos(pct(T.claudeCode)); |
| var box = $("sp-rows"); |
| box.innerHTML = ""; |
| T.bars.forEach(function (b) { |
| var isDx = /DarwinX/.test(b.label); |
| var isBase = b.label === "Base"; |
| var row = document.createElement("div"); |
| row.className = "hrow" + (isBase ? " dim" : ""); |
| var color = isDx ? "var(--blue)" : (isBase ? "#c6d3e0" : "#8fb4e8"); |
| row.innerHTML = |
| "<div class=\"hn\"" + (isDx ? " style=\"font-weight:600\"" : "") + ">" + b.label + "</div>" + |
| "<div class=\"ht\"><div class=\"hf\" style=\"width:" + pos(pct(b.solved)).toFixed(1) + |
| "%;background:" + color + "\"></div></div>" + |
| "<div class=\"hv\">" + (counts ? b.solved + "/" + T.n : pct(b.solved).toFixed(1) + "%") + "</div>"; |
| var track = row.querySelector(".ht"); |
| var ref = document.createElement("div"); |
| ref.className = "refline"; |
| ref.style.left = refPos.toFixed(1) + "%"; |
| track.appendChild(ref); |
| bindTip(track, function () { |
| return "<b>" + b.label + "</b><br>" + b.solved + " of " + T.n + " held-out tasks (" + |
| pct(b.solved).toFixed(1) + "%)<br>Claude Code reference: " + T.claudeCode + "/" + T.n + |
| " (" + pct(T.claudeCode).toFixed(1) + "%)"; |
| }); |
| box.appendChild(row); |
| }); |
| var tag = document.createElement("div"); |
| tag.className = "hrow"; |
| tag.innerHTML = "<div></div><div style=\"position:relative;height:14px\">" + |
| "<div class=\"reftag\" style=\"left:" + refPos.toFixed(1) + "%;top:0\">Claude Code " + |
| pct(T.claudeCode).toFixed(1) + "%</div></div><div></div>"; |
| box.appendChild(tag); |
| |
| $("sp-note").innerHTML = scale.id === "zoom" |
| ? "Bars start at 55%, not 0, so the arms are distinguishable \u2014 the whole spread is 9.8 points. " + |
| "Switch to the full axis to see that spread in proportion." |
| : "On the full axis every arm looks alike, which is the honest framing of the magnitude: the merge " + |
| "buys 2.4 points over the best single specialist. The ordering is what matters here, not the size."; |
| } |
| $("sp-counts").addEventListener("change", render); |
| render(); |
| })(); |
| |
| |
| (function () { |
| if (!D || !D.tb21Compute) { return; } |
| var C = D.tb21Compute; |
| var metrics = [ |
| { id: "turns", label: "turns", unit: "", fmt: function (v) { return v; } }, |
| { id: "tokens", label: "tokens", unit: "K", fmt: function (v) { return v + "K"; } } |
| ]; |
| var active = "turns"; |
| |
| var chips = $("cp-chips"); |
| metrics.forEach(function (m) { |
| var b = document.createElement("button"); |
| b.className = "chip"; |
| b.type = "button"; |
| b.textContent = m.label; |
| b.setAttribute("aria-pressed", m.id === active ? "true" : "false"); |
| b.addEventListener("click", function () { |
| active = m.id; |
| Array.prototype.forEach.call(chips.children, function (c) { |
| c.setAttribute("aria-pressed", c === b ? "true" : "false"); |
| }); |
| render(); |
| }); |
| chips.appendChild(b); |
| }); |
| |
| function render() { |
| var m = metrics.filter(function (x) { return x.id === active; })[0]; |
| var d = C[active]; |
| var max = Math.max.apply(null, d.base.concat(d.evolved)); |
| var box = $("cp-rows"); |
| box.innerHTML = ""; |
| C.groups.forEach(function (g, gi) { |
| [["Monet (base)", d.base[gi], "#c6d3e0"], ["Monet (DarwinX)", d.evolved[gi], "var(--blue)"]] |
| .forEach(function (s, si) { |
| var row = document.createElement("div"); |
| row.className = "hrow"; |
| row.innerHTML = |
| "<div class=\"hn\">" + (si === 0 ? g : "") + "</div>" + |
| "<div class=\"ht\"><div class=\"hf\" style=\"width:" + (100 * s[1] / max).toFixed(1) + |
| "%;background:" + s[2] + "\"></div></div>" + |
| "<div class=\"hv\">" + m.fmt(s[1]) + "</div>"; |
| bindTip(row.querySelector(".ht"), function () { |
| var other = si === 0 ? d.evolved[gi] : d.base[gi]; |
| var ratio = si === 0 ? (other / s[1]) : (s[1] / other); |
| return "<b>" + g + "</b><br>" + s[0] + ": " + m.fmt(s[1]) + " median " + m.label + |
| "<br>evolved / base = " + ratio.toFixed(1) + "x"; |
| }); |
| box.appendChild(row); |
| }); |
| }); |
| $("cp-note").innerHTML = active === "turns" |
| ? "On the six tasks that flip from failing to solved the evolved harness roughly doubles its turns " + |
| "(22 vs. 11). On the 69 both already solve, turns barely move (13 vs. 12): the extra effort is " + |
| "spent where it changes the outcome, not spread across the benchmark." |
| : "Tokens tell the same story more sharply: 4.3x on the newly solved six (380K vs. 89K) against " + |
| "1.4x on the 69 already solved (172K vs. 125K). This is verify-and-retry effort converting a " + |
| "near-miss into a pass, not uniformly more compute."; |
| } |
| render(); |
| })(); |
| |
| |
| (function () { |
| if (!D || !D.waiInvalid) { return; } |
| var V = D.waiInvalid; |
| var views = [{ id: "application", label: "application" }, { id: "mechanism", label: "mechanism" }]; |
| var active = "mechanism"; |
| |
| var chips = $("iv-chips"); |
| views.forEach(function (v) { |
| var b = document.createElement("button"); |
| b.className = "chip"; |
| b.type = "button"; |
| b.textContent = v.label; |
| b.setAttribute("aria-pressed", v.id === active ? "true" : "false"); |
| b.addEventListener("click", function () { |
| active = v.id; |
| Array.prototype.forEach.call(chips.children, function (c) { |
| c.setAttribute("aria-pressed", c === b ? "true" : "false"); |
| }); |
| render(); |
| }); |
| chips.appendChild(b); |
| }); |
| |
| function render() { |
| var v = V[active]; |
| var norm = $("iv-norm").checked; |
| var totalBefore = v.before.reduce(function (a, b) { return a + b; }, 0); |
| var box = $("iv-rows"); |
| box.innerHTML = ""; |
| |
| [["Before evolution", v.before], ["After evolution", v.after]].forEach(function (pair) { |
| var vals = pair[1]; |
| var sum = vals.reduce(function (a, b) { return a + b; }, 0); |
| var denom = norm ? (sum || 1) : totalBefore; |
| var row = document.createElement("div"); |
| row.className = "srow"; |
| row.innerHTML = "<div class=\"sh\"><span>" + pair[0] + |
| "</span><span><b>" + sum + "</b> invalid</span></div>"; |
| var st = document.createElement("div"); |
| st.className = "stack"; |
| vals.forEach(function (val, i) { |
| if (!val) { return; } |
| var seg = document.createElement("i"); |
| seg.style.width = (100 * val / denom).toFixed(2) + "%"; |
| seg.style.background = v.colors[i]; |
| bindTip(seg, function () { |
| return "<b>" + v.labels[i] + "</b><br>" + val + " invalid trajector" + |
| (val === 1 ? "y" : "ies") + "<br>" + (100 * val / (sum || 1)).toFixed(1) + |
| "% of this row"; |
| }); |
| st.appendChild(seg); |
| }); |
| row.appendChild(st); |
| box.appendChild(row); |
| }); |
| |
| $("iv-keys").innerHTML = v.labels.map(function (l, i) { |
| if (!v.before[i] && !v.after[i]) { return ""; } |
| return "<span><i class=\"sw\" style=\"background:" + v.colors[i] + "\"></i>" + l + "</span>"; |
| }).join(""); |
| |
| $("iv-note").innerHTML = active === "mechanism" |
| ? "Evaluation-plane access, privileged knowledge, and exploits go to <b>zero</b>. Every one of the " + |
| "17 remaining invalid trajectories is a raw-state mutation, up from 15 of 293 before \u2014 the " + |
| "category that survives is the mildest one." |
| : "The remainder concentrates: 8 of the 17 sit in a single application (Super-human), and five of " + |
| "the ten applications end with no invalid trajectories at all."; |
| } |
| $("iv-norm").addEventListener("change", render); |
| render(); |
| })(); |
| |
| |
| (function () { |
| if (!D || !D.waiAudit) { return; } |
| var A = D.waiAudit; |
| |
| function render() { |
| var showBase = $("ad-base").checked, showDx = $("ad-dx").checked; |
| var idx = A.apps.map(function (_, i) { return i; }); |
| var overall = idx.shift(); |
| if ($("ad-sort").checked) { |
| idx.sort(function (a, b) { |
| return (A.rawBase[b] - A.auditedBase[b]) - (A.rawBase[a] - A.auditedBase[a]); |
| }); |
| } |
| var box = $("ad-rows"); |
| box.innerHTML = ""; |
| |
| var ax = document.createElement("div"); |
| ax.className = "drow dax"; |
| ax.innerHTML = "<div class=\"dn\" style=\"color:var(--muted);font-size:0.7rem\">pass@1 (%)</div>" + |
| "<div class=\"dt\">" + [0, 25, 50, 75, 100].map(function (t) { |
| return "<span style=\"left:" + t + "%\">" + t + "</span>"; |
| }).join("") + "</div>"; |
| box.appendChild(ax); |
| |
| [overall].concat(idx).forEach(function (i, pos) { |
| var row = document.createElement("div"); |
| row.className = "drow" + (pos === 0 ? " tot" : ""); |
| row.innerHTML = "<div class=\"dn\">" + A.apps[i] + "</div><div class=\"dt\"></div>"; |
| var t = row.querySelector(".dt"); |
| |
| function lane(raw, aud, color, top) { |
| var x1 = Math.min(raw, aud), x2 = Math.max(raw, aud); |
| var seg = document.createElement("div"); |
| seg.className = "seg"; |
| seg.style.left = x1 + "%"; |
| seg.style.width = (x2 - x1) + "%"; |
| seg.style.top = top + "px"; |
| seg.style.background = color; |
| t.appendChild(seg); |
| var p1 = document.createElement("div"); |
| p1.className = "pt2 hollow"; |
| p1.style.left = raw + "%"; |
| p1.style.top = (top + 1.5) + "px"; |
| p1.style.boxShadow = "inset 0 0 0 2px " + color; |
| t.appendChild(p1); |
| var p2 = document.createElement("div"); |
| p2.className = "pt2"; |
| p2.style.left = aud + "%"; |
| p2.style.top = (top + 1.5) + "px"; |
| p2.style.background = color; |
| t.appendChild(p2); |
| } |
| |
| if (showBase) { lane(A.rawBase[i], A.auditedBase[i], "#8794a6", 8); } |
| if (showDx) { lane(A.rawDx[i], A.auditedDx[i], "#0176d3", 19); } |
| |
| bindTip(t, function () { |
| return "<b>" + A.apps[i] + "</b><br>base " + A.rawBase[i].toFixed(1) + " \u2192 " + |
| A.auditedBase[i].toFixed(1) + " (\u2212" + (A.rawBase[i] - A.auditedBase[i]).toFixed(1) + |
| ")<br>DarwinX " + A.rawDx[i].toFixed(1) + " \u2192 " + A.auditedDx[i].toFixed(1) + |
| " (\u2212" + (A.rawDx[i] - A.auditedDx[i]).toFixed(1) + ")"; |
| }); |
| box.appendChild(row); |
| }); |
| |
| var lossB = A.rawBase[0] - A.auditedBase[0], lossD = A.rawDx[0] - A.auditedDx[0]; |
| $("ad-note").innerHTML = "Overall the audit costs the base <b>" + lossB.toFixed(1) + |
| "</b> points and DarwinX <b>" + lossD.toFixed(1) + "</b>. The gap therefore <i>widens</i> under " + |
| "scrutiny, from " + (A.rawDx[0] - A.rawBase[0]).toFixed(1) + " points raw to " + |
| (A.auditedDx[0] - A.auditedBase[0]).toFixed(1) + " audited, which is why the page reports the " + |
| "audited number. Audit coverage is 99.0% for the base and 99.4% for the evolved harness."; |
| } |
| ["ad-base", "ad-dx", "ad-sort"].forEach(function (id) { |
| $(id).addEventListener("change", render); |
| }); |
| render(); |
| })(); |
| |
| |
| (function () { |
| if (!D || !D.tw) { return; } |
| var tasks = D.tw.tasks; |
| var counts = D.tw.counts; |
| var KEYS = ["A", "B", "C", "D"]; |
| var sel = { A: true, B: true, C: true, D: true }; |
| |
| var chips = $("tw-chips"); |
| KEYS.forEach(function (k, i) { |
| var b = document.createElement("button"); |
| b.className = "chip"; |
| b.type = "button"; |
| b.setAttribute("aria-pressed", "true"); |
| b.innerHTML = "Spec. " + k + " <span class=\"cnt\">" + counts[k] + "/41</span>"; |
| b.addEventListener("click", function () { |
| sel[k] = !sel[k]; |
| b.setAttribute("aria-pressed", sel[k] ? "true" : "false"); |
| render(); |
| }); |
| chips.appendChild(b); |
| void i; |
| }); |
| |
| var grid = $("tw-grid"); |
| var cells = tasks.map(function (t) { |
| var c = document.createElement("div"); |
| c.className = "tcell"; |
| grid.appendChild(c); |
| bindTip(c, function () { |
| var solvers = KEYS.filter(function (k) { return t.bits[KEYS.indexOf(k)] === "1"; }); |
| return "<b>" + t.id + "</b><br>solved by: " + |
| (solvers.length ? "Spec. " + solvers.join(", ") : "no specialist") + |
| "<br>realized merge: " + (t.bits[4] === "1" ? "solved" : "unsolved"); |
| }); |
| return c; |
| }); |
| |
| function unionOf(keys) { |
| return tasks.filter(function (t) { |
| return keys.some(function (k) { return t.bits[KEYS.indexOf(k)] === "1"; }); |
| }); |
| } |
| |
| function render() { |
| var on = KEYS.filter(function (k) { return sel[k]; }); |
| var u = unionOf(on); |
| var uIds = {}; |
| u.forEach(function (t) { uIds[t.id] = 1; }); |
| var cmp = $("tw-cmp").checked; |
| |
| cells.forEach(function (c, i) { |
| var t = tasks[i]; |
| var inU = !!uIds[t.id]; |
| var inM = t.bits[4] === "1"; |
| c.className = "tcell" + (inU ? " on" : "") + |
| (cmp && inM && !inU ? " gain" : "") + (cmp && inU && !inM ? " loss" : ""); |
| }); |
| |
| var best = on.length ? Math.max.apply(null, on.map(function (k) { return counts[k]; })) : 0; |
| $("tw-readout").innerHTML = |
| ro("Selected", on.length ? on.join(" + ") : "none", true) + |
| ro("Union coverage", u.length + "<small>/41</small>") + |
| ro("Best single selected", (best || "\u2014") + (best ? "<small>/41</small>" : "")) + |
| ro("Realized merge", counts.M + "<small>/41</small>"); |
| |
| var note = ""; |
| if (on.length === 4) { |
| note = "All four specialists together cover <b>29/41</b>, two more than the best single lineage " + |
| "(Spec. D at 27). The harness DarwinX actually produced solves <b>28/41</b>: recombination is not " + |
| "a free set union. Turn on the comparison to see the difference \u2014 the merge picks up " + |
| "<code>tw_448247</code>, which <i>no</i> specialist solves, and drops <code>tw_449421</code> and " + |
| "<code>tw_498533</code>. These are single attempts, so individual flips sit inside the noise band " + |
| "(one task = 2.4 points)."; |
| } else if (on.length === 0) { |
| note = "Nothing selected. 12 of the 41 held-out tasks are solved by none of the four specialists."; |
| } else if (u.length > best) { |
| note = "This selection covers <b>" + u.length + "/41</b>, which is " + (u.length - best) + |
| " more than its best individual member \u2014 the specialists solve overlapping but distinct subsets."; |
| } else { |
| note = "This selection covers <b>" + u.length + "/41</b>, no more than its best individual member: " + |
| "these lineages are not complementary."; |
| } |
| $("tw-note").innerHTML = note; |
| |
| var lg = $("tw-legend"); |
| lg.innerHTML = |
| "<span><i class=\"sw\" style=\"background:var(--blue)\"></i> solved by selection</span>" + |
| "<span><i class=\"sw\" style=\"background:#e6ecf2\"></i> unsolved</span>" + |
| (cmp ? "<span><i class=\"sw\" style=\"box-shadow:0 0 0 2px #e8a33d;background:#e6ecf2\"></i>" + |
| " merge solves, selection does not</span>" + |
| "<span><i class=\"sw\" style=\"box-shadow:0 0 0 2px #d05a4e;background:var(--blue)\"></i>" + |
| " selection solves, merge does not</span>" : ""); |
| } |
| |
| function ro(k, v, plain) { |
| return "<div class=\"ro" + (plain ? "" : " hi") + "\"><div class=\"k\">" + k + |
| "</div><div class=\"v\"" + (plain ? " style=\"font-size:0.95rem;font-weight:600\"" : "") + ">" + |
| v + "</div></div>"; |
| } |
| |
| $("tw-cmp").addEventListener("change", render); |
| render(); |
| })(); |
| |
| |
| (function () { |
| if (!D || !D.tb21Clusters) { return; } |
| var data = D.tb21Clusters.slice(); |
| var modes = [ |
| { id: "delta", label: "gain", cmp: function (a, b) { return b.delta - a.delta; } }, |
| { id: "base", label: "base rate", cmp: function (a, b) { return a.base - b.base; } }, |
| { id: "n", label: "cluster size", cmp: function (a, b) { return b.n - a.n; } }, |
| { id: "name", label: "name", cmp: function (a, b) { return a.name.localeCompare(b.name); } } |
| ]; |
| var active = "delta"; |
| |
| var chips = $("cl-chips"); |
| modes.forEach(function (m) { |
| var b = document.createElement("button"); |
| b.className = "chip"; |
| b.type = "button"; |
| b.textContent = m.label; |
| b.setAttribute("aria-pressed", m.id === active ? "true" : "false"); |
| b.addEventListener("click", function () { |
| active = m.id; |
| Array.prototype.forEach.call(chips.children, function (c) { |
| c.setAttribute("aria-pressed", c === b ? "true" : "false"); |
| }); |
| render(); |
| }); |
| chips.appendChild(b); |
| }); |
| |
| function render() { |
| var mode = modes.filter(function (m) { return m.id === active; })[0]; |
| var rows = data.slice().sort(mode.cmp); |
| var showBase = $("cl-base").checked; |
| var box = $("cl-bars"); |
| box.innerHTML = ""; |
| rows.forEach(function (d) { |
| var el = document.createElement("div"); |
| el.className = "cbar"; |
| el.innerHTML = |
| "<div class=\"cname\">" + d.name + " <span class=\"cn\">n=" + d.n + "</span></div>" + |
| "<div class=\"track\">" + |
| (showBase ? "<div class=\"fill b\" style=\"left:0;width:" + d.base + "%\"></div>" : "") + |
| "<div class=\"fill e\" style=\"left:0;width:" + d.evolved + "%;height:" + |
| (showBase ? "58%;top:21%" : "100%;top:0") + "\"></div>" + |
| "</div>" + |
| "<div class=\"cd " + (d.delta >= 0 ? "up" : "down") + "\">" + |
| (d.delta >= 0 ? "+" : "\u2212") + Math.abs(d.delta).toFixed(1) + "</div>"; |
| bindTip(el.querySelector(".track"), function () { |
| return "<b>" + d.name + "</b><br>base " + d.base.toFixed(1) + "% \u2192 evolved " + |
| d.evolved.toFixed(1) + "%<br>" + d.n + " tasks, \u0394 " + |
| (d.delta >= 0 ? "+" : "\u2212") + Math.abs(d.delta).toFixed(1) + " points"; |
| }); |
| box.appendChild(el); |
| }); |
| } |
| |
| $("cl-base").addEventListener("change", render); |
| render(); |
| })(); |
| |
| |
| (function () { |
| if (!D || !D.waiCurve) { return; } |
| var s = D.waiCurve; |
| var svg = $("cv-chart"); |
| var W = 860, H = 320, L = 44, R = 14, T = 14, B = 34; |
| var iw = W - L - R, ih = H - T - B; |
| var NS = "http://www.w3.org/2000/svg"; |
| |
| function x(i) { return L + (s.length === 1 ? iw / 2 : (i / (s.length - 1)) * iw); } |
| function y(v) { return T + ih - (v / 100) * ih; } |
| function mk(n, a) { |
| var e = document.createElementNS(NS, n); |
| Object.keys(a).forEach(function (k) { e.setAttribute(k, a[k]); }); |
| return e; |
| } |
| |
| function render() { |
| while (svg.firstChild) { svg.removeChild(svg.firstChild); } |
| |
| [0, 25, 50, 75, 100].forEach(function (v) { |
| svg.appendChild(mk("line", { class: "gl", x1: L, x2: W - R, y1: y(v), y2: y(v) })); |
| var t = mk("text", { class: "lbl", x: L - 8, y: y(v) + 3.5, "text-anchor": "end" }); |
| t.textContent = v; |
| svg.appendChild(t); |
| }); |
| svg.appendChild(mk("line", { class: "ax", x1: L, x2: L, y1: T, y2: T + ih })); |
| svg.appendChild(mk("line", { class: "ax", x1: L, x2: W - R, y1: T + ih, y2: T + ih })); |
| |
| [0, 9, 18, 27, 36].forEach(function (i) { |
| if (i >= s.length) { return; } |
| var t = mk("text", { class: "lbl", x: x(i), y: T + ih + 17, "text-anchor": "middle" }); |
| t.textContent = i; |
| svg.appendChild(t); |
| }); |
| var xl = mk("text", { class: "lbl", x: L + iw / 2, y: H - 3, "text-anchor": "middle" }); |
| xl.textContent = "archive node"; |
| svg.appendChild(xl); |
| var yl = mk("text", { class: "lbl", x: 12, y: T + ih / 2, "text-anchor": "middle", |
| transform: "rotate(-90 12 " + (T + ih / 2) + ")" }); |
| yl.textContent = "screening score (%)"; |
| svg.appendChild(yl); |
| |
| if ($("cv-best").checked) { |
| var m = -1, pts = []; |
| s.forEach(function (v, i) { |
| m = Math.max(m, v); |
| if (i > 0) { pts.push(x(i) + "," + y(m)); } |
| pts.push(x(i) + "," + y(m)); |
| }); |
| svg.appendChild(mk("polyline", { class: "best", points: pts.join(" ") })); |
| } |
| |
| if ($("cv-all").checked) { |
| s.forEach(function (v, i) { |
| var dot = mk("circle", { class: "dot", cx: x(i), cy: y(v), r: 3.4 }); |
| svg.appendChild(dot); |
| var hit = mk("circle", { class: "hit", cx: x(i), cy: y(v), r: 11 }); |
| hit.addEventListener("mousemove", function (e) { |
| dot.classList.add("act"); |
| showTip("<b>archive node " + i + "</b><br>screening score " + v.toFixed(1) + "%", e); |
| }); |
| hit.addEventListener("mouseleave", function () { dot.classList.remove("act"); hideTip(); }); |
| svg.appendChild(hit); |
| }); |
| } |
| } |
| |
| var best = Math.max.apply(null, s); |
| $("cv-note").innerHTML = "The run evaluates <b>" + s.length + "</b> variants. The screening score " + |
| "climbs from " + s[0].toFixed(1) + "% to <b>" + best.toFixed(1) + "%</b>, but not monotonically: " + |
| "individual variants score well below the running best, and those are the ones the " + |
| "preserve-and-extend contract reverts. Screening runs on synthetic intents, so this axis is the " + |
| "in-loop proxy, not the 1,260-task real result."; |
| |
| $("cv-all").addEventListener("change", render); |
| $("cv-best").addEventListener("change", render); |
| render(); |
| })(); |
| |
| |
| (function () { |
| function num(td) { |
| var v = parseFloat(td.textContent.replace(/[^0-9.\-\u2212]/g, "").replace("\u2212", "-")); |
| return isNaN(v) ? null : v; |
| } |
| Array.prototype.forEach.call(document.querySelectorAll("table.sortable"), function (tb) { |
| var ths = tb.querySelectorAll("thead th"); |
| Array.prototype.forEach.call(ths, function (th, ci) { |
| th.classList.add("sortable"); |
| th.addEventListener("click", function () { |
| var body = tb.querySelector("tbody"); |
| var rows = Array.prototype.slice.call(body.rows); |
| var pinned = rows.filter(function (r) { return r.classList.contains("total"); }); |
| var movable = rows.filter(function (r) { return !r.classList.contains("total"); }); |
| if (!movable.length) { return; } |
| |
| |
| var numeric = movable.some(function (r) { return num(r.cells[ci]) !== null; }); |
| var desc; |
| if (th.classList.contains("desc")) { desc = false; } |
| else if (th.classList.contains("asc")) { desc = true; } |
| else { desc = numeric; } |
| |
| Array.prototype.forEach.call(ths, function (o) { o.classList.remove("asc", "desc"); }); |
| th.classList.add(desc ? "desc" : "asc"); |
| |
| movable.sort(function (a, b) { |
| var x = num(a.cells[ci]), y = num(b.cells[ci]); |
| if (x === null || y === null) { |
| var av = a.cells[ci].textContent.trim(), bv = b.cells[ci].textContent.trim(); |
| return desc ? bv.localeCompare(av) : av.localeCompare(bv); |
| } |
| return desc ? y - x : x - y; |
| }); |
| movable.concat(pinned).forEach(function (r) { body.appendChild(r); }); |
| }); |
| }); |
| }); |
| })(); |
| |
| |
| $("copybtn").addEventListener("click", function () { |
| var btn = this; |
| navigator.clipboard.writeText($("bib").textContent).then( |
| function () { |
| btn.textContent = "Copied"; |
| setTimeout(function () { btn.textContent = "Copy"; }, 1600); |
| }, |
| function () { btn.textContent = "Press \u2318C"; } |
| ); |
| }); |
| })(); |
| </script> |
|
|
| </body> |
| </html> |
|
|