File size: 10,222 Bytes
7a3b3da
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
#!/usr/bin/env python3
"""Emit assets/data.js for the DarwinX project page from the original run artifacts.

Every number the interactive figures display comes from the files listed in SOURCES,
so the page can be re-derived rather than hand-transcribed. Run from anywhere:

    python3 tools/build_data.py
"""

import json
import os
import sys
from datetime import datetime, timezone

TW_DIR = ("/Users/yifan.zhang/Desktop/techreport/darwinx_zeyuan_feedback/"
          "version_control/version_0717_local/results/tw/results")
NOTES = "/Users/yifan.zhang/Desktop/techreport/darwinx_zeyuan_feedback/notes"

TW_ARMS = [
    ("A", "Spec. A", "opus_node_34f71461_run.json"),
    ("B", "Spec. B", "opus_node_cab04955_run.json"),
    ("C", "Spec. C", "opus_node_56fc77cc_run.json"),
    ("D", "Spec. D", "opus_node_eb39901b_run.json"),
    ("M", "Merged",  "TW_HEADLINE_merge_opus_28of41_run.json"),
]

# From notes/TB21_RESULTS.md (paired protocol, 88 tasks). Deltas are carried over as
# reported rather than recomputed: the source rounds the delta from unrounded rates,
# so evolved-minus-base on the displayed 1-decimal values differs by up to 0.1.
TB21_CLUSTERS = [
    ("ML & scientific computing", 23, 60.1, 74.9, 14.8),
    ("Data & databases",           9, 83.9, 97.8, 13.8),
    ("Algorithms & coding",       14, 83.1, 90.0,  6.9),
    ("System admin & networking", 11, 92.0, 97.6,  5.7),
    ("Parsing, text & media",     11, 67.3, 72.2,  5.0),
    ("Systems & build",           11, 79.3, 81.8,  2.5),
    ("Security & crypto",          9, 85.1, 84.4, -0.6),
]

# The four headline panels, from scripts/gen_summary_figure.py GROUPS. Each panel is
# matched-model: base and evolved share one frozen base, so the gap is the harness.
# The prior-agent value is context, not a controlled comparison, because those
# systems use different models and effort settings. SWE-V has no prior-agent bar and
# its "base" is the fix-skill reference rather than an unevolved Monet, so the label
# travels with the number.
HEADLINE = [
    {"short": "TB 2.1", "name": "Terminal-Bench 2.1", "metric": "avg@5",
     "model": "GPT-5.5", "scope": "89 tasks",
     "baseLabel": "Monet (base)", "base": 75.5,
     "evolvedLabel": "Monet (DarwinX)", "evolved": 83.2,
     "priorLabel": "Codex", "prior": 83.1,
     "delta": 7.7, "lo": 70, "hi": 88},
    {"short": "T-World", "name": "TerminalWorld", "metric": "pass@1",
     "model": "Opus 4.8", "scope": "41 held-out tasks",
     "baseLabel": "Monet (base)", "base": 61.0,
     "evolvedLabel": "Monet (DarwinX)", "evolved": 68.3,
     "priorLabel": "Claude Code", "prior": 65.9,
     "delta": 7.3, "lo": 55, "hi": 75},
    {"short": "WA-Inf.", "name": "WebArena-Infinity", "metric": "audit-clean pass@1",
     "model": "GPT-5.5", "scope": "1,260 real tasks",
     "baseLabel": "Monet (base)", "base": 43.5,
     "evolvedLabel": "Monet (DarwinX)", "evolved": 93.0,
     "priorLabel": "GPT-5.5 + Browser Use", "prior": 86.1,
     "delta": 49.5, "lo": 35, "hi": 100},
    {"short": "SWE-V", "name": "SWE-bench Verified", "metric": "official pass@1",
     "model": "Opus 4.8", "scope": "500 instances, transfer only",
     "baseLabel": "Fix-skill reference", "base": 80.8,
     "evolvedLabel": "Monet (DarwinX), transferred", "evolved": 84.2,
     "priorLabel": None, "prior": None,
     "delta": 3.4, "lo": 75, "hi": 88},
]

# scripts/gen_tw_dynamics.py + notes/tw_dynamics.json. Base is the unevolved v1.0.0
# harness on Opus 4.8 (25/41); its per-task run.json is not on this machine, so the
# merge explorer covers only the arms that have per-task records.
TW_HELDOUT_BARS = [("Base", 25), ("Spec. A", 24), ("Spec. B", 25),
                   ("Spec. C", 26), ("Spec. D", 27), ("Monet (DarwinX)", 28)]
TW_CLAUDE_CODE = 27          # Claude Code (Opus 4.8) reference, 27/41
TW_N = 41

# scripts/gen_tb21_compute.py: medians precomputed from the trial archive.
TB21_COMPUTE = {
    "groups": ["Newly solved (6 tasks)", "Already solved (69 tasks)"],
    "turns": {"base": [11, 12], "evolved": [22, 13]},
    "tokens": {"base": [89, 125], "evolved": [380, 172]},   # thousands
}

# scripts/gen_wai_invalid_composition.py: 293 invalid trajectories before evolution,
# 17 after, decomposed two ways over the same totals.
WAI_INVALID = {
    "application": {
        "labels": ["Elation Clinical", "Elation Prescription", "GitLab Plan & Track",
                   "Gmail", "Gmail Acct & Contact", "Handshake", "Linear Acct Mgmt",
                   "PayPal Wallet", "Super-human", "Xero Invoicing"],
        "before": [29, 27, 38, 21, 44, 54, 26, 10, 44, 0],
        "after":  [2, 3, 0, 0, 3, 0, 0, 0, 8, 1],
        "colors": ["#4C72B0", "#A6C8E0", "#DD8452", "#E3B23C", "#55A868",
                   "#8FD19E", "#C44E52", "#E79EA6", "#8172B3", "#C7BFE0"],
    },
    "mechanism": {
        "labels": ["Evaluation plane access", "Privileged knowledge",
                   "Exploit or privilege escalation", "Raw state mutation"],
        "before": [155, 97, 26, 15],
        "after":  [0, 0, 0, 17],
        "colors": ["#55A868", "#DD8452", "#CE6DBD", "#8FA6D1"],
    },
}

# scripts/gen_wai_audit_by_app.py: pass@1 before and after the validity audit.
WAI_AUDIT = {
    "apps": ["Overall", "Elation-Clin", "Elation-Rx", "GitLab", "Gmail",
             "Gmail-Acct", "Handshake", "Linear", "PayPal", "Superhuman", "Xero"],
    "rawBase":      [53.0, 95.8, 28.3, 75.7, 45.0, 33.3, 44.0, 54.2, 50.0, 53.3, 49.2],
    "rawDx":        [94.4, 98.3, 97.5, 97.9, 98.3, 94.2, 84.0, 94.2, 95.7, 94.2, 97.5],
    "auditedBase":  [43.5, 95.8, 20.0, 63.6, 25.0, 21.7, 36.5, 43.3, 49.3, 31.7, 39.2],
    "auditedDx":    [93.0, 96.7, 95.0, 97.9, 98.3, 91.7, 84.0, 94.2, 95.7, 87.5, 96.7],
}


def load_tw():
    """Per-task resolved flags for the four specialists and the merged harness.

    The run files list the 41 tasks in different orders, so every arm is joined on
    task_id and the emitted order is canonical (the first arm's file order).
    """
    order, universe, sets = None, None, {}
    for key, _label, fname in TW_ARMS:
        path = os.path.join(TW_DIR, fname)
        if not os.path.exists(path):
            sys.exit(f"missing TerminalWorld run file: {path}")
        rows = json.load(open(path))["per_task_results"]
        ids = {r["task_id"] for r in rows}
        if len(ids) != len(rows):
            sys.exit(f"duplicate task_id in {fname}")
        if order is None:
            order, universe = [r["task_id"] for r in rows], ids
        elif ids != universe:
            sys.exit(f"task set differs in {fname}: "
                     f"{sorted(ids ^ universe)} not shared; refusing to emit")
        sets[key] = {r["task_id"] for r in rows if r["resolved"]}

    tasks = [{"id": t, "bits": "".join("1" if t in sets[k] else "0"
                                       for k, _l, _f in TW_ARMS)} for t in order]
    counts = {k: len(sets[k]) for k, _l, _f in TW_ARMS}
    if [counts[k] for k in "ABCD"] != [24, 25, 26, 27] or counts["M"] != 28:
        sys.exit(f"unexpected TerminalWorld totals {counts}; paper reports 24/25/26/27 and 28")
    return tasks, counts


def load_wai_curve():
    path = os.path.join(NOTES, "tw_dynamics.json")
    if not os.path.exists(path):
        sys.exit(f"missing {path}")
    scores = json.load(open(path))["wai_adaptive_scores"]
    return [round(float(s), 2) for s in scores]


def main():
    tasks, counts = load_tw()
    curve = load_wai_curve()

    payload = {
        "generated": datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M UTC"),
        "tw": {
            "arms": [{"key": k, "label": l} for k, l, _f in TW_ARMS],
            "tasks": tasks,
            "counts": counts,
        },
        "tb21Clusters": [
            {"name": n, "n": c, "base": b, "evolved": e, "delta": d}
            for n, c, b, e, d in TB21_CLUSTERS
        ],
        "waiCurve": curve,
        "headline": HEADLINE,
        "twBars": {
            "n": TW_N,
            "claudeCode": TW_CLAUDE_CODE,
            "bars": [{"label": l, "solved": s} for l, s in TW_HELDOUT_BARS],
        },
        "tb21Compute": TB21_COMPUTE,
        "waiInvalid": WAI_INVALID,
        "waiAudit": WAI_AUDIT,
    }

    # Guard the invariants the page states in prose.
    for view in ("application", "mechanism"):
        v = WAI_INVALID[view]
        if sum(v["before"]) != 293 or sum(v["after"]) != 17:
            sys.exit(f"WAI_INVALID[{view}] must total 293 before and 17 after, got "
                     f"{sum(v['before'])} and {sum(v['after'])}")
        if not (len(v["labels"]) == len(v["before"]) == len(v["after"]) == len(v["colors"])):
            sys.exit(f"WAI_INVALID[{view}] has ragged columns")
    for k in ("rawBase", "rawDx", "auditedBase", "auditedDx"):
        if len(WAI_AUDIT[k]) != len(WAI_AUDIT["apps"]):
            sys.exit(f"WAI_AUDIT[{k}] length does not match apps")
    # The audited numbers must agree with the per-application table on the page.
    page_table = [96.7, 95.0, 97.9, 98.3, 91.7, 84.0, 94.2, 95.7, 87.5, 96.7]
    if WAI_AUDIT["auditedDx"][1:] != page_table:
        sys.exit("WAI_AUDIT auditedDx disagrees with the page's per-application table")
    if [b["solved"] for b in payload["twBars"]["bars"]] != [25, 24, 25, 26, 27, 28]:
        sys.exit("twBars solved counts changed unexpectedly")

    out = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
                       "assets", "data.js")
    with open(out, "w") as f:
        f.write("/* Generated by tools/build_data.py - do not edit by hand. */\n")
        f.write("window.DARWINX = ")
        json.dump(payload, f, indent=1, sort_keys=False)
        f.write(";\n")

    union = {t["id"] for t in tasks if "1" in t["bits"][:4]}
    merged = {t["id"] for t in tasks if t["bits"][4] == "1"}
    print(f"wrote {out}")
    print(f"  TerminalWorld: {len(tasks)} tasks, counts={counts}")
    print(f"    union(A..D)={len(union)}  merged={len(merged)}")
    print(f"    merged only: {sorted(merged - union)}")
    print(f"    union only : {sorted(union - merged)}")
    print(f"  TB2.1 clusters: {len(payload['tb21Clusters'])}")
    print(f"  WAI curve: {len(curve)} nodes, max={max(curve)}")


if __name__ == "__main__":
    main()