File size: 27,476 Bytes
e0414df
58b23e0
 
 
 
 
e0414df
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
efae0a1
 
 
 
 
 
 
 
e0414df
 
 
 
 
 
 
 
 
0c38e08
e0414df
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
9018414
 
 
 
efae0a1
9018414
e0414df
 
 
 
 
 
 
 
 
efae0a1
e0414df
 
58b23e0
e0414df
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
'use strict';
function recordingRepairHistory(t,e){
 const old=t.recording_replacements||[], diagnostics=t.predicate_diagnostics||[];
 if(!old.length&&!diagnostics.length)return '';
 return panel('Recording repair and native diagnostics',`<div class="repair-history">${old.map(h=>`<p>Earlier native ${h.success?'success':'failure'} · ${num(h.steps)} steps. Original native verdict and evidence retained; video repair preserves every available source frame.</p><div class="downloads">${['video','native_session','verdict','protocol','usage','provenance','terminal_frame'].filter(k=>h.links[k]).map(k=>external(h.links[k],k.replaceAll('_',' '))).join('')}</div>`).join('')}${diagnostics.map(h=>`<p>Independent seed-0 diagnostic · native ${h.success?'success':'failure'} · ${num(h.steps)} steps. Same instruction and native verifier; excluded from formal-selection totals.</p><p>${esc(h.analysis.diagnosis||h.analysis.diagnostics||'')}</p><div class="video-panel"><video controls playsinline preload="metadata" src="${esc(h.links.video)}" poster="${esc(h.links.poster)}"></video></div><div class="downloads">${['owner_diagnostics','transcript','native_session','verdict','protocol','usage','provenance','terminal_frame'].filter(k=>h.links[k]).map(k=>external(h.links[k],k.replaceAll('_',' '))).join('')}</div>`).join('')}</div>`);
}
const $=s=>document.querySelector(s), esc=v=>String(v??'').replace(/[&<>"']/g,c=>({'&':'&amp;','<':'&lt;','>':'&gt;','"':'&quot;',"'":'&#39;'}[c]));
const num=v=>v==null?'—':Number(v).toLocaleString('en-US');
let data,routeEpoch=0;
const symbol='<svg viewBox="0 0 100 100" aria-hidden="true"><path d="M50 9 87 30v40L50 91 13 70V30Z M13 30l37 22 37-22 M50 52v39 M31 19l38 22v20 M31 60l19 11 19-11"/><circle cx="50" cy="52" r="5"/></svg>';
const familyName=id=>data.families.find(f=>f.id===id).name;
const external=(url,label)=>`<a href="${esc(url)}" target="_blank" rel="noopener">${esc(label)} ↗</a>`;
const measuredInstruction=e=>e?.instruction ?? e?.effective_instruction ?? e?.native_instruction;
const outcome=e=>e.success===true?'Successful':e.success===false?'Unsuccessful':'Unknown';
const elapsed=s=>s==null?'—':`${Math.floor(s/60)}m ${Math.round(s%60)}s`;

function modifiedText(t){
 const r=t.instruction_revision;if(!r)return esc(t.native_instruction||t.catalog_instruction);
 return r.diff.filter(p=>p.modified).map(p=>p.op==='equal'?esc(p.modified):`<mark>${esc(p.modified)}</mark>`).join(r.preserve_whitespace?'':' ');
}
function instructionPanel(t,e){
 const r=t.instruction_revision;
 if(!r)return panel(e?'Original native instruction':'Upstream catalog description',`<blockquote class="native-goal original">${esc(e?.native_instruction||t.catalog_instruction)}</blockquote>${e?`<p class="source-note">${external(e.links.native_goal,'Instruction evidence')}</p>`:''}`);
 const earlier=e?.instruction_policy==='modified' && measuredInstruction(e)!==r.instruction;
 const measured=earlier?'This recording used an earlier modified instruction. The current revised definition is shown for comparison.':e?(e.instruction_policy==='modified'?'This recording used the modified instruction.':'This recording used the original instruction. The revised definition is shown for comparison.'):'This revised task has not been evaluated yet.';
 return panel('Modified instruction',`<p class="instruction-context">${measured}</p><blockquote class="native-goal modified">${modifiedText(t)}</blockquote><p class="source-note">Highlighted words show additions and replacements. ${external(r.source_url,'Instruction source')}</p><details class="instruction-original"><summary>${r.native_instruction_kind==='episode_template'?'Original instruction template':'Original native instruction'}</summary><blockquote class="native-goal original">${esc(t.family==='task04'?(e?.native_instruction||r.native_instruction):r.native_instruction)}</blockquote>${r.native_instruction_kind==='episode_template'?'<p class="source-note">Object names and target assignments are supplied by the current scene.</p>':''}</details>${e && t.family!=="task04" && e.native_instruction!==r.native_instruction?`<details class="instruction-recorded-native"><summary>Native instruction recorded in this scene</summary><blockquote class="native-goal recorded">${esc(e.native_instruction)}</blockquote></details>`:""}${earlier?`<details class="instruction-measured"><summary>Instruction used in this recording</summary><blockquote class="native-goal measured">${esc(measuredInstruction(e))}</blockquote></details>`:''}${e?`<p class="source-note">${external(e.links.native_goal,'Measured instruction evidence')}</p>`:''}`);
}


function visualReviewPanel(e){
 const r=e.visual_review;if(!r)return '';
 const labels={consistent_pass:'画面与原生通过一致',confirmed_false_positive:'误通过:画面未完成,但原生 verifier 判定通过',consistent_failure:'画面与原生失败一致'};
 const label=labels[r.status]||'原生条件与画面复核';
 const text=e.analysis?.visible_completion||r.review||'';
 return panel('独立视频复核',`<p><strong>${esc(label)}</strong></p><p>${esc(text)}</p><p class="source-note">原生分数保持不变。${external(e.links.independent_review||e.links.analysis,'完整复核记录')}</p>`,'wide-panel');
}

function failureBadge(r){return `<span class="failure-badge ${esc(r.category)}">${r.status==='provisional'?'可能 · ':''}${esc(r.label)}</span>`;}
function failureTable(episodes,id){
 return `<details id="${id}" class="failure-list"><summary>查看全部 ${episodes.length} 项失败说明</summary><div class="table-wrap"><table><thead><tr><th>Task</th><th>分类</th><th>失败原因</th></tr></thead><tbody>${episodes.map(e=>`<tr><td><a href="#task/${esc(e.task_key)}">${esc(data.tasks.find(t=>t.key===e.task_key)?.title || e.task_key)}</a><small>${esc(data.tasks.find(t=>t.key===e.task_key)?.display_key || e.task_key)}</small></td><td>${failureBadge(e.failure_review)}</td><td>${esc(e.failure_review.reason)}</td></tr>`).join('')}</tbody></table></div></details>`;
}
function failureCategories(review){
 return `<div class="failure-categories">${review.categories.map(c=>`<button class="failure-category" data-category="${esc(c.id)}"><b>${esc(c.label)} <span>${c.count}</span></b><p>${esc(c.description)}</p><small>${c.exact_predicate_count!==undefined?`运行记录已核对 · ${c.exact_predicate_count} 项唯一谓词已定位`:c.provisional?`${c.provisional} 项具体根因仍待确认`:'运行证据支持'}</small></button>`).join('')}</div>`;
}
function renderFailureOverview(){
 const failures=data.episodes.filter(e=>!e.success);
 $('#failure-overview').innerHTML=`<div class="section-top"><h2>当前结果与失败分析</h2><a href="attempt-history.json" target="_blank" rel="noopener">Attempt history ↗</a></div><p class="section-note">${data.tasks.length} 个任务 · ${data.episodes.filter(e=>e.success).length} 个成功 · ${failures.length} 个原生失败 · ${data.tasks.length-data.episodes.length} 个待发布。原生判定与完整 session 已保留;具体失败机制尚未逐项定位。服务端中断及其补跑记录见尝试历史。</p>`;
 $('#failure-filter').innerHTML='<option value="all">All results</option><option value="failed">All unsuccessful results</option>';
}
const successPercent=value=>value==null?'—':(value*100).toFixed(2)+'%';
function successSummary(f){
 return `<span class="rate">${successPercent(f.success_rate)}</span><small>${f.valid_results?`${f.successes} / ${f.valid_results} valid results`:'Not evaluated'}</small>`;
}
function renderOverview(){
 $('#readiness-note').hidden=!data.selection_pending?.length;
 $('#readiness-note').innerHTML=data.selection_pending?.length?'RoboTwin task03/10 is awaiting selection after seed-0 initialization failed. No model episode or score is recorded. <a href="#task/task03/10">View preparation evidence ↗</a>':'';
 $('#edition-count').innerHTML=`<strong>${data.episodes.length} / ${data.tasks.length}</strong> results available · ${data.tasks.length-data.episodes.length} pending`;
 $('#edition-label').textContent=`${data.episodes.filter(e=>e.phase==='formal').length} formal results · ${data.episodes.filter(e=>e.phase==='preflight').length} preserved preflight`;

 $('#overview').innerHTML=data.families.map(f=>`<tr data-family="${f.id}"><td><a href="#${f.id}">${f.id.replace('task','Task')} · ${esc(f.name)}</a></td><td>${f.completed}/${f.total}</td><td>${successSummary(f)}</td><td>${num(f.input_tokens)}<small>${f.cached_input_tokens==null?'':num(f.cached_input_tokens)+' cached'}</small></td><td>${num(f.output_tokens)}</td><td>${f.completed?`${f.usage_complete} / ${f.completed}<small>Complete</small>`:'—<small>No usage yet</small>'}</td></tr>`).join('');
 $('#family-nav').innerHTML=data.families.map(f=>`<a href="#${f.id}">${f.id.replace('task','Task')} · ${esc(f.name)} <span class="muted">${f.total}</span></a>`).join('');
 renderFailureOverview();renderCards();
}
function renderCards(){
 const query=$('#search').value.toLowerCase(),status=$('#status-filter').value,category=$('#failure-filter').value;
 const filtered=data.tasks.filter(t=>{const e=data.episodes.find(e=>e.id===t.episode_id),r=e?.failure_review;return (status==='all'||t.status===status)&&(category==='all'||category==='failed'&&e&&!e.success||r?.category===category)&&[t.display_key || t.key,t.title,t.native_id,t.catalog_instruction,t.native_instruction||'',r?.label||'',r?.reason||''].join(' ').toLowerCase().includes(query);});
 $('#families').innerHTML=data.families.map(f=>{
  const tasks=filtered.filter(t=>t.family===f.id);if(!tasks.length)return '';
  return `<section class="family-section" id="${f.id}"><div class="family-heading"><div><h2>${f.id.replace('task','Task')} · ${esc(f.name)}</h2><p>${f.total} tasks · ${f.completed} result${f.completed===1?'':'s'} available · ${f.pending} pending${f.completed?' · '+f.successes+' native successes':''}</p></div><span class="family-number">${f.id.slice(-2)}</span></div><div class="cards">${tasks.map(t=>{
   const e=data.episodes.find(e=>e.id===t.episode_id);
   return `<a class="card" href="#task/${t.key}" data-task="${t.key}" aria-label="${esc(t.title)} — ${e?'View result':'Pending'}"><div class="card-media">${e?`<img loading="lazy" src="${esc(e.links.poster)}" alt="${esc(f.name)} native episode recording"><span class="media-label">4× VIDEO · ${e.phase.toUpperCase()}</span>`:`<div class="pending-art">${symbol}<span>${esc(f.name.toUpperCase())} / PENDING</span></div>`}</div><div class="card-body"><div class="slot">${t.family} / ${t.display_slot || t.slot}</div><h3>${esc(t.title)}</h3><div class="instruction-label">${t.instruction_revision?'Modified instruction':e?'Original native instruction':'Upstream catalog description'}</div><p class="instruction">${modifiedText(t)}</p>${t.instruction_revision&&(!e || measuredInstruction(e)!==t.instruction_revision.instruction)?`<small class="measured-version">${e?(e.instruction_policy==="modified"?"Recorded result: earlier modified instruction":"Recorded result: original instruction"):"Revised task · not yet evaluated"}</small>`:""}${e?.failure_review?`<div class="card-failure">${failureBadge(e.failure_review)}<p>${esc(e.failure_review.reason)}</p></div>`:''}</div><div class="card-footer"><span class="${e?(e.success?'status-success':'status-fail'):'muted'}">${e?outcome(e)+' · '+(e.valid?'valid':'review required'):(t.readiness?'Pending · setup blocked':'Pending · no published result')}</span><span>${e?'View result':'View task'} ↗</span></div></a>`;
  }).join('')}</div></section>`;
 }).join('');
 $('#filter-count').textContent=`${filtered.length} / ${data.tasks.length} tasks`;$('#empty').hidden=filtered.length>0;
}
function panel(title,body,cls=''){return `<section class="panel ${cls}"><h3>${title}</h3>${body}</section>`;}
function simpleAttemptHistory(t,e){
 if(!t.simple_rerun)return '';
 const history=t.attempt_history||[];
 const current=e.provenance.max_steps||t.planned_protocol.max_control_steps;
 return panel('Attempt history',`<div class="simple-attempt-history"><p>Current result: attempt ${e.provenance.attempt||1} · ${num(current)} action-step limit.</p><p>${esc(t.simple_rerun.note)}</p>${history.map(h=>`<div class="previous-attempt"><p>Previous attempt · ${num(h.max_steps)} action-step limit · native ${h.success?'success':'failure'} · ${num(h.steps)} steps used. Retained outside current-selection totals.</p>${h.visual_review?`<p>独立复核:${esc(h.visual_review.review)}</p>`:''}<div class="downloads">${['video','native_session','verdict','protocol','usage','provenance','episode_record','independent_review'].filter(k=>h.links[k]).map(k=>external(h.links[k],k.replaceAll('_',' '))).join('')}</div></div>`).join('')}</div>`);
}
function taskHeader(t,e){return `<a class="back" href="#${t.family}">← Back to ${esc(familyName(t.family))}</a><div class="detail-header"><div><div class="slot">${t.family} / ${t.display_slot || t.slot} · ${esc(familyName(t.family))}</div><h2>${esc(t.title)}</h2><div class="detail-meta">Codex ${esc(e?'v'+e.codex_version:'benchmark')} · GPT-6 Astra · high · seed 0 · ${e?'1 measured episode':'1 planned episode'}</div><div class="badge-row">${e?`<span class="badge ${e.success?'':'fail'}">Native: ${outcome(e).toLowerCase()}</span><span class="badge">Execution: ${esc(e.execution.status)}</span><span class="badge">Evidence: ${e.verdict.evidence_valid?'valid':'invalid'}</span><span class="badge">Usage: ${e.usage.audit_complete?'complete':'incomplete'}</span>`:'<span class="badge">Pending · no published result</span>'}</div></div>${e?`<span class="chip">${e.phase==='preflight'?'Preflight record':'Formal evaluation'}</span>`:''}</div>`;}
function renderPending(t){
 const p=t.planned_protocol,r=t.readiness;
 return taskHeader(t,null)+`<section class="pending-detail">${symbol}<h3>${r?'Seed 0 initialization blocked':'Evaluation pending'}</h3><p>${r?esc(r.message):esc(t.status_note || 'Queued for evaluation.')}</p>${r?`<p>${external(r.evidence,'Native initialization evidence')}</p>`:''}<div class="pending-protocol"><span class="badge">Seed 0</span><span class="badge">${num(p.max_control_steps)} control steps · ${p.control_frequency_hz} Hz</span><span class="badge">8-hour limit</span><span class="badge">Stepped</span></div></section>`+instructionPanel(t,null)+(t.attempt_history?.length?panel('Preserved interrupted attempts',t.attempt_history.map(h=>external(h.links.provenance,'Attempt provenance')+' · '+external(h.links.native_session,'Native session')).join('<br>')):'');
}
async function renderResult(t,e,epoch){
 const u=e.usage,m=e.media,views=m.view_names||m.recording.views.map(v=>v.name);
 const metrics=[[`${num(e.steps)} / ${num(t.planned_protocol.max_control_steps)}`,'Native control steps',e.simulation_time_s==null?`${m.recording.end_time_s.toFixed(2)} recorded simulation seconds`:`${e.simulation_time_s.toFixed(2)} simulation seconds`],[elapsed(e.wall_time_s),'Agent wall time','8-hour safety limit'],[num(u.accounting==='reported-responses'?u.response_count:(u.request_attempts??u.response_count)),u.accounting==='reported-responses'?'Model responses':'Model requests',u.accounting==='reported-responses'?'Provider-reported responses':`${num(u.reported_responses?.input_tokens)} / ${num(u.request_attempts??u.response_count)} terminal usage reported`],[num(e.call_activity.model_tool_calls),'Model tool calls','Nested RPC count unavailable'],[num(u.input_tokens),'Input tokens',num(u.cached_input_tokens)+' cached'],[num(u.output_tokens),'Output tokens',num(u.reasoning_output_tokens)+' reasoning']];
 const r=e.failure_review;const reviewEvidence=r?`<details><summary>失败判定依据 · ${esc(r.status_label)}</summary><div class="evidence-note"><p>${esc(r.basis)}</p><p>${external(e.links.review_frame,'查看录像末帧')} · ${external(e.links.failure_review,'完整判定记录')}</p>${r.source_links.map(s=>`<p>${external(s.url,s.path)}</p>`).join('')}</div></details>`:'';
 $('#detail').innerHTML=taskHeader(t,e)+visualReviewPanel(e)+simpleAttemptHistory(t,e)+(e.usage_note?`<div class="usage-note"><strong>Usage incomplete</strong><p>${esc(e.usage_note)}</p></div>`:'')+`<div class="video-panel"><div class="view-labels">${views.map(v=>`<span>${esc(v.replaceAll('_',' ').toUpperCase())}</span>`).join('')}</div><video id="episode-video" style="aspect-ratio:${m.width}/${m.height}" controls playsinline preload="metadata" poster="${esc(e.links.poster)}" src="${esc(e.links.video)}"></video><div class="video-toolbar"><span>Native spectator recording · ${m.width} × ${m.height} · ${m.duration_s.toFixed(2)} s</span><label>Playback <select id="playback-speed" aria-label="Video playback speed"><option value="0.25">1× simulation / 0.25× video</option><option value="0.5">2× simulation / 0.5× video</option><option value="1" selected>4× simulation / 1× video</option><option value="2">8× simulation / 2× video</option></select></label></div></div><p class="video-note">${views.length} native ${m.recording.views[0].height}p views captured at ${m.source_fps} FPS; ${m.recording.dropped_samples?num(m.recording.dropped_samples)+' capture samples dropped; ':''}edited to 4× simulation speed and ${m.output_fps} FPS. ${m.terminal_hold_s?`All ${num(m.native_frames)} recorded source frames retained; the true final frame is held for ${m.terminal_hold_s} seconds without advancing simulation. ${e.links.terminal_frame?external(e.links.terminal_frame,"Final recorded frame"):""}`:""} Agent events below use wall time and are not frame-synchronized.</p><div class="detail-grid"><div>${instructionPanel(t,e)}${data.instruction_correction?.task_key===t.key?panel("屋顶方向说明的纠正",`<div class="instruction-correction"><p>${esc(data.instruction_correction.summary)}</p><p>${external(data.instruction_correction.evidence,"网格与原生条件核对")} · ${external(data.instruction_correction.video,"错误措辞回合的视频")} · ${external(data.instruction_correction.session,"完整 session")} · ${external(data.instruction_correction.native_verdict,"原生判定")}</p></div>`):""}${data.supplemental_verifications?.[t.key]?panel('Verifier 与环境不对齐',`<div class="button-verification"><p>${esc(data.supplemental_verifications[t.key].summary)}</p><p>独立物理验证 · 0 次模型调用 · 不计入正式成功率</p><a href="${esc(data.supplemental_verifications[t.key].report)}">查看按下、撤手与回弹记录 →</a></div>`):''}${panel('运行分析',`${r?`<div class="failure-verdict">${failureBadge(r)}<p>${esc(r.reason)}</p></div>`:''}<div class="analysis">${Object.entries(e.analysis).filter(([k])=>k!=='preflight').map(([,v])=>`<p>${esc(v)}</p>`).join('')}</div>${r?.trace_event_ids?`<div class="review-events"><p>查看引用的 session 事件:</p>${r.trace_event_ids.map(id=>`<button class="review-event" data-event-id="${esc(id)}">${esc(id)}</button>`).join('')}</div>`:''}`,'wide-panel')}</div><div>${panel('Episode metrics',`<div class="metrics">${metrics.map(([v,l,s])=>`<div class="metric"><b>${v}</b><span>${l}</span><small>${s}</small></div>`).join('')}</div><p class="metric-note">Native reward: ${num(e.native_reward??(e.success?1:0))} · termination: ${esc(e.verdict.termination)}<br>Uncached input: ${num(u.uncached_input_tokens)} · cost: ${u.cost_usd==null?'unavailable':u.cost_usd}<br>Execution: ${esc(e.execution.status)}${e.execution.reason?' · '+esc(e.execution.reason):''}.</p>`)}${e.phase==='preflight'?`<div class="preflight-note"><strong>Preserved preflight result</strong><p>${esc(e.analysis.preflight)}</p>${external(e.links.provenance,'Build provenance')}</div>`:''}</div></div>`+
 panel('Session & observations',`<p class="section-note" id="session-counts">Loading session counts… Tool inputs and outputs are expandable; the native session download retains the full event format. Event timestamps are in UTC.</p><div class="timeline-controls"><label><span class="sr-only">Filter session events</span><select id="event-filter"><option value="all">All events</option><option value="tools">Tool results</option><option value="errors">Tool errors</option><option value="assistant">Assistant</option><option value="images">With images</option></select></label><input id="event-search" type="search" aria-label="Search session" placeholder="Search text, tool or event ID…"><span id="event-count" aria-live="polite">Loading…</span></div><div id="timeline"></div><button class="plain-button" id="more-events">Load more events</button><details id="gallery"><summary>Browse agent observation images</summary><div class="image-grid" id="observation-grid"></div></details>`,'wide-panel')+
 panel('Agent resources',e.resources.length?`<p class="section-note">Final tool and memo files written during this episode. Native session edits preserve their development history.</p><div class="resource-toolbar"><label class="sr-only" for="resource-select">Resource file</label><select id="resource-select">${e.resources.map(r=>`<option value="${esc(r.file)}">${esc(r.name)}</option>`).join('')}</select><a id="resource-download" target="_blank" rel="noopener">Open file ↗</a></div><pre id="resource-preview">Loading…</pre>`:'<p>No tool or memo files were saved in this episode. The session and workspace evidence remain available below.','wide-panel')+
 panel('Evidence & provenance',`<p class="section-note">Lightweight public evidence is stored in the Dataset. Host paths are redacted in copies; original event IDs, native verdicts and usage are preserved.</p>${reviewEvidence}<div class="downloads">${Object.entries(e.links).filter(([k])=>!['poster','transcript'].includes(k)).map(([k,v])=>external(v,k.replaceAll('_',' '))).join('')}</div><details><summary>Source revisions and measured build identities</summary><pre>${esc(JSON.stringify(e.provenance,null,2))}</pre></details>`,'wide-panel')+recordingRepairHistory(t,e);
 $('#playback-speed').onchange=()=>{$('#episode-video').playbackRate=Number($('#playback-speed').value);};
 if(e.resources.length){
  const loadResource=async()=>{const file=$('#resource-select').value;$('#resource-download').href=file;const r=await fetch(file);if(!r.ok)throw Error('Resource unavailable');const txt=await r.text();if(epoch===routeEpoch&&$('#resource-select').value===file)$('#resource-preview').textContent=txt;};
  $('#resource-select').onchange=()=>loadResource().catch(showError);await loadResource();if(epoch!==routeEpoch)return;
 }
 const response=await fetch(e.links.transcript);if(!response.ok)throw Error('Could not load transcript');const events=await response.json();if(epoch!==routeEpoch)return;
 const images=events.flatMap(v=>v.parts.filter(p=>p.type==='image'));
 $('#session-counts').textContent=`${events.length} visible session events · ${images.length} observed images. Tool inputs and outputs are expandable; the native session download retains the full event format. Event timestamps are in UTC.`;
 $('#gallery summary').textContent=`Browse ${images.length} agent observation images`;
 const base=e.asset_base||e.links.transcript.slice(0,e.links.transcript.lastIndexOf('/')+1);let count=24;
 function renderEvents(){
  const filter=$('#event-filter').value,q=$('#event-search').value.toLowerCase();
  const selected=events.filter(v=>(filter==='all'||filter==='tools'&&v.role==='toolResult'||filter==='errors'&&v.error||filter==='assistant'&&v.role==='assistant'||filter==='images'&&v.parts.some(p=>p.type==='image'))&&JSON.stringify(v).toLowerCase().includes(q));
  $('#event-count').textContent=`${Math.min(count,selected.length)} / ${selected.length} events`;
  $('#timeline').innerHTML=selected.slice(0,count).map(v=>{
   const text=v.parts.map(p=>p.type==='text'?p.text:p.type==='tool'?p.name:'[image]').join(' ').trim();
   return `<details class="event ${v.error?'error':''}" data-event="${esc(v.id)}"><summary><time>${esc((v.timestamp||'').slice(11,19))}</time><span class="role">${esc(v.tool||v.role)}</span><span class="event-text">${esc(text.slice(0,160)||v.id)}</span>${v.error?'<span class="status-fail">error</span>':''}</summary><div class="event-content"><div class="source-note">Event ${esc(v.id)} · ${esc(v.timestamp)}</div>${v.parts.map(p=>p.type==='image'?`<a href="${esc(base+p.src)}" target="_blank" rel="noopener"><img class="event-image" loading="lazy" src="${esc(base+p.src)}" alt="Agent observation at event ${esc(v.id)}"></a>`:`<pre>${esc(p.type==='tool'?JSON.stringify({tool:p.name,input:p.input},null,2):p.text)}</pre>`).join('')}</div></details>`;
  }).join('');$('#more-events').hidden=count>=selected.length;
 }
 document.querySelectorAll('.review-event').forEach(button=>button.onclick=()=>{$('#event-filter').value='all';$('#event-search').value=button.dataset.eventId;count=24;renderEvents();const event=document.querySelector('.event[data-event="'+button.dataset.eventId+'"]');if(event){event.open=true;event.scrollIntoView({block:'center'});}});
 $('#event-filter').onchange=()=>{count=24;renderEvents();};$('#event-search').oninput=()=>{count=24;renderEvents();};$('#more-events').onclick=()=>{count+=24;renderEvents();};renderEvents();
 $('#observation-grid').innerHTML=events.flatMap(v=>v.parts.filter(p=>p.type==='image').map(p=>`<a href="${esc(base+p.src)}" target="_blank" rel="noopener"><img loading="lazy" src="${esc(base+p.src)}" alt="Observed image, event ${esc(v.id)}"><span>${esc(v.id)}</span></a>`)).join('');
 $('#detail').dataset.ready='true';
}
function showError(e){console.error(e);const p=document.createElement('p');p.setAttribute('role','alert');p.textContent='An asset could not be loaded. Please refresh or open the Dataset evidence.';$('#detail').append(p);}
async function route(){
 const epoch=++routeEpoch,hash=decodeURIComponent(location.hash.slice(1));
 const video=$('#episode-video');if(video)video.pause();
 if(hash.startsWith('task/')){
  const [taskKey,query]=hash.slice(5).split("?");const task=data.tasks.find(t=>t.key===taskKey);$('#browse').hidden=true;$('#detail').hidden=false;$('#detail').dataset.ready='false';$('#detail').dataset.taskKey=taskKey;window.scrollTo(0,0);
  if(!task){$('#detail').innerHTML='<a class="back" href="#tasks">← All tasks</a><h2>Task not found</h2>';return;}
  document.title=task.title+' · Codex Benchmark';if(query)window.history.replaceState(null,'','#task/'+task.key);const e=data.episodes.find(e=>e.task_key===task.key&&e.id===task.episode_id);
  if(e)await renderResult(task,e,epoch);else{$('#detail').innerHTML=renderPending(task);$('#detail').dataset.ready='true';}
 }else{
  $('#detail').hidden=true;$('#detail').innerHTML='';$('#browse').hidden=false;document.title=data.title+' · RLE-Bench';
  if(hash){if(!document.getElementById(hash)&&data.families.some(f=>f.id===hash)){$('#search').value='';$('#status-filter').value='all';$('#failure-filter').value='all';renderCards();}requestAnimationFrame(()=>document.getElementById(hash)?.scrollIntoView());}else window.scrollTo(0,0);
 }
}
(async()=>{try{const r=await fetch('data.json');if(!r.ok)throw Error('Data index unavailable');data=await r.json();renderOverview();$('#loading').hidden=true;$('#search').oninput=renderCards;$('#status-filter').onchange=renderCards;$('#failure-filter').onchange=renderCards;window.addEventListener('hashchange',()=>route().catch(showError));await route();document.body.dataset.ready='true';}catch(e){$('#loading').textContent='Could not load benchmark data. Please refresh or open the Dataset.';console.error(e);}})();