Spaces:
Running
Running
Download app.js from RLE-Bench/codex-benchmark: direct link, hf CLI and curl.
- Browser
- Download file 27.5 kB
-
https://huggingface.co/spaces/RLE-Bench/codex-benchmark/resolve/main/app.js
- Command line
-
hf download hf://spaces/RLE-Bench/codex-benchmark/app.js
-
curl -L -o app.js https://huggingface.co/spaces/RLE-Bench/codex-benchmark/resolve/main/app.js
27.5 kB
| ; | |
| function recordingRepairHistory(t,e){ | |
| const old=t.recording_replacements||[], diagnostics=t.predicate_diagnostics||[]; | |
| if(!old.length&&!diagnostics.length)return ''; | |
| return panel('Recording repair and native diagnostics',`<div class="repair-history">${old.map(h=>`<p>Earlier native ${h.success?'success':'failure'} · ${num(h.steps)} steps. Original native verdict and evidence retained; video repair preserves every available source frame.</p><div class="downloads">${['video','native_session','verdict','protocol','usage','provenance','terminal_frame'].filter(k=>h.links[k]).map(k=>external(h.links[k],k.replaceAll('_',' '))).join('')}</div>`).join('')}${diagnostics.map(h=>`<p>Independent seed-0 diagnostic · native ${h.success?'success':'failure'} · ${num(h.steps)} steps. Same instruction and native verifier; excluded from formal-selection totals.</p><p>${esc(h.analysis.diagnosis||h.analysis.diagnostics||'')}</p><div class="video-panel"><video controls playsinline preload="metadata" src="${esc(h.links.video)}" poster="${esc(h.links.poster)}"></video></div><div class="downloads">${['owner_diagnostics','transcript','native_session','verdict','protocol','usage','provenance','terminal_frame'].filter(k=>h.links[k]).map(k=>external(h.links[k],k.replaceAll('_',' '))).join('')}</div>`).join('')}</div>`); | |
| } | |
| const $=s=>document.querySelector(s), esc=v=>String(v??'').replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c])); | |
| const num=v=>v==null?'—':Number(v).toLocaleString('en-US'); | |
| let data,routeEpoch=0; | |
| const symbol='<svg viewBox="0 0 100 100" aria-hidden="true"><path d="M50 9 87 30v40L50 91 13 70V30Z M13 30l37 22 37-22 M50 52v39 M31 19l38 22v20 M31 60l19 11 19-11"/><circle cx="50" cy="52" r="5"/></svg>'; | |
| const familyName=id=>data.families.find(f=>f.id===id).name; | |
| const external=(url,label)=>`<a href="${esc(url)}" target="_blank" rel="noopener">${esc(label)} ↗</a>`; | |
| const measuredInstruction=e=>e?.instruction ?? e?.effective_instruction ?? e?.native_instruction; | |
| const outcome=e=>e.success===true?'Successful':e.success===false?'Unsuccessful':'Unknown'; | |
| const elapsed=s=>s==null?'—':`${Math.floor(s/60)}m ${Math.round(s%60)}s`; | |
| function modifiedText(t){ | |
| const r=t.instruction_revision;if(!r)return esc(t.native_instruction||t.catalog_instruction); | |
| return r.diff.filter(p=>p.modified).map(p=>p.op==='equal'?esc(p.modified):`<mark>${esc(p.modified)}</mark>`).join(r.preserve_whitespace?'':' '); | |
| } | |
| function instructionPanel(t,e){ | |
| const r=t.instruction_revision; | |
| if(!r)return panel(e?'Original native instruction':'Upstream catalog description',`<blockquote class="native-goal original">${esc(e?.native_instruction||t.catalog_instruction)}</blockquote>${e?`<p class="source-note">${external(e.links.native_goal,'Instruction evidence')}</p>`:''}`); | |
| const earlier=e?.instruction_policy==='modified' && measuredInstruction(e)!==r.instruction; | |
| const measured=earlier?'This recording used an earlier modified instruction. The current revised definition is shown for comparison.':e?(e.instruction_policy==='modified'?'This recording used the modified instruction.':'This recording used the original instruction. The revised definition is shown for comparison.'):'This revised task has not been evaluated yet.'; | |
| return panel('Modified instruction',`<p class="instruction-context">${measured}</p><blockquote class="native-goal modified">${modifiedText(t)}</blockquote><p class="source-note">Highlighted words show additions and replacements. ${external(r.source_url,'Instruction source')}</p><details class="instruction-original"><summary>${r.native_instruction_kind==='episode_template'?'Original instruction template':'Original native instruction'}</summary><blockquote class="native-goal original">${esc(t.family==='task04'?(e?.native_instruction||r.native_instruction):r.native_instruction)}</blockquote>${r.native_instruction_kind==='episode_template'?'<p class="source-note">Object names and target assignments are supplied by the current scene.</p>':''}</details>${e && t.family!=="task04" && e.native_instruction!==r.native_instruction?`<details class="instruction-recorded-native"><summary>Native instruction recorded in this scene</summary><blockquote class="native-goal recorded">${esc(e.native_instruction)}</blockquote></details>`:""}${earlier?`<details class="instruction-measured"><summary>Instruction used in this recording</summary><blockquote class="native-goal measured">${esc(measuredInstruction(e))}</blockquote></details>`:''}${e?`<p class="source-note">${external(e.links.native_goal,'Measured instruction evidence')}</p>`:''}`); | |
| } | |
| function visualReviewPanel(e){ | |
| const r=e.visual_review;if(!r)return ''; | |
| const labels={consistent_pass:'画面与原生通过一致',confirmed_false_positive:'误通过:画面未完成,但原生 verifier 判定通过',consistent_failure:'画面与原生失败一致'}; | |
| const label=labels[r.status]||'原生条件与画面复核'; | |
| const text=e.analysis?.visible_completion||r.review||''; | |
| return panel('独立视频复核',`<p><strong>${esc(label)}</strong></p><p>${esc(text)}</p><p class="source-note">原生分数保持不变。${external(e.links.independent_review||e.links.analysis,'完整复核记录')}</p>`,'wide-panel'); | |
| } | |
| function failureBadge(r){return `<span class="failure-badge ${esc(r.category)}">${r.status==='provisional'?'可能 · ':''}${esc(r.label)}</span>`;} | |
| function failureTable(episodes,id){ | |
| return `<details id="${id}" class="failure-list"><summary>查看全部 ${episodes.length} 项失败说明</summary><div class="table-wrap"><table><thead><tr><th>Task</th><th>分类</th><th>失败原因</th></tr></thead><tbody>${episodes.map(e=>`<tr><td><a href="#task/${esc(e.task_key)}">${esc(data.tasks.find(t=>t.key===e.task_key)?.title || e.task_key)}</a><small>${esc(data.tasks.find(t=>t.key===e.task_key)?.display_key || e.task_key)}</small></td><td>${failureBadge(e.failure_review)}</td><td>${esc(e.failure_review.reason)}</td></tr>`).join('')}</tbody></table></div></details>`; | |
| } | |
| function failureCategories(review){ | |
| return `<div class="failure-categories">${review.categories.map(c=>`<button class="failure-category" data-category="${esc(c.id)}"><b>${esc(c.label)} <span>${c.count}</span></b><p>${esc(c.description)}</p><small>${c.exact_predicate_count!==undefined?`运行记录已核对 · ${c.exact_predicate_count} 项唯一谓词已定位`:c.provisional?`${c.provisional} 项具体根因仍待确认`:'运行证据支持'}</small></button>`).join('')}</div>`; | |
| } | |
| function renderFailureOverview(){ | |
| const failures=data.episodes.filter(e=>!e.success); | |
| $('#failure-overview').innerHTML=`<div class="section-top"><h2>当前结果与失败分析</h2><a href="attempt-history.json" target="_blank" rel="noopener">Attempt history ↗</a></div><p class="section-note">${data.tasks.length} 个任务 · ${data.episodes.filter(e=>e.success).length} 个成功 · ${failures.length} 个原生失败 · ${data.tasks.length-data.episodes.length} 个待发布。原生判定与完整 session 已保留;具体失败机制尚未逐项定位。服务端中断及其补跑记录见尝试历史。</p>`; | |
| $('#failure-filter').innerHTML='<option value="all">All results</option><option value="failed">All unsuccessful results</option>'; | |
| } | |
| const successPercent=value=>value==null?'—':(value*100).toFixed(2)+'%'; | |
| function successSummary(f){ | |
| return `<span class="rate">${successPercent(f.success_rate)}</span><small>${f.valid_results?`${f.successes} / ${f.valid_results} valid results`:'Not evaluated'}</small>`; | |
| } | |
| function renderOverview(){ | |
| $('#readiness-note').hidden=!data.selection_pending?.length; | |
| $('#readiness-note').innerHTML=data.selection_pending?.length?'RoboTwin task03/10 is awaiting selection after seed-0 initialization failed. No model episode or score is recorded. <a href="#task/task03/10">View preparation evidence ↗</a>':''; | |
| $('#edition-count').innerHTML=`<strong>${data.episodes.length} / ${data.tasks.length}</strong> results available · ${data.tasks.length-data.episodes.length} pending`; | |
| $('#edition-label').textContent=`${data.episodes.filter(e=>e.phase==='formal').length} formal results · ${data.episodes.filter(e=>e.phase==='preflight').length} preserved preflight`; | |
| $('#overview').innerHTML=data.families.map(f=>`<tr data-family="${f.id}"><td><a href="#${f.id}">${f.id.replace('task','Task')} · ${esc(f.name)}</a></td><td>${f.completed}/${f.total}</td><td>${successSummary(f)}</td><td>${num(f.input_tokens)}<small>${f.cached_input_tokens==null?'':num(f.cached_input_tokens)+' cached'}</small></td><td>${num(f.output_tokens)}</td><td>${f.completed?`${f.usage_complete} / ${f.completed}<small>Complete</small>`:'—<small>No usage yet</small>'}</td></tr>`).join(''); | |
| $('#family-nav').innerHTML=data.families.map(f=>`<a href="#${f.id}">${f.id.replace('task','Task')} · ${esc(f.name)} <span class="muted">${f.total}</span></a>`).join(''); | |
| renderFailureOverview();renderCards(); | |
| } | |
| function renderCards(){ | |
| const query=$('#search').value.toLowerCase(),status=$('#status-filter').value,category=$('#failure-filter').value; | |
| const filtered=data.tasks.filter(t=>{const e=data.episodes.find(e=>e.id===t.episode_id),r=e?.failure_review;return (status==='all'||t.status===status)&&(category==='all'||category==='failed'&&e&&!e.success||r?.category===category)&&[t.display_key || t.key,t.title,t.native_id,t.catalog_instruction,t.native_instruction||'',r?.label||'',r?.reason||''].join(' ').toLowerCase().includes(query);}); | |
| $('#families').innerHTML=data.families.map(f=>{ | |
| const tasks=filtered.filter(t=>t.family===f.id);if(!tasks.length)return ''; | |
| return `<section class="family-section" id="${f.id}"><div class="family-heading"><div><h2>${f.id.replace('task','Task')} · ${esc(f.name)}</h2><p>${f.total} tasks · ${f.completed} result${f.completed===1?'':'s'} available · ${f.pending} pending${f.completed?' · '+f.successes+' native successes':''}</p></div><span class="family-number">${f.id.slice(-2)}</span></div><div class="cards">${tasks.map(t=>{ | |
| const e=data.episodes.find(e=>e.id===t.episode_id); | |
| return `<a class="card" href="#task/${t.key}" data-task="${t.key}" aria-label="${esc(t.title)} — ${e?'View result':'Pending'}"><div class="card-media">${e?`<img loading="lazy" src="${esc(e.links.poster)}" alt="${esc(f.name)} native episode recording"><span class="media-label">4× VIDEO · ${e.phase.toUpperCase()}</span>`:`<div class="pending-art">${symbol}<span>${esc(f.name.toUpperCase())} / PENDING</span></div>`}</div><div class="card-body"><div class="slot">${t.family} / ${t.display_slot || t.slot}</div><h3>${esc(t.title)}</h3><div class="instruction-label">${t.instruction_revision?'Modified instruction':e?'Original native instruction':'Upstream catalog description'}</div><p class="instruction">${modifiedText(t)}</p>${t.instruction_revision&&(!e || measuredInstruction(e)!==t.instruction_revision.instruction)?`<small class="measured-version">${e?(e.instruction_policy==="modified"?"Recorded result: earlier modified instruction":"Recorded result: original instruction"):"Revised task · not yet evaluated"}</small>`:""}${e?.failure_review?`<div class="card-failure">${failureBadge(e.failure_review)}<p>${esc(e.failure_review.reason)}</p></div>`:''}</div><div class="card-footer"><span class="${e?(e.success?'status-success':'status-fail'):'muted'}">${e?outcome(e)+' · '+(e.valid?'valid':'review required'):(t.readiness?'Pending · setup blocked':'Pending · no published result')}</span><span>${e?'View result':'View task'} ↗</span></div></a>`; | |
| }).join('')}</div></section>`; | |
| }).join(''); | |
| $('#filter-count').textContent=`${filtered.length} / ${data.tasks.length} tasks`;$('#empty').hidden=filtered.length>0; | |
| } | |
| function panel(title,body,cls=''){return `<section class="panel ${cls}"><h3>${title}</h3>${body}</section>`;} | |
| function simpleAttemptHistory(t,e){ | |
| if(!t.simple_rerun)return ''; | |
| const history=t.attempt_history||[]; | |
| const current=e.provenance.max_steps||t.planned_protocol.max_control_steps; | |
| return panel('Attempt history',`<div class="simple-attempt-history"><p>Current result: attempt ${e.provenance.attempt||1} · ${num(current)} action-step limit.</p><p>${esc(t.simple_rerun.note)}</p>${history.map(h=>`<div class="previous-attempt"><p>Previous attempt · ${num(h.max_steps)} action-step limit · native ${h.success?'success':'failure'} · ${num(h.steps)} steps used. Retained outside current-selection totals.</p>${h.visual_review?`<p>独立复核:${esc(h.visual_review.review)}</p>`:''}<div class="downloads">${['video','native_session','verdict','protocol','usage','provenance','episode_record','independent_review'].filter(k=>h.links[k]).map(k=>external(h.links[k],k.replaceAll('_',' '))).join('')}</div></div>`).join('')}</div>`); | |
| } | |
| function taskHeader(t,e){return `<a class="back" href="#${t.family}">← Back to ${esc(familyName(t.family))}</a><div class="detail-header"><div><div class="slot">${t.family} / ${t.display_slot || t.slot} · ${esc(familyName(t.family))}</div><h2>${esc(t.title)}</h2><div class="detail-meta">Codex ${esc(e?'v'+e.codex_version:'benchmark')} · GPT-6 Astra · high · seed 0 · ${e?'1 measured episode':'1 planned episode'}</div><div class="badge-row">${e?`<span class="badge ${e.success?'':'fail'}">Native: ${outcome(e).toLowerCase()}</span><span class="badge">Execution: ${esc(e.execution.status)}</span><span class="badge">Evidence: ${e.verdict.evidence_valid?'valid':'invalid'}</span><span class="badge">Usage: ${e.usage.audit_complete?'complete':'incomplete'}</span>`:'<span class="badge">Pending · no published result</span>'}</div></div>${e?`<span class="chip">${e.phase==='preflight'?'Preflight record':'Formal evaluation'}</span>`:''}</div>`;} | |
| function renderPending(t){ | |
| const p=t.planned_protocol,r=t.readiness; | |
| return taskHeader(t,null)+`<section class="pending-detail">${symbol}<h3>${r?'Seed 0 initialization blocked':'Evaluation pending'}</h3><p>${r?esc(r.message):esc(t.status_note || 'Queued for evaluation.')}</p>${r?`<p>${external(r.evidence,'Native initialization evidence')}</p>`:''}<div class="pending-protocol"><span class="badge">Seed 0</span><span class="badge">${num(p.max_control_steps)} control steps · ${p.control_frequency_hz} Hz</span><span class="badge">8-hour limit</span><span class="badge">Stepped</span></div></section>`+instructionPanel(t,null)+(t.attempt_history?.length?panel('Preserved interrupted attempts',t.attempt_history.map(h=>external(h.links.provenance,'Attempt provenance')+' · '+external(h.links.native_session,'Native session')).join('<br>')):''); | |
| } | |
| async function renderResult(t,e,epoch){ | |
| const u=e.usage,m=e.media,views=m.view_names||m.recording.views.map(v=>v.name); | |
| const metrics=[[`${num(e.steps)} / ${num(t.planned_protocol.max_control_steps)}`,'Native control steps',e.simulation_time_s==null?`${m.recording.end_time_s.toFixed(2)} recorded simulation seconds`:`${e.simulation_time_s.toFixed(2)} simulation seconds`],[elapsed(e.wall_time_s),'Agent wall time','8-hour safety limit'],[num(u.accounting==='reported-responses'?u.response_count:(u.request_attempts??u.response_count)),u.accounting==='reported-responses'?'Model responses':'Model requests',u.accounting==='reported-responses'?'Provider-reported responses':`${num(u.reported_responses?.input_tokens)} / ${num(u.request_attempts??u.response_count)} terminal usage reported`],[num(e.call_activity.model_tool_calls),'Model tool calls','Nested RPC count unavailable'],[num(u.input_tokens),'Input tokens',num(u.cached_input_tokens)+' cached'],[num(u.output_tokens),'Output tokens',num(u.reasoning_output_tokens)+' reasoning']]; | |
| const r=e.failure_review;const reviewEvidence=r?`<details><summary>失败判定依据 · ${esc(r.status_label)}</summary><div class="evidence-note"><p>${esc(r.basis)}</p><p>${external(e.links.review_frame,'查看录像末帧')} · ${external(e.links.failure_review,'完整判定记录')}</p>${r.source_links.map(s=>`<p>${external(s.url,s.path)}</p>`).join('')}</div></details>`:''; | |
| $('#detail').innerHTML=taskHeader(t,e)+visualReviewPanel(e)+simpleAttemptHistory(t,e)+(e.usage_note?`<div class="usage-note"><strong>Usage incomplete</strong><p>${esc(e.usage_note)}</p></div>`:'')+`<div class="video-panel"><div class="view-labels">${views.map(v=>`<span>${esc(v.replaceAll('_',' ').toUpperCase())}</span>`).join('')}</div><video id="episode-video" style="aspect-ratio:${m.width}/${m.height}" controls playsinline preload="metadata" poster="${esc(e.links.poster)}" src="${esc(e.links.video)}"></video><div class="video-toolbar"><span>Native spectator recording · ${m.width} × ${m.height} · ${m.duration_s.toFixed(2)} s</span><label>Playback <select id="playback-speed" aria-label="Video playback speed"><option value="0.25">1× simulation / 0.25× video</option><option value="0.5">2× simulation / 0.5× video</option><option value="1" selected>4× simulation / 1× video</option><option value="2">8× simulation / 2× video</option></select></label></div></div><p class="video-note">${views.length} native ${m.recording.views[0].height}p views captured at ${m.source_fps} FPS; ${m.recording.dropped_samples?num(m.recording.dropped_samples)+' capture samples dropped; ':''}edited to 4× simulation speed and ${m.output_fps} FPS. ${m.terminal_hold_s?`All ${num(m.native_frames)} recorded source frames retained; the true final frame is held for ${m.terminal_hold_s} seconds without advancing simulation. ${e.links.terminal_frame?external(e.links.terminal_frame,"Final recorded frame"):""}`:""} Agent events below use wall time and are not frame-synchronized.</p><div class="detail-grid"><div>${instructionPanel(t,e)}${data.instruction_correction?.task_key===t.key?panel("屋顶方向说明的纠正",`<div class="instruction-correction"><p>${esc(data.instruction_correction.summary)}</p><p>${external(data.instruction_correction.evidence,"网格与原生条件核对")} · ${external(data.instruction_correction.video,"错误措辞回合的视频")} · ${external(data.instruction_correction.session,"完整 session")} · ${external(data.instruction_correction.native_verdict,"原生判定")}</p></div>`):""}${data.supplemental_verifications?.[t.key]?panel('Verifier 与环境不对齐',`<div class="button-verification"><p>${esc(data.supplemental_verifications[t.key].summary)}</p><p>独立物理验证 · 0 次模型调用 · 不计入正式成功率</p><a href="${esc(data.supplemental_verifications[t.key].report)}">查看按下、撤手与回弹记录 →</a></div>`):''}${panel('运行分析',`${r?`<div class="failure-verdict">${failureBadge(r)}<p>${esc(r.reason)}</p></div>`:''}<div class="analysis">${Object.entries(e.analysis).filter(([k])=>k!=='preflight').map(([,v])=>`<p>${esc(v)}</p>`).join('')}</div>${r?.trace_event_ids?`<div class="review-events"><p>查看引用的 session 事件:</p>${r.trace_event_ids.map(id=>`<button class="review-event" data-event-id="${esc(id)}">${esc(id)}</button>`).join('')}</div>`:''}`,'wide-panel')}</div><div>${panel('Episode metrics',`<div class="metrics">${metrics.map(([v,l,s])=>`<div class="metric"><b>${v}</b><span>${l}</span><small>${s}</small></div>`).join('')}</div><p class="metric-note">Native reward: ${num(e.native_reward??(e.success?1:0))} · termination: ${esc(e.verdict.termination)}<br>Uncached input: ${num(u.uncached_input_tokens)} · cost: ${u.cost_usd==null?'unavailable':u.cost_usd}<br>Execution: ${esc(e.execution.status)}${e.execution.reason?' · '+esc(e.execution.reason):''}.</p>`)}${e.phase==='preflight'?`<div class="preflight-note"><strong>Preserved preflight result</strong><p>${esc(e.analysis.preflight)}</p>${external(e.links.provenance,'Build provenance')}</div>`:''}</div></div>`+ | |
| panel('Session & observations',`<p class="section-note" id="session-counts">Loading session counts… Tool inputs and outputs are expandable; the native session download retains the full event format. Event timestamps are in UTC.</p><div class="timeline-controls"><label><span class="sr-only">Filter session events</span><select id="event-filter"><option value="all">All events</option><option value="tools">Tool results</option><option value="errors">Tool errors</option><option value="assistant">Assistant</option><option value="images">With images</option></select></label><input id="event-search" type="search" aria-label="Search session" placeholder="Search text, tool or event ID…"><span id="event-count" aria-live="polite">Loading…</span></div><div id="timeline"></div><button class="plain-button" id="more-events">Load more events</button><details id="gallery"><summary>Browse agent observation images</summary><div class="image-grid" id="observation-grid"></div></details>`,'wide-panel')+ | |
| panel('Agent resources',e.resources.length?`<p class="section-note">Final tool and memo files written during this episode. Native session edits preserve their development history.</p><div class="resource-toolbar"><label class="sr-only" for="resource-select">Resource file</label><select id="resource-select">${e.resources.map(r=>`<option value="${esc(r.file)}">${esc(r.name)}</option>`).join('')}</select><a id="resource-download" target="_blank" rel="noopener">Open file ↗</a></div><pre id="resource-preview">Loading…</pre>`:'<p>No tool or memo files were saved in this episode. The session and workspace evidence remain available below.','wide-panel')+ | |
| panel('Evidence & provenance',`<p class="section-note">Lightweight public evidence is stored in the Dataset. Host paths are redacted in copies; original event IDs, native verdicts and usage are preserved.</p>${reviewEvidence}<div class="downloads">${Object.entries(e.links).filter(([k])=>!['poster','transcript'].includes(k)).map(([k,v])=>external(v,k.replaceAll('_',' '))).join('')}</div><details><summary>Source revisions and measured build identities</summary><pre>${esc(JSON.stringify(e.provenance,null,2))}</pre></details>`,'wide-panel')+recordingRepairHistory(t,e); | |
| $('#playback-speed').onchange=()=>{$('#episode-video').playbackRate=Number($('#playback-speed').value);}; | |
| if(e.resources.length){ | |
| const loadResource=async()=>{const file=$('#resource-select').value;$('#resource-download').href=file;const r=await fetch(file);if(!r.ok)throw Error('Resource unavailable');const txt=await r.text();if(epoch===routeEpoch&&$('#resource-select').value===file)$('#resource-preview').textContent=txt;}; | |
| $('#resource-select').onchange=()=>loadResource().catch(showError);await loadResource();if(epoch!==routeEpoch)return; | |
| } | |
| const response=await fetch(e.links.transcript);if(!response.ok)throw Error('Could not load transcript');const events=await response.json();if(epoch!==routeEpoch)return; | |
| const images=events.flatMap(v=>v.parts.filter(p=>p.type==='image')); | |
| $('#session-counts').textContent=`${events.length} visible session events · ${images.length} observed images. Tool inputs and outputs are expandable; the native session download retains the full event format. Event timestamps are in UTC.`; | |
| $('#gallery summary').textContent=`Browse ${images.length} agent observation images`; | |
| const base=e.asset_base||e.links.transcript.slice(0,e.links.transcript.lastIndexOf('/')+1);let count=24; | |
| function renderEvents(){ | |
| const filter=$('#event-filter').value,q=$('#event-search').value.toLowerCase(); | |
| const selected=events.filter(v=>(filter==='all'||filter==='tools'&&v.role==='toolResult'||filter==='errors'&&v.error||filter==='assistant'&&v.role==='assistant'||filter==='images'&&v.parts.some(p=>p.type==='image'))&&JSON.stringify(v).toLowerCase().includes(q)); | |
| $('#event-count').textContent=`${Math.min(count,selected.length)} / ${selected.length} events`; | |
| $('#timeline').innerHTML=selected.slice(0,count).map(v=>{ | |
| const text=v.parts.map(p=>p.type==='text'?p.text:p.type==='tool'?p.name:'[image]').join(' ').trim(); | |
| return `<details class="event ${v.error?'error':''}" data-event="${esc(v.id)}"><summary><time>${esc((v.timestamp||'').slice(11,19))}</time><span class="role">${esc(v.tool||v.role)}</span><span class="event-text">${esc(text.slice(0,160)||v.id)}</span>${v.error?'<span class="status-fail">error</span>':''}</summary><div class="event-content"><div class="source-note">Event ${esc(v.id)} · ${esc(v.timestamp)}</div>${v.parts.map(p=>p.type==='image'?`<a href="${esc(base+p.src)}" target="_blank" rel="noopener"><img class="event-image" loading="lazy" src="${esc(base+p.src)}" alt="Agent observation at event ${esc(v.id)}"></a>`:`<pre>${esc(p.type==='tool'?JSON.stringify({tool:p.name,input:p.input},null,2):p.text)}</pre>`).join('')}</div></details>`; | |
| }).join('');$('#more-events').hidden=count>=selected.length; | |
| } | |
| document.querySelectorAll('.review-event').forEach(button=>button.onclick=()=>{$('#event-filter').value='all';$('#event-search').value=button.dataset.eventId;count=24;renderEvents();const event=document.querySelector('.event[data-event="'+button.dataset.eventId+'"]');if(event){event.open=true;event.scrollIntoView({block:'center'});}}); | |
| $('#event-filter').onchange=()=>{count=24;renderEvents();};$('#event-search').oninput=()=>{count=24;renderEvents();};$('#more-events').onclick=()=>{count+=24;renderEvents();};renderEvents(); | |
| $('#observation-grid').innerHTML=events.flatMap(v=>v.parts.filter(p=>p.type==='image').map(p=>`<a href="${esc(base+p.src)}" target="_blank" rel="noopener"><img loading="lazy" src="${esc(base+p.src)}" alt="Observed image, event ${esc(v.id)}"><span>${esc(v.id)}</span></a>`)).join(''); | |
| $('#detail').dataset.ready='true'; | |
| } | |
| function showError(e){console.error(e);const p=document.createElement('p');p.setAttribute('role','alert');p.textContent='An asset could not be loaded. Please refresh or open the Dataset evidence.';$('#detail').append(p);} | |
| async function route(){ | |
| const epoch=++routeEpoch,hash=decodeURIComponent(location.hash.slice(1)); | |
| const video=$('#episode-video');if(video)video.pause(); | |
| if(hash.startsWith('task/')){ | |
| const [taskKey,query]=hash.slice(5).split("?");const task=data.tasks.find(t=>t.key===taskKey);$('#browse').hidden=true;$('#detail').hidden=false;$('#detail').dataset.ready='false';$('#detail').dataset.taskKey=taskKey;window.scrollTo(0,0); | |
| if(!task){$('#detail').innerHTML='<a class="back" href="#tasks">← All tasks</a><h2>Task not found</h2>';return;} | |
| document.title=task.title+' · Codex Benchmark';if(query)window.history.replaceState(null,'','#task/'+task.key);const e=data.episodes.find(e=>e.task_key===task.key&&e.id===task.episode_id); | |
| if(e)await renderResult(task,e,epoch);else{$('#detail').innerHTML=renderPending(task);$('#detail').dataset.ready='true';} | |
| }else{ | |
| $('#detail').hidden=true;$('#detail').innerHTML='';$('#browse').hidden=false;document.title=data.title+' · RLE-Bench'; | |
| if(hash){if(!document.getElementById(hash)&&data.families.some(f=>f.id===hash)){$('#search').value='';$('#status-filter').value='all';$('#failure-filter').value='all';renderCards();}requestAnimationFrame(()=>document.getElementById(hash)?.scrollIntoView());}else window.scrollTo(0,0); | |
| } | |
| } | |
| (async()=>{try{const r=await fetch('data.json');if(!r.ok)throw Error('Data index unavailable');data=await r.json();renderOverview();$('#loading').hidden=true;$('#search').oninput=renderCards;$('#status-filter').onchange=renderCards;$('#failure-filter').onchange=renderCards;window.addEventListener('hashchange',()=>route().catch(showError));await route();document.body.dataset.ready='true';}catch(e){$('#loading').textContent='Could not load benchmark data. Please refresh or open the Dataset.';console.error(e);}})(); | |