Upload folder using huggingface_hub
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- app/src/je_validation/console/static/app.js +28 -0
- app/src/je_validation/console/static/content/tasks.html +94 -11
- app/src/je_validation/console/static/landing.css +53 -0
- app/src/je_validation/console/static/styles.css +23 -0
- app/src/je_validation/console/static/tabs.js +15 -2
- app/src/je_validation/envir/briefs.py +24 -0
- app/src/je_validation/scoring/gates.py +37 -2
- app/src/je_validation/scoring/verifiers/je10.py +7 -7
- data/runs/console/c5-grid-je01/baseline_flag_everything/seed1/JE-01_seed1.jsonl +0 -0
- data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_flag_everything/seed1/score.json +0 -0
- data/runs/console/c5-grid-je01/baseline_flag_everything/seed2/JE-01_seed2.jsonl +0 -0
- data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_flag_everything/seed2/score.json +0 -0
- data/runs/console/c5-grid-je01/baseline_flag_everything/seed3/JE-01_seed3.jsonl +0 -0
- data/runs/console/{run-932f8101/baseline_flag_everything/seed1 → c5-grid-je01/baseline_flag_everything/seed3}/score.json +0 -0
- data/runs/console/c5-grid-je01/baseline_flag_everything/seed4/JE-01_seed4.jsonl +0 -0
- data/runs/console/{run-932f8101/baseline_flag_everything/seed2 → c5-grid-je01/baseline_flag_everything/seed4}/score.json +0 -0
- data/runs/console/c5-grid-je01/baseline_flag_everything/seed5/JE-01_seed5.jsonl +0 -0
- data/runs/console/{run-a5453c3d/baseline_flag_everything/seed1 → c5-grid-je01/baseline_flag_everything/seed5}/score.json +0 -0
- data/runs/console/c5-grid-je01/baseline_no_evidence/seed1/JE-01_seed1.jsonl +0 -0
- data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_no_evidence/seed1/score.json +0 -0
- data/runs/console/c5-grid-je01/baseline_no_evidence/seed2/JE-01_seed2.jsonl +0 -0
- data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_no_evidence/seed2/score.json +0 -0
- data/runs/console/c5-grid-je01/baseline_no_evidence/seed3/JE-01_seed3.jsonl +0 -0
- data/runs/console/{run-932f8101/baseline_no_evidence/seed1 → c5-grid-je01/baseline_no_evidence/seed3}/score.json +0 -0
- data/runs/console/c5-grid-je01/baseline_no_evidence/seed4/JE-01_seed4.jsonl +0 -0
- data/runs/console/{run-932f8101/baseline_no_evidence/seed2 → c5-grid-je01/baseline_no_evidence/seed4}/score.json +0 -0
- data/runs/console/c5-grid-je01/baseline_no_evidence/seed5/JE-01_seed5.jsonl +0 -0
- data/runs/console/{run-a5453c3d/baseline_no_evidence/seed1 → c5-grid-je01/baseline_no_evidence/seed5}/score.json +0 -0
- data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed1/JE-01_seed1.jsonl +0 -0
- data/runs/console/{run-a5453c3d/baseline_no_evidence/seed3 → c5-grid-je01/deepseek_deepseek-v4-flash/seed1}/score.json +4 -4
- data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed2/JE-01_seed2.jsonl +0 -0
- data/runs/console/{run-a5453c3d/baseline_no_evidence/seed5 → c5-grid-je01/deepseek_deepseek-v4-flash/seed2}/score.json +4 -4
- data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed3/JE-01_seed3.jsonl +0 -0
- data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed3/score.json +16 -0
- data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed4/JE-01_seed4.jsonl +0 -0
- data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed4/score.json +16 -0
- data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed5/JE-01_seed5.jsonl +0 -0
- data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed5/score.json +16 -0
- data/runs/console/{run-a5453c3d → c5-grid-je01}/results.json +24 -125
- data/runs/console/{run-a5453c3d → c5-grid-je01}/run.json +17 -31
- data/runs/console/c5-grid-je01/seed1/bundle/heldout/defect_ledger.json +1 -0
- data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed1/bundle/instance/documents.jsonl +0 -0
- data/runs/console/c5-grid-je01/seed1/bundle/instance/entries.jsonl +0 -0
- data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed1/bundle/instance/master_data.json +0 -0
- data/runs/console/c5-grid-je01/seed1/bundle/instance/meta.json +1 -0
- data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed1/bundle/instance/policies.json +0 -0
- data/runs/console/c5-grid-je01/seed1/bundle/manifest.json +1 -0
- data/runs/console/c5-grid-je01/seed2/bundle/heldout/defect_ledger.json +1 -0
- data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed2/bundle/instance/documents.jsonl +0 -0
- data/runs/console/c5-grid-je01/seed2/bundle/instance/entries.jsonl +0 -0
app/src/je_validation/console/static/app.js
CHANGED
|
@@ -793,6 +793,34 @@ async function openPastRun(runId) {
|
|
| 793 |
openLiveView(runId, snap.contract);
|
| 794 |
}
|
| 795 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 796 |
// ---- live view: SSE client -------------------------------------------------
|
| 797 |
//
|
| 798 |
// A fresh EventSource cannot send Last-Event-ID (the browser only adds it on
|
|
|
|
| 793 |
openLiveView(runId, snap.contract);
|
| 794 |
}
|
| 795 |
|
| 796 |
+
// ---- deep links from the Tasks tab -----------------------------------------
|
| 797 |
+
//
|
| 798 |
+
// tabs.js translates #console/run/<run_id> into a "je:open-run" event (it is
|
| 799 |
+
// the only hash reader). Opening the run scrolls to the live view; the pulse
|
| 800 |
+
// on the panel and on the run's history row says "this is the one you
|
| 801 |
+
// clicked". Unknown run ids degrade to the plain console (openPastRun
|
| 802 |
+
// already swallows the failed fetch).
|
| 803 |
+
|
| 804 |
+
function pulseOnce(el) {
|
| 805 |
+
if (!el) return;
|
| 806 |
+
el.classList.remove('deep-flash');
|
| 807 |
+
void el.offsetWidth; // restart the animation on repeat clicks
|
| 808 |
+
el.classList.add('deep-flash');
|
| 809 |
+
setTimeout(() => el.classList.remove('deep-flash'), 1600);
|
| 810 |
+
}
|
| 811 |
+
|
| 812 |
+
async function openDeepLinkRun(runId) {
|
| 813 |
+
await loadRunHistory(); // ensure the row exists before pulsing it
|
| 814 |
+
await openPastRun(runId);
|
| 815 |
+
if (!Console.live || Console.live.runId !== runId) return;
|
| 816 |
+
pulseOnce(document.querySelector('#live .panel'));
|
| 817 |
+
const idEl = [...document.querySelectorAll('#history-list .history-id')]
|
| 818 |
+
.find((el) => el.textContent === runId);
|
| 819 |
+
if (idEl) pulseOnce(idEl.closest('.history-row'));
|
| 820 |
+
}
|
| 821 |
+
|
| 822 |
+
window.addEventListener('je:open-run', (e) => openDeepLinkRun(e.detail));
|
| 823 |
+
|
| 824 |
// ---- live view: SSE client -------------------------------------------------
|
| 825 |
//
|
| 826 |
// A fresh EventSource cannot send Last-Event-ID (the browser only adds it on
|
app/src/je_validation/console/static/content/tasks.html
CHANGED
|
@@ -15,24 +15,65 @@
|
|
| 15 |
<div class="task-card"><h3><span class="je-id">JE-01</span> Broken entries
|
| 16 |
<span class="badge-live">Live</span></h3>
|
| 17 |
<p>An overnight interface load has left a batch of entries damaged. Find
|
| 18 |
-
every entry that no longer balances or is missing required pieces.</p>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 19 |
<div class="task-card"><h3><span class="je-id">JE-02</span> Coding validity
|
| 20 |
<span class="badge-live">Live</span></h3>
|
| 21 |
<p>Account, fund, and department codes must exist, be active in the
|
| 22 |
period, and be an allowed combination. Find the entries that violate any
|
| 23 |
-
of it.</p>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 24 |
<div class="task-card"><h3><span class="je-id">JE-03</span> Period cut-off
|
| 25 |
<span class="badge-live">Live</span></h3>
|
| 26 |
<p>Entries must be posted to the period their dates belong to. Find the
|
| 27 |
-
ones recorded in the wrong period.</p>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 28 |
<div class="task-card"><h3><span class="je-id">JE-04</span> Duplicate payments
|
| 29 |
<span class="badge-live">Live</span></h3>
|
| 30 |
<p>A vendor claims they were paid twice. Separate true duplicates from
|
| 31 |
-
legitimate split payments and properly re-issued checks.</p>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 32 |
<div class="task-card"><h3><span class="je-id">JE-05</span> Foreign currency
|
| 33 |
<span class="badge-live">Live</span></h3>
|
| 34 |
<p>Foreign-currency entries must translate at the official period rate.
|
| 35 |
-
Find wrong rates and translation errors.</p>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 36 |
</div>
|
| 37 |
|
| 38 |
<div class="tier-head tier-l2"><span class="tier-tag">L2</span><h2>Analytical checks</h2>
|
|
@@ -44,23 +85,64 @@
|
|
| 44 |
<span class="badge-live">Live</span></h3>
|
| 45 |
<p>Payments above the approval threshold need a purchase order, an
|
| 46 |
invoice, and a payment that agree. Find the mismatches and explain each
|
| 47 |
-
one.</p>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 48 |
<div class="task-card"><h3><span class="je-id">JE-07</span> Subledger reconciliation
|
| 49 |
<span class="badge-live">Live</span></h3>
|
| 50 |
<p>The payables control account doesn't agree with its subledger.
|
| 51 |
-
Decompose the difference into the specific items that cause it.</p>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 52 |
<div class="task-card"><h3><span class="je-id">JE-08</span> Accrual reversals
|
| 53 |
<span class="badge-live">Live</span></h3>
|
| 54 |
<p>Period-end accruals should reverse in the next period. Find the ones
|
| 55 |
-
that never did, and recompute the ones reversed incorrectly.</p>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 56 |
<div class="task-card"><h3><span class="je-id">JE-09</span> Interfund transfers
|
| 57 |
<span class="badge-live">Live</span></h3>
|
| 58 |
<p>A transfer between funds needs a matching counterparty leg. Find the
|
| 59 |
-
unmatched legs before consolidation.</p>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 60 |
<div class="task-card"><h3><span class="je-id">JE-10</span> Budget integrity
|
| 61 |
<span class="badge-live">Live</span></h3>
|
| 62 |
<p>Obligations should liquidate as payments are made, and spending must
|
| 63 |
-
stay inside budget authority. Find where either has broken down.</p>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 64 |
</div>
|
| 65 |
|
| 66 |
<div class="tier-head tier-l3"><span class="tier-tag">L3</span><h2>Cross-document evidence</h2>
|
|
@@ -121,5 +203,6 @@
|
|
| 121 |
</section>
|
| 122 |
|
| 123 |
<footer class="land-footer">
|
| 124 |
-
<span>
|
|
|
|
| 125 |
</footer>
|
|
|
|
| 15 |
<div class="task-card"><h3><span class="je-id">JE-01</span> Broken entries
|
| 16 |
<span class="badge-live">Live</span></h3>
|
| 17 |
<p>An overnight interface load has left a batch of entries damaged. Find
|
| 18 |
+
every entry that no longer balances or is missing required pieces.</p>
|
| 19 |
+
<!-- example:JE-01 -->
|
| 20 |
+
<div class="example-run">
|
| 21 |
+
<span class="ex-model">deepseek-v4-flash</span>
|
| 22 |
+
<span class="ex-score">final 0.04 ± 0.02 <small>(5 seeds)</small></span>
|
| 23 |
+
<span class="ex-base">baselines: flag-everything −1.80 · no-evidence 0.00</span>
|
| 24 |
+
<a class="ex-link" href="#console/run/c5-grid-je01">View the full run →</a>
|
| 25 |
+
</div>
|
| 26 |
+
<!-- /example:JE-01 --></div>
|
| 27 |
<div class="task-card"><h3><span class="je-id">JE-02</span> Coding validity
|
| 28 |
<span class="badge-live">Live</span></h3>
|
| 29 |
<p>Account, fund, and department codes must exist, be active in the
|
| 30 |
period, and be an allowed combination. Find the entries that violate any
|
| 31 |
+
of it.</p>
|
| 32 |
+
<!-- example:JE-02 -->
|
| 33 |
+
<div class="example-run">
|
| 34 |
+
<span class="ex-model">deepseek-v4-flash</span>
|
| 35 |
+
<span class="ex-score">final 1.00 ± 0.00 <small>(5 seeds)</small></span>
|
| 36 |
+
<span class="ex-base">baselines: flag-everything −1.80 · no-evidence 0.22</span>
|
| 37 |
+
<a class="ex-link" href="#console/run/c5-grid-je02">View the full run →</a>
|
| 38 |
+
</div>
|
| 39 |
+
<!-- /example:JE-02 --></div>
|
| 40 |
<div class="task-card"><h3><span class="je-id">JE-03</span> Period cut-off
|
| 41 |
<span class="badge-live">Live</span></h3>
|
| 42 |
<p>Entries must be posted to the period their dates belong to. Find the
|
| 43 |
+
ones recorded in the wrong period.</p>
|
| 44 |
+
<!-- example:JE-03 -->
|
| 45 |
+
<div class="example-run">
|
| 46 |
+
<span class="ex-model">deepseek-v4-flash</span>
|
| 47 |
+
<span class="ex-score">final 0.21 ± 0.02 <small>(5 seeds)</small></span>
|
| 48 |
+
<span class="ex-base">baselines: flag-everything −1.80 · no-evidence 0.00</span>
|
| 49 |
+
<a class="ex-link" href="#console/run/c5-grid-je03">View the full run →</a>
|
| 50 |
+
</div>
|
| 51 |
+
<!-- /example:JE-03 --></div>
|
| 52 |
<div class="task-card"><h3><span class="je-id">JE-04</span> Duplicate payments
|
| 53 |
<span class="badge-live">Live</span></h3>
|
| 54 |
<p>A vendor claims they were paid twice. Separate true duplicates from
|
| 55 |
+
legitimate split payments and properly re-issued checks.</p>
|
| 56 |
+
<!-- example:JE-04 -->
|
| 57 |
+
<div class="example-run">
|
| 58 |
+
<span class="ex-model">deepseek-v4-flash</span>
|
| 59 |
+
<span class="ex-score">final 0.10 ± 0.30 <small>(5 seeds)</small></span>
|
| 60 |
+
<span class="ex-base">baselines: flag-everything −1.80 · no-evidence 0.00</span>
|
| 61 |
+
<a class="ex-link" href="#console/run/c5-grid-je04">View the full run →</a>
|
| 62 |
+
</div>
|
| 63 |
+
<!-- /example:JE-04 --></div>
|
| 64 |
<div class="task-card"><h3><span class="je-id">JE-05</span> Foreign currency
|
| 65 |
<span class="badge-live">Live</span></h3>
|
| 66 |
<p>Foreign-currency entries must translate at the official period rate.
|
| 67 |
+
Find wrong rates and translation errors.</p>
|
| 68 |
+
<!-- example:JE-05 -->
|
| 69 |
+
<div class="example-run">
|
| 70 |
+
<span class="ex-model">deepseek-v4-flash</span>
|
| 71 |
+
<span class="ex-score">final 0.16 ± 0.02 <small>(5 seeds)</small></span>
|
| 72 |
+
<span class="ex-base">baselines: flag-everything −1.71 · no-evidence 0.18</span>
|
| 73 |
+
<span class="ex-fail">1 of 5 seeds auto-failed (fabrication)</span>
|
| 74 |
+
<a class="ex-link" href="#console/run/c5-grid-je05">View the full run →</a>
|
| 75 |
+
</div>
|
| 76 |
+
<!-- /example:JE-05 --></div>
|
| 77 |
</div>
|
| 78 |
|
| 79 |
<div class="tier-head tier-l2"><span class="tier-tag">L2</span><h2>Analytical checks</h2>
|
|
|
|
| 85 |
<span class="badge-live">Live</span></h3>
|
| 86 |
<p>Payments above the approval threshold need a purchase order, an
|
| 87 |
invoice, and a payment that agree. Find the mismatches and explain each
|
| 88 |
+
one.</p>
|
| 89 |
+
<!-- example:JE-06 -->
|
| 90 |
+
<div class="example-run">
|
| 91 |
+
<span class="ex-model">deepseek-v4-flash</span>
|
| 92 |
+
<span class="ex-score">final 0.83 ± 0.14 <small>(5 seeds)</small></span>
|
| 93 |
+
<span class="ex-base">baselines: flag-everything −1.76 · no-evidence 0.00</span>
|
| 94 |
+
<a class="ex-link" href="#console/run/c5-grid-je06">View the full run →</a>
|
| 95 |
+
</div>
|
| 96 |
+
<!-- /example:JE-06 --></div>
|
| 97 |
<div class="task-card"><h3><span class="je-id">JE-07</span> Subledger reconciliation
|
| 98 |
<span class="badge-live">Live</span></h3>
|
| 99 |
<p>The payables control account doesn't agree with its subledger.
|
| 100 |
+
Decompose the difference into the specific items that cause it.</p>
|
| 101 |
+
<!-- example:JE-07 -->
|
| 102 |
+
<div class="example-run">
|
| 103 |
+
<span class="ex-model">deepseek-v4-flash</span>
|
| 104 |
+
<span class="ex-score">final 0.17 ± 0.08 <small>(5 seeds)</small></span>
|
| 105 |
+
<span class="ex-base">baselines: flag-everything −1.80 · no-evidence 0.00</span>
|
| 106 |
+
<a class="ex-link" href="#console/run/c5-grid-je07">View the full run →</a>
|
| 107 |
+
</div>
|
| 108 |
+
<!-- /example:JE-07 --></div>
|
| 109 |
<div class="task-card"><h3><span class="je-id">JE-08</span> Accrual reversals
|
| 110 |
<span class="badge-live">Live</span></h3>
|
| 111 |
<p>Period-end accruals should reverse in the next period. Find the ones
|
| 112 |
+
that never did, and recompute the ones reversed incorrectly.</p>
|
| 113 |
+
<!-- example:JE-08 -->
|
| 114 |
+
<div class="example-run">
|
| 115 |
+
<span class="ex-model">deepseek-v4-flash</span>
|
| 116 |
+
<span class="ex-score">final 0.68 ± 0.08 <small>(5 seeds)</small></span>
|
| 117 |
+
<span class="ex-base">baselines: flag-everything −1.79 · no-evidence 0.00</span>
|
| 118 |
+
<a class="ex-link" href="#console/run/c5-grid-je08">View the full run →</a>
|
| 119 |
+
</div>
|
| 120 |
+
<!-- /example:JE-08 --></div>
|
| 121 |
<div class="task-card"><h3><span class="je-id">JE-09</span> Interfund transfers
|
| 122 |
<span class="badge-live">Live</span></h3>
|
| 123 |
<p>A transfer between funds needs a matching counterparty leg. Find the
|
| 124 |
+
unmatched legs before consolidation.</p>
|
| 125 |
+
<!-- example:JE-09 -->
|
| 126 |
+
<div class="example-run">
|
| 127 |
+
<span class="ex-model">deepseek-v4-flash</span>
|
| 128 |
+
<span class="ex-score">final 0.59 ± 0.20 <small>(5 seeds)</small></span>
|
| 129 |
+
<span class="ex-base">baselines: flag-everything −1.63 · no-evidence 0.31</span>
|
| 130 |
+
<a class="ex-link" href="#console/run/c5-grid-je09">View the full run →</a>
|
| 131 |
+
</div>
|
| 132 |
+
<!-- /example:JE-09 --></div>
|
| 133 |
<div class="task-card"><h3><span class="je-id">JE-10</span> Budget integrity
|
| 134 |
<span class="badge-live">Live</span></h3>
|
| 135 |
<p>Obligations should liquidate as payments are made, and spending must
|
| 136 |
+
stay inside budget authority. Find where either has broken down.</p>
|
| 137 |
+
<!-- example:JE-10 -->
|
| 138 |
+
<div class="example-run">
|
| 139 |
+
<span class="ex-model">deepseek-v4-flash</span>
|
| 140 |
+
<span class="ex-score">final 0.91 ± 0.07 <small>(5 seeds)</small></span>
|
| 141 |
+
<span class="ex-base">baselines: flag-everything −1.80 · no-evidence 0.25</span>
|
| 142 |
+
<span class="ex-fail">1 of 5 seeds auto-failed (fabrication)</span>
|
| 143 |
+
<a class="ex-link" href="#console/run/c5-grid-je10">View the full run →</a>
|
| 144 |
+
</div>
|
| 145 |
+
<!-- /example:JE-10 --></div>
|
| 146 |
</div>
|
| 147 |
|
| 148 |
<div class="tier-head tier-l3"><span class="tier-tag">L3</span><h2>Cross-document evidence</h2>
|
|
|
|
| 203 |
</section>
|
| 204 |
|
| 205 |
<footer class="land-footer">
|
| 206 |
+
<span>Recorded 2026-08-11 · open any run in the console's Past runs panel —
|
| 207 |
+
the same task and seed always reproduces the same case.</span>
|
| 208 |
</footer>
|
app/src/je_validation/console/static/landing.css
CHANGED
|
@@ -189,6 +189,10 @@
|
|
| 189 |
.tier-l3 .tier-tag { color: var(--l3); background: var(--l3bg); }
|
| 190 |
.tier-l4 .tier-tag { color: var(--l4); background: var(--l4bg); }
|
| 191 |
.task-grid { display: grid; grid-template-columns: 1fr 1fr; gap: 12px; }
|
|
|
|
|
|
|
|
|
|
|
|
|
| 192 |
.task-card h3 { display: flex; gap: 8px; align-items: baseline; }
|
| 193 |
.task-card .je-id { font-family: var(--mono); font-size: 11px; color: var(--ink-3); }
|
| 194 |
.badge-live, .badge-roadmap {
|
|
@@ -203,6 +207,55 @@
|
|
| 203 |
.badge-live { color: var(--l1); background: var(--l1bg); }
|
| 204 |
.badge-roadmap { color: var(--ink-3); background: var(--panel-3); }
|
| 205 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 206 |
/* ---- scoring ---- */
|
| 207 |
.formula {
|
| 208 |
background: var(--panel);
|
|
|
|
| 189 |
.tier-l3 .tier-tag { color: var(--l3); background: var(--l3bg); }
|
| 190 |
.tier-l4 .tier-tag { color: var(--l4); background: var(--l4bg); }
|
| 191 |
.task-grid { display: grid; grid-template-columns: 1fr 1fr; gap: 12px; }
|
| 192 |
+
/* column + auto margin: the example run sits on the card floor, so the score
|
| 193 |
+
* lines of neighbouring cards align however long the description runs. */
|
| 194 |
+
.task-card { display: flex; flex-direction: column; }
|
| 195 |
+
.task-card > p { margin-bottom: auto; }
|
| 196 |
.task-card h3 { display: flex; gap: 8px; align-items: baseline; }
|
| 197 |
.task-card .je-id { font-family: var(--mono); font-size: 11px; color: var(--ink-3); }
|
| 198 |
.badge-live, .badge-roadmap {
|
|
|
|
| 207 |
.badge-live { color: var(--l1); background: var(--l1bg); }
|
| 208 |
.badge-roadmap { color: var(--ink-3); background: var(--panel-3); }
|
| 209 |
|
| 210 |
+
/* ---- example runs on task cards ---- */
|
| 211 |
+
.example-run {
|
| 212 |
+
margin-top: 14px;
|
| 213 |
+
padding-top: 10px;
|
| 214 |
+
border-top: 1px dashed var(--line-2);
|
| 215 |
+
display: grid;
|
| 216 |
+
gap: 3px;
|
| 217 |
+
font-size: 12px;
|
| 218 |
+
}
|
| 219 |
+
.example-run .ex-model { font-family: var(--mono); font-size: 11px; color: var(--ink-3); }
|
| 220 |
+
.example-run .ex-score {
|
| 221 |
+
font-weight: 700;
|
| 222 |
+
font-size: 14px;
|
| 223 |
+
color: var(--ink);
|
| 224 |
+
font-variant-numeric: tabular-nums;
|
| 225 |
+
}
|
| 226 |
+
.example-run .ex-score small { font-weight: 600; font-size: 11.5px; color: var(--ink-3); }
|
| 227 |
+
.example-run .ex-base { font-family: var(--mono); font-size: 11.5px; color: var(--ink-2); }
|
| 228 |
+
/* a flagged outcome, not an alarm: dot + danger ink, quieter than the link */
|
| 229 |
+
.example-run .ex-fail {
|
| 230 |
+
display: flex;
|
| 231 |
+
align-items: center;
|
| 232 |
+
gap: 6px;
|
| 233 |
+
font-size: 11.5px;
|
| 234 |
+
font-weight: 600;
|
| 235 |
+
color: var(--l4);
|
| 236 |
+
}
|
| 237 |
+
.example-run .ex-fail::before {
|
| 238 |
+
content: "";
|
| 239 |
+
flex: none;
|
| 240 |
+
width: 6px;
|
| 241 |
+
height: 6px;
|
| 242 |
+
border-radius: 50%;
|
| 243 |
+
background: currentColor;
|
| 244 |
+
}
|
| 245 |
+
.example-run .ex-link {
|
| 246 |
+
justify-self: start; /* hug the text so the focus ring frames the link */
|
| 247 |
+
margin-top: 4px;
|
| 248 |
+
font-weight: 700;
|
| 249 |
+
color: var(--accent);
|
| 250 |
+
text-decoration: none;
|
| 251 |
+
}
|
| 252 |
+
.example-run .ex-link:hover { text-decoration: underline; }
|
| 253 |
+
/* under ~440px the mono line cannot fit; cap the measure so it breaks at the
|
| 254 |
+
* separator instead of inside "no-evidence". */
|
| 255 |
+
@media (max-width: 440px) {
|
| 256 |
+
.example-run .ex-base { max-width: 36ch; }
|
| 257 |
+
}
|
| 258 |
+
|
| 259 |
/* ---- scoring ---- */
|
| 260 |
.formula {
|
| 261 |
background: var(--panel);
|
app/src/je_validation/console/static/styles.css
CHANGED
|
@@ -530,3 +530,26 @@ input[type="checkbox"] { accent-color: var(--accent); }
|
|
| 530 |
.step-detail-row pre { white-space: pre-wrap; word-break: break-word; }
|
| 531 |
|
| 532 |
.breakdown { margin-bottom: 14px; }
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 530 |
.step-detail-row pre { white-space: pre-wrap; word-break: break-word; }
|
| 531 |
|
| 532 |
.breakdown { margin-bottom: 14px; }
|
| 533 |
+
|
| 534 |
+
/* ---- deep-link arrival pulse (Tasks tab -> a specific run) ---- */
|
| 535 |
+
/* Outline-only on the big live panel (a full wash would tint the whole step
|
| 536 |
+
* log); the compact history row gets the wash too so it reads at a glance. */
|
| 537 |
+
.deep-flash { animation: deep-flash 1.4s ease-out 1; }
|
| 538 |
+
.history-row.deep-flash { animation-name: deep-flash-fill; }
|
| 539 |
+
@keyframes deep-flash {
|
| 540 |
+
0% { outline: 2px solid var(--accent); outline-offset: 2px; }
|
| 541 |
+
60% { outline-color: var(--accent); }
|
| 542 |
+
100% { outline: 2px solid transparent; outline-offset: 2px; }
|
| 543 |
+
}
|
| 544 |
+
@keyframes deep-flash-fill {
|
| 545 |
+
0% { outline: 2px solid var(--accent); outline-offset: 2px;
|
| 546 |
+
background-color: var(--accent-soft); }
|
| 547 |
+
60% { outline-color: var(--accent); }
|
| 548 |
+
100% { outline: 2px solid transparent; outline-offset: 2px;
|
| 549 |
+
background-color: transparent; }
|
| 550 |
+
}
|
| 551 |
+
@media (prefers-reduced-motion: reduce) {
|
| 552 |
+
/* no motion: a still outline that app.js clears with the class */
|
| 553 |
+
.deep-flash { animation: none; outline: 2px solid var(--accent);
|
| 554 |
+
outline-offset: 2px; }
|
| 555 |
+
}
|
app/src/je_validation/console/static/tabs.js
CHANGED
|
@@ -1,13 +1,22 @@
|
|
| 1 |
/* Hash-routed tabs. app.js owns everything inside #tab-console and never
|
| 2 |
-
* reads location.hash (verified), so routing is isolated here.
|
|
|
|
|
|
|
|
|
|
| 3 |
(() => {
|
| 4 |
const TABS = ["home", "datasets", "tasks", "scoring", "console"];
|
| 5 |
|
| 6 |
const current = () => {
|
| 7 |
-
const h = location.hash.replace("#", "");
|
| 8 |
return TABS.includes(h) ? h : "home";
|
| 9 |
};
|
| 10 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 11 |
async function show(tab) {
|
| 12 |
document.querySelectorAll(".tabbar a").forEach((a) =>
|
| 13 |
a.classList.toggle("active", a.dataset.tab === tab));
|
|
@@ -27,6 +36,10 @@
|
|
| 27 |
}
|
| 28 |
}
|
| 29 |
window.scrollTo(0, 0);
|
|
|
|
|
|
|
|
|
|
|
|
|
| 30 |
}
|
| 31 |
|
| 32 |
window.addEventListener("hashchange", () => show(current()));
|
|
|
|
| 1 |
/* Hash-routed tabs. app.js owns everything inside #tab-console and never
|
| 2 |
+
* reads location.hash (verified), so routing is isolated here. A deep link
|
| 3 |
+
* of the form #console/run/<run_id> opens the console tab and hands the run
|
| 4 |
+
* id to app.js via a "je:open-run" event (app.js loads first, so its
|
| 5 |
+
* listener exists before the initial dispatch). */
|
| 6 |
(() => {
|
| 7 |
const TABS = ["home", "datasets", "tasks", "scoring", "console"];
|
| 8 |
|
| 9 |
const current = () => {
|
| 10 |
+
const h = location.hash.replace("#", "").split("/")[0];
|
| 11 |
return TABS.includes(h) ? h : "home";
|
| 12 |
};
|
| 13 |
|
| 14 |
+
const runTarget = () => {
|
| 15 |
+
const parts = location.hash.replace("#", "").split("/");
|
| 16 |
+
return parts[0] === "console" && parts[1] === "run" && parts[2]
|
| 17 |
+
? parts[2] : null;
|
| 18 |
+
};
|
| 19 |
+
|
| 20 |
async function show(tab) {
|
| 21 |
document.querySelectorAll(".tabbar a").forEach((a) =>
|
| 22 |
a.classList.toggle("active", a.dataset.tab === tab));
|
|
|
|
| 36 |
}
|
| 37 |
}
|
| 38 |
window.scrollTo(0, 0);
|
| 39 |
+
const run = runTarget();
|
| 40 |
+
if (tab === "console" && run) {
|
| 41 |
+
window.dispatchEvent(new CustomEvent("je:open-run", { detail: run }));
|
| 42 |
+
}
|
| 43 |
}
|
| 44 |
|
| 45 |
window.addEventListener("hashchange", () => show(current()));
|
app/src/je_validation/envir/briefs.py
CHANGED
|
@@ -59,6 +59,28 @@ TASK_RULES: dict[str, str] = {
|
|
| 59 |
"have actually seen in tool results) earns no outcome credit."),
|
| 60 |
}
|
| 61 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 62 |
PROMPT_VERSIONS: tuple[str, ...] = ("minimal", "standard", "detailed")
|
| 63 |
|
| 64 |
# standard-level additions. The stated rule must match the SCORED rule
|
|
@@ -103,6 +125,8 @@ def _minimal(task: dict) -> str:
|
|
| 103 |
parts = [task["brief"]]
|
| 104 |
if (rule := TASK_RULES.get(task["id"])) is not None:
|
| 105 |
parts.append(rule)
|
|
|
|
|
|
|
| 106 |
extra = CLAIM_INSTRUCTIONS.get(task["id"])
|
| 107 |
if extra is not None:
|
| 108 |
parts.append(f"{CLAIM_FORMAT.lstrip()}\n\n{extra}")
|
|
|
|
| 59 |
"have actually seen in tool results) earns no outcome credit."),
|
| 60 |
}
|
| 61 |
|
| 62 |
+
# Per-task reporting requirements (c5): how findings must be REPRESENTED in
|
| 63 |
+
# the submission — flag targets and evidence citations. A versioned output
|
| 64 |
+
# contract, kept separate from TASK_RULES on purpose (sol 2026-08-11): a
|
| 65 |
+
# seed-invariant procedural rule is still not world ontology, and admitting
|
| 66 |
+
# it there would widen the world-rules exception. Same leak test applies:
|
| 67 |
+
# each sentence must be true and identical for every seed and every defect
|
| 68 |
+
# placement — id formats and fixed row conventions only, never per-instance
|
| 69 |
+
# facts. Scored by the JE-10 outcome verifier (flag target, via seeded
|
| 70 |
+
# defect ids) and the JE-10 process gate (citation factor), both c5.
|
| 71 |
+
REPORTING_RULES: dict[str, str] = {
|
| 72 |
+
"JE-10": (
|
| 73 |
+
"Reporting requirements: every budgeted account has one BUDGET "
|
| 74 |
+
"STATUS SUMMARY entry, id 'JE10-BUDGET-<account>'. Raise a "
|
| 75 |
+
"budget_exceeded exception by flagging that summary entry - not "
|
| 76 |
+
"the individual payments posted to the account. If you submit "
|
| 77 |
+
"issue type wrong_document, cite in evidence_ids the obligation "
|
| 78 |
+
"register's obligating_doc_id field value (the OD-... id you "
|
| 79 |
+
"compared with the flagged entry); the master:obligation:... "
|
| 80 |
+
"record key may also be cited as provenance, but it does not "
|
| 81 |
+
"replace that field-value citation."),
|
| 82 |
+
}
|
| 83 |
+
|
| 84 |
PROMPT_VERSIONS: tuple[str, ...] = ("minimal", "standard", "detailed")
|
| 85 |
|
| 86 |
# standard-level additions. The stated rule must match the SCORED rule
|
|
|
|
| 125 |
parts = [task["brief"]]
|
| 126 |
if (rule := TASK_RULES.get(task["id"])) is not None:
|
| 127 |
parts.append(rule)
|
| 128 |
+
if (rep := REPORTING_RULES.get(task["id"])) is not None:
|
| 129 |
+
parts.append(rep)
|
| 130 |
extra = CLAIM_INSTRUCTIONS.get(task["id"])
|
| 131 |
if extra is not None:
|
| 132 |
parts.append(f"{CLAIM_FORMAT.lstrip()}\n\n{extra}")
|
app/src/je_validation/scoring/gates.py
CHANGED
|
@@ -32,6 +32,12 @@ def _is_flag(s: StepRecord) -> bool:
|
|
| 32 |
return s.tool == "disposition" and s.args.get("verdict") in FLAGGING
|
| 33 |
|
| 34 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 35 |
def _token_in(token: str, expr: str) -> bool:
|
| 36 |
"""Whole-numeric-token match: plain substring would let entry amount 100
|
| 37 |
"match" expression "1000-900"."""
|
|
@@ -309,8 +315,25 @@ def je09_next_period(steps: list[StepRecord], dispositions: dict,
|
|
| 309 |
|
| 310 |
def je10_account_aggregate(steps: list[StepRecord], dispositions: dict,
|
| 311 |
instance: Instance) -> GateResult:
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 312 |
account_of = {e.entry_id: str(e.header.get("account", ""))
|
| 313 |
for e in instance.entries}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 314 |
grouped = False
|
| 315 |
filtered_accounts: set[str] = set()
|
| 316 |
b_flagged = b_supported = 0
|
|
@@ -328,8 +351,20 @@ def je10_account_aggregate(steps: list[StepRecord], dispositions: dict,
|
|
| 328 |
b_supported += (grouped or account_of.get(
|
| 329 |
s.args.get("entry_id")) in filtered_accounts)
|
| 330 |
b_rate = b_supported / b_flagged if b_flagged else 1.0
|
| 331 |
-
|
| 332 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 333 |
|
| 334 |
|
| 335 |
def je05_rate_recompute(steps: list[StepRecord], dispositions: dict,
|
|
|
|
| 32 |
return s.tool == "disposition" and s.args.get("verdict") in FLAGGING
|
| 33 |
|
| 34 |
|
| 35 |
+
def _dget(d, key):
|
| 36 |
+
"""Disposition field access: production passes state.Disposition objects,
|
| 37 |
+
tests pass plain dicts (same duality as verifiers._get)."""
|
| 38 |
+
return d[key] if isinstance(d, dict) else getattr(d, key)
|
| 39 |
+
|
| 40 |
+
|
| 41 |
def _token_in(token: str, expr: str) -> bool:
|
| 42 |
"""Whole-numeric-token match: plain substring would let entry amount 100
|
| 43 |
"match" expression "1000-900"."""
|
|
|
|
| 315 |
|
| 316 |
def je10_account_aggregate(steps: list[StepRecord], dispositions: dict,
|
| 317 |
instance: Instance) -> GateResult:
|
| 318 |
+
"""Two per-flag factors, multiplied (standing gate algebra). Budget
|
| 319 |
+
factor (c4), order-sensitive per spec: a budget_exceeded flag needs an
|
| 320 |
+
account-level aggregate EARLIER in the episode. Citation factor (c5),
|
| 321 |
+
order-free over episode-final dispositions: a wrong_document flag is
|
| 322 |
+
supported only if its disposition cites the obligation register's
|
| 323 |
+
obligating_doc_id field value for the flagged entry - the reporting
|
| 324 |
+
requirement REPORTING_RULES['JE-10'] states. Agent-visible data only:
|
| 325 |
+
the entry header names its obligation_id and the register row carries
|
| 326 |
+
the correct doc id; a wrong_document flag on an entry with no
|
| 327 |
+
obligation has no correct doc to cite and is unsupported. Entries are
|
| 328 |
+
identified by the dispositions dict key - the authoritative id
|
| 329 |
+
everywhere in scoring."""
|
| 330 |
account_of = {e.entry_id: str(e.header.get("account", ""))
|
| 331 |
for e in instance.entries}
|
| 332 |
+
register = instance.master_data.get("obligation", {})
|
| 333 |
+
correct_doc_of = {
|
| 334 |
+
e.entry_id: (register.get(e.header.get("obligation_id")) or {}
|
| 335 |
+
).get("obligating_doc_id")
|
| 336 |
+
for e in instance.entries}
|
| 337 |
grouped = False
|
| 338 |
filtered_accounts: set[str] = set()
|
| 339 |
b_flagged = b_supported = 0
|
|
|
|
| 351 |
b_supported += (grouped or account_of.get(
|
| 352 |
s.args.get("entry_id")) in filtered_accounts)
|
| 353 |
b_rate = b_supported / b_flagged if b_flagged else 1.0
|
| 354 |
+
wd_flagged = wd_supported = 0
|
| 355 |
+
for eid, d in dispositions.items():
|
| 356 |
+
if (_dget(d, "verdict") in FLAGGING
|
| 357 |
+
and _dget(d, "issue_type") == "wrong_document"):
|
| 358 |
+
wd_flagged += 1
|
| 359 |
+
correct = correct_doc_of.get(eid)
|
| 360 |
+
wd_supported += (correct is not None
|
| 361 |
+
and correct in set(_dget(d, "evidence_ids") or ()))
|
| 362 |
+
wd_rate = wd_supported / wd_flagged if wd_flagged else 1.0
|
| 363 |
+
return GateResult(b_rate * wd_rate,
|
| 364 |
+
{"budget_flags": b_flagged,
|
| 365 |
+
"with_account_aggregate": b_supported,
|
| 366 |
+
"wrong_document_flags": wd_flagged,
|
| 367 |
+
"with_doc_citation": wd_supported})
|
| 368 |
|
| 369 |
|
| 370 |
def je05_rate_recompute(steps: list[StepRecord], dispositions: dict,
|
app/src/je_validation/scoring/verifiers/je10.py
CHANGED
|
@@ -1,11 +1,15 @@
|
|
| 1 |
-
"""JE-10: encumbrance liquidation. F1 x issue accuracy with exact-overage
|
| 2 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
from __future__ import annotations
|
| 4 |
|
| 5 |
from collections.abc import Iterable
|
| 6 |
|
| 7 |
from je_validation.scoring.claims import claims_of
|
| 8 |
-
from je_validation.scoring.verifiers import OutcomeVerdict,
|
| 9 |
|
| 10 |
|
| 11 |
def verify(dispositions: dict, entry_ids: Iterable[str], heldout) -> OutcomeVerdict:
|
|
@@ -24,10 +28,6 @@ def verify(dispositions: dict, entry_ids: Iterable[str], heldout) -> OutcomeVerd
|
|
| 24 |
# spec sign: budget - (actual + encumbrance), negative when over
|
| 25 |
claim = claims_of(dispositions, eid).get("available_cents")
|
| 26 |
ok = claim == heldout.claims[eid]["available_cents"]
|
| 27 |
-
if ok and dtype == "wrong_document":
|
| 28 |
-
cited = set(_get(dispositions[eid], "evidence_ids"))
|
| 29 |
-
ok = any(x in cited for x in heldout.evidence_map[eid]
|
| 30 |
-
if not x.startswith("master:"))
|
| 31 |
issue_correct += ok
|
| 32 |
fn = len(set(seeded) & ids) - tp
|
| 33 |
precision, recall, f1 = f1_stats(tp, fp, fn)
|
|
|
|
| 1 |
+
"""JE-10: encumbrance liquidation. F1 x issue accuracy with the exact-overage
|
| 2 |
+
condition folded into issue correctness (spec: stated overage must equal
|
| 3 |
+
budget - (actual + encumbrance), exact). The obligating-document citation is
|
| 4 |
+
a REPORTING requirement priced by the process gate (c5,
|
| 5 |
+
gates.je10_account_aggregate) - the spec's outcome line is "exception set and
|
| 6 |
+
type matched against seed", so citation selection never alters outcome."""
|
| 7 |
from __future__ import annotations
|
| 8 |
|
| 9 |
from collections.abc import Iterable
|
| 10 |
|
| 11 |
from je_validation.scoring.claims import claims_of
|
| 12 |
+
from je_validation.scoring.verifiers import OutcomeVerdict, f1_stats, flagged_ids, issue_of
|
| 13 |
|
| 14 |
|
| 15 |
def verify(dispositions: dict, entry_ids: Iterable[str], heldout) -> OutcomeVerdict:
|
|
|
|
| 28 |
# spec sign: budget - (actual + encumbrance), negative when over
|
| 29 |
claim = claims_of(dispositions, eid).get("available_cents")
|
| 30 |
ok = claim == heldout.claims[eid]["available_cents"]
|
|
|
|
|
|
|
|
|
|
|
|
|
| 31 |
issue_correct += ok
|
| 32 |
fn = len(set(seeded) & ids) - tp
|
| 33 |
precision, recall, f1 = f1_stats(tp, fp, fn)
|
data/runs/console/c5-grid-je01/baseline_flag_everything/seed1/JE-01_seed1.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_flag_everything/seed1/score.json
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/baseline_flag_everything/seed2/JE-01_seed2.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_flag_everything/seed2/score.json
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/baseline_flag_everything/seed3/JE-01_seed3.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-932f8101/baseline_flag_everything/seed1 → c5-grid-je01/baseline_flag_everything/seed3}/score.json
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/baseline_flag_everything/seed4/JE-01_seed4.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-932f8101/baseline_flag_everything/seed2 → c5-grid-je01/baseline_flag_everything/seed4}/score.json
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/baseline_flag_everything/seed5/JE-01_seed5.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-a5453c3d/baseline_flag_everything/seed1 → c5-grid-je01/baseline_flag_everything/seed5}/score.json
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/baseline_no_evidence/seed1/JE-01_seed1.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_no_evidence/seed1/score.json
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/baseline_no_evidence/seed2/JE-01_seed2.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_no_evidence/seed2/score.json
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/baseline_no_evidence/seed3/JE-01_seed3.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-932f8101/baseline_no_evidence/seed1 → c5-grid-je01/baseline_no_evidence/seed3}/score.json
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/baseline_no_evidence/seed4/JE-01_seed4.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-932f8101/baseline_no_evidence/seed2 → c5-grid-je01/baseline_no_evidence/seed4}/score.json
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/baseline_no_evidence/seed5/JE-01_seed5.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-a5453c3d/baseline_no_evidence/seed1 → c5-grid-je01/baseline_no_evidence/seed5}/score.json
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed1/JE-01_seed1.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-a5453c3d/baseline_no_evidence/seed3 → c5-grid-je01/deepseek_deepseek-v4-flash/seed1}/score.json
RENAMED
|
@@ -1,16 +1,16 @@
|
|
| 1 |
{
|
| 2 |
"outcome": 0.0,
|
| 3 |
-
"process": 0.
|
| 4 |
"rationale": null,
|
| 5 |
-
"combined": 0.
|
| 6 |
"fp_rate": 0.0,
|
| 7 |
"fp_penalty": 0.0,
|
| 8 |
-
"final": 0.
|
| 9 |
"fabrications": [],
|
| 10 |
"auto_fail": false,
|
| 11 |
"hard_fail": false,
|
| 12 |
"judge_pass": null,
|
| 13 |
"passed": null,
|
| 14 |
"judge": "absent",
|
| 15 |
-
"cost_usd": 0.
|
| 16 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"outcome": 0.0,
|
| 3 |
+
"process": 0.5,
|
| 4 |
"rationale": null,
|
| 5 |
+
"combined": 0.05555555555555556,
|
| 6 |
"fp_rate": 0.0,
|
| 7 |
"fp_penalty": 0.0,
|
| 8 |
+
"final": 0.05555555555555556,
|
| 9 |
"fabrications": [],
|
| 10 |
"auto_fail": false,
|
| 11 |
"hard_fail": false,
|
| 12 |
"judge_pass": null,
|
| 13 |
"passed": null,
|
| 14 |
"judge": "absent",
|
| 15 |
+
"cost_usd": 0.22998800869240005
|
| 16 |
}
|
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed2/JE-01_seed2.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-a5453c3d/baseline_no_evidence/seed5 → c5-grid-je01/deepseek_deepseek-v4-flash/seed2}/score.json
RENAMED
|
@@ -1,16 +1,16 @@
|
|
| 1 |
{
|
| 2 |
"outcome": 0.0,
|
| 3 |
-
"process": 0.
|
| 4 |
"rationale": null,
|
| 5 |
-
"combined": 0.
|
| 6 |
"fp_rate": 0.0,
|
| 7 |
"fp_penalty": 0.0,
|
| 8 |
-
"final": 0.
|
| 9 |
"fabrications": [],
|
| 10 |
"auto_fail": false,
|
| 11 |
"hard_fail": false,
|
| 12 |
"judge_pass": null,
|
| 13 |
"passed": null,
|
| 14 |
"judge": "absent",
|
| 15 |
-
"cost_usd": 0.
|
| 16 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"outcome": 0.0,
|
| 3 |
+
"process": 0.495,
|
| 4 |
"rationale": null,
|
| 5 |
+
"combined": 0.055,
|
| 6 |
"fp_rate": 0.0,
|
| 7 |
"fp_penalty": 0.0,
|
| 8 |
+
"final": 0.055,
|
| 9 |
"fabrications": [],
|
| 10 |
"auto_fail": false,
|
| 11 |
"hard_fail": false,
|
| 12 |
"judge_pass": null,
|
| 13 |
"passed": null,
|
| 14 |
"judge": "absent",
|
| 15 |
+
"cost_usd": 0.45799382661999993
|
| 16 |
}
|
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed3/JE-01_seed3.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed3/score.json
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"outcome": 0.0,
|
| 3 |
+
"process": 0.3333333333333333,
|
| 4 |
+
"rationale": null,
|
| 5 |
+
"combined": 0.037037037037037035,
|
| 6 |
+
"fp_rate": 0.0,
|
| 7 |
+
"fp_penalty": 0.0,
|
| 8 |
+
"final": 0.037037037037037035,
|
| 9 |
+
"fabrications": [],
|
| 10 |
+
"auto_fail": false,
|
| 11 |
+
"hard_fail": false,
|
| 12 |
+
"judge_pass": null,
|
| 13 |
+
"passed": null,
|
| 14 |
+
"judge": "absent",
|
| 15 |
+
"cost_usd": 0.39998567448000016
|
| 16 |
+
}
|
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed4/JE-01_seed4.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed4/score.json
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"outcome": 0.0,
|
| 3 |
+
"process": 0.16666666666666666,
|
| 4 |
+
"rationale": null,
|
| 5 |
+
"combined": 0.018518518518518517,
|
| 6 |
+
"fp_rate": 0.0,
|
| 7 |
+
"fp_penalty": 0.0,
|
| 8 |
+
"final": 0.018518518518518517,
|
| 9 |
+
"fabrications": [],
|
| 10 |
+
"auto_fail": false,
|
| 11 |
+
"hard_fail": false,
|
| 12 |
+
"judge_pass": null,
|
| 13 |
+
"passed": null,
|
| 14 |
+
"judge": "absent",
|
| 15 |
+
"cost_usd": 0.36706333776000016
|
| 16 |
+
}
|
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed5/JE-01_seed5.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed5/score.json
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"outcome": 0.0,
|
| 3 |
+
"process": 0.14285714285714285,
|
| 4 |
+
"rationale": null,
|
| 5 |
+
"combined": 0.015873015873015872,
|
| 6 |
+
"fp_rate": 0.0,
|
| 7 |
+
"fp_penalty": 0.0,
|
| 8 |
+
"final": 0.015873015873015872,
|
| 9 |
+
"fabrications": [],
|
| 10 |
+
"auto_fail": false,
|
| 11 |
+
"hard_fail": false,
|
| 12 |
+
"judge_pass": null,
|
| 13 |
+
"passed": null,
|
| 14 |
+
"judge": "absent",
|
| 15 |
+
"cost_usd": 0.38351098015999985
|
| 16 |
+
}
|
data/runs/console/{run-a5453c3d → c5-grid-je01}/results.json
RENAMED
|
@@ -8,130 +8,29 @@
|
|
| 8 |
"std": 0.0
|
| 9 |
},
|
| 10 |
"process": {
|
| 11 |
-
"mean": 0.
|
| 12 |
-
"std": 0.
|
| 13 |
},
|
| 14 |
"rationale": "pending judge calibration",
|
| 15 |
"combined": {
|
| 16 |
-
"mean": 0.
|
| 17 |
-
"std": 0.
|
| 18 |
},
|
| 19 |
"final": {
|
| 20 |
-
"mean": 0.
|
| 21 |
-
"std": 0.
|
| 22 |
},
|
| 23 |
"fail_rate": 0.0
|
| 24 |
},
|
| 25 |
"reports": [
|
| 26 |
-
{
|
| 27 |
-
"outcome": 0.0,
|
| 28 |
-
"process": 0.6666666666666666,
|
| 29 |
-
"rationale": "pending judge calibration",
|
| 30 |
-
"combined": 0.11764705882352941,
|
| 31 |
-
"fp_rate": 0.02,
|
| 32 |
-
"fp_penalty": 0.0,
|
| 33 |
-
"final": 0.11764705882352941,
|
| 34 |
-
"fabrications": [],
|
| 35 |
-
"auto_fail": false,
|
| 36 |
-
"hard_fail": false,
|
| 37 |
-
"judge_pass": null,
|
| 38 |
-
"passed": null,
|
| 39 |
-
"judge": "absent"
|
| 40 |
-
},
|
| 41 |
{
|
| 42 |
"outcome": 0.0,
|
| 43 |
"process": 0.5,
|
| 44 |
"rationale": "pending judge calibration",
|
| 45 |
-
"combined": 0.
|
| 46 |
-
"fp_rate": 0.
|
| 47 |
-
"fp_penalty": 0.0,
|
| 48 |
-
"final": 0.08823529411764706,
|
| 49 |
-
"fabrications": [],
|
| 50 |
-
"auto_fail": false,
|
| 51 |
-
"hard_fail": false,
|
| 52 |
-
"judge_pass": null,
|
| 53 |
-
"passed": null,
|
| 54 |
-
"judge": "absent"
|
| 55 |
-
},
|
| 56 |
-
{
|
| 57 |
-
"outcome": 0.0,
|
| 58 |
-
"process": 0.9500000000000001,
|
| 59 |
-
"rationale": "pending judge calibration",
|
| 60 |
-
"combined": 0.16764705882352943,
|
| 61 |
-
"fp_rate": 0.02,
|
| 62 |
-
"fp_penalty": 0.0,
|
| 63 |
-
"final": 0.16764705882352943,
|
| 64 |
-
"fabrications": [],
|
| 65 |
-
"auto_fail": false,
|
| 66 |
-
"hard_fail": false,
|
| 67 |
-
"judge_pass": null,
|
| 68 |
-
"passed": null,
|
| 69 |
-
"judge": "absent"
|
| 70 |
-
},
|
| 71 |
-
{
|
| 72 |
-
"outcome": 0.0,
|
| 73 |
-
"process": 0.6466666666666667,
|
| 74 |
-
"rationale": "pending judge calibration",
|
| 75 |
-
"combined": 0.11411764705882353,
|
| 76 |
-
"fp_rate": 0.02,
|
| 77 |
-
"fp_penalty": 0.0,
|
| 78 |
-
"final": 0.11411764705882353,
|
| 79 |
-
"fabrications": [],
|
| 80 |
-
"auto_fail": false,
|
| 81 |
-
"hard_fail": false,
|
| 82 |
-
"judge_pass": null,
|
| 83 |
-
"passed": null,
|
| 84 |
-
"judge": "absent"
|
| 85 |
-
},
|
| 86 |
-
{
|
| 87 |
-
"outcome": 0.0,
|
| 88 |
-
"process": 0.672,
|
| 89 |
-
"rationale": "pending judge calibration",
|
| 90 |
-
"combined": 0.11858823529411765,
|
| 91 |
-
"fp_rate": 0.06,
|
| 92 |
-
"fp_penalty": 0.0,
|
| 93 |
-
"final": 0.11858823529411765,
|
| 94 |
-
"fabrications": [],
|
| 95 |
-
"auto_fail": false,
|
| 96 |
-
"hard_fail": false,
|
| 97 |
-
"judge_pass": null,
|
| 98 |
-
"passed": null,
|
| 99 |
-
"judge": "absent"
|
| 100 |
-
}
|
| 101 |
-
]
|
| 102 |
-
},
|
| 103 |
-
"openai/gpt-5.6-sol": {
|
| 104 |
-
"aggregate": {
|
| 105 |
-
"n": 5,
|
| 106 |
-
"variance_measurable": true,
|
| 107 |
-
"outcome": {
|
| 108 |
-
"mean": 0.0,
|
| 109 |
-
"std": 0.0
|
| 110 |
-
},
|
| 111 |
-
"process": {
|
| 112 |
-
"mean": 0.8142857142857143,
|
| 113 |
-
"std": 0.18454300975376878
|
| 114 |
-
},
|
| 115 |
-
"rationale": "pending judge calibration",
|
| 116 |
-
"combined": {
|
| 117 |
-
"mean": 0.14369747899159663,
|
| 118 |
-
"std": 0.0325664134859592
|
| 119 |
-
},
|
| 120 |
-
"final": {
|
| 121 |
-
"mean": 0.14369747899159663,
|
| 122 |
-
"std": 0.0325664134859592
|
| 123 |
-
},
|
| 124 |
-
"fail_rate": 0.0
|
| 125 |
-
},
|
| 126 |
-
"reports": [
|
| 127 |
-
{
|
| 128 |
-
"outcome": 0.0,
|
| 129 |
-
"process": 0.75,
|
| 130 |
-
"rationale": "pending judge calibration",
|
| 131 |
-
"combined": 0.1323529411764706,
|
| 132 |
-
"fp_rate": 0.08,
|
| 133 |
"fp_penalty": 0.0,
|
| 134 |
-
"final": 0.
|
| 135 |
"fabrications": [],
|
| 136 |
"auto_fail": false,
|
| 137 |
"hard_fail": false,
|
|
@@ -141,12 +40,12 @@
|
|
| 141 |
},
|
| 142 |
{
|
| 143 |
"outcome": 0.0,
|
| 144 |
-
"process": 0.
|
| 145 |
"rationale": "pending judge calibration",
|
| 146 |
-
"combined": 0.
|
| 147 |
-
"fp_rate": 0.
|
| 148 |
"fp_penalty": 0.0,
|
| 149 |
-
"final": 0.
|
| 150 |
"fabrications": [],
|
| 151 |
"auto_fail": false,
|
| 152 |
"hard_fail": false,
|
|
@@ -156,12 +55,12 @@
|
|
| 156 |
},
|
| 157 |
{
|
| 158 |
"outcome": 0.0,
|
| 159 |
-
"process":
|
| 160 |
"rationale": "pending judge calibration",
|
| 161 |
-
"combined": 0.
|
| 162 |
-
"fp_rate": 0.
|
| 163 |
"fp_penalty": 0.0,
|
| 164 |
-
"final": 0.
|
| 165 |
"fabrications": [],
|
| 166 |
"auto_fail": false,
|
| 167 |
"hard_fail": false,
|
|
@@ -171,12 +70,12 @@
|
|
| 171 |
},
|
| 172 |
{
|
| 173 |
"outcome": 0.0,
|
| 174 |
-
"process": 0.
|
| 175 |
"rationale": "pending judge calibration",
|
| 176 |
-
"combined": 0.
|
| 177 |
-
"fp_rate": 0.
|
| 178 |
"fp_penalty": 0.0,
|
| 179 |
-
"final": 0.
|
| 180 |
"fabrications": [],
|
| 181 |
"auto_fail": false,
|
| 182 |
"hard_fail": false,
|
|
@@ -186,12 +85,12 @@
|
|
| 186 |
},
|
| 187 |
{
|
| 188 |
"outcome": 0.0,
|
| 189 |
-
"process":
|
| 190 |
"rationale": "pending judge calibration",
|
| 191 |
-
"combined": 0.
|
| 192 |
"fp_rate": 0.0,
|
| 193 |
"fp_penalty": 0.0,
|
| 194 |
-
"final": 0.
|
| 195 |
"fabrications": [],
|
| 196 |
"auto_fail": false,
|
| 197 |
"hard_fail": false,
|
|
|
|
| 8 |
"std": 0.0
|
| 9 |
},
|
| 10 |
"process": {
|
| 11 |
+
"mean": 0.32757142857142857,
|
| 12 |
+
"std": 0.17161501105130808
|
| 13 |
},
|
| 14 |
"rationale": "pending judge calibration",
|
| 15 |
"combined": {
|
| 16 |
+
"mean": 0.03639682539682539,
|
| 17 |
+
"std": 0.019068334561256454
|
| 18 |
},
|
| 19 |
"final": {
|
| 20 |
+
"mean": 0.03639682539682539,
|
| 21 |
+
"std": 0.019068334561256454
|
| 22 |
},
|
| 23 |
"fail_rate": 0.0
|
| 24 |
},
|
| 25 |
"reports": [
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 26 |
{
|
| 27 |
"outcome": 0.0,
|
| 28 |
"process": 0.5,
|
| 29 |
"rationale": "pending judge calibration",
|
| 30 |
+
"combined": 0.05555555555555556,
|
| 31 |
+
"fp_rate": 0.0,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 32 |
"fp_penalty": 0.0,
|
| 33 |
+
"final": 0.05555555555555556,
|
| 34 |
"fabrications": [],
|
| 35 |
"auto_fail": false,
|
| 36 |
"hard_fail": false,
|
|
|
|
| 40 |
},
|
| 41 |
{
|
| 42 |
"outcome": 0.0,
|
| 43 |
+
"process": 0.495,
|
| 44 |
"rationale": "pending judge calibration",
|
| 45 |
+
"combined": 0.055,
|
| 46 |
+
"fp_rate": 0.0,
|
| 47 |
"fp_penalty": 0.0,
|
| 48 |
+
"final": 0.055,
|
| 49 |
"fabrications": [],
|
| 50 |
"auto_fail": false,
|
| 51 |
"hard_fail": false,
|
|
|
|
| 55 |
},
|
| 56 |
{
|
| 57 |
"outcome": 0.0,
|
| 58 |
+
"process": 0.3333333333333333,
|
| 59 |
"rationale": "pending judge calibration",
|
| 60 |
+
"combined": 0.037037037037037035,
|
| 61 |
+
"fp_rate": 0.0,
|
| 62 |
"fp_penalty": 0.0,
|
| 63 |
+
"final": 0.037037037037037035,
|
| 64 |
"fabrications": [],
|
| 65 |
"auto_fail": false,
|
| 66 |
"hard_fail": false,
|
|
|
|
| 70 |
},
|
| 71 |
{
|
| 72 |
"outcome": 0.0,
|
| 73 |
+
"process": 0.16666666666666666,
|
| 74 |
"rationale": "pending judge calibration",
|
| 75 |
+
"combined": 0.018518518518518517,
|
| 76 |
+
"fp_rate": 0.0,
|
| 77 |
"fp_penalty": 0.0,
|
| 78 |
+
"final": 0.018518518518518517,
|
| 79 |
"fabrications": [],
|
| 80 |
"auto_fail": false,
|
| 81 |
"hard_fail": false,
|
|
|
|
| 85 |
},
|
| 86 |
{
|
| 87 |
"outcome": 0.0,
|
| 88 |
+
"process": 0.14285714285714285,
|
| 89 |
"rationale": "pending judge calibration",
|
| 90 |
+
"combined": 0.015873015873015872,
|
| 91 |
"fp_rate": 0.0,
|
| 92 |
"fp_penalty": 0.0,
|
| 93 |
+
"final": 0.015873015873015872,
|
| 94 |
"fabrications": [],
|
| 95 |
"auto_fail": false,
|
| 96 |
"hard_fail": false,
|
data/runs/console/{run-a5453c3d → c5-grid-je01}/run.json
RENAMED
|
@@ -1,18 +1,17 @@
|
|
| 1 |
{
|
| 2 |
-
"run_id": "
|
| 3 |
"status": "finished",
|
| 4 |
"request": {
|
| 5 |
-
"run_id": "
|
| 6 |
-
"task_id": "JE-
|
| 7 |
"models": [
|
| 8 |
"deepseek/deepseek-v4-flash",
|
| 9 |
-
"openai/gpt-5.6-sol",
|
| 10 |
"baseline:flag_everything",
|
| 11 |
"baseline:no_evidence"
|
| 12 |
],
|
| 13 |
"seed_count": 5,
|
| 14 |
"knobs": {
|
| 15 |
-
"population":
|
| 16 |
"defect_density": 28,
|
| 17 |
"distractor_ratio": 2,
|
| 18 |
"doc_noise": 10
|
|
@@ -20,56 +19,49 @@
|
|
| 20 |
"tools_enabled": [
|
| 21 |
"query_ledger",
|
| 22 |
"get_entry",
|
| 23 |
-
"get_master_data",
|
| 24 |
"recompute",
|
| 25 |
"disposition",
|
| 26 |
"submit"
|
| 27 |
],
|
| 28 |
-
"step_budget":
|
| 29 |
-
"token_ceiling":
|
| 30 |
"pricing": {
|
| 31 |
"deepseek/deepseek-v4-flash": {
|
| 32 |
-
"prompt_price":
|
| 33 |
-
"completion_price":
|
| 34 |
-
},
|
| 35 |
-
"openai/gpt-5.6-sol": {
|
| 36 |
-
"prompt_price": 5e-06,
|
| 37 |
-
"completion_price": 3e-05
|
| 38 |
}
|
| 39 |
},
|
| 40 |
"prompt_version": "standard"
|
| 41 |
},
|
| 42 |
"contract": {
|
| 43 |
-
"task_id": "JE-
|
| 44 |
"tier": "L1",
|
| 45 |
"seed": 1,
|
| 46 |
"generator": {
|
| 47 |
"knobs": {
|
| 48 |
-
"population":
|
| 49 |
"defect_density": 28,
|
| 50 |
"distractor_ratio": 2,
|
| 51 |
"doc_noise": 10
|
| 52 |
},
|
| 53 |
"sources": [
|
| 54 |
-
"DS-A"
|
| 55 |
-
"DS-J"
|
| 56 |
]
|
| 57 |
},
|
| 58 |
"tools_enabled": [
|
| 59 |
"query_ledger",
|
| 60 |
"get_entry",
|
| 61 |
-
"get_master_data",
|
| 62 |
"recompute",
|
| 63 |
"disposition",
|
| 64 |
"submit"
|
| 65 |
],
|
| 66 |
"tools_withheld": [],
|
| 67 |
-
"step_budget":
|
| 68 |
"scoring": {
|
| 69 |
"weights": {
|
| 70 |
-
"outcome": 0.
|
| 71 |
-
"process": 0.
|
| 72 |
-
"rationale": 0.
|
| 73 |
},
|
| 74 |
"judge": "global_rubric_v1",
|
| 75 |
"judges_per_episode": 1
|
|
@@ -77,12 +69,11 @@
|
|
| 77 |
"sweep": {
|
| 78 |
"models": [
|
| 79 |
"deepseek/deepseek-v4-flash",
|
| 80 |
-
"openai/gpt-5.6-sol",
|
| 81 |
"baseline:flag_everything",
|
| 82 |
"baseline:no_evidence"
|
| 83 |
],
|
| 84 |
"seeds": 5,
|
| 85 |
-
"episodes":
|
| 86 |
},
|
| 87 |
"prompt_version": "standard"
|
| 88 |
},
|
|
@@ -92,11 +83,6 @@
|
|
| 92 |
"deepseek_deepseek-v4-flash/seed3": "finished",
|
| 93 |
"deepseek_deepseek-v4-flash/seed4": "finished",
|
| 94 |
"deepseek_deepseek-v4-flash/seed5": "finished",
|
| 95 |
-
"openai_gpt-5.6-sol/seed1": "finished",
|
| 96 |
-
"openai_gpt-5.6-sol/seed2": "finished",
|
| 97 |
-
"openai_gpt-5.6-sol/seed3": "finished",
|
| 98 |
-
"openai_gpt-5.6-sol/seed4": "finished",
|
| 99 |
-
"openai_gpt-5.6-sol/seed5": "finished",
|
| 100 |
"baseline_flag_everything/seed1": "finished",
|
| 101 |
"baseline_flag_everything/seed2": "finished",
|
| 102 |
"baseline_flag_everything/seed3": "finished",
|
|
@@ -108,5 +94,5 @@
|
|
| 108 |
"baseline_no_evidence/seed4": "finished",
|
| 109 |
"baseline_no_evidence/seed5": "finished"
|
| 110 |
},
|
| 111 |
-
"cost_usd":
|
| 112 |
}
|
|
|
|
| 1 |
{
|
| 2 |
+
"run_id": "c5-grid-je01",
|
| 3 |
"status": "finished",
|
| 4 |
"request": {
|
| 5 |
+
"run_id": "c5-grid-je01",
|
| 6 |
+
"task_id": "JE-01",
|
| 7 |
"models": [
|
| 8 |
"deepseek/deepseek-v4-flash",
|
|
|
|
| 9 |
"baseline:flag_everything",
|
| 10 |
"baseline:no_evidence"
|
| 11 |
],
|
| 12 |
"seed_count": 5,
|
| 13 |
"knobs": {
|
| 14 |
+
"population": 400,
|
| 15 |
"defect_density": 28,
|
| 16 |
"distractor_ratio": 2,
|
| 17 |
"doc_noise": 10
|
|
|
|
| 19 |
"tools_enabled": [
|
| 20 |
"query_ledger",
|
| 21 |
"get_entry",
|
|
|
|
| 22 |
"recompute",
|
| 23 |
"disposition",
|
| 24 |
"submit"
|
| 25 |
],
|
| 26 |
+
"step_budget": 1800,
|
| 27 |
+
"token_ceiling": 1530000000,
|
| 28 |
"pricing": {
|
| 29 |
"deepseek/deepseek-v4-flash": {
|
| 30 |
+
"prompt_price": 1.4e-07,
|
| 31 |
+
"completion_price": 2.8e-07
|
|
|
|
|
|
|
|
|
|
|
|
|
| 32 |
}
|
| 33 |
},
|
| 34 |
"prompt_version": "standard"
|
| 35 |
},
|
| 36 |
"contract": {
|
| 37 |
+
"task_id": "JE-01",
|
| 38 |
"tier": "L1",
|
| 39 |
"seed": 1,
|
| 40 |
"generator": {
|
| 41 |
"knobs": {
|
| 42 |
+
"population": 400,
|
| 43 |
"defect_density": 28,
|
| 44 |
"distractor_ratio": 2,
|
| 45 |
"doc_noise": 10
|
| 46 |
},
|
| 47 |
"sources": [
|
| 48 |
+
"DS-A"
|
|
|
|
| 49 |
]
|
| 50 |
},
|
| 51 |
"tools_enabled": [
|
| 52 |
"query_ledger",
|
| 53 |
"get_entry",
|
|
|
|
| 54 |
"recompute",
|
| 55 |
"disposition",
|
| 56 |
"submit"
|
| 57 |
],
|
| 58 |
"tools_withheld": [],
|
| 59 |
+
"step_budget": 1800,
|
| 60 |
"scoring": {
|
| 61 |
"weights": {
|
| 62 |
+
"outcome": 0.8,
|
| 63 |
+
"process": 0.1,
|
| 64 |
+
"rationale": 0.1
|
| 65 |
},
|
| 66 |
"judge": "global_rubric_v1",
|
| 67 |
"judges_per_episode": 1
|
|
|
|
| 69 |
"sweep": {
|
| 70 |
"models": [
|
| 71 |
"deepseek/deepseek-v4-flash",
|
|
|
|
| 72 |
"baseline:flag_everything",
|
| 73 |
"baseline:no_evidence"
|
| 74 |
],
|
| 75 |
"seeds": 5,
|
| 76 |
+
"episodes": 15
|
| 77 |
},
|
| 78 |
"prompt_version": "standard"
|
| 79 |
},
|
|
|
|
| 83 |
"deepseek_deepseek-v4-flash/seed3": "finished",
|
| 84 |
"deepseek_deepseek-v4-flash/seed4": "finished",
|
| 85 |
"deepseek_deepseek-v4-flash/seed5": "finished",
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 86 |
"baseline_flag_everything/seed1": "finished",
|
| 87 |
"baseline_flag_everything/seed2": "finished",
|
| 88 |
"baseline_flag_everything/seed3": "finished",
|
|
|
|
| 94 |
"baseline_no_evidence/seed4": "finished",
|
| 95 |
"baseline_no_evidence/seed5": "finished"
|
| 96 |
},
|
| 97 |
+
"cost_usd": 1.838541827712399
|
| 98 |
}
|
data/runs/console/c5-grid-je01/seed1/bundle/heldout/defect_ledger.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"claims":{},"defects":[{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0000219389","revealing_evidence_id":"2024-0000219389:lines"},{"correct_verdict":"flag","defect_type":"dup_line_seq","entry_id":"2024-0018941205","revealing_evidence_id":"2024-0018941205:lines"},{"correct_verdict":"flag","defect_type":"orphan_leg","entry_id":"2024-0018944498","revealing_evidence_id":"2024-0018944498:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018945399","revealing_evidence_id":"2024-0018945399:lines"},{"correct_verdict":"flag","defect_type":"orphan_leg","entry_id":"2024-0018950877","revealing_evidence_id":"2024-0018950877:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018952717","revealing_evidence_id":"2024-0018952717:lines"},{"correct_verdict":"flag","defect_type":"orphan_leg","entry_id":"2024-0018953134","revealing_evidence_id":"2024-0018953134:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018955782","revealing_evidence_id":"2024-0018955782:lines"},{"correct_verdict":"flag","defect_type":"zero_amount_line","entry_id":"2024-0018958794","revealing_evidence_id":"2024-0018958794:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018966176","revealing_evidence_id":"2024-0018966176:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018967117","revealing_evidence_id":"2024-0018967117:lines"},{"correct_verdict":"flag","defect_type":"zero_amount_line","entry_id":"2024-0018972058","revealing_evidence_id":"2024-0018972058:lines"}],"evidence_map":{"2024-0000219389":["2024-0000219389:lines"],"2024-0018941205":["2024-0018941205:lines"],"2024-0018944498":["2024-0018944498:lines"],"2024-0018945399":["2024-0018945399:lines"],"2024-0018950877":["2024-0018950877:lines"],"2024-0018952717":["2024-0018952717:lines"],"2024-0018953134":["2024-0018953134:lines"],"2024-0018955782":["2024-0018955782:lines"],"2024-0018958794":["2024-0018958794:lines"],"2024-0018966176":["2024-0018966176:lines"],"2024-0018967117":["2024-0018967117:lines"],"2024-0018972058":["2024-0018972058:lines"]},"required_docs":[]}
|
data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed1/bundle/instance/documents.jsonl
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/seed1/bundle/instance/entries.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed1/bundle/instance/master_data.json
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/seed1/bundle/instance/meta.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed":1,"task_id":"JE-01"}
|
data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed1/bundle/instance/policies.json
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/seed1/bundle/manifest.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"generator_version":"4c.1","knobs":{"n_entries":400},"seed":1,"snapshot_id":"2026-08-04","spec_sha256":"90d55c077a0acea9a57b95ecd63a70df0316fc271d873a38afa195a766f23978","task_id":"JE-01"}
|
data/runs/console/c5-grid-je01/seed2/bundle/heldout/defect_ledger.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"claims":{},"defects":[{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018942198","revealing_evidence_id":"2024-0018942198:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018944317","revealing_evidence_id":"2024-0018944317:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018946773","revealing_evidence_id":"2024-0018946773:lines"},{"correct_verdict":"flag","defect_type":"orphan_leg","entry_id":"2024-0018950131","revealing_evidence_id":"2024-0018950131:lines"},{"correct_verdict":"flag","defect_type":"orphan_leg","entry_id":"2024-0018951225","revealing_evidence_id":"2024-0018951225:lines"},{"correct_verdict":"flag","defect_type":"zero_amount_line","entry_id":"2024-0018952858","revealing_evidence_id":"2024-0018952858:lines"},{"correct_verdict":"flag","defect_type":"orphan_leg","entry_id":"2024-0018960305","revealing_evidence_id":"2024-0018960305:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018965646","revealing_evidence_id":"2024-0018965646:lines"},{"correct_verdict":"flag","defect_type":"dup_line_seq","entry_id":"2024-0018965792","revealing_evidence_id":"2024-0018965792:lines"},{"correct_verdict":"flag","defect_type":"zero_amount_line","entry_id":"2024-0018969623","revealing_evidence_id":"2024-0018969623:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018969733","revealing_evidence_id":"2024-0018969733:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018969865","revealing_evidence_id":"2024-0018969865:lines"}],"evidence_map":{"2024-0018942198":["2024-0018942198:lines"],"2024-0018944317":["2024-0018944317:lines"],"2024-0018946773":["2024-0018946773:lines"],"2024-0018950131":["2024-0018950131:lines"],"2024-0018951225":["2024-0018951225:lines"],"2024-0018952858":["2024-0018952858:lines"],"2024-0018960305":["2024-0018960305:lines"],"2024-0018965646":["2024-0018965646:lines"],"2024-0018965792":["2024-0018965792:lines"],"2024-0018969623":["2024-0018969623:lines"],"2024-0018969733":["2024-0018969733:lines"],"2024-0018969865":["2024-0018969865:lines"]},"required_docs":[]}
|
data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed2/bundle/instance/documents.jsonl
RENAMED
|
File without changes
|
data/runs/console/c5-grid-je01/seed2/bundle/instance/entries.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|