ParthKulshreshtha commited on
Commit
8bccf1e
·
verified ·
1 Parent(s): 1c43ec2

Upload folder using huggingface_hub

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. app/src/je_validation/console/static/app.js +28 -0
  2. app/src/je_validation/console/static/content/tasks.html +94 -11
  3. app/src/je_validation/console/static/landing.css +53 -0
  4. app/src/je_validation/console/static/styles.css +23 -0
  5. app/src/je_validation/console/static/tabs.js +15 -2
  6. app/src/je_validation/envir/briefs.py +24 -0
  7. app/src/je_validation/scoring/gates.py +37 -2
  8. app/src/je_validation/scoring/verifiers/je10.py +7 -7
  9. data/runs/console/c5-grid-je01/baseline_flag_everything/seed1/JE-01_seed1.jsonl +0 -0
  10. data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_flag_everything/seed1/score.json +0 -0
  11. data/runs/console/c5-grid-je01/baseline_flag_everything/seed2/JE-01_seed2.jsonl +0 -0
  12. data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_flag_everything/seed2/score.json +0 -0
  13. data/runs/console/c5-grid-je01/baseline_flag_everything/seed3/JE-01_seed3.jsonl +0 -0
  14. data/runs/console/{run-932f8101/baseline_flag_everything/seed1 → c5-grid-je01/baseline_flag_everything/seed3}/score.json +0 -0
  15. data/runs/console/c5-grid-je01/baseline_flag_everything/seed4/JE-01_seed4.jsonl +0 -0
  16. data/runs/console/{run-932f8101/baseline_flag_everything/seed2 → c5-grid-je01/baseline_flag_everything/seed4}/score.json +0 -0
  17. data/runs/console/c5-grid-je01/baseline_flag_everything/seed5/JE-01_seed5.jsonl +0 -0
  18. data/runs/console/{run-a5453c3d/baseline_flag_everything/seed1 → c5-grid-je01/baseline_flag_everything/seed5}/score.json +0 -0
  19. data/runs/console/c5-grid-je01/baseline_no_evidence/seed1/JE-01_seed1.jsonl +0 -0
  20. data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_no_evidence/seed1/score.json +0 -0
  21. data/runs/console/c5-grid-je01/baseline_no_evidence/seed2/JE-01_seed2.jsonl +0 -0
  22. data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_no_evidence/seed2/score.json +0 -0
  23. data/runs/console/c5-grid-je01/baseline_no_evidence/seed3/JE-01_seed3.jsonl +0 -0
  24. data/runs/console/{run-932f8101/baseline_no_evidence/seed1 → c5-grid-je01/baseline_no_evidence/seed3}/score.json +0 -0
  25. data/runs/console/c5-grid-je01/baseline_no_evidence/seed4/JE-01_seed4.jsonl +0 -0
  26. data/runs/console/{run-932f8101/baseline_no_evidence/seed2 → c5-grid-je01/baseline_no_evidence/seed4}/score.json +0 -0
  27. data/runs/console/c5-grid-je01/baseline_no_evidence/seed5/JE-01_seed5.jsonl +0 -0
  28. data/runs/console/{run-a5453c3d/baseline_no_evidence/seed1 → c5-grid-je01/baseline_no_evidence/seed5}/score.json +0 -0
  29. data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed1/JE-01_seed1.jsonl +0 -0
  30. data/runs/console/{run-a5453c3d/baseline_no_evidence/seed3 → c5-grid-je01/deepseek_deepseek-v4-flash/seed1}/score.json +4 -4
  31. data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed2/JE-01_seed2.jsonl +0 -0
  32. data/runs/console/{run-a5453c3d/baseline_no_evidence/seed5 → c5-grid-je01/deepseek_deepseek-v4-flash/seed2}/score.json +4 -4
  33. data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed3/JE-01_seed3.jsonl +0 -0
  34. data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed3/score.json +16 -0
  35. data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed4/JE-01_seed4.jsonl +0 -0
  36. data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed4/score.json +16 -0
  37. data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed5/JE-01_seed5.jsonl +0 -0
  38. data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed5/score.json +16 -0
  39. data/runs/console/{run-a5453c3d → c5-grid-je01}/results.json +24 -125
  40. data/runs/console/{run-a5453c3d → c5-grid-je01}/run.json +17 -31
  41. data/runs/console/c5-grid-je01/seed1/bundle/heldout/defect_ledger.json +1 -0
  42. data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed1/bundle/instance/documents.jsonl +0 -0
  43. data/runs/console/c5-grid-je01/seed1/bundle/instance/entries.jsonl +0 -0
  44. data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed1/bundle/instance/master_data.json +0 -0
  45. data/runs/console/c5-grid-je01/seed1/bundle/instance/meta.json +1 -0
  46. data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed1/bundle/instance/policies.json +0 -0
  47. data/runs/console/c5-grid-je01/seed1/bundle/manifest.json +1 -0
  48. data/runs/console/c5-grid-je01/seed2/bundle/heldout/defect_ledger.json +1 -0
  49. data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed2/bundle/instance/documents.jsonl +0 -0
  50. data/runs/console/c5-grid-je01/seed2/bundle/instance/entries.jsonl +0 -0
app/src/je_validation/console/static/app.js CHANGED
@@ -793,6 +793,34 @@ async function openPastRun(runId) {
793
  openLiveView(runId, snap.contract);
794
  }
795
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
796
  // ---- live view: SSE client -------------------------------------------------
797
  //
798
  // A fresh EventSource cannot send Last-Event-ID (the browser only adds it on
 
793
  openLiveView(runId, snap.contract);
794
  }
795
 
796
+ // ---- deep links from the Tasks tab -----------------------------------------
797
+ //
798
+ // tabs.js translates #console/run/<run_id> into a "je:open-run" event (it is
799
+ // the only hash reader). Opening the run scrolls to the live view; the pulse
800
+ // on the panel and on the run's history row says "this is the one you
801
+ // clicked". Unknown run ids degrade to the plain console (openPastRun
802
+ // already swallows the failed fetch).
803
+
804
+ function pulseOnce(el) {
805
+ if (!el) return;
806
+ el.classList.remove('deep-flash');
807
+ void el.offsetWidth; // restart the animation on repeat clicks
808
+ el.classList.add('deep-flash');
809
+ setTimeout(() => el.classList.remove('deep-flash'), 1600);
810
+ }
811
+
812
+ async function openDeepLinkRun(runId) {
813
+ await loadRunHistory(); // ensure the row exists before pulsing it
814
+ await openPastRun(runId);
815
+ if (!Console.live || Console.live.runId !== runId) return;
816
+ pulseOnce(document.querySelector('#live .panel'));
817
+ const idEl = [...document.querySelectorAll('#history-list .history-id')]
818
+ .find((el) => el.textContent === runId);
819
+ if (idEl) pulseOnce(idEl.closest('.history-row'));
820
+ }
821
+
822
+ window.addEventListener('je:open-run', (e) => openDeepLinkRun(e.detail));
823
+
824
  // ---- live view: SSE client -------------------------------------------------
825
  //
826
  // A fresh EventSource cannot send Last-Event-ID (the browser only adds it on
app/src/je_validation/console/static/content/tasks.html CHANGED
@@ -15,24 +15,65 @@
15
  <div class="task-card"><h3><span class="je-id">JE-01</span> Broken entries
16
  <span class="badge-live">Live</span></h3>
17
  <p>An overnight interface load has left a batch of entries damaged. Find
18
- every entry that no longer balances or is missing required pieces.</p></div>
 
 
 
 
 
 
 
 
19
  <div class="task-card"><h3><span class="je-id">JE-02</span> Coding validity
20
  <span class="badge-live">Live</span></h3>
21
  <p>Account, fund, and department codes must exist, be active in the
22
  period, and be an allowed combination. Find the entries that violate any
23
- of it.</p></div>
 
 
 
 
 
 
 
 
24
  <div class="task-card"><h3><span class="je-id">JE-03</span> Period cut-off
25
  <span class="badge-live">Live</span></h3>
26
  <p>Entries must be posted to the period their dates belong to. Find the
27
- ones recorded in the wrong period.</p></div>
 
 
 
 
 
 
 
 
28
  <div class="task-card"><h3><span class="je-id">JE-04</span> Duplicate payments
29
  <span class="badge-live">Live</span></h3>
30
  <p>A vendor claims they were paid twice. Separate true duplicates from
31
- legitimate split payments and properly re-issued checks.</p></div>
 
 
 
 
 
 
 
 
32
  <div class="task-card"><h3><span class="je-id">JE-05</span> Foreign currency
33
  <span class="badge-live">Live</span></h3>
34
  <p>Foreign-currency entries must translate at the official period rate.
35
- Find wrong rates and translation errors.</p></div>
 
 
 
 
 
 
 
 
 
36
  </div>
37
 
38
  <div class="tier-head tier-l2"><span class="tier-tag">L2</span><h2>Analytical checks</h2>
@@ -44,23 +85,64 @@
44
  <span class="badge-live">Live</span></h3>
45
  <p>Payments above the approval threshold need a purchase order, an
46
  invoice, and a payment that agree. Find the mismatches and explain each
47
- one.</p></div>
 
 
 
 
 
 
 
 
48
  <div class="task-card"><h3><span class="je-id">JE-07</span> Subledger reconciliation
49
  <span class="badge-live">Live</span></h3>
50
  <p>The payables control account doesn't agree with its subledger.
51
- Decompose the difference into the specific items that cause it.</p></div>
 
 
 
 
 
 
 
 
52
  <div class="task-card"><h3><span class="je-id">JE-08</span> Accrual reversals
53
  <span class="badge-live">Live</span></h3>
54
  <p>Period-end accruals should reverse in the next period. Find the ones
55
- that never did, and recompute the ones reversed incorrectly.</p></div>
 
 
 
 
 
 
 
 
56
  <div class="task-card"><h3><span class="je-id">JE-09</span> Interfund transfers
57
  <span class="badge-live">Live</span></h3>
58
  <p>A transfer between funds needs a matching counterparty leg. Find the
59
- unmatched legs before consolidation.</p></div>
 
 
 
 
 
 
 
 
60
  <div class="task-card"><h3><span class="je-id">JE-10</span> Budget integrity
61
  <span class="badge-live">Live</span></h3>
62
  <p>Obligations should liquidate as payments are made, and spending must
63
- stay inside budget authority. Find where either has broken down.</p></div>
 
 
 
 
 
 
 
 
 
64
  </div>
65
 
66
  <div class="tier-head tier-l3"><span class="tier-tag">L3</span><h2>Cross-document evidence</h2>
@@ -121,5 +203,6 @@
121
  </section>
122
 
123
  <footer class="land-footer">
124
- <span>Example runs for each task will appear here as recorded runs are published.</span>
 
125
  </footer>
 
15
  <div class="task-card"><h3><span class="je-id">JE-01</span> Broken entries
16
  <span class="badge-live">Live</span></h3>
17
  <p>An overnight interface load has left a batch of entries damaged. Find
18
+ every entry that no longer balances or is missing required pieces.</p>
19
+ <!-- example:JE-01 -->
20
+ <div class="example-run">
21
+ <span class="ex-model">deepseek-v4-flash</span>
22
+ <span class="ex-score">final 0.04 ± 0.02 <small>(5 seeds)</small></span>
23
+ <span class="ex-base">baselines: flag-everything −1.80 · no-evidence 0.00</span>
24
+ <a class="ex-link" href="#console/run/c5-grid-je01">View the full run →</a>
25
+ </div>
26
+ <!-- /example:JE-01 --></div>
27
  <div class="task-card"><h3><span class="je-id">JE-02</span> Coding validity
28
  <span class="badge-live">Live</span></h3>
29
  <p>Account, fund, and department codes must exist, be active in the
30
  period, and be an allowed combination. Find the entries that violate any
31
+ of it.</p>
32
+ <!-- example:JE-02 -->
33
+ <div class="example-run">
34
+ <span class="ex-model">deepseek-v4-flash</span>
35
+ <span class="ex-score">final 1.00 ± 0.00 <small>(5 seeds)</small></span>
36
+ <span class="ex-base">baselines: flag-everything −1.80 · no-evidence 0.22</span>
37
+ <a class="ex-link" href="#console/run/c5-grid-je02">View the full run →</a>
38
+ </div>
39
+ <!-- /example:JE-02 --></div>
40
  <div class="task-card"><h3><span class="je-id">JE-03</span> Period cut-off
41
  <span class="badge-live">Live</span></h3>
42
  <p>Entries must be posted to the period their dates belong to. Find the
43
+ ones recorded in the wrong period.</p>
44
+ <!-- example:JE-03 -->
45
+ <div class="example-run">
46
+ <span class="ex-model">deepseek-v4-flash</span>
47
+ <span class="ex-score">final 0.21 ± 0.02 <small>(5 seeds)</small></span>
48
+ <span class="ex-base">baselines: flag-everything −1.80 · no-evidence 0.00</span>
49
+ <a class="ex-link" href="#console/run/c5-grid-je03">View the full run →</a>
50
+ </div>
51
+ <!-- /example:JE-03 --></div>
52
  <div class="task-card"><h3><span class="je-id">JE-04</span> Duplicate payments
53
  <span class="badge-live">Live</span></h3>
54
  <p>A vendor claims they were paid twice. Separate true duplicates from
55
+ legitimate split payments and properly re-issued checks.</p>
56
+ <!-- example:JE-04 -->
57
+ <div class="example-run">
58
+ <span class="ex-model">deepseek-v4-flash</span>
59
+ <span class="ex-score">final 0.10 ± 0.30 <small>(5 seeds)</small></span>
60
+ <span class="ex-base">baselines: flag-everything −1.80 · no-evidence 0.00</span>
61
+ <a class="ex-link" href="#console/run/c5-grid-je04">View the full run →</a>
62
+ </div>
63
+ <!-- /example:JE-04 --></div>
64
  <div class="task-card"><h3><span class="je-id">JE-05</span> Foreign currency
65
  <span class="badge-live">Live</span></h3>
66
  <p>Foreign-currency entries must translate at the official period rate.
67
+ Find wrong rates and translation errors.</p>
68
+ <!-- example:JE-05 -->
69
+ <div class="example-run">
70
+ <span class="ex-model">deepseek-v4-flash</span>
71
+ <span class="ex-score">final 0.16 ± 0.02 <small>(5 seeds)</small></span>
72
+ <span class="ex-base">baselines: flag-everything −1.71 · no-evidence 0.18</span>
73
+ <span class="ex-fail">1 of 5 seeds auto-failed (fabrication)</span>
74
+ <a class="ex-link" href="#console/run/c5-grid-je05">View the full run →</a>
75
+ </div>
76
+ <!-- /example:JE-05 --></div>
77
  </div>
78
 
79
  <div class="tier-head tier-l2"><span class="tier-tag">L2</span><h2>Analytical checks</h2>
 
85
  <span class="badge-live">Live</span></h3>
86
  <p>Payments above the approval threshold need a purchase order, an
87
  invoice, and a payment that agree. Find the mismatches and explain each
88
+ one.</p>
89
+ <!-- example:JE-06 -->
90
+ <div class="example-run">
91
+ <span class="ex-model">deepseek-v4-flash</span>
92
+ <span class="ex-score">final 0.83 ± 0.14 <small>(5 seeds)</small></span>
93
+ <span class="ex-base">baselines: flag-everything −1.76 · no-evidence 0.00</span>
94
+ <a class="ex-link" href="#console/run/c5-grid-je06">View the full run →</a>
95
+ </div>
96
+ <!-- /example:JE-06 --></div>
97
  <div class="task-card"><h3><span class="je-id">JE-07</span> Subledger reconciliation
98
  <span class="badge-live">Live</span></h3>
99
  <p>The payables control account doesn't agree with its subledger.
100
+ Decompose the difference into the specific items that cause it.</p>
101
+ <!-- example:JE-07 -->
102
+ <div class="example-run">
103
+ <span class="ex-model">deepseek-v4-flash</span>
104
+ <span class="ex-score">final 0.17 ± 0.08 <small>(5 seeds)</small></span>
105
+ <span class="ex-base">baselines: flag-everything −1.80 · no-evidence 0.00</span>
106
+ <a class="ex-link" href="#console/run/c5-grid-je07">View the full run →</a>
107
+ </div>
108
+ <!-- /example:JE-07 --></div>
109
  <div class="task-card"><h3><span class="je-id">JE-08</span> Accrual reversals
110
  <span class="badge-live">Live</span></h3>
111
  <p>Period-end accruals should reverse in the next period. Find the ones
112
+ that never did, and recompute the ones reversed incorrectly.</p>
113
+ <!-- example:JE-08 -->
114
+ <div class="example-run">
115
+ <span class="ex-model">deepseek-v4-flash</span>
116
+ <span class="ex-score">final 0.68 ± 0.08 <small>(5 seeds)</small></span>
117
+ <span class="ex-base">baselines: flag-everything −1.79 · no-evidence 0.00</span>
118
+ <a class="ex-link" href="#console/run/c5-grid-je08">View the full run →</a>
119
+ </div>
120
+ <!-- /example:JE-08 --></div>
121
  <div class="task-card"><h3><span class="je-id">JE-09</span> Interfund transfers
122
  <span class="badge-live">Live</span></h3>
123
  <p>A transfer between funds needs a matching counterparty leg. Find the
124
+ unmatched legs before consolidation.</p>
125
+ <!-- example:JE-09 -->
126
+ <div class="example-run">
127
+ <span class="ex-model">deepseek-v4-flash</span>
128
+ <span class="ex-score">final 0.59 ± 0.20 <small>(5 seeds)</small></span>
129
+ <span class="ex-base">baselines: flag-everything −1.63 · no-evidence 0.31</span>
130
+ <a class="ex-link" href="#console/run/c5-grid-je09">View the full run →</a>
131
+ </div>
132
+ <!-- /example:JE-09 --></div>
133
  <div class="task-card"><h3><span class="je-id">JE-10</span> Budget integrity
134
  <span class="badge-live">Live</span></h3>
135
  <p>Obligations should liquidate as payments are made, and spending must
136
+ stay inside budget authority. Find where either has broken down.</p>
137
+ <!-- example:JE-10 -->
138
+ <div class="example-run">
139
+ <span class="ex-model">deepseek-v4-flash</span>
140
+ <span class="ex-score">final 0.91 ± 0.07 <small>(5 seeds)</small></span>
141
+ <span class="ex-base">baselines: flag-everything −1.80 · no-evidence 0.25</span>
142
+ <span class="ex-fail">1 of 5 seeds auto-failed (fabrication)</span>
143
+ <a class="ex-link" href="#console/run/c5-grid-je10">View the full run →</a>
144
+ </div>
145
+ <!-- /example:JE-10 --></div>
146
  </div>
147
 
148
  <div class="tier-head tier-l3"><span class="tier-tag">L3</span><h2>Cross-document evidence</h2>
 
203
  </section>
204
 
205
  <footer class="land-footer">
206
+ <span>Recorded 2026-08-11 · open any run in the console's Past runs panel
207
+ the same task and seed always reproduces the same case.</span>
208
  </footer>
app/src/je_validation/console/static/landing.css CHANGED
@@ -189,6 +189,10 @@
189
  .tier-l3 .tier-tag { color: var(--l3); background: var(--l3bg); }
190
  .tier-l4 .tier-tag { color: var(--l4); background: var(--l4bg); }
191
  .task-grid { display: grid; grid-template-columns: 1fr 1fr; gap: 12px; }
 
 
 
 
192
  .task-card h3 { display: flex; gap: 8px; align-items: baseline; }
193
  .task-card .je-id { font-family: var(--mono); font-size: 11px; color: var(--ink-3); }
194
  .badge-live, .badge-roadmap {
@@ -203,6 +207,55 @@
203
  .badge-live { color: var(--l1); background: var(--l1bg); }
204
  .badge-roadmap { color: var(--ink-3); background: var(--panel-3); }
205
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
206
  /* ---- scoring ---- */
207
  .formula {
208
  background: var(--panel);
 
189
  .tier-l3 .tier-tag { color: var(--l3); background: var(--l3bg); }
190
  .tier-l4 .tier-tag { color: var(--l4); background: var(--l4bg); }
191
  .task-grid { display: grid; grid-template-columns: 1fr 1fr; gap: 12px; }
192
+ /* column + auto margin: the example run sits on the card floor, so the score
193
+ * lines of neighbouring cards align however long the description runs. */
194
+ .task-card { display: flex; flex-direction: column; }
195
+ .task-card > p { margin-bottom: auto; }
196
  .task-card h3 { display: flex; gap: 8px; align-items: baseline; }
197
  .task-card .je-id { font-family: var(--mono); font-size: 11px; color: var(--ink-3); }
198
  .badge-live, .badge-roadmap {
 
207
  .badge-live { color: var(--l1); background: var(--l1bg); }
208
  .badge-roadmap { color: var(--ink-3); background: var(--panel-3); }
209
 
210
+ /* ---- example runs on task cards ---- */
211
+ .example-run {
212
+ margin-top: 14px;
213
+ padding-top: 10px;
214
+ border-top: 1px dashed var(--line-2);
215
+ display: grid;
216
+ gap: 3px;
217
+ font-size: 12px;
218
+ }
219
+ .example-run .ex-model { font-family: var(--mono); font-size: 11px; color: var(--ink-3); }
220
+ .example-run .ex-score {
221
+ font-weight: 700;
222
+ font-size: 14px;
223
+ color: var(--ink);
224
+ font-variant-numeric: tabular-nums;
225
+ }
226
+ .example-run .ex-score small { font-weight: 600; font-size: 11.5px; color: var(--ink-3); }
227
+ .example-run .ex-base { font-family: var(--mono); font-size: 11.5px; color: var(--ink-2); }
228
+ /* a flagged outcome, not an alarm: dot + danger ink, quieter than the link */
229
+ .example-run .ex-fail {
230
+ display: flex;
231
+ align-items: center;
232
+ gap: 6px;
233
+ font-size: 11.5px;
234
+ font-weight: 600;
235
+ color: var(--l4);
236
+ }
237
+ .example-run .ex-fail::before {
238
+ content: "";
239
+ flex: none;
240
+ width: 6px;
241
+ height: 6px;
242
+ border-radius: 50%;
243
+ background: currentColor;
244
+ }
245
+ .example-run .ex-link {
246
+ justify-self: start; /* hug the text so the focus ring frames the link */
247
+ margin-top: 4px;
248
+ font-weight: 700;
249
+ color: var(--accent);
250
+ text-decoration: none;
251
+ }
252
+ .example-run .ex-link:hover { text-decoration: underline; }
253
+ /* under ~440px the mono line cannot fit; cap the measure so it breaks at the
254
+ * separator instead of inside "no-evidence". */
255
+ @media (max-width: 440px) {
256
+ .example-run .ex-base { max-width: 36ch; }
257
+ }
258
+
259
  /* ---- scoring ---- */
260
  .formula {
261
  background: var(--panel);
app/src/je_validation/console/static/styles.css CHANGED
@@ -530,3 +530,26 @@ input[type="checkbox"] { accent-color: var(--accent); }
530
  .step-detail-row pre { white-space: pre-wrap; word-break: break-word; }
531
 
532
  .breakdown { margin-bottom: 14px; }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
530
  .step-detail-row pre { white-space: pre-wrap; word-break: break-word; }
531
 
532
  .breakdown { margin-bottom: 14px; }
533
+
534
+ /* ---- deep-link arrival pulse (Tasks tab -> a specific run) ---- */
535
+ /* Outline-only on the big live panel (a full wash would tint the whole step
536
+ * log); the compact history row gets the wash too so it reads at a glance. */
537
+ .deep-flash { animation: deep-flash 1.4s ease-out 1; }
538
+ .history-row.deep-flash { animation-name: deep-flash-fill; }
539
+ @keyframes deep-flash {
540
+ 0% { outline: 2px solid var(--accent); outline-offset: 2px; }
541
+ 60% { outline-color: var(--accent); }
542
+ 100% { outline: 2px solid transparent; outline-offset: 2px; }
543
+ }
544
+ @keyframes deep-flash-fill {
545
+ 0% { outline: 2px solid var(--accent); outline-offset: 2px;
546
+ background-color: var(--accent-soft); }
547
+ 60% { outline-color: var(--accent); }
548
+ 100% { outline: 2px solid transparent; outline-offset: 2px;
549
+ background-color: transparent; }
550
+ }
551
+ @media (prefers-reduced-motion: reduce) {
552
+ /* no motion: a still outline that app.js clears with the class */
553
+ .deep-flash { animation: none; outline: 2px solid var(--accent);
554
+ outline-offset: 2px; }
555
+ }
app/src/je_validation/console/static/tabs.js CHANGED
@@ -1,13 +1,22 @@
1
  /* Hash-routed tabs. app.js owns everything inside #tab-console and never
2
- * reads location.hash (verified), so routing is isolated here. */
 
 
 
3
  (() => {
4
  const TABS = ["home", "datasets", "tasks", "scoring", "console"];
5
 
6
  const current = () => {
7
- const h = location.hash.replace("#", "");
8
  return TABS.includes(h) ? h : "home";
9
  };
10
 
 
 
 
 
 
 
11
  async function show(tab) {
12
  document.querySelectorAll(".tabbar a").forEach((a) =>
13
  a.classList.toggle("active", a.dataset.tab === tab));
@@ -27,6 +36,10 @@
27
  }
28
  }
29
  window.scrollTo(0, 0);
 
 
 
 
30
  }
31
 
32
  window.addEventListener("hashchange", () => show(current()));
 
1
  /* Hash-routed tabs. app.js owns everything inside #tab-console and never
2
+ * reads location.hash (verified), so routing is isolated here. A deep link
3
+ * of the form #console/run/<run_id> opens the console tab and hands the run
4
+ * id to app.js via a "je:open-run" event (app.js loads first, so its
5
+ * listener exists before the initial dispatch). */
6
  (() => {
7
  const TABS = ["home", "datasets", "tasks", "scoring", "console"];
8
 
9
  const current = () => {
10
+ const h = location.hash.replace("#", "").split("/")[0];
11
  return TABS.includes(h) ? h : "home";
12
  };
13
 
14
+ const runTarget = () => {
15
+ const parts = location.hash.replace("#", "").split("/");
16
+ return parts[0] === "console" && parts[1] === "run" && parts[2]
17
+ ? parts[2] : null;
18
+ };
19
+
20
  async function show(tab) {
21
  document.querySelectorAll(".tabbar a").forEach((a) =>
22
  a.classList.toggle("active", a.dataset.tab === tab));
 
36
  }
37
  }
38
  window.scrollTo(0, 0);
39
+ const run = runTarget();
40
+ if (tab === "console" && run) {
41
+ window.dispatchEvent(new CustomEvent("je:open-run", { detail: run }));
42
+ }
43
  }
44
 
45
  window.addEventListener("hashchange", () => show(current()));
app/src/je_validation/envir/briefs.py CHANGED
@@ -59,6 +59,28 @@ TASK_RULES: dict[str, str] = {
59
  "have actually seen in tool results) earns no outcome credit."),
60
  }
61
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
62
  PROMPT_VERSIONS: tuple[str, ...] = ("minimal", "standard", "detailed")
63
 
64
  # standard-level additions. The stated rule must match the SCORED rule
@@ -103,6 +125,8 @@ def _minimal(task: dict) -> str:
103
  parts = [task["brief"]]
104
  if (rule := TASK_RULES.get(task["id"])) is not None:
105
  parts.append(rule)
 
 
106
  extra = CLAIM_INSTRUCTIONS.get(task["id"])
107
  if extra is not None:
108
  parts.append(f"{CLAIM_FORMAT.lstrip()}\n\n{extra}")
 
59
  "have actually seen in tool results) earns no outcome credit."),
60
  }
61
 
62
+ # Per-task reporting requirements (c5): how findings must be REPRESENTED in
63
+ # the submission — flag targets and evidence citations. A versioned output
64
+ # contract, kept separate from TASK_RULES on purpose (sol 2026-08-11): a
65
+ # seed-invariant procedural rule is still not world ontology, and admitting
66
+ # it there would widen the world-rules exception. Same leak test applies:
67
+ # each sentence must be true and identical for every seed and every defect
68
+ # placement — id formats and fixed row conventions only, never per-instance
69
+ # facts. Scored by the JE-10 outcome verifier (flag target, via seeded
70
+ # defect ids) and the JE-10 process gate (citation factor), both c5.
71
+ REPORTING_RULES: dict[str, str] = {
72
+ "JE-10": (
73
+ "Reporting requirements: every budgeted account has one BUDGET "
74
+ "STATUS SUMMARY entry, id 'JE10-BUDGET-<account>'. Raise a "
75
+ "budget_exceeded exception by flagging that summary entry - not "
76
+ "the individual payments posted to the account. If you submit "
77
+ "issue type wrong_document, cite in evidence_ids the obligation "
78
+ "register's obligating_doc_id field value (the OD-... id you "
79
+ "compared with the flagged entry); the master:obligation:... "
80
+ "record key may also be cited as provenance, but it does not "
81
+ "replace that field-value citation."),
82
+ }
83
+
84
  PROMPT_VERSIONS: tuple[str, ...] = ("minimal", "standard", "detailed")
85
 
86
  # standard-level additions. The stated rule must match the SCORED rule
 
125
  parts = [task["brief"]]
126
  if (rule := TASK_RULES.get(task["id"])) is not None:
127
  parts.append(rule)
128
+ if (rep := REPORTING_RULES.get(task["id"])) is not None:
129
+ parts.append(rep)
130
  extra = CLAIM_INSTRUCTIONS.get(task["id"])
131
  if extra is not None:
132
  parts.append(f"{CLAIM_FORMAT.lstrip()}\n\n{extra}")
app/src/je_validation/scoring/gates.py CHANGED
@@ -32,6 +32,12 @@ def _is_flag(s: StepRecord) -> bool:
32
  return s.tool == "disposition" and s.args.get("verdict") in FLAGGING
33
 
34
 
 
 
 
 
 
 
35
  def _token_in(token: str, expr: str) -> bool:
36
  """Whole-numeric-token match: plain substring would let entry amount 100
37
  "match" expression "1000-900"."""
@@ -309,8 +315,25 @@ def je09_next_period(steps: list[StepRecord], dispositions: dict,
309
 
310
  def je10_account_aggregate(steps: list[StepRecord], dispositions: dict,
311
  instance: Instance) -> GateResult:
 
 
 
 
 
 
 
 
 
 
 
 
312
  account_of = {e.entry_id: str(e.header.get("account", ""))
313
  for e in instance.entries}
 
 
 
 
 
314
  grouped = False
315
  filtered_accounts: set[str] = set()
316
  b_flagged = b_supported = 0
@@ -328,8 +351,20 @@ def je10_account_aggregate(steps: list[StepRecord], dispositions: dict,
328
  b_supported += (grouped or account_of.get(
329
  s.args.get("entry_id")) in filtered_accounts)
330
  b_rate = b_supported / b_flagged if b_flagged else 1.0
331
- return GateResult(b_rate, {"budget_flags": b_flagged,
332
- "with_account_aggregate": b_supported})
 
 
 
 
 
 
 
 
 
 
 
 
333
 
334
 
335
  def je05_rate_recompute(steps: list[StepRecord], dispositions: dict,
 
32
  return s.tool == "disposition" and s.args.get("verdict") in FLAGGING
33
 
34
 
35
+ def _dget(d, key):
36
+ """Disposition field access: production passes state.Disposition objects,
37
+ tests pass plain dicts (same duality as verifiers._get)."""
38
+ return d[key] if isinstance(d, dict) else getattr(d, key)
39
+
40
+
41
  def _token_in(token: str, expr: str) -> bool:
42
  """Whole-numeric-token match: plain substring would let entry amount 100
43
  "match" expression "1000-900"."""
 
315
 
316
  def je10_account_aggregate(steps: list[StepRecord], dispositions: dict,
317
  instance: Instance) -> GateResult:
318
+ """Two per-flag factors, multiplied (standing gate algebra). Budget
319
+ factor (c4), order-sensitive per spec: a budget_exceeded flag needs an
320
+ account-level aggregate EARLIER in the episode. Citation factor (c5),
321
+ order-free over episode-final dispositions: a wrong_document flag is
322
+ supported only if its disposition cites the obligation register's
323
+ obligating_doc_id field value for the flagged entry - the reporting
324
+ requirement REPORTING_RULES['JE-10'] states. Agent-visible data only:
325
+ the entry header names its obligation_id and the register row carries
326
+ the correct doc id; a wrong_document flag on an entry with no
327
+ obligation has no correct doc to cite and is unsupported. Entries are
328
+ identified by the dispositions dict key - the authoritative id
329
+ everywhere in scoring."""
330
  account_of = {e.entry_id: str(e.header.get("account", ""))
331
  for e in instance.entries}
332
+ register = instance.master_data.get("obligation", {})
333
+ correct_doc_of = {
334
+ e.entry_id: (register.get(e.header.get("obligation_id")) or {}
335
+ ).get("obligating_doc_id")
336
+ for e in instance.entries}
337
  grouped = False
338
  filtered_accounts: set[str] = set()
339
  b_flagged = b_supported = 0
 
351
  b_supported += (grouped or account_of.get(
352
  s.args.get("entry_id")) in filtered_accounts)
353
  b_rate = b_supported / b_flagged if b_flagged else 1.0
354
+ wd_flagged = wd_supported = 0
355
+ for eid, d in dispositions.items():
356
+ if (_dget(d, "verdict") in FLAGGING
357
+ and _dget(d, "issue_type") == "wrong_document"):
358
+ wd_flagged += 1
359
+ correct = correct_doc_of.get(eid)
360
+ wd_supported += (correct is not None
361
+ and correct in set(_dget(d, "evidence_ids") or ()))
362
+ wd_rate = wd_supported / wd_flagged if wd_flagged else 1.0
363
+ return GateResult(b_rate * wd_rate,
364
+ {"budget_flags": b_flagged,
365
+ "with_account_aggregate": b_supported,
366
+ "wrong_document_flags": wd_flagged,
367
+ "with_doc_citation": wd_supported})
368
 
369
 
370
  def je05_rate_recompute(steps: list[StepRecord], dispositions: dict,
app/src/je_validation/scoring/verifiers/je10.py CHANGED
@@ -1,11 +1,15 @@
1
- """JE-10: encumbrance liquidation. F1 x issue accuracy with exact-overage and
2
- obligating-document-citation conditions folded into issue correctness."""
 
 
 
 
3
  from __future__ import annotations
4
 
5
  from collections.abc import Iterable
6
 
7
  from je_validation.scoring.claims import claims_of
8
- from je_validation.scoring.verifiers import OutcomeVerdict, _get, f1_stats, flagged_ids, issue_of
9
 
10
 
11
  def verify(dispositions: dict, entry_ids: Iterable[str], heldout) -> OutcomeVerdict:
@@ -24,10 +28,6 @@ def verify(dispositions: dict, entry_ids: Iterable[str], heldout) -> OutcomeVerd
24
  # spec sign: budget - (actual + encumbrance), negative when over
25
  claim = claims_of(dispositions, eid).get("available_cents")
26
  ok = claim == heldout.claims[eid]["available_cents"]
27
- if ok and dtype == "wrong_document":
28
- cited = set(_get(dispositions[eid], "evidence_ids"))
29
- ok = any(x in cited for x in heldout.evidence_map[eid]
30
- if not x.startswith("master:"))
31
  issue_correct += ok
32
  fn = len(set(seeded) & ids) - tp
33
  precision, recall, f1 = f1_stats(tp, fp, fn)
 
1
+ """JE-10: encumbrance liquidation. F1 x issue accuracy with the exact-overage
2
+ condition folded into issue correctness (spec: stated overage must equal
3
+ budget - (actual + encumbrance), exact). The obligating-document citation is
4
+ a REPORTING requirement priced by the process gate (c5,
5
+ gates.je10_account_aggregate) - the spec's outcome line is "exception set and
6
+ type matched against seed", so citation selection never alters outcome."""
7
  from __future__ import annotations
8
 
9
  from collections.abc import Iterable
10
 
11
  from je_validation.scoring.claims import claims_of
12
+ from je_validation.scoring.verifiers import OutcomeVerdict, f1_stats, flagged_ids, issue_of
13
 
14
 
15
  def verify(dispositions: dict, entry_ids: Iterable[str], heldout) -> OutcomeVerdict:
 
28
  # spec sign: budget - (actual + encumbrance), negative when over
29
  claim = claims_of(dispositions, eid).get("available_cents")
30
  ok = claim == heldout.claims[eid]["available_cents"]
 
 
 
 
31
  issue_correct += ok
32
  fn = len(set(seeded) & ids) - tp
33
  precision, recall, f1 = f1_stats(tp, fp, fn)
data/runs/console/c5-grid-je01/baseline_flag_everything/seed1/JE-01_seed1.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_flag_everything/seed1/score.json RENAMED
File without changes
data/runs/console/c5-grid-je01/baseline_flag_everything/seed2/JE-01_seed2.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_flag_everything/seed2/score.json RENAMED
File without changes
data/runs/console/c5-grid-je01/baseline_flag_everything/seed3/JE-01_seed3.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-932f8101/baseline_flag_everything/seed1 → c5-grid-je01/baseline_flag_everything/seed3}/score.json RENAMED
File without changes
data/runs/console/c5-grid-je01/baseline_flag_everything/seed4/JE-01_seed4.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-932f8101/baseline_flag_everything/seed2 → c5-grid-je01/baseline_flag_everything/seed4}/score.json RENAMED
File without changes
data/runs/console/c5-grid-je01/baseline_flag_everything/seed5/JE-01_seed5.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-a5453c3d/baseline_flag_everything/seed1 → c5-grid-je01/baseline_flag_everything/seed5}/score.json RENAMED
File without changes
data/runs/console/c5-grid-je01/baseline_no_evidence/seed1/JE-01_seed1.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_no_evidence/seed1/score.json RENAMED
File without changes
data/runs/console/c5-grid-je01/baseline_no_evidence/seed2/JE-01_seed2.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-1b6438a7 → c5-grid-je01}/baseline_no_evidence/seed2/score.json RENAMED
File without changes
data/runs/console/c5-grid-je01/baseline_no_evidence/seed3/JE-01_seed3.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-932f8101/baseline_no_evidence/seed1 → c5-grid-je01/baseline_no_evidence/seed3}/score.json RENAMED
File without changes
data/runs/console/c5-grid-je01/baseline_no_evidence/seed4/JE-01_seed4.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-932f8101/baseline_no_evidence/seed2 → c5-grid-je01/baseline_no_evidence/seed4}/score.json RENAMED
File without changes
data/runs/console/c5-grid-je01/baseline_no_evidence/seed5/JE-01_seed5.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-a5453c3d/baseline_no_evidence/seed1 → c5-grid-je01/baseline_no_evidence/seed5}/score.json RENAMED
File without changes
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed1/JE-01_seed1.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-a5453c3d/baseline_no_evidence/seed3 → c5-grid-je01/deepseek_deepseek-v4-flash/seed1}/score.json RENAMED
@@ -1,16 +1,16 @@
1
  {
2
  "outcome": 0.0,
3
- "process": 0.0,
4
  "rationale": null,
5
- "combined": 0.0,
6
  "fp_rate": 0.0,
7
  "fp_penalty": 0.0,
8
- "final": 0.0,
9
  "fabrications": [],
10
  "auto_fail": false,
11
  "hard_fail": false,
12
  "judge_pass": null,
13
  "passed": null,
14
  "judge": "absent",
15
- "cost_usd": 0.0
16
  }
 
1
  {
2
  "outcome": 0.0,
3
+ "process": 0.5,
4
  "rationale": null,
5
+ "combined": 0.05555555555555556,
6
  "fp_rate": 0.0,
7
  "fp_penalty": 0.0,
8
+ "final": 0.05555555555555556,
9
  "fabrications": [],
10
  "auto_fail": false,
11
  "hard_fail": false,
12
  "judge_pass": null,
13
  "passed": null,
14
  "judge": "absent",
15
+ "cost_usd": 0.22998800869240005
16
  }
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed2/JE-01_seed2.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-a5453c3d/baseline_no_evidence/seed5 → c5-grid-je01/deepseek_deepseek-v4-flash/seed2}/score.json RENAMED
@@ -1,16 +1,16 @@
1
  {
2
  "outcome": 0.0,
3
- "process": 0.0,
4
  "rationale": null,
5
- "combined": 0.0,
6
  "fp_rate": 0.0,
7
  "fp_penalty": 0.0,
8
- "final": 0.0,
9
  "fabrications": [],
10
  "auto_fail": false,
11
  "hard_fail": false,
12
  "judge_pass": null,
13
  "passed": null,
14
  "judge": "absent",
15
- "cost_usd": 0.0
16
  }
 
1
  {
2
  "outcome": 0.0,
3
+ "process": 0.495,
4
  "rationale": null,
5
+ "combined": 0.055,
6
  "fp_rate": 0.0,
7
  "fp_penalty": 0.0,
8
+ "final": 0.055,
9
  "fabrications": [],
10
  "auto_fail": false,
11
  "hard_fail": false,
12
  "judge_pass": null,
13
  "passed": null,
14
  "judge": "absent",
15
+ "cost_usd": 0.45799382661999993
16
  }
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed3/JE-01_seed3.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed3/score.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "outcome": 0.0,
3
+ "process": 0.3333333333333333,
4
+ "rationale": null,
5
+ "combined": 0.037037037037037035,
6
+ "fp_rate": 0.0,
7
+ "fp_penalty": 0.0,
8
+ "final": 0.037037037037037035,
9
+ "fabrications": [],
10
+ "auto_fail": false,
11
+ "hard_fail": false,
12
+ "judge_pass": null,
13
+ "passed": null,
14
+ "judge": "absent",
15
+ "cost_usd": 0.39998567448000016
16
+ }
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed4/JE-01_seed4.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed4/score.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "outcome": 0.0,
3
+ "process": 0.16666666666666666,
4
+ "rationale": null,
5
+ "combined": 0.018518518518518517,
6
+ "fp_rate": 0.0,
7
+ "fp_penalty": 0.0,
8
+ "final": 0.018518518518518517,
9
+ "fabrications": [],
10
+ "auto_fail": false,
11
+ "hard_fail": false,
12
+ "judge_pass": null,
13
+ "passed": null,
14
+ "judge": "absent",
15
+ "cost_usd": 0.36706333776000016
16
+ }
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed5/JE-01_seed5.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/c5-grid-je01/deepseek_deepseek-v4-flash/seed5/score.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "outcome": 0.0,
3
+ "process": 0.14285714285714285,
4
+ "rationale": null,
5
+ "combined": 0.015873015873015872,
6
+ "fp_rate": 0.0,
7
+ "fp_penalty": 0.0,
8
+ "final": 0.015873015873015872,
9
+ "fabrications": [],
10
+ "auto_fail": false,
11
+ "hard_fail": false,
12
+ "judge_pass": null,
13
+ "passed": null,
14
+ "judge": "absent",
15
+ "cost_usd": 0.38351098015999985
16
+ }
data/runs/console/{run-a5453c3d → c5-grid-je01}/results.json RENAMED
@@ -8,130 +8,29 @@
8
  "std": 0.0
9
  },
10
  "process": {
11
- "mean": 0.6870666666666667,
12
- "std": 0.16309751684191887
13
  },
14
  "rationale": "pending judge calibration",
15
  "combined": {
16
- "mean": 0.12124705882352942,
17
- "std": 0.028781914736809213
18
  },
19
  "final": {
20
- "mean": 0.12124705882352942,
21
- "std": 0.028781914736809213
22
  },
23
  "fail_rate": 0.0
24
  },
25
  "reports": [
26
- {
27
- "outcome": 0.0,
28
- "process": 0.6666666666666666,
29
- "rationale": "pending judge calibration",
30
- "combined": 0.11764705882352941,
31
- "fp_rate": 0.02,
32
- "fp_penalty": 0.0,
33
- "final": 0.11764705882352941,
34
- "fabrications": [],
35
- "auto_fail": false,
36
- "hard_fail": false,
37
- "judge_pass": null,
38
- "passed": null,
39
- "judge": "absent"
40
- },
41
  {
42
  "outcome": 0.0,
43
  "process": 0.5,
44
  "rationale": "pending judge calibration",
45
- "combined": 0.08823529411764706,
46
- "fp_rate": 0.02,
47
- "fp_penalty": 0.0,
48
- "final": 0.08823529411764706,
49
- "fabrications": [],
50
- "auto_fail": false,
51
- "hard_fail": false,
52
- "judge_pass": null,
53
- "passed": null,
54
- "judge": "absent"
55
- },
56
- {
57
- "outcome": 0.0,
58
- "process": 0.9500000000000001,
59
- "rationale": "pending judge calibration",
60
- "combined": 0.16764705882352943,
61
- "fp_rate": 0.02,
62
- "fp_penalty": 0.0,
63
- "final": 0.16764705882352943,
64
- "fabrications": [],
65
- "auto_fail": false,
66
- "hard_fail": false,
67
- "judge_pass": null,
68
- "passed": null,
69
- "judge": "absent"
70
- },
71
- {
72
- "outcome": 0.0,
73
- "process": 0.6466666666666667,
74
- "rationale": "pending judge calibration",
75
- "combined": 0.11411764705882353,
76
- "fp_rate": 0.02,
77
- "fp_penalty": 0.0,
78
- "final": 0.11411764705882353,
79
- "fabrications": [],
80
- "auto_fail": false,
81
- "hard_fail": false,
82
- "judge_pass": null,
83
- "passed": null,
84
- "judge": "absent"
85
- },
86
- {
87
- "outcome": 0.0,
88
- "process": 0.672,
89
- "rationale": "pending judge calibration",
90
- "combined": 0.11858823529411765,
91
- "fp_rate": 0.06,
92
- "fp_penalty": 0.0,
93
- "final": 0.11858823529411765,
94
- "fabrications": [],
95
- "auto_fail": false,
96
- "hard_fail": false,
97
- "judge_pass": null,
98
- "passed": null,
99
- "judge": "absent"
100
- }
101
- ]
102
- },
103
- "openai/gpt-5.6-sol": {
104
- "aggregate": {
105
- "n": 5,
106
- "variance_measurable": true,
107
- "outcome": {
108
- "mean": 0.0,
109
- "std": 0.0
110
- },
111
- "process": {
112
- "mean": 0.8142857142857143,
113
- "std": 0.18454300975376878
114
- },
115
- "rationale": "pending judge calibration",
116
- "combined": {
117
- "mean": 0.14369747899159663,
118
- "std": 0.0325664134859592
119
- },
120
- "final": {
121
- "mean": 0.14369747899159663,
122
- "std": 0.0325664134859592
123
- },
124
- "fail_rate": 0.0
125
- },
126
- "reports": [
127
- {
128
- "outcome": 0.0,
129
- "process": 0.75,
130
- "rationale": "pending judge calibration",
131
- "combined": 0.1323529411764706,
132
- "fp_rate": 0.08,
133
  "fp_penalty": 0.0,
134
- "final": 0.1323529411764706,
135
  "fabrications": [],
136
  "auto_fail": false,
137
  "hard_fail": false,
@@ -141,12 +40,12 @@
141
  },
142
  {
143
  "outcome": 0.0,
144
- "process": 0.5714285714285714,
145
  "rationale": "pending judge calibration",
146
- "combined": 0.10084033613445377,
147
- "fp_rate": 0.04,
148
  "fp_penalty": 0.0,
149
- "final": 0.10084033613445377,
150
  "fabrications": [],
151
  "auto_fail": false,
152
  "hard_fail": false,
@@ -156,12 +55,12 @@
156
  },
157
  {
158
  "outcome": 0.0,
159
- "process": 1.0,
160
  "rationale": "pending judge calibration",
161
- "combined": 0.17647058823529413,
162
- "fp_rate": 0.02,
163
  "fp_penalty": 0.0,
164
- "final": 0.17647058823529413,
165
  "fabrications": [],
166
  "auto_fail": false,
167
  "hard_fail": false,
@@ -171,12 +70,12 @@
171
  },
172
  {
173
  "outcome": 0.0,
174
- "process": 0.75,
175
  "rationale": "pending judge calibration",
176
- "combined": 0.1323529411764706,
177
- "fp_rate": 0.08,
178
  "fp_penalty": 0.0,
179
- "final": 0.1323529411764706,
180
  "fabrications": [],
181
  "auto_fail": false,
182
  "hard_fail": false,
@@ -186,12 +85,12 @@
186
  },
187
  {
188
  "outcome": 0.0,
189
- "process": 1.0,
190
  "rationale": "pending judge calibration",
191
- "combined": 0.17647058823529413,
192
  "fp_rate": 0.0,
193
  "fp_penalty": 0.0,
194
- "final": 0.17647058823529413,
195
  "fabrications": [],
196
  "auto_fail": false,
197
  "hard_fail": false,
 
8
  "std": 0.0
9
  },
10
  "process": {
11
+ "mean": 0.32757142857142857,
12
+ "std": 0.17161501105130808
13
  },
14
  "rationale": "pending judge calibration",
15
  "combined": {
16
+ "mean": 0.03639682539682539,
17
+ "std": 0.019068334561256454
18
  },
19
  "final": {
20
+ "mean": 0.03639682539682539,
21
+ "std": 0.019068334561256454
22
  },
23
  "fail_rate": 0.0
24
  },
25
  "reports": [
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
26
  {
27
  "outcome": 0.0,
28
  "process": 0.5,
29
  "rationale": "pending judge calibration",
30
+ "combined": 0.05555555555555556,
31
+ "fp_rate": 0.0,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
32
  "fp_penalty": 0.0,
33
+ "final": 0.05555555555555556,
34
  "fabrications": [],
35
  "auto_fail": false,
36
  "hard_fail": false,
 
40
  },
41
  {
42
  "outcome": 0.0,
43
+ "process": 0.495,
44
  "rationale": "pending judge calibration",
45
+ "combined": 0.055,
46
+ "fp_rate": 0.0,
47
  "fp_penalty": 0.0,
48
+ "final": 0.055,
49
  "fabrications": [],
50
  "auto_fail": false,
51
  "hard_fail": false,
 
55
  },
56
  {
57
  "outcome": 0.0,
58
+ "process": 0.3333333333333333,
59
  "rationale": "pending judge calibration",
60
+ "combined": 0.037037037037037035,
61
+ "fp_rate": 0.0,
62
  "fp_penalty": 0.0,
63
+ "final": 0.037037037037037035,
64
  "fabrications": [],
65
  "auto_fail": false,
66
  "hard_fail": false,
 
70
  },
71
  {
72
  "outcome": 0.0,
73
+ "process": 0.16666666666666666,
74
  "rationale": "pending judge calibration",
75
+ "combined": 0.018518518518518517,
76
+ "fp_rate": 0.0,
77
  "fp_penalty": 0.0,
78
+ "final": 0.018518518518518517,
79
  "fabrications": [],
80
  "auto_fail": false,
81
  "hard_fail": false,
 
85
  },
86
  {
87
  "outcome": 0.0,
88
+ "process": 0.14285714285714285,
89
  "rationale": "pending judge calibration",
90
+ "combined": 0.015873015873015872,
91
  "fp_rate": 0.0,
92
  "fp_penalty": 0.0,
93
+ "final": 0.015873015873015872,
94
  "fabrications": [],
95
  "auto_fail": false,
96
  "hard_fail": false,
data/runs/console/{run-a5453c3d → c5-grid-je01}/run.json RENAMED
@@ -1,18 +1,17 @@
1
  {
2
- "run_id": "run-a5453c3d",
3
  "status": "finished",
4
  "request": {
5
- "run_id": "run-a5453c3d",
6
- "task_id": "JE-05",
7
  "models": [
8
  "deepseek/deepseek-v4-flash",
9
- "openai/gpt-5.6-sol",
10
  "baseline:flag_everything",
11
  "baseline:no_evidence"
12
  ],
13
  "seed_count": 5,
14
  "knobs": {
15
- "population": 60,
16
  "defect_density": 28,
17
  "distractor_ratio": 2,
18
  "doc_noise": 10
@@ -20,56 +19,49 @@
20
  "tools_enabled": [
21
  "query_ledger",
22
  "get_entry",
23
- "get_master_data",
24
  "recompute",
25
  "disposition",
26
  "submit"
27
  ],
28
- "step_budget": 270,
29
- "token_ceiling": 229500000,
30
  "pricing": {
31
  "deepseek/deepseek-v4-flash": {
32
- "prompt_price": 8.82e-08,
33
- "completion_price": 1.764e-07
34
- },
35
- "openai/gpt-5.6-sol": {
36
- "prompt_price": 5e-06,
37
- "completion_price": 3e-05
38
  }
39
  },
40
  "prompt_version": "standard"
41
  },
42
  "contract": {
43
- "task_id": "JE-05",
44
  "tier": "L1",
45
  "seed": 1,
46
  "generator": {
47
  "knobs": {
48
- "population": 60,
49
  "defect_density": 28,
50
  "distractor_ratio": 2,
51
  "doc_noise": 10
52
  },
53
  "sources": [
54
- "DS-A",
55
- "DS-J"
56
  ]
57
  },
58
  "tools_enabled": [
59
  "query_ledger",
60
  "get_entry",
61
- "get_master_data",
62
  "recompute",
63
  "disposition",
64
  "submit"
65
  ],
66
  "tools_withheld": [],
67
- "step_budget": 270,
68
  "scoring": {
69
  "weights": {
70
- "outcome": 0.7,
71
- "process": 0.15,
72
- "rationale": 0.15
73
  },
74
  "judge": "global_rubric_v1",
75
  "judges_per_episode": 1
@@ -77,12 +69,11 @@
77
  "sweep": {
78
  "models": [
79
  "deepseek/deepseek-v4-flash",
80
- "openai/gpt-5.6-sol",
81
  "baseline:flag_everything",
82
  "baseline:no_evidence"
83
  ],
84
  "seeds": 5,
85
- "episodes": 20
86
  },
87
  "prompt_version": "standard"
88
  },
@@ -92,11 +83,6 @@
92
  "deepseek_deepseek-v4-flash/seed3": "finished",
93
  "deepseek_deepseek-v4-flash/seed4": "finished",
94
  "deepseek_deepseek-v4-flash/seed5": "finished",
95
- "openai_gpt-5.6-sol/seed1": "finished",
96
- "openai_gpt-5.6-sol/seed2": "finished",
97
- "openai_gpt-5.6-sol/seed3": "finished",
98
- "openai_gpt-5.6-sol/seed4": "finished",
99
- "openai_gpt-5.6-sol/seed5": "finished",
100
  "baseline_flag_everything/seed1": "finished",
101
  "baseline_flag_everything/seed2": "finished",
102
  "baseline_flag_everything/seed3": "finished",
@@ -108,5 +94,5 @@
108
  "baseline_no_evidence/seed4": "finished",
109
  "baseline_no_evidence/seed5": "finished"
110
  },
111
- "cost_usd": 12.381930227651985
112
  }
 
1
  {
2
+ "run_id": "c5-grid-je01",
3
  "status": "finished",
4
  "request": {
5
+ "run_id": "c5-grid-je01",
6
+ "task_id": "JE-01",
7
  "models": [
8
  "deepseek/deepseek-v4-flash",
 
9
  "baseline:flag_everything",
10
  "baseline:no_evidence"
11
  ],
12
  "seed_count": 5,
13
  "knobs": {
14
+ "population": 400,
15
  "defect_density": 28,
16
  "distractor_ratio": 2,
17
  "doc_noise": 10
 
19
  "tools_enabled": [
20
  "query_ledger",
21
  "get_entry",
 
22
  "recompute",
23
  "disposition",
24
  "submit"
25
  ],
26
+ "step_budget": 1800,
27
+ "token_ceiling": 1530000000,
28
  "pricing": {
29
  "deepseek/deepseek-v4-flash": {
30
+ "prompt_price": 1.4e-07,
31
+ "completion_price": 2.8e-07
 
 
 
 
32
  }
33
  },
34
  "prompt_version": "standard"
35
  },
36
  "contract": {
37
+ "task_id": "JE-01",
38
  "tier": "L1",
39
  "seed": 1,
40
  "generator": {
41
  "knobs": {
42
+ "population": 400,
43
  "defect_density": 28,
44
  "distractor_ratio": 2,
45
  "doc_noise": 10
46
  },
47
  "sources": [
48
+ "DS-A"
 
49
  ]
50
  },
51
  "tools_enabled": [
52
  "query_ledger",
53
  "get_entry",
 
54
  "recompute",
55
  "disposition",
56
  "submit"
57
  ],
58
  "tools_withheld": [],
59
+ "step_budget": 1800,
60
  "scoring": {
61
  "weights": {
62
+ "outcome": 0.8,
63
+ "process": 0.1,
64
+ "rationale": 0.1
65
  },
66
  "judge": "global_rubric_v1",
67
  "judges_per_episode": 1
 
69
  "sweep": {
70
  "models": [
71
  "deepseek/deepseek-v4-flash",
 
72
  "baseline:flag_everything",
73
  "baseline:no_evidence"
74
  ],
75
  "seeds": 5,
76
+ "episodes": 15
77
  },
78
  "prompt_version": "standard"
79
  },
 
83
  "deepseek_deepseek-v4-flash/seed3": "finished",
84
  "deepseek_deepseek-v4-flash/seed4": "finished",
85
  "deepseek_deepseek-v4-flash/seed5": "finished",
 
 
 
 
 
86
  "baseline_flag_everything/seed1": "finished",
87
  "baseline_flag_everything/seed2": "finished",
88
  "baseline_flag_everything/seed3": "finished",
 
94
  "baseline_no_evidence/seed4": "finished",
95
  "baseline_no_evidence/seed5": "finished"
96
  },
97
+ "cost_usd": 1.838541827712399
98
  }
data/runs/console/c5-grid-je01/seed1/bundle/heldout/defect_ledger.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"claims":{},"defects":[{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0000219389","revealing_evidence_id":"2024-0000219389:lines"},{"correct_verdict":"flag","defect_type":"dup_line_seq","entry_id":"2024-0018941205","revealing_evidence_id":"2024-0018941205:lines"},{"correct_verdict":"flag","defect_type":"orphan_leg","entry_id":"2024-0018944498","revealing_evidence_id":"2024-0018944498:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018945399","revealing_evidence_id":"2024-0018945399:lines"},{"correct_verdict":"flag","defect_type":"orphan_leg","entry_id":"2024-0018950877","revealing_evidence_id":"2024-0018950877:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018952717","revealing_evidence_id":"2024-0018952717:lines"},{"correct_verdict":"flag","defect_type":"orphan_leg","entry_id":"2024-0018953134","revealing_evidence_id":"2024-0018953134:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018955782","revealing_evidence_id":"2024-0018955782:lines"},{"correct_verdict":"flag","defect_type":"zero_amount_line","entry_id":"2024-0018958794","revealing_evidence_id":"2024-0018958794:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018966176","revealing_evidence_id":"2024-0018966176:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018967117","revealing_evidence_id":"2024-0018967117:lines"},{"correct_verdict":"flag","defect_type":"zero_amount_line","entry_id":"2024-0018972058","revealing_evidence_id":"2024-0018972058:lines"}],"evidence_map":{"2024-0000219389":["2024-0000219389:lines"],"2024-0018941205":["2024-0018941205:lines"],"2024-0018944498":["2024-0018944498:lines"],"2024-0018945399":["2024-0018945399:lines"],"2024-0018950877":["2024-0018950877:lines"],"2024-0018952717":["2024-0018952717:lines"],"2024-0018953134":["2024-0018953134:lines"],"2024-0018955782":["2024-0018955782:lines"],"2024-0018958794":["2024-0018958794:lines"],"2024-0018966176":["2024-0018966176:lines"],"2024-0018967117":["2024-0018967117:lines"],"2024-0018972058":["2024-0018972058:lines"]},"required_docs":[]}
data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed1/bundle/instance/documents.jsonl RENAMED
File without changes
data/runs/console/c5-grid-je01/seed1/bundle/instance/entries.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed1/bundle/instance/master_data.json RENAMED
File without changes
data/runs/console/c5-grid-je01/seed1/bundle/instance/meta.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed":1,"task_id":"JE-01"}
data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed1/bundle/instance/policies.json RENAMED
File without changes
data/runs/console/c5-grid-je01/seed1/bundle/manifest.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"generator_version":"4c.1","knobs":{"n_entries":400},"seed":1,"snapshot_id":"2026-08-04","spec_sha256":"90d55c077a0acea9a57b95ecd63a70df0316fc271d873a38afa195a766f23978","task_id":"JE-01"}
data/runs/console/c5-grid-je01/seed2/bundle/heldout/defect_ledger.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"claims":{},"defects":[{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018942198","revealing_evidence_id":"2024-0018942198:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018944317","revealing_evidence_id":"2024-0018944317:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018946773","revealing_evidence_id":"2024-0018946773:lines"},{"correct_verdict":"flag","defect_type":"orphan_leg","entry_id":"2024-0018950131","revealing_evidence_id":"2024-0018950131:lines"},{"correct_verdict":"flag","defect_type":"orphan_leg","entry_id":"2024-0018951225","revealing_evidence_id":"2024-0018951225:lines"},{"correct_verdict":"flag","defect_type":"zero_amount_line","entry_id":"2024-0018952858","revealing_evidence_id":"2024-0018952858:lines"},{"correct_verdict":"flag","defect_type":"orphan_leg","entry_id":"2024-0018960305","revealing_evidence_id":"2024-0018960305:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018965646","revealing_evidence_id":"2024-0018965646:lines"},{"correct_verdict":"flag","defect_type":"dup_line_seq","entry_id":"2024-0018965792","revealing_evidence_id":"2024-0018965792:lines"},{"correct_verdict":"flag","defect_type":"zero_amount_line","entry_id":"2024-0018969623","revealing_evidence_id":"2024-0018969623:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018969733","revealing_evidence_id":"2024-0018969733:lines"},{"correct_verdict":"flag","defect_type":"unbalanced","entry_id":"2024-0018969865","revealing_evidence_id":"2024-0018969865:lines"}],"evidence_map":{"2024-0018942198":["2024-0018942198:lines"],"2024-0018944317":["2024-0018944317:lines"],"2024-0018946773":["2024-0018946773:lines"],"2024-0018950131":["2024-0018950131:lines"],"2024-0018951225":["2024-0018951225:lines"],"2024-0018952858":["2024-0018952858:lines"],"2024-0018960305":["2024-0018960305:lines"],"2024-0018965646":["2024-0018965646:lines"],"2024-0018965792":["2024-0018965792:lines"],"2024-0018969623":["2024-0018969623:lines"],"2024-0018969733":["2024-0018969733:lines"],"2024-0018969865":["2024-0018969865:lines"]},"required_docs":[]}
data/runs/console/{run-1b6438a7 → c5-grid-je01}/seed2/bundle/instance/documents.jsonl RENAMED
File without changes
data/runs/console/c5-grid-je01/seed2/bundle/instance/entries.jsonl ADDED
The diff for this file is too large to render. See raw diff