File size: 1,393 Bytes
599718a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
{
  "harness": "deliverable_landing_gate",
  "transfer": [
    {
      "model": "DeepSeek V4 Pro",
      "model_short": "DeepSeek-V4-Pro-together",
      "baseline_pooled": 0.6339,
      "baseline_all_pass": 0.0,
      "baseline_n": 100,
      "harness_pooled": 0.7686,
      "harness_all_pass": 0.02,
      "harness_n": 100,
      "delta_pp": 13.5,
      "evaluable": true
    },
    {
      "model": "DeepSeek V4 Flash",
      "model_short": "DeepSeek-V4-Flash",
      "baseline_pooled": 0.6463,
      "baseline_all_pass": 0.0,
      "baseline_n": 100,
      "harness_pooled": 0.79,
      "harness_all_pass": 0.02,
      "harness_n": 100,
      "delta_pp": 14.4,
      "evaluable": true
    },
    {
      "model": "Nemotron-3 Ultra 550B",
      "model_short": "NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4",
      "baseline_pooled": 0.6384,
      "baseline_all_pass": 0.0,
      "baseline_n": 100,
      "harness_pooled": 0.6421,
      "harness_all_pass": 0.03,
      "harness_n": 100,
      "delta_pp": 0.4,
      "evaluable": true
    },
    {
      "model": "GLM-5.2",
      "model_short": "GLM-5-2",
      "baseline_pooled": 0.3446,
      "baseline_all_pass": 0.02,
      "baseline_n": 100,
      "evaluable": false
    }
  ],
  "frontier_on_tuned_model": {
    "model": "DeepSeek V4 Pro",
    "harness": "matter_audit_allwork",
    "pooled": 0.8008,
    "all_pass": 0.05,
    "n": 100
  }
}