File size: 6,338 Bytes
6ca9813
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
{
  "artifact_kind": "frozen_vibethinker_v1_readout_reference",
  "claim_boundary": "This reference records the V1 readout results. It excludes causal-assay results and does not establish free-generation steering or a global workspace.",
  "classification": "validated_readout_only",
  "coverage": {
    "locked_test_items": 377,
    "readout_items_completed": 539,
    "readout_items_no_eligible_targets": 12,
    "released_prompts": 551
  },
  "coverage_scope": "frozen_readout_source_run_not_the_100_prompt_ui_test_pack",
  "task_definition": {
    "aggregate_mean_reciprocal_rank": "mean_across_eligible_target_terms_of_reciprocal_best_rank",
    "eligible_target": "at_least_one_candidate_form_tokenizes_to_exactly_one_token",
    "final_model": "rank_target_terms_in_the_model_next_token_logits_at_the_score_position",
    "item": "one_prompt_with_one_or_more_target_terms",
    "layer_scope_reduction": "best_target_rank_across_layers_in_the_reported_scope",
    "no_eligible_target_item": "none_of_the_item_target_terms_has_an_eligible_single_token_form",
    "paired_bootstrap_mean_reciprocal_rank": "within_item_mean_of_reciprocal_best_rank_across_eligible_target_terms",
    "pass_at_k": "mean_across_items_of_the_fraction_of_item_target_terms_with_best_rank_at_most_k",
    "score_position": {
      "default": "final_prompt_token",
      "poetry": "last_newline_token"
    },
    "split": {
      "dev_fraction": 0.3,
      "method": "sha256_stable_split",
      "seed": "vibethinker-jlens-v1"
    },
    "suites": ["association", "multihop", "multilingual", "order-ops", "poetry", "typo"],
    "target_candidate_forms": [
      "original_lowercase_and_capitalized_forms_with_and_without_leading_space",
      "order_ops_also_adds_configured_operation_synonyms_and_number_word_digit_forms"
    ],
    "target_rank": "best_one_based_vocabulary_rank_among_eligible_target_token_ids"
  },
  "evidence": {
    "incremental_over_ordinary_logit_lens": false,
    "layer_mapping_specific_vs_shuffled_jacobian": true,
    "paired_bootstrap": {
      "confidence": 0.95,
      "input": "paired_item_metric_differences",
      "interval": "percentile",
      "metrics": ["pass@10", "mean_reciprocal_rank"],
      "samples": 2000,
      "scope": {
        "kind": "selected_band",
        "source_layers": [24, 26, 28, 30, 32, 34]
      },
      "split": "test"
    },
    "paired_bootstrap_decisions": {
      "incremental_over_ordinary_logit_lens": {
        "comparison": "jlens_band_minus_logit_lens_band",
        "criterion": "at_least_one_lower_bound_greater_than_zero",
        "met": false
      },
      "layer_mapping_specific_vs_shuffled_jacobian": {
        "comparison": "jlens_band_minus_shuffled_layer_band",
        "criterion": "both_lower_bounds_greater_than_zero",
        "met": true
      },
      "token_specific_vs_shuffled_target": {
        "comparison": "jlens_band_minus_shuffled_token_band",
        "criterion": "both_lower_bounds_greater_than_zero",
        "met": true
      }
    },
    "specificity_scope": {
      "kind": "selected_band",
      "source_layers": [24, 26, 28, 30, 32, 34]
    },
    "token_specific_vs_shuffled_target": true,
    "validated_readout_signal": true
  },
  "lens_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
  "lens_variant": "fp32_evaluation_safetensors_included",
  "released_lens_artifact": {
    "filename": "evaluation.safetensors",
    "format": "safetensors",
    "sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
    "size_bytes": 301991904,
    "source_checkpoint_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
    "tensor_conversion": "lossless_fp32_reserialization"
  },
  "source_artifact_availability": {
    "fp32_compatibility_check_output_included": true,
    "fp32_compatibility_check_output_path": "evaluation_compatibility.json",
    "fp32_derivation_record_included": true,
    "fp32_derivation_record_path": "evaluation_provenance.json",
    "fp32_evaluation_lens_included": true,
    "fp32_evaluation_lens_path": "evaluation.safetensors",
    "item_level_evaluation_rows_included": false,
    "item_level_evaluation_rows_location": "companion_trace_repository:data/evaluation-results/readout-trials.jsonl",
    "paired_bootstrap_interval_bounds_included": false,
    "paired_bootstrap_interval_bounds_location": "companion_trace_repository:data/evaluation-results/readout-bootstrap-intervals.json",
    "source_evaluation_bundle_included": false,
    "source_hashes_included": true,
    "source_hashes_location": "evaluation_provenance.json_and_companion_trace_repository:data/readout-reference.json",
    "release_content": "fp16_trace_lens_fp32_evaluation_lens_aggregate_reference_provenance_and_compatibility"
  },
  "metrics": {
    "jlens_all_layers": {
      "mean_reciprocal_rank": 0.027840488103301416,
      "n_items": 377,
      "pass_at_1": 0.009283819628647215,
      "pass_at_10": 0.08819628647214854,
      "pass_at_50": 0.1655614500442087
    },
    "jlens_selected_band": {
      "mean_reciprocal_rank": 0.01888596779741093,
      "n_items": 377,
      "pass_at_1": 0.003978779840848806,
      "pass_at_10": 0.0629973474801061,
      "pass_at_50": 0.11914235190097258
    },
    "logit_lens_all_layers": {
      "mean_reciprocal_rank": 0.027249274540932285,
      "n_items": 377,
      "pass_at_1": 0.007294429708222812,
      "pass_at_10": 0.08377541998231654,
      "pass_at_50": 0.16114058355437666
    },
    "shuffled_layer_all_layers": {
      "mean_reciprocal_rank": 0.042073366910739256,
      "n_items": 377,
      "pass_at_1": 0.04509283819628647,
      "pass_at_10": 0.08819628647214854,
      "pass_at_50": 0.1823607427055703
    },
    "final_model": {
      "mean_reciprocal_rank": 0.019635164837169684,
      "n_items": 377,
      "pass_at_1": 0.003978779840848806,
      "pass_at_10": 0.0665340406719717,
      "pass_at_50": 0.10587975243147656
    }
  },
  "model": "WeiboAI/VibeThinker-3B",
  "model_revision": "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c",
  "note": "The recorded readout metrics bind to evaluation.safetensors, the FP32 evaluation lens in this repository. They do not evaluate model.safetensors, the FP16 lens used for the captured traces.",
  "selected_band": [24, 26, 28, 30, 32, 34],
  "schema_version": 1
}