vibethinker-3b-jlens-model / evaluation.json
jvogan
Prepare VibeThinker-3B J Lens model release
6ca9813
Raw
History Blame Contribute Delete
6.34 kB
{
"artifact_kind": "frozen_vibethinker_v1_readout_reference",
"claim_boundary": "This reference records the V1 readout results. It excludes causal-assay results and does not establish free-generation steering or a global workspace.",
"classification": "validated_readout_only",
"coverage": {
"locked_test_items": 377,
"readout_items_completed": 539,
"readout_items_no_eligible_targets": 12,
"released_prompts": 551
},
"coverage_scope": "frozen_readout_source_run_not_the_100_prompt_ui_test_pack",
"task_definition": {
"aggregate_mean_reciprocal_rank": "mean_across_eligible_target_terms_of_reciprocal_best_rank",
"eligible_target": "at_least_one_candidate_form_tokenizes_to_exactly_one_token",
"final_model": "rank_target_terms_in_the_model_next_token_logits_at_the_score_position",
"item": "one_prompt_with_one_or_more_target_terms",
"layer_scope_reduction": "best_target_rank_across_layers_in_the_reported_scope",
"no_eligible_target_item": "none_of_the_item_target_terms_has_an_eligible_single_token_form",
"paired_bootstrap_mean_reciprocal_rank": "within_item_mean_of_reciprocal_best_rank_across_eligible_target_terms",
"pass_at_k": "mean_across_items_of_the_fraction_of_item_target_terms_with_best_rank_at_most_k",
"score_position": {
"default": "final_prompt_token",
"poetry": "last_newline_token"
},
"split": {
"dev_fraction": 0.3,
"method": "sha256_stable_split",
"seed": "vibethinker-jlens-v1"
},
"suites": ["association", "multihop", "multilingual", "order-ops", "poetry", "typo"],
"target_candidate_forms": [
"original_lowercase_and_capitalized_forms_with_and_without_leading_space",
"order_ops_also_adds_configured_operation_synonyms_and_number_word_digit_forms"
],
"target_rank": "best_one_based_vocabulary_rank_among_eligible_target_token_ids"
},
"evidence": {
"incremental_over_ordinary_logit_lens": false,
"layer_mapping_specific_vs_shuffled_jacobian": true,
"paired_bootstrap": {
"confidence": 0.95,
"input": "paired_item_metric_differences",
"interval": "percentile",
"metrics": ["pass@10", "mean_reciprocal_rank"],
"samples": 2000,
"scope": {
"kind": "selected_band",
"source_layers": [24, 26, 28, 30, 32, 34]
},
"split": "test"
},
"paired_bootstrap_decisions": {
"incremental_over_ordinary_logit_lens": {
"comparison": "jlens_band_minus_logit_lens_band",
"criterion": "at_least_one_lower_bound_greater_than_zero",
"met": false
},
"layer_mapping_specific_vs_shuffled_jacobian": {
"comparison": "jlens_band_minus_shuffled_layer_band",
"criterion": "both_lower_bounds_greater_than_zero",
"met": true
},
"token_specific_vs_shuffled_target": {
"comparison": "jlens_band_minus_shuffled_token_band",
"criterion": "both_lower_bounds_greater_than_zero",
"met": true
}
},
"specificity_scope": {
"kind": "selected_band",
"source_layers": [24, 26, 28, 30, 32, 34]
},
"token_specific_vs_shuffled_target": true,
"validated_readout_signal": true
},
"lens_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
"lens_variant": "fp32_evaluation_safetensors_included",
"released_lens_artifact": {
"filename": "evaluation.safetensors",
"format": "safetensors",
"sha256": "0cc184eb65d273ac8bfee5450a141c8cdf8dd6ec8caa68d7d47260ef621777d1",
"size_bytes": 301991904,
"source_checkpoint_sha256": "8f752032a26a5196c1cb447ef63f01e8a29820ff0c57178dd80c6e26d32b12a9",
"tensor_conversion": "lossless_fp32_reserialization"
},
"source_artifact_availability": {
"fp32_compatibility_check_output_included": true,
"fp32_compatibility_check_output_path": "evaluation_compatibility.json",
"fp32_derivation_record_included": true,
"fp32_derivation_record_path": "evaluation_provenance.json",
"fp32_evaluation_lens_included": true,
"fp32_evaluation_lens_path": "evaluation.safetensors",
"item_level_evaluation_rows_included": false,
"item_level_evaluation_rows_location": "companion_trace_repository:data/evaluation-results/readout-trials.jsonl",
"paired_bootstrap_interval_bounds_included": false,
"paired_bootstrap_interval_bounds_location": "companion_trace_repository:data/evaluation-results/readout-bootstrap-intervals.json",
"source_evaluation_bundle_included": false,
"source_hashes_included": true,
"source_hashes_location": "evaluation_provenance.json_and_companion_trace_repository:data/readout-reference.json",
"release_content": "fp16_trace_lens_fp32_evaluation_lens_aggregate_reference_provenance_and_compatibility"
},
"metrics": {
"jlens_all_layers": {
"mean_reciprocal_rank": 0.027840488103301416,
"n_items": 377,
"pass_at_1": 0.009283819628647215,
"pass_at_10": 0.08819628647214854,
"pass_at_50": 0.1655614500442087
},
"jlens_selected_band": {
"mean_reciprocal_rank": 0.01888596779741093,
"n_items": 377,
"pass_at_1": 0.003978779840848806,
"pass_at_10": 0.0629973474801061,
"pass_at_50": 0.11914235190097258
},
"logit_lens_all_layers": {
"mean_reciprocal_rank": 0.027249274540932285,
"n_items": 377,
"pass_at_1": 0.007294429708222812,
"pass_at_10": 0.08377541998231654,
"pass_at_50": 0.16114058355437666
},
"shuffled_layer_all_layers": {
"mean_reciprocal_rank": 0.042073366910739256,
"n_items": 377,
"pass_at_1": 0.04509283819628647,
"pass_at_10": 0.08819628647214854,
"pass_at_50": 0.1823607427055703
},
"final_model": {
"mean_reciprocal_rank": 0.019635164837169684,
"n_items": 377,
"pass_at_1": 0.003978779840848806,
"pass_at_10": 0.0665340406719717,
"pass_at_50": 0.10587975243147656
}
},
"model": "WeiboAI/VibeThinker-3B",
"model_revision": "77bd2cced09193c8b9a59a32bd8577bbd1f3e01c",
"note": "The recorded readout metrics bind to evaluation.safetensors, the FP32 evaluation lens in this repository. They do not evaluate model.safetensors, the FP16 lens used for the captured traces.",
"selected_band": [24, 26, 28, 30, 32, 34],
"schema_version": 1
}