File size: 2,674 Bytes
96a1862
0172581
96a1862
0172581
96a1862
0172581
 
96a1862
0172581
96a1862
 
 
 
0172581
 
 
96a1862
0172581
96a1862
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0172581
 
 
 
 
 
96a1862
 
 
 
0172581
 
 
96a1862
 
 
 
 
 
0172581
96a1862
 
0172581
 
 
96a1862
0172581
96a1862
0172581
96a1862
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
{
  "source_model": "/tmp/bench_advanced",
  "technique": "refusal_direction_ablation",
  "method": "advanced",
  "method_config": {
    "n_directions": 4,
    "direction_method": "svd",
    "norm_preserve": true,
    "regularization": 0.5,
    "refinement_passes": 3,
    "project_biases": true,
    "use_chat_template": true,
    "use_whitened_svd": false,
    "true_iterative_refinement": false,
    "winsorize_activations": false,
    "float_layer_interpolation": false,
    "cot_aware": false,
    "use_kl_optimization": false,
    "use_lora_ablation": false,
    "som_iterations": null,
    "som_learning_rate": null,
    "som_sigma": null,
    "som_candidate_count": null,
    "som_harmless_pc_count": null,
    "som_distortion_aware": null,
    "som_diversity_penalty": null,
    "som_min_signal_to_noise": null,
    "layer_selection": "knee_cosmic",
    "min_layer_fraction": null,
    "max_layer_fraction": null,
    "harmless_pc_count": 0,
    "shield_concept_count": 0,
    "shield_ridge": 0.05,
    "shield_residualize": false,
    "shield_layer_penalty": 0.0,
    "projection_target": "all",
    "projection_row_fraction": 1.0,
    "som_contiguous_layer_budget": null,
    "spectral_cascade": false,
    "spectral_bands": 3,
    "spectral_threshold": 0.05
  },
  "references": [
    "Arditi et al., Refusal in Language Models Is Mediated by a Single Direction (NeurIPS 2024)",
    "Gabliteration: SVD-based multi-direction extraction (arXiv:2512.18901)",
    "Norm-Preserving Biprojected Abliteration (grimjim, 2025)",
    "Young, Comparative Analysis of LLM Abliteration Methods (arXiv:2512.13655)",
    "Joad et al., More to Refusal than a Single Direction (2026)",
    "Piras et al., SOM Directions Are Better than One (AAAI 2026)",
    "Heretic (p-e-w, 2025): Bayesian optimization, LoRA-mediated ablation, winsorization",
    "OBLITERATUS: Whitened SVD, EGA, CoT-aware, KL co-optimization, float interpolation (novel)"
  ],
  "strong_layers": [
    31,
    30,
    29,
    28,
    27,
    26
  ],
  "n_harmful_prompts": 33,
  "n_harmless_prompts": 33,
  "quality_metrics": {
    "perplexity": 44.838846540806735,
    "coherence": 0.6,
    "capability_score": 0.16666666666666666,
    "capability_results": {
      "tool_call": false,
      "json_schema": false,
      "chain_of_thought": false,
      "code_function": false,
      "visual_description": false,
      "instruction_following": true
    },
    "refusal_rate": 0.0,
    "degenerate_count": 19,
    "kl_divergence": 1.8895667791366577,
    "spectral_certification": "RED"
  },
  "kl_contributions": {},
  "cot_preserved_layers": [],
  "float_layer_weights": {},
  "lora_adapters_saved": false
}