| { |
| "source_model": "/tmp/bench_advanced", |
| "technique": "refusal_direction_ablation", |
| "method": "advanced", |
| "method_config": { |
| "n_directions": 4, |
| "direction_method": "svd", |
| "norm_preserve": true, |
| "regularization": 0.5, |
| "refinement_passes": 3, |
| "project_biases": true, |
| "use_chat_template": true, |
| "use_whitened_svd": false, |
| "true_iterative_refinement": false, |
| "winsorize_activations": false, |
| "float_layer_interpolation": false, |
| "cot_aware": false, |
| "use_kl_optimization": false, |
| "use_lora_ablation": false, |
| "som_iterations": null, |
| "som_learning_rate": null, |
| "som_sigma": null, |
| "som_candidate_count": null, |
| "som_harmless_pc_count": null, |
| "som_distortion_aware": null, |
| "som_diversity_penalty": null, |
| "som_min_signal_to_noise": null, |
| "layer_selection": "knee_cosmic", |
| "min_layer_fraction": null, |
| "max_layer_fraction": null, |
| "harmless_pc_count": 0, |
| "shield_concept_count": 0, |
| "shield_ridge": 0.05, |
| "shield_residualize": false, |
| "shield_layer_penalty": 0.0, |
| "projection_target": "all", |
| "projection_row_fraction": 1.0, |
| "som_contiguous_layer_budget": null, |
| "spectral_cascade": false, |
| "spectral_bands": 3, |
| "spectral_threshold": 0.05 |
| }, |
| "references": [ |
| "Arditi et al., Refusal in Language Models Is Mediated by a Single Direction (NeurIPS 2024)", |
| "Gabliteration: SVD-based multi-direction extraction (arXiv:2512.18901)", |
| "Norm-Preserving Biprojected Abliteration (grimjim, 2025)", |
| "Young, Comparative Analysis of LLM Abliteration Methods (arXiv:2512.13655)", |
| "Joad et al., More to Refusal than a Single Direction (2026)", |
| "Piras et al., SOM Directions Are Better than One (AAAI 2026)", |
| "Heretic (p-e-w, 2025): Bayesian optimization, LoRA-mediated ablation, winsorization", |
| "OBLITERATUS: Whitened SVD, EGA, CoT-aware, KL co-optimization, float interpolation (novel)" |
| ], |
| "strong_layers": [ |
| 31, |
| 30, |
| 29, |
| 28, |
| 27, |
| 26 |
| ], |
| "n_harmful_prompts": 33, |
| "n_harmless_prompts": 33, |
| "quality_metrics": { |
| "perplexity": 44.838846540806735, |
| "coherence": 0.6, |
| "capability_score": 0.16666666666666666, |
| "capability_results": { |
| "tool_call": false, |
| "json_schema": false, |
| "chain_of_thought": false, |
| "code_function": false, |
| "visual_description": false, |
| "instruction_following": true |
| }, |
| "refusal_rate": 0.0, |
| "degenerate_count": 19, |
| "kl_divergence": 1.8895667791366577, |
| "spectral_certification": "RED" |
| }, |
| "kl_contributions": {}, |
| "cot_preserved_layers": [], |
| "float_layer_weights": {}, |
| "lora_adapters_saved": false |
| } |