{ "source_model": "outputs/qwen3.8-27b-s30_s23_l63_b070", "technique": "refusal_direction_ablation", "method": "aggressive", "method_config": { "n_directions": 5, "direction_method": "svd", "norm_preserve": true, "regularization": 0.04, "refinement_passes": 3, "project_biases": true, "use_chat_template": true, "use_whitened_svd": true, "true_iterative_refinement": true, "winsorize_activations": true, "float_layer_interpolation": false, "cot_aware": false, "use_kl_optimization": false, "use_lora_ablation": false, "som_iterations": null, "som_learning_rate": null, "som_sigma": null, "som_candidate_count": null, "som_harmless_pc_count": null, "som_distortion_aware": null, "som_diversity_penalty": null, "som_min_signal_to_noise": null, "layer_selection": "knee_cosmic", "min_layer_fraction": 0.4, "max_layer_fraction": null, "harmless_pc_count": 0, "shield_concept_count": 0, "shield_ridge": 0.05, "shield_residualize": false, "shield_layer_penalty": 0.0, "projection_target": "all", "projection_row_fraction": 1.0, "som_contiguous_layer_budget": null, "spectral_cascade": false, "spectral_bands": 3, "spectral_threshold": 0.05 }, "references": [ "Arditi et al., Refusal in Language Models Is Mediated by a Single Direction (NeurIPS 2024)", "Gabliteration: SVD-based multi-direction extraction (arXiv:2512.18901)", "Norm-Preserving Biprojected Abliteration (grimjim, 2025)", "Young, Comparative Analysis of LLM Abliteration Methods (arXiv:2512.13655)", "Joad et al., More to Refusal than a Single Direction (2026)", "Piras et al., SOM Directions Are Better than One (AAAI 2026)", "Heretic (p-e-w, 2025): Bayesian optimization, LoRA-mediated ablation, winsorization", "OBLITERATUS: Whitened SVD, EGA, CoT-aware, KL co-optimization, float interpolation (novel)" ], "strong_layers": [ 35, 34, 36, 33, 37, 39, 38, 43, 40, 42, 32, 46, 44, 45, 41, 31, 47, 26, 27, 30, 29, 48, 25, 28, 49, 50, 62, 61, 60, 51, 59, 52, 57, 56, 55, 53, 63, 54, 58 ], "n_harmful_prompts": 1007, "n_harmless_prompts": 1007, "quality_metrics": { "perplexity": 4.258847364718222, "coherence": 1.0, "capability_score": 1.0, "capability_results": { "tool_call": true, "json_schema": true, "chain_of_thought": true, "code_function": true, "visual_description": true, "instruction_following": true }, "refusal_rate": 0.0, "kl_divergence": 0.9624386429786682, "spectral_certification": "RED" }, "kl_contributions": {}, "cot_preserved_layers": [], "float_layer_weights": {}, "lora_adapters_saved": false }