diff --git a/configs/full_14b.yaml b/configs/full_14b.yaml index d04ab7d..31b8bef 100644 --- a/configs/full_14b.yaml +++ b/configs/full_14b.yaml @@ -2,10 +2,11 @@ # 100 open factual questions, mid-depth injection, training strengths to the # measured attack peak. These are the headline runs. # -# 48 layers, so 24 is mid-depth. THIS IS A PLACEHOLDER: run the depth and -# strength sweeps first (scripts/sweep_layers.py) and set layer/alphas from -# what they show, as we did for 3B and 7B. The 7B's behaviour differed from -# the 3B's, so mid-depth is an assumption until measured. +# Layer 24 and the strength range are MEASURED, not assumed. The depth sweep +# (results/sweep_layers_14b) found layer 24 admits 63/68 viable attacks against +# 46 at layer 12 and 31 at layer 36, and steers hardest at moderate strength. +# The strength sweep at layer 24 (results/sweep_alpha_14b) peaks at alpha 1.0 +# (94% steered) and degenerates above 1.2, so training brackets the peak. # Needs a 48GB card: 14B in bf16 is ~28GB before activations. @@ -31,7 +32,7 @@ steered_frac: 0.7 eval_question_frac: 0.2 repeats_per_question: 6 alpaca_replay_frac: 0.5 -eval_concepts_per_split: 40 # seeded subsample per split for the eval sweep +eval_concepts_per_split: {train: 40, heldout: null} # Training lora_r: 16 diff --git a/configs/full_3b.yaml b/configs/full_3b.yaml index 754685a..e1f1ea2 100644 --- a/configs/full_3b.yaml +++ b/configs/full_3b.yaml @@ -25,7 +25,7 @@ steered_frac: 0.7 eval_question_frac: 0.2 repeats_per_question: 6 alpaca_replay_frac: 0.5 -eval_concepts_per_split: 40 # seeded subsample per split for the eval sweep +eval_concepts_per_split: {train: 40, heldout: null} # Training lora_r: 16