steer-full_7b / run /config.yaml
JacoDuToit's picture
run directory (provenance + evals)
72f0db1 verified
Raw
History Blame Contribute Delete
2 kB
# HEADLINE RUN: full 500-concept bank on Qwen2.5-7B. Layer 14 = mid-depth,
# the largest viable-attack pool per the 7B sweeps (see results/sweep_*_7b).
# Settings fixed by the sweeps (see results/sweep_layers_paper_3b, sweep_alpha_3b):
# * layer 18 = mid-depth, the strongest attack surface (worst case for us)
# * train alphas extend to 1.2 = the steered-rate peak on the base model
# * eval alphas cover past the peak into the degeneracy regime
# Task = open factual questions; eval concepts subsampled (seeded) to keep the
# eval tractable at 500 concepts.
model_id: Qwen/Qwen2.5-7B-Instruct
device: auto
dtype: auto
seed: 0
# Steering
layer: 14
train_alphas: [0.4, 0.8, 1.2]
eval_alphas: [0.4, 0.8, 1.0, 1.2, 1.6]
efficacy_alpha: 0.8
efficacy_min_rate: 0.3
n_contrast_pairs: 10
# Data — FULL paper bank + open factual questions
qa_source: json
concepts_path: data/concepts_paper.json
qa_path: data/qa_open.json
relevant_frac: 0.6
steered_frac: 0.7
eval_question_frac: 0.2
repeats_per_question: 6
alpaca_replay_frac: 0.5
eval_concepts_per_split: 40 # seeded subsample per split for the eval sweep
# Training
lora_r: 16
lora_alpha: 32
lora_dropout: 0.05
lora_targets: [q_proj, k_proj, v_proj, o_proj, gate_proj, up_proj, down_proj]
lr: 1.0e-4
epochs: 1
effective_batch_size: 16
train_batch_size: 2
max_seq_len: 256
# Eval
max_new_tokens: 64
eval_batch_size: 4
bootstrap_resamples: 10000
# Capability retention — same pinned subset as every other run
capability_tasks: {mmlu: 5, gsm8k_cot: 8}
capability_apply_chat_template: true
capability_limit: "mmlu=15,gsm8k_cot=200"
# Tracking & backup — hub_repo_id set by scripts/runpod.sh at run time
wandb_project: null
wandb_entity: null
hub_repo_id: JacoDuToit/steer-full_7b
hub_private: true
# Paths
vectors_path: results/full_7b/vectors.pt
train_examples_path: results/full_7b/train_examples.json
eval_questions_path: results/full_7b/eval_questions.json
adapter_dir: results/full_7b/m1_resist_adapter
results_dir: results/full_7b