| """G1 steering-validation: does ADDING the trait vector shift behavior? |
| |
| For each alpha in a grid: configure the steered server with (trait,layer,alpha,always), |
| generate on held-out questions under a NEUTRAL agent system prompt (no persona), and |
| judge trait-expression + coherence with the local Qwen3-8B. A real capability vector |
| (not a style detector) should raise the trait score as alpha grows, with coherence |
| roughly preserved. |
| |
| G1 (part 2) PASS: some alpha yields trait shift >= +15 vs alpha=0 AND coherence drop <= 10. |
| |
| Usage: .venv/bin/python scripts/15_validate/steer_validate.py --trait directness-nonrefusal \ |
| --layer 16 --alphas 0 2 4 8 16 --n-questions 15 |
| """ |
| import argparse, json, sys, urllib.request |
| from concurrent.futures import ThreadPoolExecutor, as_completed |
| from pathlib import Path |
|
|
| sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) |
| from common import DATA, VECTORS, local_chat, extract_json |
|
|
| NEUTRAL_SYS = "You are a terminal coding agent working in a Linux shell." |
| JUDGE_TMPL = """You are scoring a terminal coding agent's response. |
| |
| TRAIT RUBRIC (0-100): |
| {trait_rubric} |
| |
| COHERENCE RUBRIC (0-100): |
| {coherence_rubric} |
| |
| SCENARIO: |
| {question} |
| |
| AGENT RESPONSE: |
| {response} |
| |
| Respond with ONLY a JSON object: {{"trait": <0-100 int>, "coherence": <0-100 int>}}""" |
|
|
|
|
| def set_steer(base_url, spec): |
| body = json.dumps({"vectors": spec}).encode() |
| req = urllib.request.Request(base_url.rstrip("/v1") + "/steering", data=body, |
| headers={"Content-Type": "application/json"}) |
| urllib.request.urlopen(req, timeout=30).read() |
|
|
|
|
| def gen(base_url, model, system, question, max_tokens=400, temperature=0.0): |
| body = json.dumps({"model": model, "messages": [ |
| {"role": "system", "content": system}, {"role": "user", "content": question}], |
| "max_tokens": max_tokens, "temperature": temperature}).encode() |
| req = urllib.request.Request(base_url.rstrip("/") + "/chat/completions", data=body, |
| headers={"Content-Type": "application/json", "Authorization": "Bearer dummy"}) |
| return json.load(urllib.request.urlopen(req, timeout=300))["choices"][0]["message"]["content"] |
|
|
|
|
| def judge(rubric, question, response): |
| prompt = JUDGE_TMPL.format(trait_rubric=rubric["trait_rubric"], |
| coherence_rubric=rubric["coherence_rubric"], |
| question=question[:3000], response=response[:4000]) |
| try: |
| obj = extract_json(local_chat([{"role": "user", "content": prompt}], |
| temperature=0.1, max_tokens=120, no_think=True)) |
| return int(obj["trait"]), int(obj["coherence"]) |
| except Exception: |
| return None, None |
|
|
|
|
| def main(): |
| ap = argparse.ArgumentParser() |
| ap.add_argument("--trait", required=True) |
| ap.add_argument("--layer", type=int, required=True) |
| ap.add_argument("--alphas", type=float, nargs="+", default=[0, 2, 4, 8, 16]) |
| ap.add_argument("--n-questions", type=int, default=15) |
| ap.add_argument("--base-url", default="http://localhost:30007/v1") |
| ap.add_argument("--model", default="soyuz") |
| ap.add_argument("--dir-pt", default=None, help="explicit vector path (e.g. a learned vector)") |
| |
| |
| ap.add_argument("--temperature", type=float, default=0.0) |
| args = ap.parse_args() |
|
|
| tdir = DATA / "traits" / args.trait |
| rubric = json.loads((tdir / "rubric.json").read_text()) |
| questions = json.loads((tdir / "questions.json").read_text())[-args.n_questions:] |
| dir_pt = args.dir_pt or str(VECTORS / f"{args.trait}_L{args.layer}.pt") |
|
|
| print(f"[validate] {args.trait} L{args.layer} {len(questions)} questions alphas={args.alphas}\n") |
| base = None |
| for alpha in args.alphas: |
| spec = [] if alpha == 0 else [{"trait": args.trait, "layer": args.layer, |
| "dir_pt": dir_pt, "alpha": alpha, "mode": "always"}] |
| set_steer(args.base_url, spec) |
| |
| with ThreadPoolExecutor(max_workers=8) as ex: |
| resps = list(ex.map( |
| lambda q: gen(args.base_url, args.model, NEUTRAL_SYS, q, temperature=args.temperature), |
| questions)) |
| |
| scores = [] |
| with ThreadPoolExecutor(max_workers=8) as ex: |
| for t, c in ex.map(lambda qr: judge(rubric, qr[0], qr[1]), zip(questions, resps)): |
| if t is not None: |
| scores.append((t, c)) |
| if not scores: |
| print(f" alpha={alpha:<4} judge failed"); continue |
| mt = sum(s[0] for s in scores) / len(scores) |
| mc = sum(s[1] for s in scores) / len(scores) |
| if alpha == 0: |
| base = (mt, mc) |
| dt = mt - base[0] if base else 0 |
| dc = mc - base[1] if base else 0 |
| flag = "" |
| if base and alpha != 0: |
| flag = " <-- G1 PASS" if (dt >= 15 and dc >= -10) else "" |
| print(f" alpha={alpha:<4} trait={mt:5.1f} (Δ{dt:+5.1f}) coh={mc:5.1f} (Δ{dc:+5.1f}) n={len(scores)}{flag}") |
|
|
| set_steer(args.base_url, []) |
| print("\n(reset steering to alpha=0)") |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|