"""Build v9 SFT mix: deterministic synthetic evidence-comparison (4000) + real curated claim-vs-evidence (138) + a slice of v8 for format/story retention. Goal: teach input->verdict conditioning (the v8 failure mode) while keeping general chat fluency.""" import json, random from pathlib import Path rng = random.Random(20260803) OUT = Path("data/sft_mix_v9.jsonl") V9_SLICE = 2500 # rows sampled from v8 for format/story retention def load(p): return [json.loads(l) for l in open(p, encoding="utf-8") if l.strip()] def verdict_word(v): return {"not_enough_info": "not enough information"}.get(v, v) def main(): synth = load("data/synth_evidence_v1.jsonl") real = [] for r in load("data/evidence_judge.jsonl"): if "Given the evidence" in r.get("user", ""): v = r.get("verdict", "") if v in ("supports", "refutes", "not_enough_info"): vw = verdict_word(v) real.append({"persona": "analyst", "user": r["user"], "assistant": (f"<|scratchpad|>Compare claim against evidence. " f"The evidence directly addresses the claim. " f"<|final|>Verdict: {vw}. Confidence: MEDIUM. " f"Reasoning: The evidence was weighed against the claim and " f"{'supports it' if vw=='supports' else ('contradicts it' if vw=='refutes' else 'does not settle it')}.")}) v8 = [r for r in load("data/sft_mix_v8.jsonl") if "Evaluate this claim for accuracy" not in r.get("user", "")] keep = rng.sample(v8, min(V9_SLICE, len(v8))) mix = synth + real + keep rng.shuffle(mix) with open(OUT, "w", encoding="utf-8") as f: for r in mix: f.write(json.dumps(r, ensure_ascii=False) + "\n") print(f"synth {len(synth)} real {len(real)} v8-slice {len(keep)} total {len(mix)}", flush=True) if __name__ == "__main__": main()