fsi-anomaly / data /build_v9.py
FerrellSyntheticIntelligence's picture
backup all: 100 files (batch)
1c0d385 verified
Raw
History Blame Contribute Delete
2.06 kB
"""Build v9 SFT mix: deterministic synthetic evidence-comparison (4000) +
real curated claim-vs-evidence (138) + a slice of v8 for format/story retention.
Goal: teach input->verdict conditioning (the v8 failure mode) while keeping
general chat fluency."""
import json, random
from pathlib import Path
rng = random.Random(20260803)
OUT = Path("data/sft_mix_v9.jsonl")
V9_SLICE = 2500 # rows sampled from v8 for format/story retention
def load(p):
return [json.loads(l) for l in open(p, encoding="utf-8") if l.strip()]
def verdict_word(v):
return {"not_enough_info": "not enough information"}.get(v, v)
def main():
synth = load("data/synth_evidence_v1.jsonl")
real = []
for r in load("data/evidence_judge.jsonl"):
if "Given the evidence" in r.get("user", ""):
v = r.get("verdict", "")
if v in ("supports", "refutes", "not_enough_info"):
vw = verdict_word(v)
real.append({"persona": "analyst",
"user": r["user"],
"assistant": (f"<|scratchpad|>Compare claim against evidence. "
f"The evidence directly addresses the claim. "
f"<|final|>Verdict: {vw}. Confidence: MEDIUM. "
f"Reasoning: The evidence was weighed against the claim and "
f"{'supports it' if vw=='supports' else ('contradicts it' if vw=='refutes' else 'does not settle it')}.")})
v8 = [r for r in load("data/sft_mix_v8.jsonl")
if "Evaluate this claim for accuracy" not in r.get("user", "")]
keep = rng.sample(v8, min(V9_SLICE, len(v8)))
mix = synth + real + keep
rng.shuffle(mix)
with open(OUT, "w", encoding="utf-8") as f:
for r in mix:
f.write(json.dumps(r, ensure_ascii=False) + "\n")
print(f"synth {len(synth)} real {len(real)} v8-slice {len(keep)} total {len(mix)}", flush=True)
if __name__ == "__main__":
main()