File size: 2,055 Bytes
1c0d385
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
"""Build v9 SFT mix: deterministic synthetic evidence-comparison (4000) +
real curated claim-vs-evidence (138) + a slice of v8 for format/story retention.
Goal: teach input->verdict conditioning (the v8 failure mode) while keeping
general chat fluency."""
import json, random
from pathlib import Path

rng = random.Random(20260803)
OUT = Path("data/sft_mix_v9.jsonl")
V9_SLICE = 2500  # rows sampled from v8 for format/story retention

def load(p):
    return [json.loads(l) for l in open(p, encoding="utf-8") if l.strip()]

def verdict_word(v):
    return {"not_enough_info": "not enough information"}.get(v, v)

def main():
    synth = load("data/synth_evidence_v1.jsonl")
    real = []
    for r in load("data/evidence_judge.jsonl"):
        if "Given the evidence" in r.get("user", ""):
            v = r.get("verdict", "")
            if v in ("supports", "refutes", "not_enough_info"):
                vw = verdict_word(v)
                real.append({"persona": "analyst",
                             "user": r["user"],
                             "assistant": (f"<|scratchpad|>Compare claim against evidence. "
                                           f"The evidence directly addresses the claim. "
                                           f"<|final|>Verdict: {vw}. Confidence: MEDIUM. "
                                           f"Reasoning: The evidence was weighed against the claim and "
                                           f"{'supports it' if vw=='supports' else ('contradicts it' if vw=='refutes' else 'does not settle it')}.")})
    v8 = [r for r in load("data/sft_mix_v8.jsonl")
          if "Evaluate this claim for accuracy" not in r.get("user", "")]
    keep = rng.sample(v8, min(V9_SLICE, len(v8)))
    mix = synth + real + keep
    rng.shuffle(mix)
    with open(OUT, "w", encoding="utf-8") as f:
        for r in mix:
            f.write(json.dumps(r, ensure_ascii=False) + "\n")
    print(f"synth {len(synth)} real {len(real)} v8-slice {len(keep)} total {len(mix)}", flush=True)

if __name__ == "__main__":
    main()