fsi-anomaly / data /build_evidence.py
FerrellSyntheticIntelligence's picture
backup all: 100 files (batch)
1c0d385 verified
Raw
History Blame Contribute Delete
2.11 kB
"""Build evidence-in-context judgment dataset: claim + evidence -> verdict.
Dedupe across all SFT corpora; keeps rows where the user text contains both a
claim and a separate evidence marker, with a parseable Verdict label."""
import json, re
from collections import Counter
from pathlib import Path
EVID = re.compile(r"records? show|permit|account [ab]|accounts? [ab]:|according to|filing|inspection|report[sed]? (by|that|the)|data show|documents|registered|confirmed|assessor|timeline|witness|source|wire|blog|notes? that|second investigation|reassessment|raised? [0-9]|rose [0-9]|increas[ed]? [0-9]", re.I)
VALS = ["true statement", "false statement", "supports", "refutes", "not_enough_info"]
def parse_v(text):
m = re.search(r"Verdict\s*:\s*([^.]+)\.", text, re.I)
if not m: return None
lab = m.group(1).strip().lower()
for cand in VALS:
if lab.startswith(cand): return cand
if lab.startswith("true"): return "true statement"
if lab.startswith("false"): return "false statement"
if lab.startswith("not"): return "not_enough_info"
if lab.startswith("support"): return "supports"
if lab.startswith("refut"): return "refutes"
return None
def load(p):
return [json.loads(l) for l in open(p, encoding="utf-8") if l.strip()]
files = ["sft_forensic.jsonl","sft_sop_mix.jsonl","sft_sop.jsonl","sft_distill_mix.jsonl",
"distill_method.jsonl","distill_analyst_a.jsonl","distill_analyst_b.jsonl",
"sft_mix_v2.jsonl","sft_mix_v3.jsonl"]
seen, out = set(), []
for f in files:
for r in load(f"data/{f}"):
u, a = r.get("user", ""), r.get("assistant", "")
v = parse_v(a)
if not v or not EVID.search(u):
continue
key = u[:200]
if key in seen:
continue
seen.add(key)
out.append({"persona": r.get("persona", "analyst"), "user": u, "verdict": v})
print("total", len(out), dict(Counter(o["verdict"] for o in out)), flush=True)
with open("data/evidence_judge.jsonl", "w", encoding="utf-8") as f:
for o in out:
f.write(json.dumps(o, ensure_ascii=False) + "\n")