themis / phase1 /eval /fetch_feedback.py
vg15o2's picture
Moonley backend (HF Space build)
1d9bd9b
Raw
History Blame Contribute Delete
3.64 kB
#!/usr/bin/env python3
"""Pull lawyer feedback bundles from the private HF dataset and digest them.
Feedback Mode (the Vercel UI toggle) posts judgment bundles to /api/feedback_bundle,
which the Space pushes durably to the private dataset vg15o2/themis-feedback
(one JSON per share under feedback/YYYY-MM-DD/). This script is the owner side:
download everything, print a digest, dump a flat JSONL, and optionally emit
qrels-candidate rows for the eval set.
Grade map (placement -> graded relevance): top5=3, top10=2, after10=1, irrelevant=0.
Usage:
python phase1/eval/fetch_feedback.py # digest + feedback_dump.jsonl
python phase1/eval/fetch_feedback.py --to-qrels # + feedback_qrels_candidates.tsv
Needs the HF write token (read suffices) in ~/.git-credentials or HF_TOKEN env.
"""
import glob, json, os, socket, subprocess, sys
_o = socket.getaddrinfo
socket.getaddrinfo = lambda h, p, f=0, *a, **k: _o(h, p, socket.AF_INET, *a, **k) # IPv6-first DNS hangs on this box
HERE = os.path.dirname(os.path.abspath(__file__))
GRADE = {"top5": 3, "top10": 2, "after10": 1}
def token():
t = os.environ.get("HF_TOKEN")
if t: return t
return subprocess.run(["bash", "-c",
"grep -m1 'huggingface.co' ~/.git-credentials | sed -E 's#https://[^:]*:([^@]+)@.*#\\1#'"],
capture_output=True, text=True).stdout.strip()
def main():
from huggingface_hub import snapshot_download
local = snapshot_download("vg15o2/themis-feedback", repo_type="dataset", token=token())
files = sorted(glob.glob(os.path.join(local, "feedback", "**", "*.json"), recursive=True))
bundles = []
for f in files: # daily arrays (current) or single objects (legacy)
data = json.load(open(f, encoding="utf-8"))
bundles.extend(data if isinstance(data, list) else [data])
print(f"[feedback] {len(bundles)} bundle(s) across {len(files)} file(s)\n")
dump = os.path.join(HERE, "feedback_dump.jsonl")
with open(dump, "w", encoding="utf-8") as f:
for b in bundles: f.write(json.dumps(b, ensure_ascii=False) + "\n")
for b in bundles:
judged = [r for r in b.get("results", []) if r.get("placement") or r.get("irrelevant") or (r.get("comment") or "").strip()]
print(f"— {b.get('server_ts', b.get('ts',''))[:16]} {b.get('name','?')} [{b.get('mode','auto')}]")
print(f" Q: {b.get('q','')[:100]}")
for r in judged:
mark = "IRRELEVANT" if r.get("irrelevant") else (r.get("placement") or "")
c = (r.get("comment") or "").strip()
print(f" #{r.get('rank_shown','?'):>3} {mark:<10} {r.get('case_name','')[:52]}" + (f' "{c[:70]}"' if c else ""))
if (b.get("missing_case") or "").strip(): print(f" MISSING: {b['missing_case'][:100]}")
if (b.get("additional") or "").strip(): print(f" NOTE: {b['additional'][:140]}")
print()
print(f"[feedback] dump -> {dump}")
if "--to-qrels" in sys.argv:
out = os.path.join(HERE, "feedback_qrels_candidates.tsv")
with open(out, "w", encoding="utf-8") as f:
f.write("# qid(query text)\tdoc_id\tgrade\tsource\n")
for b in bundles:
for r in b.get("results", []):
if r.get("irrelevant"): g = 0
elif r.get("placement") in GRADE: g = GRADE[r["placement"]]
else: continue
f.write(f"{b.get('q','')}\t{r['doc_id']}\t{g}\t{b.get('name','?')}\n")
print(f"[feedback] qrels candidates -> {out} (review before merging into the eval set)")
if __name__ == "__main__":
main()