File size: 5,209 Bytes
548b2a6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
#!/usr/bin/env python3
"""Build viewer data for the PhantomWiki viewer.

PhantomWiki (https://github.com/kilian-group/phantom-wiki, arXiv:2502.20377)
generates synthetic "universes" of people (family + friendship graphs) with
Wikipedia-style articles, plus multi-hop reasoning questions over them.

The pre-generated HF dataset ``kilian-group/phantom-wiki-v1`` has, for each of 9
universes (``depth_20_size_{50,500,5000}_seed_{1,2,3}``), two relevant configs:

  text-corpus     {title, article (markdown), facts}   — one document per person
  question-answer {id, question, answer (list), intermediate_answers,
                   prolog{query, answer}, template, type, difficulty}

Each universe has exactly 500 questions (paper's "500 per instance"); the corpus
has one document per person (≈ the universe size). This script writes, per
universe:

  corpus_<u>.json   list[{title, article}]                    (documents, inline)
  eval_<u>.json     list[{id, question, answer, supporting_titles,
                          template, prolog, meta{Difficulty, Type}}]

plus a ``sets.json`` manifest (universe selector + Difficulty/Type facets).
Answers that name people resolve to corpus documents (clickable supporting
docs); answers that are attribute values (dates, counts, jobs) have none.

Run from the viewer repo root:

    python scripts/build_data.py --data-dir /mnt/tmp/phantom-wiki/pw-v1
"""
import argparse
import json
import os

try:
    import pyarrow.parquet as pq
except ImportError:
    pq = None

ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
DEFAULT_DATA_DIR = "/mnt/tmp/phantom-wiki/pw-v1"
SIZES = [50, 500, 5000]
SEEDS = [1, 2, 3]


def parquet_path(data_dir, cfg, size, seed):
    return os.path.join(data_dir, cfg,
                        f"depth_20_size_{size}_seed_{seed}-00000-of-00001.parquet")


def build_universe(data_dir, size, seed):
    uid = f"size{size}_seed{seed}"
    tc_path = parquet_path(data_dir, "text-corpus", size, seed)
    qa_path = parquet_path(data_dir, "question-answer", size, seed)
    if not (os.path.exists(tc_path) and os.path.exists(qa_path)):
        print(f"[{uid}] missing parquet — skipping")
        return None

    tc = pq.read_table(tc_path).to_pydict()
    corpus_rows = []
    titles = set()
    for i in range(len(tc["title"])):
        t = tc["title"][i]
        titles.add(t)
        corpus_rows.append({"title": t, "article": tc["article"][i] or ""})
    corpus_rows.sort(key=lambda d: d["title"].lower())

    qa = pq.read_table(qa_path).to_pydict()
    eval_rows = []
    diffs, types = set(), set()
    for i in range(len(qa["id"])):
        answer = list(qa["answer"][i]) if qa["answer"][i] is not None else []
        supporting = [a for a in answer if a in titles]
        prolog = qa["prolog"][i] or {}
        template = qa["template"][i] or []
        d = int(qa["difficulty"][i]) if qa["difficulty"][i] is not None else None
        ty = int(qa["type"][i]) if qa["type"][i] is not None else None
        diffs.add(d)
        types.add(ty)
        eval_rows.append({
            "id": qa["id"][i],
            "question": (qa["question"][i] or "").strip(),
            "answer": answer,
            "supporting_titles": supporting,
            "template": " ".join(template),
            "prolog": list(prolog.get("query") or []),
            "prolog_answer": prolog.get("answer", ""),
            "meta": {"Difficulty": str(d), "Type": str(ty)},
        })
    eval_rows.sort(key=lambda r: (int(r["meta"]["Type"]), r["id"]))

    corpus_path = os.path.join(ROOT, f"corpus_{uid}.json")
    eval_path = os.path.join(ROOT, f"eval_{uid}.json")
    with open(corpus_path, "w", encoding="utf-8") as f:
        json.dump(corpus_rows, f, ensure_ascii=False)
    with open(eval_path, "w", encoding="utf-8") as f:
        json.dump(eval_rows, f, ensure_ascii=False)

    facets = [
        {"key": "Difficulty", "values": [str(x) for x in sorted(v for v in diffs if v is not None)]},
        {"key": "Type", "values": [str(x) for x in sorted(v for v in types if v is not None)]},
    ]
    print(f"[{uid}] docs={len(corpus_rows)} questions={len(eval_rows)} "
          f"corpus={os.path.getsize(corpus_path)/1e6:.2f}MB eval={os.path.getsize(eval_path)/1e6:.2f}MB")
    return {
        "set": uid,
        "label": f"size {size} · seed {seed}",
        "size": size, "seed": seed,
        "n_docs": len(corpus_rows),
        "n_questions": len(eval_rows),
        "corpus_file": f"corpus_{uid}.json",
        "eval_file": f"eval_{uid}.json",
        "facets": facets,
    }


def main():
    ap = argparse.ArgumentParser(description=__doc__)
    ap.add_argument("--data-dir", default=DEFAULT_DATA_DIR)
    args = ap.parse_args()
    if pq is None:
        raise SystemExit("pyarrow is required")
    manifest = []
    for size in SIZES:
        for seed in SEEDS:
            m = build_universe(args.data_dir, size, seed)
            if m:
                manifest.append(m)
    with open(os.path.join(ROOT, "sets.json"), "w", encoding="utf-8") as f:
        json.dump(manifest, f, ensure_ascii=False, indent=2)
    print("wrote sets.json:", [s["set"] for s in manifest])


if __name__ == "__main__":
    main()