opus-max-record / harness /scripts /diversify_paths.py
simonycl's picture
Upload folder using huggingface_hub
b2ebc95 verified
Raw
History Blame Contribute Delete
3.72 kB
#!/usr/bin/env python3
"""Path-diversify an SFT corpus.
99% of the converted trajectories live at /testbed, because that is where SWE-bench-style
container images put the repo. That is correct for swe-bench-verified (all 500 tasks) but it
would teach an unconditional "/testbed" reflex, which is wrong for terminal-bench-2 (whose
tasks live at /app and name their paths in the prompt).
For a fraction of rows this rewrites /testbed to another absolute root, consistently across
every message, tool argument and observation in the trajectory, AND states the location in the
user turn. Rows left alone keep /testbed with a bare prompt. The pair teaches:
prompt names a location -> work there
prompt names nothing -> the repo is at /testbed
Nothing is invented: only a path string is substituted, and the added sentence is true of the
rewritten trajectory.
"""
import argparse
import glob
import json
import os
import random
import re
ROOTS = [
"/app",
"/app/{repo}",
"/workspace",
"/workspace/{repo}",
"/srv/{repo}",
"/opt/{repo}",
"/home/user/{repo}",
"/code",
"/repo",
"/project/{repo}",
]
SAY = [
"The project is at {root}.",
"The repository you need to work on is checked out at {root}.",
"Work in {root}.",
"You will find the code in {root}.",
"The codebase lives at {root} — use absolute paths under it.",
]
REPO_WORDS = [
"src", "pkg", "core", "lib", "service", "backend", "engine", "toolkit",
"app", "server", "client", "runtime", "sdk", "platform",
]
def sub_all(obj, old, new):
if isinstance(obj, str):
return obj.replace(old, new)
if isinstance(obj, list):
return [sub_all(x, old, new) for x in obj]
if isinstance(obj, dict):
return {k: sub_all(v, old, new) for k, v in obj.items()}
return obj
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--src", required=True)
ap.add_argument("--out", required=True)
ap.add_argument("--p", type=float, default=0.30, help="fraction of rows to relocate")
ap.add_argument("--seed", type=int, default=17)
args = ap.parse_args()
rng = random.Random(args.seed)
import pyarrow.parquet as pq
from datasets import Dataset
rows = []
n_moved = 0
for f in sorted(glob.glob(args.src + "/*.parquet")):
for batch in pq.ParquetFile(f).iter_batches(batch_size=256):
for row in batch.to_pylist():
msgs = row["messages"]
blob = json.dumps(msgs)
if "/testbed" in blob and rng.random() < args.p:
repo = f"{rng.choice(REPO_WORDS)}{rng.choice(['', '-' + rng.choice(REPO_WORDS), '_' + str(rng.randint(2, 99))])}"
root = rng.choice(ROOTS).format(repo=repo)
msgs = sub_all(msgs, "/testbed", root)
# state the location in the first user turn
for m in msgs:
if m["role"] == "user":
m["content"] = rng.choice(SAY).format(root=root) + "\n\n" + (m.get("content") or "")
break
n_moved += 1
row = {**row, "messages": msgs, "source": row["source"] + "+moved"}
rows.append(row)
rng.shuffle(rows)
os.makedirs(args.out, exist_ok=True)
ds = Dataset.from_list(rows)
shards = max(1, len(rows) // 20000 + 1)
for i in range(shards):
ds.shard(num_shards=shards, index=i).to_parquet(
os.path.join(args.out, f"train-{i:05d}-of-{shards:05d}.parquet")
)
print(f"wrote {len(rows)} rows ({n_moved} relocated) -> {args.out}")
if __name__ == "__main__":
main()