bert_simpson / forgebench /code /baselines /omni_reference /omni_sched_reference.py
Ronaldo-GOAT's picture
forgebench: cropK Pixal3D fix, cammap_fb150, scoring wrappers, Omni real-image VGGT code (creds removed), FB150 + Omni 4v input tars, README
d4011d8 verified
Raw History Blame Contribute Delete
17.3 kB
#!/usr/bin/env python3
"""eval4v_300 scheduler: 4v generation for 6 methods x {toys4k300_rand, omni3d300_rand} on GPUs 4,5 ONLY,
then scoring. VRAM admission is MEASURED (sliding-window max of nvidia-smi used) + bookings of my
recently-launched jobs; never touches other processes. Restart-safe (drivers skip existing glbs;
finished shards recorded in logs/jobs.jsonl)."""
import os, sys, json, time, subprocess, collections
R = "/lp-dev/jonghoon/mv-mesh"; E = f"{R}/.debug/eval4v_300"; FB = f"{R}/.debug/forgebench_eval/fb150"
GPUS = [4, 5, 6]
UUID = {}
LIMIT_G = {4: 60.0, 5: 45.0, 6: 60.0} # 5 shared w/ VGGT agent -> conservative (2026-09-25 KST)
LIMIT = 76.0 # GB total per GPU incl. other users
FB_PIX_GATE = 52.0 # while FB150 still has to launch a Pixal lane: keep non-Pixal usage <= this
PIX_RESERVE = 40.0 # after FB150 Pixal is done: keep this free for my Pixal lane on a GPU without Pixal
WINDOW = 600 # s of memory history
PENDING_S = 300 # a job younger than this is booked at NEED (its memory may not show yet)
MAXMINE = 5 # my procs per GPU (excl. scoring)
NEED = {"pixal3d": 40, "rvg": 16, "amodal3r": 15, "hy3d2mv": 16, "cupid": 14, "ours": 22, "oursfill": 14, "score": 12}
NSH = {"pixal3d": 10, "rvg": 10, "amodal3r": 10, "hy3d2mv": 10, "cupid": 10}
DSS = ["toys4k300", "omni3d300"]
SS = "/data/mv_mesh_data/ckpt/ssflow_seedcond_v3_20260916/step_0080000.pt"
SLAT = "/lp-dev/jonghoon/mv-mesh/ckpt/slatflow_mv8_200k_4gpu_nockpt_4567_from21k_20260923/step_0032000.pt"
ENV = dict(os.environ, HF_HOME=f"{R}/hf_cache", TORCH_HOME=f"{R}/hf_cache/torch", MUJOCO_GL="egl",
PYOPENGL_PLATFORM="egl", OMP_NUM_THREADS="4", MKL_NUM_THREADS="4",
PYTORCH_CUDA_ALLOC_CONF="expandable_segments:True", MPLBACKEND="Agg", HF_HUB_DISABLE_XET="1")
def exp(ds): return f"{R}/exp_faithfulness/{ds}_rand"
def objs(ds): return [s["object"] for s in json.load(open(f"{exp(ds)}/selection.json"))["selections"]]
def gendir(m, ds): return f"{E}/ours_{ds}/runs/step_0032000/{ds}_rand/seed/glb" if m == "ours" else f"{E}/gen/{m}_{ds}"
def missing(m, ds): return [o for o in objs(ds) if not os.path.exists(f"{gendir(m, ds)}/{o}.glb")]
def cmd(m, ds, sh, n):
X = exp(ds); SEL = f"{X}/selection.json"; INP = f"{X}/inputs"; out = gendir(m, ds); os.makedirs(out, exist_ok=True)
s = ["--shard", str(sh), "--nshards", str(n)]
if m == "rvg":
return [f"{R}/envs/reconviagen/bin/python", "metrics/batch_reconviagen.py", "--selection", SEL, "--inputs", INP,
"--out", out, "--views", "4"] + s, R, {}
if m == "amodal3r":
return [f"{R}/envs/reconviagen/bin/python", "scratchpad_extbaselines/batch_amodal3r.py", "--selection", SEL,
"--inputs", INP, "--exp", X, "--out", out, "--views", "4", "--gpu", "0"] + s, R, {}
if m == "hy3d2mv": # slot-permuted inputs (random companions -> right/back/left by relative azimuth)
HX = f"{X}_hyslots"
return [f"{R}/hunyuan3d-2mv/env/bin/python", "scratchpad_extbaselines/batch_hy3d_2mv.py", "--selection",
f"{HX}/selection.json", "--inputs", f"{HX}/inputs", "--exp", X, "--out", out, "--views", "4",
"--seed", "42", "--simplify-faces", "40000"] + s, R, {}
if m == "pixal3d":
# Omni rand views are NOT object-centred (origin off image centre, median 16 px) -> look-at re-centring driver;
# Toys views are exactly centred (0 px) -> original official-MV driver.
drv = "pixal3d/batch_pixal3d_mv.py" # user decision 2026-09-25: NO re-centring, GT cameras as-is (rc driver reverted)
return [f"{R}/envs/pixal3d/bin/python", drv, "--exp", X, "--selection", SEL,
"--out", out, "--views", "4", "--resolution", "1536", "--fallback_resolution", "1024",
"--retries", "2"] + s, R, {}
if m == "cupid":
CB = f"{R}/baselines/cupid"
return [f"{CB}/env/bin/python", f"{CB}/batch_cupid_mv.py", "--exp", X, "--out", out, "--views", "4"] + s, \
f"{CB}/repo", {"ATTN_BACKEND": "flash_attn", "SPCONV_ALGO": "native"}
raise ValueError(m)
def log(msg):
line = f"{time.strftime('%H:%M:%S', time.gmtime())} {msg}"
print(line, flush=True); open(f"{E}/logs/sched.log", "a").write(line + "\n")
def jlog(rec): open(f"{E}/logs/jobs.jsonl", "a").write(json.dumps(rec) + "\n")
def smi():
q = subprocess.run(["nvidia-smi", "--query-gpu=index,uuid,memory.used", "--format=csv,noheader,nounits"],
capture_output=True, text=True).stdout
used = {}
for l in q.strip().splitlines():
i, u, m = [x.strip() for x in l.split(",")]
UUID[u] = int(i); used[int(i)] = int(m) / 1024
q = subprocess.run(["nvidia-smi", "--query-compute-apps=pid,gpu_uuid,used_memory", "--format=csv,noheader,nounits"],
capture_output=True, text=True).stdout
procs = []
for l in q.strip().splitlines():
try:
p, u, m = [x.strip() for x in l.split(",")]; procs.append((int(p), UUID.get(u, -1), int(m) / 1024))
except ValueError:
pass
return used, procs
def cmdline(p):
try: return open(f"/proc/{p}/cmdline").read().replace("\0", " ")
except OSError: return ""
def fb_pix_state():
"""'pending' while FB150 may still launch a Pixal lane (#5 or a retry); 'done' once its pixal task is gen-complete."""
lg = open(f"{FB}/logs/sched4.log").read() if os.path.exists(f"{FB}/logs/sched4.log") else ""
if os.path.exists(f"{FB}/DONE_pixal3dmv_4v") or "pixal3dmv_4v: gen complete" in lg:
return "done"
if not os.path.exists(f"/proc/{open(f'{FB}/sched4.pid').read().strip()}"):
return "done" # scheduler gone -> nothing more will be launched by it
return "pending"
def ppid(p):
try: return int(open(f"/proc/{p}/stat").read().rsplit(")", 1)[1].split()[1])
except Exception: return 0
def owner(p, mine_pids):
"""walk up the process tree; return my job pid owning p, or None"""
for _ in range(12):
if p in mine_pids: return p
p = ppid(p)
if p <= 1: return None
return None
def estimate(g, used_g, procs):
"""booked GB on GPU g: Pixal procs at >=45, my jobs at max(NEED, current), foreign procs at current,
+ non-process overhead; my jobs without a visible GPU proc yet are booked at NEED."""
mine_pids = {pid: j for pid, (j, _, gg, _) in running.items()}
est, seen, tot = 0.0, collections.defaultdict(float), 0.0
for p, gg, mm in procs:
if gg != g: continue
tot += mm
o = owner(p, mine_pids)
if "batch_pixal3d" in cmdline(p):
est += max(40.0, mm)
if o is not None: seen[o] = -1.0 # my Pixal job with a visible proc: already counted
elif o is not None: seen[o] += mm
else: est += mm
for pid, (j, _, gg, _) in running.items():
if gg != g: continue
need = NEED[j["kind"]] if j["kind"] in ("score", "oursfill") else NEED[j["m"]]
if j["m"] == "pixal3d" and j["kind"] != "score":
if seen.get(pid) != -1.0: est += 40.0 # not visible yet (loading): book it
continue
est += max(need, seen.get(pid, 0.0))
return est + max(0.0, used_g - tot)
PAUSE_UNTIL = 0 # gate lifted 2026-09-25
MY_CAP = {} # GPU4 reservation lifted 04:41 UTC # coordinator 2026-09-25 04:13 UTC: my total use on GPU 4 < ~45 GB (main needs ~35 GB free)
def mine_est(g, procs):
mine_pids = {pid: j for pid, (j, _, gg, _) in running.items()}
seen = collections.defaultdict(float)
for p, gg, mm in procs:
if gg == g:
o = owner(p, mine_pids)
if o is not None: seen[o] += mm
tot = 0.0
for pid, (j, _, gg, _) in running.items():
if gg != g: continue
need = NEED[j["kind"]] if j["kind"] in ("score", "oursfill") else NEED[j["m"]]
tot += max(need, seen.get(pid, 0.0))
return tot
# ---------------- queues ----------------
done_names = set()
if os.path.exists(f"{E}/logs/jobs.jsonl"):
for l in open(f"{E}/logs/jobs.jsonl"):
r = json.loads(l)
if r["ev"] == "end" and r.get("rc") == 0: done_names.add(r["name"])
Q = {"P": [], "O": [], "S": []}
tasks = {} # (m,ds) -> phase
for ds in DSS:
for m in ["ours", "rvg", "amodal3r", "hy3d2mv", "cupid", "pixal3d"]:
tasks[(m, ds)] = "gen"
if os.path.exists(f"{E}/eval/{m}_{ds}/SCORED"):
tasks[(m, ds)] = "done"; continue
if m == "ours":
if os.path.exists(f"{E}/DONE_ours_ours_{ds}"):
tasks[(m, ds)] = "retry" if (not missing(m, ds) or os.path.exists(f"{E}/logs/oursfill_{ds}.log")) else "gen"
if tasks[(m, ds)] == "gen": Q["O"].insert(0, dict(name=f"ours_{ds}#retry", m=m, ds=ds, kind="oursfill"))
continue
Q["O"].append(dict(name=f"ours_{ds}#0", m=m, ds=ds, kind="ours")); continue
lane = "P" if m == "pixal3d" else "O"
mis = set(missing(m, ds)); allo = objs(ds)
for i in range(NSH[m]):
if mis & set(allo[i::NSH[m]]):
Q[lane].append(dict(name=f"{m}_{ds}#{i}", m=m, ds=ds, kind="gen", shard=i, nsh=NSH[m]))
if not any(j["m"] == m and j["ds"] == ds for j in Q[lane]):
tasks[(m, ds)] = "retry" # all present -> straight to scoring via idle handler
# interleave "O" queue: ours first, then round-robin by (method, ds)
ours = [j for j in Q["O"] if j["m"] == "ours"]
rest = [j for j in Q["O"] if j["m"] != "ours"]
rr = []
for dsx in DSS: # dataset-major: finish all toys rows first; round-robin over methods within a dataset
by = collections.OrderedDict()
for j in rest:
if j["ds"] == dsx: by.setdefault(j["m"], []).append(j)
while any(by.values()):
for k in list(by):
if by[k]: rr.append(by[k].pop(0))
Q["O"] = ours + rr
# Pixal: alternate datasets
pt, po = [j for j in Q["P"] if j["ds"] == DSS[0]], [j for j in Q["P"] if j["ds"] == DSS[1]]
Q["P"] = [x for pair in zip(pt, po) for x in pair] + pt[len(po):] + po[len(pt):]
running = {} # pid -> (job, popen|None, gpu, t0)
class _Adopted:
def __init__(s, pid): s.pid = pid; s.returncode = None
def poll(s):
if os.path.exists(f"/proc/{s.pid}"): return None
s.returncode = "adopted"; return s.returncode
if os.path.exists(f"{E}/logs/jobs.jsonl"):
alive = {}
for l in open(f"{E}/logs/jobs.jsonl"):
r = json.loads(l)
if r["ev"] == "start": alive[r["pid"]] = r
else: alive.pop(r["pid"], None)
for pid, r in alive.items():
if os.path.exists(f"/proc/{pid}") and r["name"].split("#")[0] in cmdline(pid) + " " + r["name"].split("#")[0] \
and ("python" in cmdline(pid) or "bash" in cmdline(pid)):
job = dict(name=r["name"], m=r["m"], ds=r["ds"], kind=r["kind"])
running[pid] = (job, _Adopted(pid), r["gpu"], r["t"])
for q in Q.values():
q[:] = [j for j in q if j["name"] != r["name"]]
if r["kind"] == "score": tasks[(r["m"], r["ds"])] = "score"
elif r["kind"] == "retry": tasks[(r["m"], r["ds"])] = "retry"
hist = {g: collections.deque() for g in GPUS}
def launch(job, g):
m, ds = job["m"], job["ds"]
env = dict(ENV, CUDA_VISIBLE_DEVICES=str(g), EGL_DEVICE_ID=str(g))
if job["kind"] == "ours":
argv = ["bash", f"{E}/ours_combined_cell.sh", f"{g} {g}", exp(ds), f"{ds}_rand", "4", SS, SLAT,
f"{E}/ours_{ds}", f"ours_{ds}"]; cwd = E; env.pop("CUDA_VISIBLE_DEVICES")
lf = f"{E}/logs/ours_{ds}.out"
elif job["kind"] == "oursfill":
argv = ["bash", f"{E}/ours_fill.sh", ds, str(g)]; cwd = E; env.pop("CUDA_VISIBLE_DEVICES")
lf = f"{E}/logs/oursfill_{ds}.out"
elif job["kind"] == "score":
argv = ["bash", f"{E}/score4v.sh", m, ds, str(g)]; cwd = E; lf = f"{E}/logs/score_{m}_{ds}.out"
else:
sh, n = (0, 1) if job["kind"] == "retry" else (job["shard"], job["nsh"])
argv, cwd, extra = cmd(m, ds, sh, n); env.update(extra)
lf = f"{E}/logs/gen_{m}_{ds}_{job['kind']}_{sh}.log"
p = subprocess.Popen(["taskset", "-c", "0-15,32-63"] + argv, cwd=cwd, env=env, stdout=open(lf, "a"),
stderr=subprocess.STDOUT, start_new_session=True)
running[p.pid] = (job, p, g, time.time())
jlog(dict(ev="start", name=job["name"], kind=job["kind"], m=m, ds=ds, gpu=g, pid=p.pid, t=time.time()))
log(f"LAUNCH {job['name']} ({job['kind']}) gpu{g} pid{p.pid}")
def idle(m, ds):
ph = tasks[(m, ds)]
if ph == "gen":
mis = missing(m, ds)
tasks[(m, ds)] = "retry"
if mis and not (m == "ours" and os.path.exists(f"{E}/logs/oursfill_{ds}.log")):
log(f"{m}_{ds}: {len(mis)} missing after gen -> retry pass")
lane = "P" if m == "pixal3d" else "O"
Q[lane].insert(0, dict(name=f"{m}_{ds}#retry", m=m, ds=ds, kind="oursfill" if m == "ours" else "retry")); return
if tasks[(m, ds)] == "retry":
mis = missing(m, ds)
json.dump(mis, open(f"{E}/gen/{m}_{ds}.missing.json", "w"))
log(f"{m}_{ds}: gen complete, missing={len(mis)} {mis[:6]} -> score")
tasks[(m, ds)] = "score"
Q["S"].append(dict(name=f"score_{m}_{ds}", m=m, ds=ds, kind="score"))
elif tasks[(m, ds)] == "score":
tasks[(m, ds)] = "done"; log(f"{m}_{ds}: DONE (scored)")
for (m, ds), ph in list(tasks.items()):
if ph == "retry" and not any(j[0]["m"] == m and j[0]["ds"] == ds for j in running.values()): idle(m, ds)
log(f"START queues P={len(Q['P'])} O={len(Q['O'])} S={len(Q['S'])} :: O={[j['name'] for j in Q['O']][:12]}...")
open(f"{E}/sched.pid", "w").write(str(os.getpid()))
last_launch = {g: time.time() + 60 for g in GPUS} # warm-up: >=2 min of memory history before any launch
while any(Q.values()) or running:
now = time.time()
for pid in list(running):
job, p, g, t0 = running[pid]
if p.poll() is None: continue
del running[pid]
jlog(dict(ev="end", name=job["name"], kind=job["kind"], m=job["m"], ds=job["ds"], gpu=g, pid=pid,
rc=p.returncode, t=now, dur=now - t0))
log(f"END {job['name']} rc={p.returncode} {now - t0:.0f}s")
m, ds = job["m"], job["ds"]
if job["kind"] == "ours":
open(f"{E}/DONE_ours_ours_{ds}", "w").write(time.strftime("%FT%TZ", time.gmtime()))
tasks[(m, ds)] = "gen"; idle(m, ds); continue
busy = [1 for j in running.values() if j[0]["m"] == m and j[0]["ds"] == ds] + \
[1 for q in Q.values() for j in q if j["m"] == m and j["ds"] == ds]
if not busy: idle(m, ds)
used, procs = smi()
for g in GPUS:
hist[g].append((now, used[g]))
while hist[g] and hist[g][0][0] < now - WINDOW: hist[g].popleft()
fbp = fb_pix_state()
for g in GPUS:
if now - last_launch[g] < 60: continue # one launch per GPU per minute: let memory show up
# Pixal3D spikes to ~45 GB transiently: assume every resident Pixal proc may be at its peak
_pix = [(p, mm) for p, gg, mm in procs if gg == g and "batch_pixal3d" in cmdline(p)]
hmax = max(u for _, u in hist[g])
wmax = estimate(g, used[g], procs) # booking-based (replaces measured window max)
pend = 0.0
pix_here = [p for p, gg, _ in procs if gg == g and "batch_pixal3d" in cmdline(p)]
pix_mem = sum(mm for p, gg, mm in procs if gg == g and "batch_pixal3d" in cmdline(p))
mine = [j for (j, _, gg, _) in running.values() if gg == g and j["kind"] != "score"]
mcap = MY_CAP.get(g, 1e9) - mine_est(g, procs)
LIM = 55.0 if now < PAUSE_UNTIL else LIMIT_G[g] # remaining room under my own cap on this GPU
# --- scoring lane (max 2 concurrent overall) ---
if Q["S"] and sum(1 for j in running.values() if j[0]["kind"] == "score") < 2 and \
wmax + pend + NEED["score"] <= LIM and NEED["score"] <= mcap:
launch(Q["S"].pop(0), g); last_launch[g] = now; continue
# --- my Pixal lane: only once FB150 Pixal is finished, 1 Pixal (any owner) per GPU ---
if Q["P"] and fbp == "done" and not pix_here and \
not any(j["m"] in ("pixal3d", "ours") for j in mine) and wmax + pend + NEED["pixal3d"] <= LIM + 4 and NEED["pixal3d"] <= mcap:
launch(Q["P"].pop(0), g); last_launch[g] = now; continue
if not Q["O"] or len(mine) >= MAXMINE: continue
cand = None
for j in Q["O"]:
if j["m"] == "cupid" and now < PAUSE_UNTIL: continue
need = NEED[j["kind"]] if j["kind"] == "oursfill" else NEED[j["m"]]
if wmax + pend + need <= LIM and need <= mcap: cand = j; break
if cand is None: continue
ok = True
if fbp == "pending" and pix_here: # FB150 Pixal must still find <= 52 GB (non-Pixal) to launch its next lane
ok = ok and (hmax - pix_mem + pend + need <= FB_PIX_GATE)
elif fbp == "done" and Q["P"] and not pix_here and not any(jj["m"] in ("pixal3d", "ours") for jj in mine):
ok = ok and (wmax + pend + need + PIX_RESERVE <= LIM)
if ok:
Q["O"].remove(cand); launch(cand, g); last_launch[g] = now
time.sleep(20)
log("ALL DONE")