forgebench: cropK Pixal3D fix, cammap_fb150, scoring wrappers, Omni real-image VGGT code (creds removed), FB150 + Omni 4v input tars, README
d4011d8 verified Download forgebench/code/baselines/omni_reference/omni_sched_reference.py from Ronaldo-GOAT/bert_simpson: direct link, hf CLI and curl.
- Browser
- Download file 17.3 kB
-
https://huggingface.co/Ronaldo-GOAT/bert_simpson/resolve/main/forgebench/code/baselines/omni_reference/omni_sched_reference.py
- Command line
-
hf download hf://Ronaldo-GOAT/bert_simpson/forgebench/code/baselines/omni_reference/omni_sched_reference.py
-
curl -L -o omni_sched_reference.py https://huggingface.co/Ronaldo-GOAT/bert_simpson/resolve/main/forgebench/code/baselines/omni_reference/omni_sched_reference.py
17.3 kB
| #!/usr/bin/env python3 | |
| """eval4v_300 scheduler: 4v generation for 6 methods x {toys4k300_rand, omni3d300_rand} on GPUs 4,5 ONLY, | |
| then scoring. VRAM admission is MEASURED (sliding-window max of nvidia-smi used) + bookings of my | |
| recently-launched jobs; never touches other processes. Restart-safe (drivers skip existing glbs; | |
| finished shards recorded in logs/jobs.jsonl).""" | |
| import os, sys, json, time, subprocess, collections | |
| R = "/lp-dev/jonghoon/mv-mesh"; E = f"{R}/.debug/eval4v_300"; FB = f"{R}/.debug/forgebench_eval/fb150" | |
| GPUS = [4, 5, 6] | |
| UUID = {} | |
| LIMIT_G = {4: 60.0, 5: 45.0, 6: 60.0} # 5 shared w/ VGGT agent -> conservative (2026-09-25 KST) | |
| LIMIT = 76.0 # GB total per GPU incl. other users | |
| FB_PIX_GATE = 52.0 # while FB150 still has to launch a Pixal lane: keep non-Pixal usage <= this | |
| PIX_RESERVE = 40.0 # after FB150 Pixal is done: keep this free for my Pixal lane on a GPU without Pixal | |
| WINDOW = 600 # s of memory history | |
| PENDING_S = 300 # a job younger than this is booked at NEED (its memory may not show yet) | |
| MAXMINE = 5 # my procs per GPU (excl. scoring) | |
| NEED = {"pixal3d": 40, "rvg": 16, "amodal3r": 15, "hy3d2mv": 16, "cupid": 14, "ours": 22, "oursfill": 14, "score": 12} | |
| NSH = {"pixal3d": 10, "rvg": 10, "amodal3r": 10, "hy3d2mv": 10, "cupid": 10} | |
| DSS = ["toys4k300", "omni3d300"] | |
| SS = "/data/mv_mesh_data/ckpt/ssflow_seedcond_v3_20260916/step_0080000.pt" | |
| SLAT = "/lp-dev/jonghoon/mv-mesh/ckpt/slatflow_mv8_200k_4gpu_nockpt_4567_from21k_20260923/step_0032000.pt" | |
| ENV = dict(os.environ, HF_HOME=f"{R}/hf_cache", TORCH_HOME=f"{R}/hf_cache/torch", MUJOCO_GL="egl", | |
| PYOPENGL_PLATFORM="egl", OMP_NUM_THREADS="4", MKL_NUM_THREADS="4", | |
| PYTORCH_CUDA_ALLOC_CONF="expandable_segments:True", MPLBACKEND="Agg", HF_HUB_DISABLE_XET="1") | |
| def exp(ds): return f"{R}/exp_faithfulness/{ds}_rand" | |
| def objs(ds): return [s["object"] for s in json.load(open(f"{exp(ds)}/selection.json"))["selections"]] | |
| def gendir(m, ds): return f"{E}/ours_{ds}/runs/step_0032000/{ds}_rand/seed/glb" if m == "ours" else f"{E}/gen/{m}_{ds}" | |
| def missing(m, ds): return [o for o in objs(ds) if not os.path.exists(f"{gendir(m, ds)}/{o}.glb")] | |
| def cmd(m, ds, sh, n): | |
| X = exp(ds); SEL = f"{X}/selection.json"; INP = f"{X}/inputs"; out = gendir(m, ds); os.makedirs(out, exist_ok=True) | |
| s = ["--shard", str(sh), "--nshards", str(n)] | |
| if m == "rvg": | |
| return [f"{R}/envs/reconviagen/bin/python", "metrics/batch_reconviagen.py", "--selection", SEL, "--inputs", INP, | |
| "--out", out, "--views", "4"] + s, R, {} | |
| if m == "amodal3r": | |
| return [f"{R}/envs/reconviagen/bin/python", "scratchpad_extbaselines/batch_amodal3r.py", "--selection", SEL, | |
| "--inputs", INP, "--exp", X, "--out", out, "--views", "4", "--gpu", "0"] + s, R, {} | |
| if m == "hy3d2mv": # slot-permuted inputs (random companions -> right/back/left by relative azimuth) | |
| HX = f"{X}_hyslots" | |
| return [f"{R}/hunyuan3d-2mv/env/bin/python", "scratchpad_extbaselines/batch_hy3d_2mv.py", "--selection", | |
| f"{HX}/selection.json", "--inputs", f"{HX}/inputs", "--exp", X, "--out", out, "--views", "4", | |
| "--seed", "42", "--simplify-faces", "40000"] + s, R, {} | |
| if m == "pixal3d": | |
| # Omni rand views are NOT object-centred (origin off image centre, median 16 px) -> look-at re-centring driver; | |
| # Toys views are exactly centred (0 px) -> original official-MV driver. | |
| drv = "pixal3d/batch_pixal3d_mv.py" # user decision 2026-09-25: NO re-centring, GT cameras as-is (rc driver reverted) | |
| return [f"{R}/envs/pixal3d/bin/python", drv, "--exp", X, "--selection", SEL, | |
| "--out", out, "--views", "4", "--resolution", "1536", "--fallback_resolution", "1024", | |
| "--retries", "2"] + s, R, {} | |
| if m == "cupid": | |
| CB = f"{R}/baselines/cupid" | |
| return [f"{CB}/env/bin/python", f"{CB}/batch_cupid_mv.py", "--exp", X, "--out", out, "--views", "4"] + s, \ | |
| f"{CB}/repo", {"ATTN_BACKEND": "flash_attn", "SPCONV_ALGO": "native"} | |
| raise ValueError(m) | |
| def log(msg): | |
| line = f"{time.strftime('%H:%M:%S', time.gmtime())} {msg}" | |
| print(line, flush=True); open(f"{E}/logs/sched.log", "a").write(line + "\n") | |
| def jlog(rec): open(f"{E}/logs/jobs.jsonl", "a").write(json.dumps(rec) + "\n") | |
| def smi(): | |
| q = subprocess.run(["nvidia-smi", "--query-gpu=index,uuid,memory.used", "--format=csv,noheader,nounits"], | |
| capture_output=True, text=True).stdout | |
| used = {} | |
| for l in q.strip().splitlines(): | |
| i, u, m = [x.strip() for x in l.split(",")] | |
| UUID[u] = int(i); used[int(i)] = int(m) / 1024 | |
| q = subprocess.run(["nvidia-smi", "--query-compute-apps=pid,gpu_uuid,used_memory", "--format=csv,noheader,nounits"], | |
| capture_output=True, text=True).stdout | |
| procs = [] | |
| for l in q.strip().splitlines(): | |
| try: | |
| p, u, m = [x.strip() for x in l.split(",")]; procs.append((int(p), UUID.get(u, -1), int(m) / 1024)) | |
| except ValueError: | |
| pass | |
| return used, procs | |
| def cmdline(p): | |
| try: return open(f"/proc/{p}/cmdline").read().replace("\0", " ") | |
| except OSError: return "" | |
| def fb_pix_state(): | |
| """'pending' while FB150 may still launch a Pixal lane (#5 or a retry); 'done' once its pixal task is gen-complete.""" | |
| lg = open(f"{FB}/logs/sched4.log").read() if os.path.exists(f"{FB}/logs/sched4.log") else "" | |
| if os.path.exists(f"{FB}/DONE_pixal3dmv_4v") or "pixal3dmv_4v: gen complete" in lg: | |
| return "done" | |
| if not os.path.exists(f"/proc/{open(f'{FB}/sched4.pid').read().strip()}"): | |
| return "done" # scheduler gone -> nothing more will be launched by it | |
| return "pending" | |
| def ppid(p): | |
| try: return int(open(f"/proc/{p}/stat").read().rsplit(")", 1)[1].split()[1]) | |
| except Exception: return 0 | |
| def owner(p, mine_pids): | |
| """walk up the process tree; return my job pid owning p, or None""" | |
| for _ in range(12): | |
| if p in mine_pids: return p | |
| p = ppid(p) | |
| if p <= 1: return None | |
| return None | |
| def estimate(g, used_g, procs): | |
| """booked GB on GPU g: Pixal procs at >=45, my jobs at max(NEED, current), foreign procs at current, | |
| + non-process overhead; my jobs without a visible GPU proc yet are booked at NEED.""" | |
| mine_pids = {pid: j for pid, (j, _, gg, _) in running.items()} | |
| est, seen, tot = 0.0, collections.defaultdict(float), 0.0 | |
| for p, gg, mm in procs: | |
| if gg != g: continue | |
| tot += mm | |
| o = owner(p, mine_pids) | |
| if "batch_pixal3d" in cmdline(p): | |
| est += max(40.0, mm) | |
| if o is not None: seen[o] = -1.0 # my Pixal job with a visible proc: already counted | |
| elif o is not None: seen[o] += mm | |
| else: est += mm | |
| for pid, (j, _, gg, _) in running.items(): | |
| if gg != g: continue | |
| need = NEED[j["kind"]] if j["kind"] in ("score", "oursfill") else NEED[j["m"]] | |
| if j["m"] == "pixal3d" and j["kind"] != "score": | |
| if seen.get(pid) != -1.0: est += 40.0 # not visible yet (loading): book it | |
| continue | |
| est += max(need, seen.get(pid, 0.0)) | |
| return est + max(0.0, used_g - tot) | |
| PAUSE_UNTIL = 0 # gate lifted 2026-09-25 | |
| MY_CAP = {} # GPU4 reservation lifted 04:41 UTC # coordinator 2026-09-25 04:13 UTC: my total use on GPU 4 < ~45 GB (main needs ~35 GB free) | |
| def mine_est(g, procs): | |
| mine_pids = {pid: j for pid, (j, _, gg, _) in running.items()} | |
| seen = collections.defaultdict(float) | |
| for p, gg, mm in procs: | |
| if gg == g: | |
| o = owner(p, mine_pids) | |
| if o is not None: seen[o] += mm | |
| tot = 0.0 | |
| for pid, (j, _, gg, _) in running.items(): | |
| if gg != g: continue | |
| need = NEED[j["kind"]] if j["kind"] in ("score", "oursfill") else NEED[j["m"]] | |
| tot += max(need, seen.get(pid, 0.0)) | |
| return tot | |
| # ---------------- queues ---------------- | |
| done_names = set() | |
| if os.path.exists(f"{E}/logs/jobs.jsonl"): | |
| for l in open(f"{E}/logs/jobs.jsonl"): | |
| r = json.loads(l) | |
| if r["ev"] == "end" and r.get("rc") == 0: done_names.add(r["name"]) | |
| Q = {"P": [], "O": [], "S": []} | |
| tasks = {} # (m,ds) -> phase | |
| for ds in DSS: | |
| for m in ["ours", "rvg", "amodal3r", "hy3d2mv", "cupid", "pixal3d"]: | |
| tasks[(m, ds)] = "gen" | |
| if os.path.exists(f"{E}/eval/{m}_{ds}/SCORED"): | |
| tasks[(m, ds)] = "done"; continue | |
| if m == "ours": | |
| if os.path.exists(f"{E}/DONE_ours_ours_{ds}"): | |
| tasks[(m, ds)] = "retry" if (not missing(m, ds) or os.path.exists(f"{E}/logs/oursfill_{ds}.log")) else "gen" | |
| if tasks[(m, ds)] == "gen": Q["O"].insert(0, dict(name=f"ours_{ds}#retry", m=m, ds=ds, kind="oursfill")) | |
| continue | |
| Q["O"].append(dict(name=f"ours_{ds}#0", m=m, ds=ds, kind="ours")); continue | |
| lane = "P" if m == "pixal3d" else "O" | |
| mis = set(missing(m, ds)); allo = objs(ds) | |
| for i in range(NSH[m]): | |
| if mis & set(allo[i::NSH[m]]): | |
| Q[lane].append(dict(name=f"{m}_{ds}#{i}", m=m, ds=ds, kind="gen", shard=i, nsh=NSH[m])) | |
| if not any(j["m"] == m and j["ds"] == ds for j in Q[lane]): | |
| tasks[(m, ds)] = "retry" # all present -> straight to scoring via idle handler | |
| # interleave "O" queue: ours first, then round-robin by (method, ds) | |
| ours = [j for j in Q["O"] if j["m"] == "ours"] | |
| rest = [j for j in Q["O"] if j["m"] != "ours"] | |
| rr = [] | |
| for dsx in DSS: # dataset-major: finish all toys rows first; round-robin over methods within a dataset | |
| by = collections.OrderedDict() | |
| for j in rest: | |
| if j["ds"] == dsx: by.setdefault(j["m"], []).append(j) | |
| while any(by.values()): | |
| for k in list(by): | |
| if by[k]: rr.append(by[k].pop(0)) | |
| Q["O"] = ours + rr | |
| # Pixal: alternate datasets | |
| pt, po = [j for j in Q["P"] if j["ds"] == DSS[0]], [j for j in Q["P"] if j["ds"] == DSS[1]] | |
| Q["P"] = [x for pair in zip(pt, po) for x in pair] + pt[len(po):] + po[len(pt):] | |
| running = {} # pid -> (job, popen|None, gpu, t0) | |
| class _Adopted: | |
| def __init__(s, pid): s.pid = pid; s.returncode = None | |
| def poll(s): | |
| if os.path.exists(f"/proc/{s.pid}"): return None | |
| s.returncode = "adopted"; return s.returncode | |
| if os.path.exists(f"{E}/logs/jobs.jsonl"): | |
| alive = {} | |
| for l in open(f"{E}/logs/jobs.jsonl"): | |
| r = json.loads(l) | |
| if r["ev"] == "start": alive[r["pid"]] = r | |
| else: alive.pop(r["pid"], None) | |
| for pid, r in alive.items(): | |
| if os.path.exists(f"/proc/{pid}") and r["name"].split("#")[0] in cmdline(pid) + " " + r["name"].split("#")[0] \ | |
| and ("python" in cmdline(pid) or "bash" in cmdline(pid)): | |
| job = dict(name=r["name"], m=r["m"], ds=r["ds"], kind=r["kind"]) | |
| running[pid] = (job, _Adopted(pid), r["gpu"], r["t"]) | |
| for q in Q.values(): | |
| q[:] = [j for j in q if j["name"] != r["name"]] | |
| if r["kind"] == "score": tasks[(r["m"], r["ds"])] = "score" | |
| elif r["kind"] == "retry": tasks[(r["m"], r["ds"])] = "retry" | |
| hist = {g: collections.deque() for g in GPUS} | |
| def launch(job, g): | |
| m, ds = job["m"], job["ds"] | |
| env = dict(ENV, CUDA_VISIBLE_DEVICES=str(g), EGL_DEVICE_ID=str(g)) | |
| if job["kind"] == "ours": | |
| argv = ["bash", f"{E}/ours_combined_cell.sh", f"{g} {g}", exp(ds), f"{ds}_rand", "4", SS, SLAT, | |
| f"{E}/ours_{ds}", f"ours_{ds}"]; cwd = E; env.pop("CUDA_VISIBLE_DEVICES") | |
| lf = f"{E}/logs/ours_{ds}.out" | |
| elif job["kind"] == "oursfill": | |
| argv = ["bash", f"{E}/ours_fill.sh", ds, str(g)]; cwd = E; env.pop("CUDA_VISIBLE_DEVICES") | |
| lf = f"{E}/logs/oursfill_{ds}.out" | |
| elif job["kind"] == "score": | |
| argv = ["bash", f"{E}/score4v.sh", m, ds, str(g)]; cwd = E; lf = f"{E}/logs/score_{m}_{ds}.out" | |
| else: | |
| sh, n = (0, 1) if job["kind"] == "retry" else (job["shard"], job["nsh"]) | |
| argv, cwd, extra = cmd(m, ds, sh, n); env.update(extra) | |
| lf = f"{E}/logs/gen_{m}_{ds}_{job['kind']}_{sh}.log" | |
| p = subprocess.Popen(["taskset", "-c", "0-15,32-63"] + argv, cwd=cwd, env=env, stdout=open(lf, "a"), | |
| stderr=subprocess.STDOUT, start_new_session=True) | |
| running[p.pid] = (job, p, g, time.time()) | |
| jlog(dict(ev="start", name=job["name"], kind=job["kind"], m=m, ds=ds, gpu=g, pid=p.pid, t=time.time())) | |
| log(f"LAUNCH {job['name']} ({job['kind']}) gpu{g} pid{p.pid}") | |
| def idle(m, ds): | |
| ph = tasks[(m, ds)] | |
| if ph == "gen": | |
| mis = missing(m, ds) | |
| tasks[(m, ds)] = "retry" | |
| if mis and not (m == "ours" and os.path.exists(f"{E}/logs/oursfill_{ds}.log")): | |
| log(f"{m}_{ds}: {len(mis)} missing after gen -> retry pass") | |
| lane = "P" if m == "pixal3d" else "O" | |
| Q[lane].insert(0, dict(name=f"{m}_{ds}#retry", m=m, ds=ds, kind="oursfill" if m == "ours" else "retry")); return | |
| if tasks[(m, ds)] == "retry": | |
| mis = missing(m, ds) | |
| json.dump(mis, open(f"{E}/gen/{m}_{ds}.missing.json", "w")) | |
| log(f"{m}_{ds}: gen complete, missing={len(mis)} {mis[:6]} -> score") | |
| tasks[(m, ds)] = "score" | |
| Q["S"].append(dict(name=f"score_{m}_{ds}", m=m, ds=ds, kind="score")) | |
| elif tasks[(m, ds)] == "score": | |
| tasks[(m, ds)] = "done"; log(f"{m}_{ds}: DONE (scored)") | |
| for (m, ds), ph in list(tasks.items()): | |
| if ph == "retry" and not any(j[0]["m"] == m and j[0]["ds"] == ds for j in running.values()): idle(m, ds) | |
| log(f"START queues P={len(Q['P'])} O={len(Q['O'])} S={len(Q['S'])} :: O={[j['name'] for j in Q['O']][:12]}...") | |
| open(f"{E}/sched.pid", "w").write(str(os.getpid())) | |
| last_launch = {g: time.time() + 60 for g in GPUS} # warm-up: >=2 min of memory history before any launch | |
| while any(Q.values()) or running: | |
| now = time.time() | |
| for pid in list(running): | |
| job, p, g, t0 = running[pid] | |
| if p.poll() is None: continue | |
| del running[pid] | |
| jlog(dict(ev="end", name=job["name"], kind=job["kind"], m=job["m"], ds=job["ds"], gpu=g, pid=pid, | |
| rc=p.returncode, t=now, dur=now - t0)) | |
| log(f"END {job['name']} rc={p.returncode} {now - t0:.0f}s") | |
| m, ds = job["m"], job["ds"] | |
| if job["kind"] == "ours": | |
| open(f"{E}/DONE_ours_ours_{ds}", "w").write(time.strftime("%FT%TZ", time.gmtime())) | |
| tasks[(m, ds)] = "gen"; idle(m, ds); continue | |
| busy = [1 for j in running.values() if j[0]["m"] == m and j[0]["ds"] == ds] + \ | |
| [1 for q in Q.values() for j in q if j["m"] == m and j["ds"] == ds] | |
| if not busy: idle(m, ds) | |
| used, procs = smi() | |
| for g in GPUS: | |
| hist[g].append((now, used[g])) | |
| while hist[g] and hist[g][0][0] < now - WINDOW: hist[g].popleft() | |
| fbp = fb_pix_state() | |
| for g in GPUS: | |
| if now - last_launch[g] < 60: continue # one launch per GPU per minute: let memory show up | |
| # Pixal3D spikes to ~45 GB transiently: assume every resident Pixal proc may be at its peak | |
| _pix = [(p, mm) for p, gg, mm in procs if gg == g and "batch_pixal3d" in cmdline(p)] | |
| hmax = max(u for _, u in hist[g]) | |
| wmax = estimate(g, used[g], procs) # booking-based (replaces measured window max) | |
| pend = 0.0 | |
| pix_here = [p for p, gg, _ in procs if gg == g and "batch_pixal3d" in cmdline(p)] | |
| pix_mem = sum(mm for p, gg, mm in procs if gg == g and "batch_pixal3d" in cmdline(p)) | |
| mine = [j for (j, _, gg, _) in running.values() if gg == g and j["kind"] != "score"] | |
| mcap = MY_CAP.get(g, 1e9) - mine_est(g, procs) | |
| LIM = 55.0 if now < PAUSE_UNTIL else LIMIT_G[g] # remaining room under my own cap on this GPU | |
| # --- scoring lane (max 2 concurrent overall) --- | |
| if Q["S"] and sum(1 for j in running.values() if j[0]["kind"] == "score") < 2 and \ | |
| wmax + pend + NEED["score"] <= LIM and NEED["score"] <= mcap: | |
| launch(Q["S"].pop(0), g); last_launch[g] = now; continue | |
| # --- my Pixal lane: only once FB150 Pixal is finished, 1 Pixal (any owner) per GPU --- | |
| if Q["P"] and fbp == "done" and not pix_here and \ | |
| not any(j["m"] in ("pixal3d", "ours") for j in mine) and wmax + pend + NEED["pixal3d"] <= LIM + 4 and NEED["pixal3d"] <= mcap: | |
| launch(Q["P"].pop(0), g); last_launch[g] = now; continue | |
| if not Q["O"] or len(mine) >= MAXMINE: continue | |
| cand = None | |
| for j in Q["O"]: | |
| if j["m"] == "cupid" and now < PAUSE_UNTIL: continue | |
| need = NEED[j["kind"]] if j["kind"] == "oursfill" else NEED[j["m"]] | |
| if wmax + pend + need <= LIM and need <= mcap: cand = j; break | |
| if cand is None: continue | |
| ok = True | |
| if fbp == "pending" and pix_here: # FB150 Pixal must still find <= 52 GB (non-Pixal) to launch its next lane | |
| ok = ok and (hmax - pix_mem + pend + need <= FB_PIX_GATE) | |
| elif fbp == "done" and Q["P"] and not pix_here and not any(jj["m"] in ("pixal3d", "ours") for jj in mine): | |
| ok = ok and (wmax + pend + need + PIX_RESERVE <= LIM) | |
| if ok: | |
| Q["O"].remove(cand); launch(cand, g); last_launch[g] = now | |
| time.sleep(20) | |
| log("ALL DONE") | |