Spaces:
Running
Running
File size: 7,509 Bytes
96718cc | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 | """Make the page Text-to-Scene only (user, 2026-10-01: drop Image-to-Scene from the video, the site and the figures).
Edits index.html, cases data and the bar script in place. Each step asserts it found what it edits, so a re-run on an
already-edited page, or on a page whose structure has moved, stops instead of silently doing half the job.
python3 scripts/t2s_only.py
"""
from __future__ import annotations
import json, re
from pathlib import Path
from bs4 import BeautifulSoup, NavigableString
ROOT = Path(__file__).resolve().parent.parent
I2S_CASES = ("comp-31-old-industrial-pallet-bay-repair-v2", "comp-61-new-york-mailbox-pair-repair-v2")
def one(found, what):
assert found is not None, f"not found: {what}"
return found
def index():
p = ROOT / "index.html"
soup = BeautifulSoup(p.read_text(), "html.parser")
# head
one(soup.title, "title").string = "Code4Scene: Benchmarking Coding Agents for Constructing 3D Scenes"
one(soup.find("meta", attrs={"name": "description"}), "meta description")["content"] = (
"A benchmark that scores the engine-native 3D scenes coding agents build in Unreal Engine from open-ended descriptions.")
# intro
intro = one(soup.find(id="top"), "intro")
h1 = one(intro.find("h1"), "intro h1")
h1.clear(); h1.append("Constructing"); h1.append(soup.new_tag("br")); em = soup.new_tag("em"); em.string = "3D Scenes with Code"; h1.append(em)
one(intro.find(class_="intro-text"), "intro text").string = (
"Benchmarking coding agents that turn open-ended scene descriptions into engine-native 3D scenes in Unreal Engine. "
"Measuring spatial reasoning, task fulfillment and physical validity.")
cap = one(intro.find(class_="featured-video-caption"), "video caption").find("span")
one(cap, "video caption span").string = "Construction and evaluation"
# the teaser is v31 (no editing subtitle on its end card), under a new name so cached copies of the old cut are not reused
for src in intro.find_all("source"):
src["src"] = "media/teaser-v31.mp4"
for a in intro.find_all("a", href="media/teaser.mp4"):
a["href"] = "media/teaser-v31.mp4"
# task format: the construction example alone, without a one-tab tab bar
tf = one(soup.find(id="task-format"), "task format")
one(tf.find(id="task-i2s"), "task-i2s").decompose()
one(tf.find(class_="task-tabs"), "task tabs").decompose()
# leaderboard: Text-to-Scene only (the overall score is half Image-to-Scene)
lb = one(soup.find(id="leaderboard"), "leaderboard")
one(lb.find(class_="lb-tabs"), "leaderboard tabs").decompose()
for el in lb.find_all(attrs={"data-view": True}):
if el.get("data-view") in ("overall", "i2s"):
el.decompose()
else:
st = (el.get("style") or "").replace("display:none", "").strip(" ;")
if st: el["style"] = st
else: del el["style"]
desc = one(lb.find(class_="section-desc"), "leaderboard desc")
desc.string = "14 coding-agent configurations · paper results on the original 20 public construction cases."
sub = one(lb.find(class_="figure-subtitle"), "pareto subtitle")
sub.string = "Quality against the cost of one case · original 20-case public evaluation"
for t in lb.find_all(class_="figure-caption"):
t.string = re.sub(r"\s*Cost is the mean USD per case; ov.*$", " Cost is the mean USD per case.", t.get_text())
# cases: the four construction cases
cs = one(soup.find(id="cases"), "cases")
removed = 0
for scene in cs.find_all(attrs={"data-case": True}):
if scene["data-case"] in I2S_CASES:
card = scene
while card.parent is not None and card.parent.name != "body" and card.parent.get("id") != "cases" and \
not any(c for c in (card.parent.get("class") or []) if "grid" in c):
card = card.parent
card.decompose(); removed += 1
assert removed == 2, removed
for p_ in cs.find_all("p"):
if "Compare repairs" in p_.get_text():
p_.string = "Drag any scene to inspect the saved 3D output, or open a case to explore every agent."
# takeaways: Spatial Composition only
tk = one(soup.find(id="takeaways"), "takeaways")
kept = 0
for card in tk.find_all(class_="tk"):
h = " ".join(one(card.find("h3"), "takeaway h3").get_text(" ").split())
if h.startswith("Spatial Composition remains"):
kept += 1
for n in card.find_all(class_="n"):
n.string = "Takeaway"
else:
card.decompose()
assert kept == 1, kept
for h in tk.find_all("h2"):
if h.get_text(strip=True) == "Takeaways":
h.string = "Takeaway"
# the case-count strip: construction counts only
av = one(soup.find(class_="availability"), "availability strip")
spans = av.find_all("span", recursive=False); assert len(spans) == 2
spans[0].b.string = "160"; spans[0].b.next_sibling.replace_with(" construction cases in the full benchmark target, including private cases")
spans[1].b.string = "129"; spans[1].b.next_sibling.replace_with(" public construction cases")
# nav label
for a in soup.find_all("a"):
if a.get_text(strip=True) == "Takeaways":
a.string = "Takeaway"
p.write_text(str(soup))
left = re.findall(r"(?i)image-to-scene|\bi2s\b|repair|corrupted", p.read_text())
print("index.html: remaining editing mentions:", len(left))
def bars():
p = ROOT / "assets/leaderboard-bars.js"
s = p.read_text()
m = re.search(r"const ALL = (\{.*?\});\n", s, re.S)
data = json.loads(one(m, "const ALL").group(1))
assert set(data) >= {"t2s"}, data.keys()
s = s[:m.start(1)] + json.dumps({"t2s": data["t2s"]}) + s[m.end(1):]
p.write_text(s)
print("leaderboard-bars.js: views", list(json.loads(re.search(r"const ALL = (\{.*?\});\n", s, re.S).group(1))))
def site_data():
p = ROOT / "data/site.js"
S = json.loads(p.read_text().split("=", 1)[1].rstrip().rstrip(";"))
before = len(S["cases"])
S["cases"] = [c for c in S["cases"] if c["track"] == "t2s"]
p.write_text("window.SITE = " + json.dumps(S) + ";\n")
print("data/site.js: cases", before, "->", len(S["cases"]))
# the cases page states the counts in its header
n3d = sum(1 for c in S["cases"] for a in c["agents"] if a.get("glb")) + sum(1 for c in S["cases"] for k in ("gt_glb", "input_glb") if c.get(k))
cp = ROOT / "cases.html"; cs = cp.read_text()
cs = re.sub(r"\d+ cases, \d+ scenes in 3D", f'{len(S["cases"])} cases, {n3d} scenes in 3D', cs); cp.write_text(cs)
# the packed model files of the two Image-to-Scene cases (repairs, ground truth, corrupted input), from data/site.js before this edit
I2S_MODELS = ["38071c69a86d174dc55f.bin", "46734878f740aae816d8.bin", "486b4742affe835723a3.bin", "66da6868d20ebb100197.bin", "6afbbaec6b7a4271b21f.bin", "7daf8107bb630cffdc18.bin", "c79715b9e6967c4f6c90.bin", "db5c0c5d2b78f9e47bb7.bin", "e6a6333939ccfe57d584.bin", "e9177d0cbf727df88100.bin", "f2b2b285e27a26ee6780.bin", "f3648a5ad9887ac0937d.bin"]
def progressive():
p = ROOT / "data/progressive-scenes.json"
d = json.loads(p.read_text()); n0 = len(d)
d = {k: v for k, v in d.items() if k not in I2S_MODELS}
p.write_text(json.dumps(d, indent=2) + "\n")
print("data/progressive-scenes.json:", n0, "->", len(d))
if __name__ == "__main__":
index(); bars(); site_data(); progressive()
|