File size: 10,745 Bytes
cca5a6d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
"""Four more Image-to-Scene cases (user, 2026-10-04: "然后image2scene多找几个case放上去").

Picked from the 75 public cases for a clean reference image and a perfect (1.000) repair whose saved scene is the one the
scored run reloaded, so it can be shown in 3D. Each card shows that repair; the case explorer adds the other exported agents.
The scenes were staged and exported to GLB on serv6 (code4scene-video/scripts/stage_i2s_more.py, export_glb_more.sh),
then compressed as website/compress_glb.sh does (meshopt geometry, 512 px WebP, Image-to-Scene nodes kept).
Scores, judge views and reference images are the film kit's (kit/host/film/i2s.json, collected by kit/collect_i2s.py);
targets and operations are the testset label.json corruption contracts; the reference cameras are the public cameras.json.

Each step asserts it found what it edits.
    GLB=<dir of compressed <case>__<key>.glb> python3 scripts/i2s_more_cases.py
"""
from __future__ import annotations
import hashlib, json, os, re, shutil, sys
from pathlib import Path
import numpy as np
from bs4 import BeautifulSoup

ROOT = Path(__file__).resolve().parent.parent
KIT = Path.home() / "Downloads/code4scene-video/kit/host/film/i2s.json"
PUB = Path.home() / "Downloads/Code4Scene/benchmark/public/image-to-scene"
GLB = Path(os.environ["GLB"])
V = "20261004i2smore"
KEY = bytes.fromhex("5c2f9e71d3a84b06e1f7c29a38d05b64a7193ce8f2046db95e81c73a2df6904b")   # website/pack_models.py

# case, title, card agent, the target actors' GLB node names, the label.json corruption contract (authored_operations), a line
# on what was done. The exporter names nodes by actor label; where that differs from the label.json label, the node is the
# one whose transform differs between the ground truth and the input GLB: comp-72's StaticMeshActor_565 is SM_Umbrella_5,
# comp-51's BP_Mailbox_01_C_443 is BP_Mailbox_01 (as build_site.py's GLB_ALIAS for comp-31).
CASES = [
    ("comp-72-suburban-umbrella-pair-repair-v2", "Suburban Umbrella Pair", "opus", ["SM_Umbrella_5"],
     [{"type": "duplication", "label": "StaticMeshActor_565", "new_label": "StaticMeshActor_565_CorruptDuplicate", "delta_cm": [-300.0, 220.0, 0.0], "yaw_deg": -28.0, "factor": 0.82},
      {"type": "rotation", "label": "StaticMeshActor_565", "yaw_deg": 46.0},
      {"type": "scaling", "label": "StaticMeshActor_565", "factor": [1.28, 1.28, 1.18]}],
     "The patio umbrella turned 46° and enlarged, plus an unintended smaller duplicate."),
    ("comp-17-rosies-service-bin-reset-v3", "Rosie's Service Bin", "astra", ["SM_Trashcan3"],
     [{"type": "translation", "label": "SM_Trashcan3", "delta_cm": [260.0, 180.0, 0.0]},
      {"type": "rotation", "label": "SM_Trashcan3", "yaw_deg": 65.0},
      {"type": "scaling", "label": "SM_Trashcan3", "factor": 1.45}],
     "A trash can moved 3.2 m, turned 65° and scaled up 1.45×."),
    ("comp-18-rosies-roadside-billboard-reset-v3", "Rosie's Roadside Billboard", "astra", ["SM_Billboard02"],
     [{"type": "translation", "label": "SM_Billboard02", "delta_cm": [650.0, 400.0, 0.0]},
      {"type": "rotation", "label": "SM_Billboard02", "yaw_deg": 55.0},
      {"type": "scaling", "label": "SM_Billboard02", "factor": 1.35}],
     "The billboard moved 7.6 m, turned 55° and scaled up 1.35×."),
    ("comp-51-suburban-driveway-mailbox-recovery-v2", "Suburban Driveway Mailbox", "fable", ["BP_Mailbox_01"],
     [{"type": "removal", "label": "BP_Mailbox_01_C_443"}],
     "The driveway mailbox removed."),
]


def one(found, what):
    assert found, f"not found: {what}"
    return found


def mid(glb_name):
    return hashlib.sha256(b"c4s-model:" + glb_name.encode()).hexdigest()[:20]


def pack():
    """website/pack_models.py for the new GLBs: models/<id>.bin, XORed with the page's keystream"""
    table = {}
    for g in sorted(GLB.glob("*.glb")):
        assert any(g.name.startswith(c[0] + "__") for c in CASES), g.name
        a = np.frombuffer(g.read_bytes(), np.uint8); idx = np.arange(a.size, dtype=np.int64)
        ks = np.frombuffer(KEY, np.uint8)[idx % 32] ^ ((idx // 32) * 131 & 255).astype(np.uint8)
        dst = ROOT / "models" / f"{mid(g.name)}.bin"; dst.write_bytes((a ^ ks).tobytes())
        table[g.name] = f"models/{dst.name}?v={int(dst.stat().st_mtime)}"
    print("packed", len(table), "models")
    return table


def ref_cam(setting, case):
    v = next(v for v in json.loads((PUB / setting / case / "cameras.json").read_text())["views"] if v["name"] == "reference")
    ue = lambda p: [p[0] / 100, p[2] / 100, p[1] / 100]   # UE (x, y, z) cm -> glTF (x, z, y) m, as build_site.py
    return {"pos": ue(v["location_cm"]), "target": ue(v["target_cm"]), "hfov": v["fov_deg"]}


def site_data(models):
    p = ROOT / "data/site.js"
    S = json.loads(p.read_text().split("=", 1)[1].rstrip().rstrip(";"))
    assert not {c[0] for c in CASES} & {c["id"] for c in S["cases"]}, "already added"
    kit = {c["case"]: c for c in json.loads(KIT.read_text())}
    q = lambda rel: f"{rel}?v={V}"
    for case, title, card, targets, ops, note in CASES:
        c = kit[case]; d = f"media/i2s/{case}"
        assert all((ROOT / d / a["img"]).exists() for a in c["agents"] if a["img"]), case
        ags = [{"key": a["key"], "s": a["s"], "f1": a["f1"], "phys": a["phys"], "tp": a["tp"], "fp": a["fp"], "fn": a["fn"],
                "img": q(f"{d}/{a['img']}") if a["img"] else None, "glb": models.get(f"{case}__{a['key']}.glb")}
               for a in sorted(c["agents"], key=lambda a: -a["s"])]
        best = next(a for a in ags if a["key"] == card)
        assert best["s"] >= 0.999 and best["glb"], (case, card)
        S["cases"].append({"id": case, "track": "i2s", "title": title, "note": note, "prompt": c["task"]["prompt"], "targets": targets,
                           "duplicates": [o["new_label"] for o in ops if o["type"] == "duplication"], "corruption": ops,
                           "gt_glb": one(models.get(f"{case}__gt.glb"), f"{case} gt"), "input_glb": one(models.get(f"{case}__input.glb"), f"{case} input"),
                           "ref_cam": ref_cam(c["setting"], case), "gt": q(f"{d}/{c['gt']}"), "input": q(f"{d}/{c['input']}") if c.get("input") else None,
                           "refs": [q(f"{d}/{r}") for r in c["refs"]], "ops": c["ops"], "agents": ags})
    p.write_text("window.SITE = " + json.dumps(S) + ";\n")
    n3d = sum(1 for c in S["cases"] for a in c["agents"] if a.get("glb")) + sum(1 for c in S["cases"] for k in ("gt_glb", "input_glb") if c.get(k))
    cp = ROOT / "cases.html"; cs = cp.read_text()
    cs, n = re.subn(r"\d+ cases, \d+ scenes in 3D", f'{len(S["cases"])} cases, {n3d} scenes in 3D', cs); assert n == 1
    cs, n = re.subn(r"(data/site\.js)\?v=\w+", lambda m: f"{m.group(1)}?v={V}", cs); assert n == 1
    cp.write_text(cs)
    print("data/site.js:", len(S["cases"]), "cases ·", n3d, "scenes in 3D")
    return S


def index(S):
    p = ROOT / "index.html"
    soup = BeautifulSoup(p.read_text(), "html.parser")
    grid = one(soup.find(id="cases").find(class_="case-grid", attrs={"data-view": "i2s"}), "Image-to-Scene case grid")
    name = {k: v["name"] for k, v in S["agents"].items()}
    for case, title, card, *_ in CASES:
        c = next(c for c in S["cases"] if c["id"] == case); a = next(a for a in c["agents"] if a["key"] == card)
        grid.append(BeautifulSoup(
            f'<article class="case-card"><div class="live-scene" data-agent="{card}" data-case="{case}"><canvas aria-label="Interactive 3D scene: '
            f'{title}. Drag to orbit, scroll or use plus and minus to zoom." tabindex="0"></canvas><span class="scene-loading" role="status">'
            'Loading 3D scene…</span><span class="scene-hint">Drag to orbit · Scroll to zoom</span><button aria-label="Pause scene rotation" '
            'class="scene-motion">Ⅱ</button></div><div aria-label="Scene state" class="scene-versions"><button class="on" data-state="repair">Repair'
            '</button><button data-state="input">Corrupted input</button><button data-state="gt">Ground truth</button></div><div class="meta">'
            f'<span class="eyebrow">Image-to-Scene</span><h4>{title}</h4><p>{name[card]} · score {a["s"]:.3f}</p>'
            f'<a class="text-button" href="cases.html#{case}">Compare all 14 agents ↗</a></div></article>', "html.parser"))
    for t in soup.find_all("script", src=re.compile(r"^data/site\.js")):
        t["src"] = f"data/site.js?v={V}"
    p.write_text(str(soup))
    print("index.html:", len(grid.find_all("article", recursive=False)), "Image-to-Scene cards")


def texture_share(packed):
    """the share of a packed GLB's bytes that are embedded images"""
    a = np.frombuffer(packed.read_bytes(), np.uint8); idx = np.arange(a.size, dtype=np.int64)
    raw = (a ^ (np.frombuffer(KEY, np.uint8)[idx % 32] ^ ((idx // 32) * 131 & 255).astype(np.uint8))).tobytes()
    n = int.from_bytes(raw[12:16], "little"); g = json.loads(raw[20:20 + n])
    views = {im["bufferView"] for im in g.get("images", []) if "bufferView" in im}
    return sum(g["bufferViews"][v]["byteLength"] for v in views) / len(raw)


def progressive(S):
    """build_progressive.py's geometry-first derivative for each new card scene (repair, input, ground truth)"""
    sys.path.insert(0, str(ROOT / "scripts"))
    import build_progressive as bp
    urls = []
    for case, _, card, *_ in CASES:
        c = next(c for c in S["cases"] if c["id"] == case)
        urls += [next(a for a in c["agents"] if a["key"] == card)["glb"], c["input_glb"], c["gt_glb"]]
    # a derivative pays off only when textures are a real share of the file: the suburban scenes are ~98% geometry, so theirs
    # would duplicate 18 MB for no earlier first frame; those cards load the packed model directly, as before progressive-v1
    urls = [u for u in urls if texture_share(ROOT / u.split("?")[0]) >= 0.1]
    for u in urls:   # the converter reads its cache before the live Space, which does not have these files yet
        shutil.copyfile(ROOT / u.split("?")[0], bp.CACHE / u.split("/")[-1].split("?")[0])
    m = ROOT / "data/progressive-scenes.json"; d = json.loads(m.read_text())
    d.update(dict(bp.convert(u) for u in urls))
    m.write_text(json.dumps(d, indent=2) + "\n")
    cu = ROOT / "scripts/progressive-card-urls.json"; cl = json.loads(cu.read_text()); cu.write_text(json.dumps(cl + urls) + "\n")
    print("data/progressive-scenes.json:", len(d), "card scenes")


if __name__ == "__main__":
    if os.environ.get("ONLY") == "progressive":   # re-run the last step on an already-edited page
        progressive(json.loads((ROOT / "data/site.js").read_text().split("=", 1)[1].rstrip().rstrip(";")))
    else:
        models = pack(); S = site_data(models); index(S); progressive(S)