lingbot-map-3d / smoke_test.py
oualidD4's picture
Community Gradio ZeroGPU demo of LingBot-Map streaming 3D reconstruction
22c23b5 verified
Raw
History Blame Contribute Delete
8.09 kB
"""Local verification + Phase-1 CLI for the LingBot-Map pipeline.
Modes
-----
--selfcheck Validate imports + that the REAL GCTStream / exporter signatures
match what pipeline.py calls. No model build. CPU, fast.
--synthetic [--out P] Fabricate a correctly-shaped vis-dict and export a GLB. Verifies
the whole export path (predictions_to_glb → .glb) on CPU, no GPU,
no checkpoint. This is the part the broken upstream Space got wrong.
(real run) --ckpt PATH (--image_folder DIR | --video FILE) [--out P]
Full inference → GLB. Needs a CUDA GPU + the checkpoint; run this on
the Space hardware or a GPU box (NOT this Mac).
Examples
--------
python smoke_test.py --selfcheck
python smoke_test.py --synthetic --out /tmp/synthetic.glb
python smoke_test.py --ckpt lingbot-map.pt --image_folder ../upstream/example/... --out /tmp/real.glb
"""
import argparse
import glob
import os
import sys
# --------------------------------------------------------------------------------------
# --selfcheck : compare the real upstream signatures against what pipeline.py relies on
# --------------------------------------------------------------------------------------
def cmd_selfcheck() -> int:
import inspect
ok = True
def check(name, params, needed):
nonlocal ok
missing = sorted(needed - set(params))
if missing:
ok = False
print(f" ✗ {name}: MISSING params {missing}")
else:
print(f" ✓ {name}: all required params present")
print("[selfcheck] importing pipeline (exercises every non-lazy import) ...")
import pipeline as P # noqa: F401
print(" ✓ import pipeline OK")
from lingbot_map.models.gct_stream import GCTStream
from lingbot_map.vis.glb_export import predictions_to_glb
check("GCTStream.__init__",
inspect.signature(GCTStream.__init__).parameters,
{"img_size", "patch_size", "enable_3d_rope", "max_frame_num",
"kv_cache_sliding_window", "kv_cache_scale_frames",
"kv_cache_cross_frame_special", "kv_cache_include_scale_frames",
"use_sdpa", "camera_num_iterations"})
check("GCTStream.inference_streaming",
inspect.signature(GCTStream.inference_streaming).parameters,
{"images", "num_scale_frames", "keyframe_interval", "output_device"})
check("predictions_to_glb",
inspect.signature(predictions_to_glb).parameters,
{"predictions", "conf_thres", "show_cam", "mask_sky", "target_dir", "prediction_mode"})
print("[selfcheck] " + ("ALL GOOD ✓" if ok else "FAILURES ✗"))
return 0 if ok else 1
# --------------------------------------------------------------------------------------
# --synthetic : build a correctly-shaped vis-dict and export a GLB (CPU only)
# --------------------------------------------------------------------------------------
def make_synthetic_vis(S: int = 5, H: int = 36, W: int = 48):
"""A wavy coloured surface seen from S cameras panning along +x — exercises every
branch of predictions_to_glb (point cloud, percentile filter, camera frustums,
scene alignment) with the exact shapes the real model produces."""
import numpy as np
ys, xs = np.meshgrid(np.linspace(-1, 1, H), np.linspace(-1, 1, W), indexing="ij")
z = 0.35 * np.sin(3 * xs) * np.cos(3 * ys) # H,W
base = np.stack([xs, ys, z], axis=-1).astype(np.float32) # H,W,3
world_points = np.broadcast_to(base, (S, H, W, 3)).copy()
# confidence: high in the centre, lower at the edges (so conf_thres actually does something)
r = np.sqrt(xs ** 2 + ys ** 2)
conf = np.clip(1.0 - 0.5 * r, 0.05, 1.0).astype(np.float32)
world_points_conf = np.broadcast_to(conf, (S, H, W)).copy()
# colours from xy position (NCHW, range [0,1]) — what predictions_to_glb expects after transpose
rgb = np.stack([(xs + 1) / 2, (ys + 1) / 2, (z - z.min()) / ((z.max() - z.min()) + 1e-6)], axis=0).astype(np.float32)
images = np.broadcast_to(rgb, (S, 3, H, W)).copy()
# w2c extrinsics [S,3,4]: identity rotation, camera centre sliding along +x, so
# t = -R @ C = -C. (Matches the convention predictions_to_glb expects.)
extrinsic = np.zeros((S, 3, 4), dtype=np.float32)
for i in range(S):
extrinsic[i, :3, :3] = np.eye(3)
cx = -0.6 + 1.2 * (i / max(1, S - 1))
extrinsic[i, :3, 3] = [-cx, 0.0, 3.0] # camera in front of the surface (+z), w2c => -C
return {
"world_points": world_points,
"world_points_conf": world_points_conf,
"images": images,
"extrinsic": extrinsic,
}
def cmd_synthetic(out_path: str) -> int:
import numpy as np
import trimesh
import pipeline as P # safe locally: load_fn import is lazy
vis = make_synthetic_vis()
print("[synthetic] vis-dict shapes:",
{k: np.asarray(v).shape for k, v in vis.items()})
kept, total = P.count_visible_points(vis, conf_thres=50.0)
print(f"[synthetic] conf_thres=50 keeps {kept}/{total} points")
path = P.export_glb(vis, out_path=out_path, conf_thres=50.0, show_cam=True)
size = os.path.getsize(path)
print(f"[synthetic] wrote {path} ({size/1024:.1f} KiB)")
# Reload to prove it's a valid GLB with real geometry.
loaded = trimesh.load(path)
geoms = list(loaded.geometry.values()) if hasattr(loaded, "geometry") else [loaded]
n_vertices = sum(int(getattr(g, "vertices", np.empty((0, 3))).shape[0]) for g in geoms)
print(f"[synthetic] reloaded GLB: {len(geoms)} geometries, {n_vertices} total vertices")
ok = size > 2048 and n_vertices > 0
print("[synthetic] " + ("EXPORT PATH VERIFIED ✓" if ok else "SOMETHING WRONG ✗"))
return 0 if ok else 1
# --------------------------------------------------------------------------------------
# real run : full inference (needs CUDA GPU + checkpoint)
# --------------------------------------------------------------------------------------
def cmd_real(args) -> int:
import torch
import pipeline as P
device = "cuda" if torch.cuda.is_available() else "cpu"
if device == "cpu":
print("[real] WARNING: no CUDA GPU — inference will be extremely slow / may not "
"match production dtype. Intended to run on the Space hardware.")
print(f"[real] building model on {device} ...")
model = P.build_model(args.ckpt, device=device)
if args.video:
paths, _ = P.extract_video_frames(args.video, fps=args.fps, max_frames=args.max_frames)
else:
paths = sorted(
p for p in glob.glob(os.path.join(args.image_folder, "*"))
if p.lower().endswith((".jpg", ".jpeg", ".png", ".bmp", ".webp"))
)[:args.max_frames]
print(f"[real] {len(paths)} frames")
images = P.preprocess_paths(paths)
vis = P.reconstruct_predictions(model, images)
out = P.export_glb(vis, out_path=args.out, conf_thres=args.conf, show_cam=True)
print(f"[real] wrote {out}")
return 0
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--selfcheck", action="store_true")
ap.add_argument("--synthetic", action="store_true")
ap.add_argument("--out", default="/tmp/lingbot_scene.glb")
ap.add_argument("--ckpt")
ap.add_argument("--image_folder")
ap.add_argument("--video")
ap.add_argument("--fps", type=int, default=6)
ap.add_argument("--max_frames", type=int, default=24)
ap.add_argument("--conf", type=float, default=50.0)
args = ap.parse_args()
if args.selfcheck:
sys.exit(cmd_selfcheck())
if args.synthetic:
sys.exit(cmd_synthetic(args.out))
if args.ckpt and (args.image_folder or args.video):
sys.exit(cmd_real(args))
ap.error("choose --selfcheck, --synthetic, or a real run (--ckpt + --image_folder/--video)")
if __name__ == "__main__":
main()