File size: 9,350 Bytes
22c23b5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
c1a519e
 
 
 
22c23b5
 
cc6d487
22c23b5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
cc6d487
22c23b5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1f5e38a
22c23b5
 
 
1f5e38a
22c23b5
1f5e38a
 
22c23b5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
cc6d487
22c23b5
 
 
cc6d487
22c23b5
 
cc6d487
22c23b5
cc6d487
22c23b5
 
cc6d487
22c23b5
cc6d487
 
22c23b5
 
cc6d487
 
22c23b5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
cc6d487
 
1f5e38a
 
22c23b5
 
 
 
a7900e8
 
 
 
 
 
 
 
 
22c23b5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2de6da3
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
"""LingBot-Map — Streaming 3D Reconstruction (community Gradio / ZeroGPU demo).

Upload a few images or a short video clip → the model reconstructs a 3D point cloud
with the camera trajectory, exported as a .glb you can orbit in the browser.

What this is
------------
A community Gradio wrapper around LingBot-Map (Robbyant Team, Apache-2.0). NOT an
official release. FlashInfer is swapped for the built-in SDPA attention fallback so it
runs on ZeroGPU's ephemeral GPUs without a per-hardware source build.

Design notes (see NOTES.md for the full rationale):
  * Model is built on CPU at startup; moved to CUDA only inside the @spaces.GPU call
    (Option A — the most predictable ZeroGPU pattern).
  * Frame count is capped BEFORE the GPU call so a job always fits the time budget.
  * Inference returns a NumPy vis-dict that we cache in gr.State, so tweaking the
    confidence filter / camera toggle re-exports the GLB on CPU — instantly, no GPU.
"""

import os
import time

import gradio as gr
import spaces
import torch
from huggingface_hub import hf_hub_download

import pipeline as P

# --------------------------------------------------------------------------------------
# Config
# --------------------------------------------------------------------------------------
CKPT_REPO = "robbyant/lingbot-map"
# Long-sequence checkpoint → globally-coherent maps. The base "lingbot-map.pt" is a causal
# streaming model whose per-frame depth doesn't fuse well (frames fan out as parallel planes);
# the README's clean reconstructions use the "-long" weights (loop closure / global consistency).
CKPT_FILE = "lingbot-map-long.pt"
NUM_SCALE_FRAMES = 8                   # initial bidirectional "scale" frames
KEYFRAME_INTERVAL = 1                  # every frame is a keyframe — best quality for short clips
MAX_FRAMES_HARD = 128                 # absolute cap (~0.2s/frame → ~25s at 128, well under the 120s budget)
GPU_DURATION = 120                    # seconds requested per GPU call (tune to your tier)

# --------------------------------------------------------------------------------------
# Startup: download checkpoint + build model on CPU (Option A)
# --------------------------------------------------------------------------------------
print(f"[startup] downloading checkpoint {CKPT_REPO}/{CKPT_FILE} ...")
CKPT_PATH = hf_hub_download(CKPT_REPO, CKPT_FILE)
print("[startup] building model on CPU ...")
MODEL = P.build_model(CKPT_PATH, device="cpu", camera_num_iterations=4, use_sdpa=True)
print("[startup] model ready.")

_AGG_BF16_DONE = {"done": False}


@spaces.GPU(duration=GPU_DURATION)
def gpu_reconstruct(images_cpu):
    """The only GPU-attached code. Move model to CUDA, run streaming inference, return
    an unbatched NumPy vis-dict (offloaded to CPU). Everything else stays on CPU."""
    MODEL.to("cuda")
    # ZeroGPU GPUs (H200 / Blackwell) are all compute-capability >= 8 → bf16. Cast the
    # aggregator once (heads stay fp32, matching upstream demo.py); inference runs under
    # autocast(bf16) inside pipeline.infer.
    if not _AGG_BF16_DONE["done"] and getattr(MODEL, "aggregator", None) is not None:
        MODEL.aggregator = MODEL.aggregator.to(dtype=torch.bfloat16)
        _AGG_BF16_DONE["done"] = True
    return P.reconstruct_points(
        MODEL, images_cpu,
        num_scale_frames=NUM_SCALE_FRAMES,
        keyframe_interval=KEYFRAME_INTERVAL,
    )


def _gather_paths(image_files, video_file, fps, max_frames):
    """Resolve the upload into a capped, ordered list of frame paths (CPU only)."""
    if video_file:
        paths, _ = P.extract_video_frames(video_file, fps=int(fps), max_frames=max_frames)
        source = f"video sampled @ ~{int(fps)} fps → {len(paths)} frames"
    else:
        paths = P.list_image_paths(image_files)[:max_frames]
        source = f"{len(paths)} uploaded image(s)"
    return paths, source


def _status_md(source, n_frames, h, w, kept, total, t_infer, t_export, conf_thres):
    pct = (100.0 * kept / total) if total else 0.0
    shown_note = f" · showing **{min(kept, 2_000_000):,}** for smooth rendering" if kept > 2_000_000 else ""
    return (
        f"**Done.** Input: {source}.\n\n"
        f"- Frames reconstructed: **{n_frames}** at **{w}×{h}**\n"
        f"- Points kept: **{kept:,} / {total:,}** ({pct:.0f}%) at conf filter **{conf_thres:.0f}**{shown_note}\n"
        f"- Inference: **{t_infer:.1f}s** · GLB export: **{t_export:.1f}s**\n\n"
        f"_Drag to orbit. Raise the confidence filter to declutter; toggle the camera track. "
        f"Re-renders instantly (no GPU needed)._"
    )


def reconstruct(image_files, video_file, fps, max_frames, conf_thres, show_cam):
    """Full path: gather → preprocess (CPU) → GPU inference → cache vis dict → export GLB."""
    max_frames = int(min(max_frames, MAX_FRAMES_HARD))
    paths, source = _gather_paths(image_files, video_file, fps, max_frames)
    if not paths:
        raise gr.Error("Please upload a few images or a short video first.")
    if len(paths) < 2:
        raise gr.Error("Need at least 2 frames — reconstruction is multi-view.")

    t0 = time.time()
    images = P.preprocess_paths(paths)            # [S, 3, H, W] on CPU
    s, _, h, w = images.shape

    rec = gpu_reconstruct(images)                 # GPU; returns a COMPACT flat point set
    t_infer = time.time() - t0

    t1 = time.time()
    glb = P.build_glb_from_points(rec, conf_thres=conf_thres, show_cam=show_cam)
    t_export = time.time() - t1

    kept, total = P.count_points(rec, conf_thres)
    status = _status_md(source, s, h, w, kept, total, t_infer, t_export, conf_thres)
    return glb, rec, status


def reexport(rec, conf_thres, show_cam):
    """CPU-only re-render when the user nudges the confidence filter / camera toggle.
    Uses the cached compact point set — no GPU, no re-inference."""
    if not rec:
        return None, "Run a reconstruction first."
    t1 = time.time()
    glb = P.build_glb_from_points(rec, conf_thres=conf_thres, show_cam=show_cam)
    kept, total = P.count_points(rec, conf_thres)
    pct = (100.0 * kept / total) if total else 0.0
    msg = (
        f"Re-rendered (CPU). Points kept: **{kept:,} / {total:,}** ({pct:.0f}%) at "
        f"conf filter **{conf_thres:.0f}** · {time.time() - t1:.2f}s"
    )
    return glb, msg


INTRO = """
# 🗺️ LingBot-Map — Streaming 3D Reconstruction
Upload a handful of images or a **short video** and get back a navigable 3D point cloud
with the camera path, exported as a GLB.

*Community Gradio demo of [LingBot-Map](https://github.com/Robbyant/lingbot-map) by the
Robbyant Team (Apache-2.0) — not an official release. Built on VGGT + DINOv2; uses the
SDPA attention fallback for ZeroGPU. Keep inputs short (a few seconds of video / up to
%d frames) so the job fits the GPU time budget.*
""" % MAX_FRAMES_HARD


def build_demo():
    with gr.Blocks(title="LingBot-Map — Streaming 3D Reconstruction") as demo:
        gr.Markdown(INTRO)
        vis_state = gr.State(None)

        with gr.Row():
            with gr.Column(scale=1):
                images_in = gr.File(
                    label="Images (ordered frames — name them 000.jpg, 001.jpg, …)",
                    file_count="multiple", file_types=["image"],
                )
                video_in = gr.Video(label="…or a short video clip")
                with gr.Accordion("Settings", open=True):
                    fps = gr.Slider(1, 12, value=6, step=1,
                                    label="Video sampling FPS (frames/sec to extract)")
                    max_frames = gr.Slider(2, MAX_FRAMES_HARD, value=80, step=1,
                                           label="Max frames — higher = denser & more complete (spread across the whole video)")
                    conf_thres = gr.Slider(0, 95, value=60, step=1,
                                           label="Confidence filter — drops the lowest-confidence % of points (raise to declutter)")
                    show_cam = gr.Checkbox(value=True, label="Show camera trajectory")
                run_btn = gr.Button("Reconstruct 3D scene", variant="primary")

            with gr.Column(scale=1):
                model_out = gr.Model3D(
                    label="Reconstructed point cloud + cameras (drag to orbit)",
                    clear_color=[0.0, 0.0, 0.0, 0.0],
                    # Open on an elevated 3/4 / bird's-eye angle so the scene's depth and the
                    # camera trajectory are visible immediately (face-on hides the track).
                    # (alpha=azimuth°, beta=polar-from-top°, radius=None → auto-fit). Drag to refine.
                    camera_position=(40, 40, None),
                    height=560,
                )
                status = gr.Markdown()

        run_btn.click(
            reconstruct,
            inputs=[images_in, video_in, fps, max_frames, conf_thres, show_cam],
            outputs=[model_out, vis_state, status],
        )
        # Re-render from the cached vis-dict on a slider/toggle change (CPU only).
        for ctrl in (conf_thres, show_cam):
            ctrl.change(reexport, inputs=[vis_state, conf_thres, show_cam],
                        outputs=[model_out, status])

    return demo


if __name__ == "__main__":
    build_demo().queue().launch(show_error=True)