room-visualizer / verify_r1_scale.py
GitHub Actions
Deploy from GitHub commit ca72656c17476e5aa37a4735af6e47ff9f94fa1a
b20c82e
Raw
History Blame Contribute Delete
5.69 kB
"""R1-3 — metric scene scale certification (local harness; needs
torch+transformers, a reference bundle, and the photo it was converted from).
Validates estimate_meters_per_unit end-to-end:
1. mpu resolves (not None) on real rooms with metric depth
2. plausibility: the visible floor's physical bottom width (plane.width *
mpu) lands in 1.5-10 m
3. independence check: a 60 cm tile's on-screen size predicted two ways
must agree within 15%:
a) through the scale chain: 0.6/mpu plane units -> homography -> pixels
b) straight from depth: 0.6 * f / Z at the same image row
4. cross-room consistency: the same physical tile, the same prediction
logic, in every supplied room.
Usage:
python verify_r1_scale.py <bundle.json[.gz]>:<photo.jpg> [more pairs...]
"""
import base64
import gzip
import json
import sys
import cv2
import numpy as np
import torch
from PIL import Image
from transformers import AutoImageProcessor, AutoModelForDepthEstimation
# --- real implementations from app.py ---------------------------------------
src = open("app.py").read()
ns = {"np": np, "cv2": cv2}
start = src.index("def depth_model_is_metric")
end = src.index("\nENABLE_DEPTH", start)
exec(compile(src[start:end], "app.py", "exec"), ns)
ns["depth_model_is_metric"] = lambda name=None: True # harness always metric
start = src.index("def estimate_meters_per_unit")
end = src.index("\n# ---", start)
exec(compile(src[start:end], "app.py", "exec"), ns)
estimate_meters_per_unit = ns["estimate_meters_per_unit"]
import re
MODEL = re.search(r'depth_model_name",\s*\n(?:\s*#.*\n)*\s*"([^"]+)"', src).group(1)
def load_bundle(path):
if path.endswith(".gz"):
with gzip.open(path, "rt") as f:
return json.load(f)
return json.load(open(path))
def run_depth(img):
inputs = run_depth.processor(images=img, return_tensors="pt")
with torch.no_grad():
out = run_depth.model(**inputs)
depth = torch.nn.functional.interpolate(
out.predicted_depth.unsqueeze(1),
size=(img.height, img.width),
mode="bicubic",
align_corners=False,
).squeeze().numpy()
return cv2.GaussianBlur(depth.astype(np.float32), (0, 0), sigmaX=3)
def main():
pairs = [a.split(":") for a in sys.argv[1:]]
if not pairs:
print(__doc__)
return 2
print(f"model: {MODEL}")
run_depth.processor = AutoImageProcessor.from_pretrained(MODEL)
run_depth.model = AutoModelForDepthEstimation.from_pretrained(MODEL).eval()
ok = True
for bundle_path, photo in pairs:
d = load_bundle(bundle_path)
w, h = d["width"], d["height"]
seg = max(d["segments"], key=lambda s: len(s["mask"]))
mask_idx = np.frombuffer(base64.b64decode(seg["mask"]), dtype=np.uint32)
mask = np.zeros(w * h, np.uint8)
mask[mask_idx] = 1
mask = mask.reshape(h, w)
H = np.asarray(seg["homography"], np.float64).reshape(3, 3)
img = Image.open(photo).convert("RGB").resize((w, h), Image.LANCZOS)
depth = run_depth(img)
mpu = estimate_meters_per_unit(depth, mask, seg["homography"], w, h)
if mpu is None:
# A clean fallback is acceptable: rooms on the synthetic-VP
# homography can't carry a trustworthy metric scale until R1-2;
# the engine then uses the heuristic repeat. FAIL is reserved for
# a returned-but-wrong scale (checked below).
print(f" [PASS] {photo.split('/')[-1]}: metersPerUnit = None "
f"(clean heuristic fallback — geometry not metric-trustworthy)")
continue
plane = seg["plane"]
floor_w_m = plane["width"] * mpu
width_ok = 1.5 <= floor_w_m <= 10.0
# independence check at a bottom-area floor row
ys, xs = np.nonzero(mask)
y_ref = int(np.percentile(ys, 92))
row_xs = xs[ys == y_ref]
x_ref = int(np.median(row_xs))
z_ref = float(depth[y_ref, x_ref])
f = float(w)
px_from_depth = f * 0.6 / z_ref
# map (0.6/mpu) plane units back through H^-1 at the same location
Hinv = np.linalg.inv(H)
den = H[2, 0] * x_ref + H[2, 1] * y_ref + H[2, 2]
px_p = (H[0, 0] * x_ref + H[0, 1] * y_ref + H[0, 2]) / den
py_p = (H[1, 0] * x_ref + H[1, 1] * y_ref + H[1, 2]) / den
def back(up, vp):
dz = Hinv[2, 0] * up + Hinv[2, 1] * vp + Hinv[2, 2]
return (
(Hinv[0, 0] * up + Hinv[0, 1] * vp + Hinv[0, 2]) / dz,
(Hinv[1, 0] * up + Hinv[1, 1] * vp + Hinv[1, 2]) / dz,
)
units = 0.6 / mpu
ax, ay = back(px_p - units / 2, py_p)
bx, by = back(px_p + units / 2, py_p)
px_from_chain = float(np.hypot(bx - ax, by - ay))
rel_err = abs(px_from_chain - px_from_depth) / px_from_depth
chain_ok = rel_err <= 0.15
passed = width_ok and chain_ok
ok &= passed
print(
f" [{'PASS' if passed else 'FAIL'}] {photo.split('/')[-1]}: "
f"mpu={mpu:.5f} m/unit | floor width = {floor_w_m:.2f} m | "
f"60cm tile @row{y_ref}: chain={px_from_chain:.0f}px vs depth={px_from_depth:.0f}px "
f"(err {rel_err * 100:.1f}%)"
)
if not width_ok:
print(" !! floor physical width implausible")
if not chain_ok:
print(" !! scale chain disagrees with direct depth prediction")
print("\n" + ("ALL R1-3 SCALE CHECKS PASSED" if ok else "R1-3 SCALE CHECKS FAILED"))
return 0 if ok else 1
if __name__ == "__main__":
raise SystemExit(main())