File size: 5,690 Bytes
b20c82e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
"""R1-3 — metric scene scale certification (local harness; needs
torch+transformers, a reference bundle, and the photo it was converted from).

Validates estimate_meters_per_unit end-to-end:

  1. mpu resolves (not None) on real rooms with metric depth
  2. plausibility: the visible floor's physical bottom width (plane.width *
     mpu) lands in 1.5-10 m
  3. independence check: a 60 cm tile's on-screen size predicted two ways
     must agree within 15%:
       a) through the scale chain: 0.6/mpu plane units -> homography -> pixels
       b) straight from depth:    0.6 * f / Z  at the same image row
  4. cross-room consistency: the same physical tile, the same prediction
     logic, in every supplied room.

Usage:
    python verify_r1_scale.py <bundle.json[.gz]>:<photo.jpg> [more pairs...]
"""

import base64
import gzip
import json
import sys

import cv2
import numpy as np
import torch
from PIL import Image
from transformers import AutoImageProcessor, AutoModelForDepthEstimation

# --- real implementations from app.py ---------------------------------------
src = open("app.py").read()
ns = {"np": np, "cv2": cv2}
start = src.index("def depth_model_is_metric")
end = src.index("\nENABLE_DEPTH", start)
exec(compile(src[start:end], "app.py", "exec"), ns)
ns["depth_model_is_metric"] = lambda name=None: True  # harness always metric
start = src.index("def estimate_meters_per_unit")
end = src.index("\n# ---", start)
exec(compile(src[start:end], "app.py", "exec"), ns)
estimate_meters_per_unit = ns["estimate_meters_per_unit"]

import re

MODEL = re.search(r'depth_model_name",\s*\n(?:\s*#.*\n)*\s*"([^"]+)"', src).group(1)


def load_bundle(path):
    if path.endswith(".gz"):
        with gzip.open(path, "rt") as f:
            return json.load(f)
    return json.load(open(path))


def run_depth(img):
    inputs = run_depth.processor(images=img, return_tensors="pt")
    with torch.no_grad():
        out = run_depth.model(**inputs)
        depth = torch.nn.functional.interpolate(
            out.predicted_depth.unsqueeze(1),
            size=(img.height, img.width),
            mode="bicubic",
            align_corners=False,
        ).squeeze().numpy()
    return cv2.GaussianBlur(depth.astype(np.float32), (0, 0), sigmaX=3)


def main():
    pairs = [a.split(":") for a in sys.argv[1:]]
    if not pairs:
        print(__doc__)
        return 2
    print(f"model: {MODEL}")
    run_depth.processor = AutoImageProcessor.from_pretrained(MODEL)
    run_depth.model = AutoModelForDepthEstimation.from_pretrained(MODEL).eval()

    ok = True
    for bundle_path, photo in pairs:
        d = load_bundle(bundle_path)
        w, h = d["width"], d["height"]
        seg = max(d["segments"], key=lambda s: len(s["mask"]))
        mask_idx = np.frombuffer(base64.b64decode(seg["mask"]), dtype=np.uint32)
        mask = np.zeros(w * h, np.uint8)
        mask[mask_idx] = 1
        mask = mask.reshape(h, w)
        H = np.asarray(seg["homography"], np.float64).reshape(3, 3)

        img = Image.open(photo).convert("RGB").resize((w, h), Image.LANCZOS)
        depth = run_depth(img)

        mpu = estimate_meters_per_unit(depth, mask, seg["homography"], w, h)
        if mpu is None:
            # A clean fallback is acceptable: rooms on the synthetic-VP
            # homography can't carry a trustworthy metric scale until R1-2;
            # the engine then uses the heuristic repeat. FAIL is reserved for
            # a returned-but-wrong scale (checked below).
            print(f"  [PASS] {photo.split('/')[-1]}: metersPerUnit = None "
                  f"(clean heuristic fallback — geometry not metric-trustworthy)")
            continue

        plane = seg["plane"]
        floor_w_m = plane["width"] * mpu
        width_ok = 1.5 <= floor_w_m <= 10.0

        # independence check at a bottom-area floor row
        ys, xs = np.nonzero(mask)
        y_ref = int(np.percentile(ys, 92))
        row_xs = xs[ys == y_ref]
        x_ref = int(np.median(row_xs))
        z_ref = float(depth[y_ref, x_ref])
        f = float(w)
        px_from_depth = f * 0.6 / z_ref

        # map (0.6/mpu) plane units back through H^-1 at the same location
        Hinv = np.linalg.inv(H)
        den = H[2, 0] * x_ref + H[2, 1] * y_ref + H[2, 2]
        px_p = (H[0, 0] * x_ref + H[0, 1] * y_ref + H[0, 2]) / den
        py_p = (H[1, 0] * x_ref + H[1, 1] * y_ref + H[1, 2]) / den

        def back(up, vp):
            dz = Hinv[2, 0] * up + Hinv[2, 1] * vp + Hinv[2, 2]
            return (
                (Hinv[0, 0] * up + Hinv[0, 1] * vp + Hinv[0, 2]) / dz,
                (Hinv[1, 0] * up + Hinv[1, 1] * vp + Hinv[1, 2]) / dz,
            )

        units = 0.6 / mpu
        ax, ay = back(px_p - units / 2, py_p)
        bx, by = back(px_p + units / 2, py_p)
        px_from_chain = float(np.hypot(bx - ax, by - ay))
        rel_err = abs(px_from_chain - px_from_depth) / px_from_depth
        chain_ok = rel_err <= 0.15

        passed = width_ok and chain_ok
        ok &= passed
        print(
            f"  [{'PASS' if passed else 'FAIL'}] {photo.split('/')[-1]}: "
            f"mpu={mpu:.5f} m/unit | floor width = {floor_w_m:.2f} m | "
            f"60cm tile @row{y_ref}: chain={px_from_chain:.0f}px vs depth={px_from_depth:.0f}px "
            f"(err {rel_err * 100:.1f}%)"
        )
        if not width_ok:
            print("        !! floor physical width implausible")
        if not chain_ok:
            print("        !! scale chain disagrees with direct depth prediction")

    print("\n" + ("ALL R1-3 SCALE CHECKS PASSED" if ok else "R1-3 SCALE CHECKS FAILED"))
    return 0 if ok else 1


if __name__ == "__main__":
    raise SystemExit(main())