Spaces:
Sleeping
Sleeping
| import asyncio | |
| import base64 | |
| import hashlib | |
| import html | |
| import io | |
| import json | |
| import mimetypes | |
| import os | |
| import shutil | |
| import subprocess | |
| import time | |
| import urllib.error | |
| import urllib.parse | |
| import urllib.request | |
| import zipfile | |
| from datetime import datetime, timedelta, timezone | |
| try: | |
| import tomllib | |
| except ImportError: | |
| try: | |
| import tomli as tomllib | |
| except ImportError: | |
| try: | |
| import tomlkit as tomllib | |
| except ImportError: | |
| raise ImportError( | |
| "No TOML library found. Please run on Python 3.11+, or run 'pip install tomli' to support Python 3.10." | |
| ) | |
| import uuid | |
| from pathlib import Path | |
| try: | |
| from dotenv import load_dotenv | |
| except ImportError: | |
| load_dotenv = None | |
| import cv2 | |
| import numpy as np | |
| import torch | |
| from fastapi import BackgroundTasks, Body, FastAPI, File, Form, HTTPException, Query, Request, Response, UploadFile | |
| from fastapi.middleware.cors import CORSMiddleware | |
| from fastapi.responses import StreamingResponse | |
| from fastapi.middleware.gzip import GZipMiddleware | |
| from fastapi.staticfiles import StaticFiles | |
| from PIL import Image, ImageOps | |
| from transformers import ( | |
| AutoImageProcessor, | |
| AutoModelForDepthEstimation, | |
| Mask2FormerForUniversalSegmentation, | |
| OneFormerForUniversalSegmentation, | |
| OneFormerProcessor, | |
| SegformerForSemanticSegmentation, | |
| ) | |
| from floor_direction import ( | |
| analyze_floor_direction as analyze_floor_direction_from_segmentation, | |
| parse_floor_segmentation, | |
| ) | |
| from direction_finder import find_floor_direction | |
| ADE20K_CLASSES = [ | |
| "wall", "building", "sky", "floor", "tree", "ceiling", "road", "bed", | |
| "window", "grass", "cabinet", "sidewalk", "person", "ground", "door", | |
| "table", "mountain", "plant", "curtain", "chair", "car", "water", | |
| "painting", "sofa", "shelf", "house", "sea", "mirror", "rug", "field", | |
| "armchair", "seat", "fence", "desk", "rock", "wardrobe", "lamp", | |
| "bathtub", "railing", "cushion", "base", "box", "column", "signboard", | |
| "chest of drawers", "counter", "sand", "sink", "skyscraper", "fireplace", | |
| "refrigerator", "stairs", "runway", "bookcase", "blind", "coffee table", | |
| "toilet", "flower", "book", "hill", "bench", "countertop", "stove", | |
| "palm", "kitchen island", "computer", "swivel chair", "boat", "bar", | |
| "arcade machine", "hovel", "bus", "towel", "light", "truck", "tower", | |
| "chandelier", "awning", "streetlight", "booth", "television", "airplane", | |
| "dirt track", "apparel", "pole", "land", "bannister", "escalator", | |
| "ottoman", "bottle", "buffet", "poster", "stage", "van", "ship", | |
| "fountain", "conveyer belt", "canopy", "washer", "plaything", | |
| "swimming pool", "stool", "barrel", "basket", "waterfall", "tent", | |
| "bag", "minibike", "cradle", "oven", "ball", "food", "step", "tank", | |
| "trade name", "microwave", "pot", "animal", "bicycle", "lake", | |
| "dishwasher", "screen", "blanket", "sculpture", "hood", "sconce", | |
| "vase", "traffic light", "tray", "ashcan", "fan", "pier", "crt screen", | |
| "plate", "monitor", "bulletin board", "shower", "radiator", "glass", | |
| "clock", "flag", | |
| ] | |
| if load_dotenv is not None: | |
| load_dotenv(Path(__file__).resolve().parent / ".env") | |
| def load_config() -> dict: | |
| config_path = os.getenv("VISUALIZER_CONFIG") | |
| if not config_path: | |
| return {} | |
| path = Path(config_path).expanduser() | |
| if not path.is_absolute(): | |
| path = Path(__file__).resolve().parent / path | |
| if not path.exists(): | |
| raise RuntimeError(f"VISUALIZER_CONFIG does not exist: {path}") | |
| with path.open("rb") as config_file: | |
| return tomllib.load(config_file) | |
| CONFIG = load_config() | |
| BACKEND_DIR = Path(__file__).resolve().parent | |
| REPO_ROOT = BACKEND_DIR.parent.parent | |
| def config_value(env_name: str, section: str, key: str, default): | |
| if env_name in os.environ: | |
| return os.environ[env_name] | |
| return CONFIG.get(section, {}).get(key, default) | |
| SEGMENTATION_MODEL = str( | |
| config_value("SEGMENTATION_MODEL", "models", "segmentation_model", "oneformer") | |
| ).lower() | |
| ONEFORMER_MODEL_NAME = str(config_value( | |
| "ONEFORMER_MODEL_NAME", | |
| "models", | |
| "oneformer_model_name", | |
| "shi-labs/oneformer_ade20k_swin_large", | |
| )) | |
| MASK2FORMER_MODEL_NAME = str(config_value( | |
| "MASK2FORMER_MODEL_NAME", | |
| "models", | |
| "mask2former_model_name", | |
| "facebook/mask2former-swin-small-ade-semantic", | |
| )) | |
| SEGFORMER_MODEL_NAME = str(config_value( | |
| "SEGFORMER_MODEL_NAME", | |
| "models", | |
| "segformer_model_name", | |
| "nvidia/segformer-b2-finetuned-ade-512-512", | |
| )) | |
| DEPTH_MODEL_NAME = str(config_value( | |
| "DEPTH_MODEL_NAME", | |
| "models", | |
| "depth_model_name", | |
| "depth-anything/Depth-Anything-V2-Metric-Indoor-Large-hf", | |
| )) | |
| GEOMETRY_MODEL = str(config_value( | |
| "GEOMETRY_MODEL", | |
| "models", | |
| "geometry_model", | |
| "depth", | |
| )).lower() | |
| MOGE_MODEL_NAME = str(config_value( | |
| "MOGE_MODEL_NAME", | |
| "models", | |
| "moge_model_name", | |
| "Ruicheng/moge-2-vitl-normal", | |
| )) | |
| SAM2_MODEL_CFG = str(config_value( | |
| "SAM2_MODEL_CFG", | |
| "models", | |
| "sam2_model_cfg", | |
| "", | |
| )) | |
| SAM2_CHECKPOINT = str(config_value( | |
| "SAM2_CHECKPOINT", | |
| "models", | |
| "sam2_checkpoint", | |
| "", | |
| )) | |
| SAM2_MODEL_NAME = str(config_value( | |
| "SAM2_MODEL_NAME", | |
| "models", | |
| "sam2_model_name", | |
| "facebook/sam2-hiera-large", | |
| )) | |
| ENABLE_MOGE_GEOMETRY = str(config_value( | |
| "ENABLE_MOGE_GEOMETRY", | |
| "runtime", | |
| "enable_moge_geometry", | |
| "0", | |
| )).lower() in {"1", "true", "yes", "on"} | |
| ENABLE_SAM_REFINEMENT = str(config_value( | |
| "ENABLE_SAM_REFINEMENT", | |
| "runtime", | |
| "enable_sam_refinement", | |
| "0", | |
| )).lower() in {"1", "true", "yes", "on"} | |
| ENABLE_FLOOR_MASK_CLEANUP = str(config_value( | |
| "ENABLE_FLOOR_MASK_CLEANUP", | |
| "runtime", | |
| "enable_floor_mask_cleanup", | |
| "1", | |
| )).lower() in {"1", "true", "yes", "on"} | |
| ENABLE_DEPTH_ESTIMATION = str(config_value( | |
| "ENABLE_DEPTH_ESTIMATION", | |
| "runtime", | |
| "enable_depth_estimation", | |
| "1", | |
| )).lower() in {"1", "true", "yes", "on"} | |
| ENABLE_PLANE_FIT = str(config_value( | |
| "ENABLE_PLANE_FIT", | |
| "runtime", | |
| "enable_plane_fit", | |
| "1", | |
| )).lower() in {"1", "true", "yes", "on"} | |
| ENABLE_MULTI_PLANE_SURFACES = str(config_value( | |
| "ENABLE_MULTI_PLANE_SURFACES", | |
| "runtime", | |
| "enable_multi_plane_surfaces", | |
| "1", | |
| )).lower() in {"1", "true", "yes", "on"} | |
| INTRINSIC_MODEL_VERSION = str(config_value( | |
| "INTRINSIC_MODEL_VERSION", | |
| "models", | |
| "intrinsic_model_version", | |
| "v2", | |
| )) | |
| ENABLE_INTRINSIC_SHADING = str(config_value( | |
| "ENABLE_INTRINSIC_SHADING", | |
| "runtime", | |
| "enable_intrinsic_shading", | |
| "0", | |
| )).lower() in {"1", "true", "yes", "on"} | |
| ENABLE_SHADING_TRANSFER = str(config_value( | |
| "ENABLE_SHADING_TRANSFER", | |
| "runtime", | |
| "enable_shading_transfer", | |
| "1", | |
| )).lower() in {"1", "true", "yes", "on"} | |
| WARM_START_MODELS = str(config_value( | |
| "WARM_START_MODELS", | |
| "runtime", | |
| "warm_start_models", | |
| "1", | |
| )).lower() in {"1", "true", "yes", "on"} | |
| BUNDLE_IMAGE_FORMAT = str(config_value( | |
| "BUNDLE_IMAGE_FORMAT", | |
| "runtime", | |
| "bundle_image_format", | |
| "jpeg", | |
| )).lower() | |
| BUNDLE_IMAGE_QUALITY = int(config_value( | |
| "BUNDLE_IMAGE_QUALITY", | |
| "runtime", | |
| "bundle_image_quality", | |
| "92", | |
| )) | |
| ROOM_UPLOAD_MAX_DIMENSION = max(720, int(config_value( | |
| "ROOM_UPLOAD_MAX_DIMENSION", | |
| "runtime", | |
| "room_upload_max_dimension", | |
| "1920", | |
| ))) | |
| ROOM_UPLOAD_JPEG_QUALITY = min(95, max(50, int(config_value( | |
| "ROOM_UPLOAD_JPEG_QUALITY", | |
| "runtime", | |
| "room_upload_jpeg_quality", | |
| "85", | |
| )))) | |
| DEBUG_ARTIFACTS_ENABLED = str(config_value( | |
| "DEBUG_ARTIFACTS_ENABLED", | |
| "runtime", | |
| "debug_artifacts_enabled", | |
| "1", | |
| )).lower() in {"1", "true", "yes", "on"} | |
| DEBUG_ARTIFACTS_MAX_DIMENSION = max(320, int(config_value( | |
| "DEBUG_ARTIFACTS_MAX_DIMENSION", | |
| "runtime", | |
| "debug_artifacts_max_dimension", | |
| "1600", | |
| ))) | |
| DEBUG_ARTIFACTS_PREVIEW_DIMENSION = max(320, int(config_value( | |
| "DEBUG_ARTIFACTS_PREVIEW_DIMENSION", | |
| "runtime", | |
| "debug_artifacts_preview_dimension", | |
| "720", | |
| ))) | |
| VISUALIZER_DATA_DIR = str(config_value( | |
| "VISUALIZER_DATA_DIR", | |
| "runtime", | |
| "data_dir", | |
| "data", | |
| )) | |
| SPACES_BUCKET = str(config_value( | |
| "SPACES_BUCKET", | |
| "spaces", | |
| "bucket", | |
| "", | |
| )).strip() | |
| SPACES_REGION = str(config_value( | |
| "SPACES_REGION", | |
| "spaces", | |
| "region", | |
| "", | |
| )).strip() | |
| SPACES_ENDPOINT_URL = str(config_value( | |
| "SPACES_ENDPOINT_URL", | |
| "spaces", | |
| "endpoint_url", | |
| "", | |
| )).strip().rstrip("/") | |
| SPACES_ACCESS_KEY_ID = str(config_value( | |
| "SPACES_ACCESS_KEY_ID", | |
| "spaces", | |
| "access_key_id", | |
| "", | |
| )).strip() | |
| SPACES_SECRET_ACCESS_KEY = str(config_value( | |
| "SPACES_SECRET_ACCESS_KEY", | |
| "spaces", | |
| "secret_access_key", | |
| "", | |
| )).strip() | |
| SPACES_UPLOAD_PREFIX = str(config_value( | |
| "SPACES_UPLOAD_PREFIX", | |
| "spaces", | |
| "upload_prefix", | |
| "room-uploads", | |
| )).strip().strip("/") | |
| SPACES_ACL = str(config_value( | |
| "SPACES_ACL", | |
| "spaces", | |
| "acl", | |
| "private", | |
| )).strip() | |
| GEMINI_API_KEY = str(config_value( | |
| "GEMINI_API_KEY", | |
| "gemini", | |
| "api_key", | |
| "", | |
| )).strip() | |
| GEMINI_IMAGE_MODEL = str(config_value( | |
| "GEMINI_IMAGE_MODEL", | |
| "gemini", | |
| "image_model", | |
| "gemini-3.1-flash-image", | |
| )).strip() | |
| GEMINI_API_BASE_URL = str(config_value( | |
| "GEMINI_API_BASE_URL", | |
| "gemini", | |
| "api_base_url", | |
| "https://generativelanguage.googleapis.com/v1/models", | |
| )).rstrip("/") | |
| GEMINI_TIMEOUT_SECONDS = float(config_value( | |
| "GEMINI_TIMEOUT_SECONDS", | |
| "gemini", | |
| "timeout_seconds", | |
| "120", | |
| )) | |
| QA_FRONTEND_URL = str(config_value( | |
| "QA_FRONTEND_URL", | |
| "qa", | |
| "frontend_url", | |
| "", | |
| )).strip().rstrip("/") | |
| QA_BACKEND_URL = str(config_value( | |
| "QA_BACKEND_URL", | |
| "qa", | |
| "backend_url", | |
| "", | |
| )).strip().rstrip("/") | |
| QA_FRONTEND_DIR = Path(str(config_value( | |
| "QA_FRONTEND_DIR", | |
| "qa", | |
| "frontend_dir", | |
| str(REPO_ROOT / "frontend" / "viz2d-demo"), | |
| ))).expanduser() | |
| QA_MAX_IMAGES = max(1, int(config_value( | |
| "QA_MAX_IMAGES", | |
| "qa", | |
| "max_images", | |
| "50", | |
| ))) | |
| QA_MAX_TESTS = max(1, int(config_value( | |
| "QA_MAX_TESTS", | |
| "qa", | |
| "max_tests", | |
| "200", | |
| ))) | |
| QA_RUN_TIMEOUT_SECONDS = max(60, int(config_value( | |
| "QA_RUN_TIMEOUT_SECONDS", | |
| "qa", | |
| "run_timeout_seconds", | |
| "7200", | |
| ))) | |
| QA_PRESIGNED_URL_EXPIRES_SECONDS = max(60, int(config_value( | |
| "QA_PRESIGNED_URL_EXPIRES_SECONDS", | |
| "qa", | |
| "presigned_url_expires_seconds", | |
| "3600", | |
| ))) | |
| SURFACE_RANSAC_DISTANCE = float(config_value( | |
| "SURFACE_RANSAC_DISTANCE", | |
| "surface_mapping", | |
| "ransac_distance_threshold", | |
| "0.04", | |
| )) | |
| SURFACE_MAX_PLANES = int(config_value( | |
| "SURFACE_MAX_PLANES", | |
| "surface_mapping", | |
| "max_planes", | |
| "5", | |
| )) | |
| SURFACE_NORMAL_ANGLE_DEGREES = float(config_value( | |
| "SURFACE_NORMAL_ANGLE_DEGREES", | |
| "surface_mapping", | |
| "normal_angle_threshold_degrees", | |
| "35", | |
| )) | |
| SURFACE_NORMAL_SMOOTHING_KERNEL = int(config_value( | |
| "SURFACE_NORMAL_SMOOTHING_KERNEL", | |
| "surface_mapping", | |
| "normal_smoothing_kernel", | |
| "5", | |
| )) | |
| SURFACE_MIN_PLANE_PIXELS = int(config_value( | |
| "SURFACE_MIN_PLANE_PIXELS", | |
| "surface_mapping", | |
| "min_plane_pixels", | |
| "1500", | |
| )) | |
| SURFACE_EDGE_CLOSE_RATIO = float(config_value( | |
| "SURFACE_EDGE_CLOSE_RATIO", | |
| "surface_mapping", | |
| "edge_close_ratio", | |
| "0.009", | |
| )) | |
| SURFACE_EDGE_KERNEL_RATIO = float(config_value( | |
| "SURFACE_EDGE_KERNEL_RATIO", | |
| "surface_mapping", | |
| "edge_kernel_ratio", | |
| "0.006", | |
| )) | |
| SURFACE_EDGE_DILATION_ITERATIONS = int(config_value( | |
| "SURFACE_EDGE_DILATION_ITERATIONS", | |
| "surface_mapping", | |
| "edge_dilation_iterations", | |
| "2", | |
| )) | |
| ENABLE_SOFT_FLOOR_COVERING_EXPANSION = str(config_value( | |
| "ENABLE_SOFT_FLOOR_COVERING_EXPANSION", | |
| "surface_mapping", | |
| "soft_floor_covering_expansion_enabled", | |
| "1", | |
| )).lower() in {"1", "true", "yes", "on"} | |
| SOFT_FLOOR_COVERING_KERNEL_RATIO = float(config_value( | |
| "SOFT_FLOOR_COVERING_KERNEL_RATIO", | |
| "surface_mapping", | |
| "soft_floor_covering_kernel_ratio", | |
| "0.008", | |
| )) | |
| SOFT_FLOOR_COVERING_DILATION_ITERATIONS = int(config_value( | |
| "SOFT_FLOOR_COVERING_DILATION_ITERATIONS", | |
| "surface_mapping", | |
| "soft_floor_covering_dilation_iterations", | |
| "2", | |
| )) | |
| SURFACE_UV_MIN_COVERAGE = float(config_value( | |
| "SURFACE_UV_MIN_COVERAGE", | |
| "surface_mapping", | |
| "uv_min_coverage", | |
| "0.995", | |
| )) | |
| SURFACE_UV_MAX_BAD_NEIGHBOR_RATIO = float(config_value( | |
| "SURFACE_UV_MAX_BAD_NEIGHBOR_RATIO", | |
| "surface_mapping", | |
| "uv_max_bad_neighbor_ratio", | |
| "0.035", | |
| )) | |
| SURFACE_UV_MAX_JUMP_RATIO = float(config_value( | |
| "SURFACE_UV_MAX_JUMP_RATIO", | |
| "surface_mapping", | |
| "uv_max_jump_ratio", | |
| "8.0", | |
| )) | |
| SURFACE_UV_ABSOLUTE_JUMP_METERS = float(config_value( | |
| "SURFACE_UV_ABSOLUTE_JUMP_METERS", | |
| "surface_mapping", | |
| "uv_absolute_jump_meters", | |
| "0.35", | |
| )) | |
| ENABLE_GEOMETRY_SURFACE_FILTER = str(config_value( | |
| "ENABLE_GEOMETRY_SURFACE_FILTER", | |
| "surface_mapping", | |
| "enable_geometry_surface_filter", | |
| "1", | |
| )).lower() in {"1", "true", "yes", "on"} | |
| SURFACE_PLANE_DISTANCE_METERS = float(config_value( | |
| "SURFACE_PLANE_DISTANCE_METERS", | |
| "surface_mapping", | |
| "plane_distance_threshold_meters", | |
| "0.12", | |
| )) | |
| SURFACE_PLANE_NORMAL_MIN_COS = float(config_value( | |
| "SURFACE_PLANE_NORMAL_MIN_COS", | |
| "surface_mapping", | |
| "plane_normal_min_cos", | |
| "0.45", | |
| )) | |
| SURFACE_PLANE_FILTER_MIN_KEEP_RATIO = float(config_value( | |
| "SURFACE_PLANE_FILTER_MIN_KEEP_RATIO", | |
| "surface_mapping", | |
| "plane_filter_min_keep_ratio", | |
| "0.45", | |
| )) | |
| device = torch.device("cuda" if torch.cuda.is_available() else "cpu") | |
| seg_processor = None | |
| seg_model = None | |
| segmentation_backend = "segformer" | |
| oneformer_polygon_processor = None | |
| oneformer_polygon_model = None | |
| depth_processor = None | |
| depth_model = None | |
| moge_model = None | |
| sam2_predictor = None | |
| intrinsic_models = None | |
| LIGHTING_TRANSFER_VERSION = 3 | |
| SHADE_MAP_MIN = 0.08 | |
| SHADE_MAP_MAX = 1.85 | |
| SHOWROOM_SHADE_MAP_MIN = 0.85 | |
| SHOWROOM_SHADE_MAP_MAX = 1.12 | |
| SHADE_MAP_ENCODING = "linear-relative" | |
| REFLECTION_RELATIVE_FLOOR = 1.04 | |
| REFLECTION_RELATIVE_CEILING = 1.95 | |
| REFLECTION_MAP_MIN = 0.0 | |
| REFLECTION_MAP_MAX = 1.0 | |
| def hf_offline() -> bool: | |
| return os.getenv("HF_HUB_OFFLINE") == "1" or os.getenv("TRANSFORMERS_OFFLINE") == "1" | |
| def _load_segmentation_model(): | |
| global seg_processor, seg_model, segmentation_backend | |
| if seg_processor is not None and seg_model is not None: | |
| return | |
| if SEGMENTATION_MODEL == "oneformer": | |
| try: | |
| print(f"Loading OneFormer: {ONEFORMER_MODEL_NAME} ...", flush=True) | |
| seg_processor = OneFormerProcessor.from_pretrained( | |
| ONEFORMER_MODEL_NAME, | |
| local_files_only=hf_offline(), | |
| ) | |
| seg_model = OneFormerForUniversalSegmentation.from_pretrained( | |
| ONEFORMER_MODEL_NAME, | |
| local_files_only=hf_offline(), | |
| ).to(device) | |
| seg_model.eval() | |
| segmentation_backend = "oneformer" | |
| print("OneFormer loaded.", flush=True) | |
| return | |
| except Exception as exc: | |
| print(f"OneFormer failed ({exc}), falling back to Mask2Former.", flush=True) | |
| if SEGMENTATION_MODEL in {"oneformer", "mask2former"}: | |
| try: | |
| print(f"Loading Mask2Former: {MASK2FORMER_MODEL_NAME} ...", flush=True) | |
| seg_processor = AutoImageProcessor.from_pretrained( | |
| MASK2FORMER_MODEL_NAME, | |
| local_files_only=hf_offline(), | |
| ) | |
| seg_model = Mask2FormerForUniversalSegmentation.from_pretrained( | |
| MASK2FORMER_MODEL_NAME, | |
| local_files_only=hf_offline(), | |
| ).to(device) | |
| seg_model.eval() | |
| segmentation_backend = "mask2former" | |
| print("Mask2Former loaded.", flush=True) | |
| return | |
| except Exception as exc: | |
| print(f"Mask2Former failed ({exc}), falling back to SegFormer.", flush=True) | |
| print(f"Loading SegFormer: {SEGFORMER_MODEL_NAME} ...", flush=True) | |
| seg_processor = AutoImageProcessor.from_pretrained( | |
| SEGFORMER_MODEL_NAME, | |
| local_files_only=hf_offline(), | |
| ) | |
| seg_model = SegformerForSemanticSegmentation.from_pretrained( | |
| SEGFORMER_MODEL_NAME, | |
| local_files_only=hf_offline(), | |
| ).to(device) | |
| seg_model.eval() | |
| segmentation_backend = "segformer" | |
| print("SegFormer loaded.", flush=True) | |
| def _load_oneformer_polygon_model(): | |
| global seg_processor, seg_model, segmentation_backend | |
| global oneformer_polygon_processor, oneformer_polygon_model | |
| if segmentation_backend == "oneformer" and seg_processor is not None and seg_model is not None: | |
| return seg_processor, seg_model | |
| if oneformer_polygon_processor is None or oneformer_polygon_model is None: | |
| print(f"Loading OneFormer polygon model: {ONEFORMER_MODEL_NAME} ...", flush=True) | |
| oneformer_polygon_processor = OneFormerProcessor.from_pretrained( | |
| ONEFORMER_MODEL_NAME, | |
| local_files_only=hf_offline(), | |
| ) | |
| oneformer_polygon_model = OneFormerForUniversalSegmentation.from_pretrained( | |
| ONEFORMER_MODEL_NAME, | |
| local_files_only=hf_offline(), | |
| ).to(device) | |
| oneformer_polygon_model.eval() | |
| print("OneFormer polygon model loaded.", flush=True) | |
| if SEGMENTATION_MODEL == "oneformer" and seg_model is None: | |
| seg_processor = oneformer_polygon_processor | |
| seg_model = oneformer_polygon_model | |
| segmentation_backend = "oneformer" | |
| return oneformer_polygon_processor, oneformer_polygon_model | |
| def _load_intrinsic_model(): | |
| global intrinsic_models | |
| if ENABLE_INTRINSIC_SHADING and intrinsic_models is None: | |
| try: | |
| print(f"Loading Intrinsic Image Decomposition model: {INTRINSIC_MODEL_VERSION} ...", flush=True) | |
| from intrinsic.pipeline import load_models | |
| intrinsic_models = load_models(INTRINSIC_MODEL_VERSION, device=str(device)) | |
| print("Intrinsic model loaded.", flush=True) | |
| except Exception as exc: | |
| print(f"Intrinsic model failed to load ({exc}). Falling back to luminance shading.", flush=True) | |
| def _load_depth_model() -> bool: | |
| global depth_processor, depth_model | |
| if not ENABLE_DEPTH_ESTIMATION: | |
| return False | |
| if depth_processor is not None and depth_model is not None: | |
| return True | |
| print(f"Loading depth model: {DEPTH_MODEL_NAME} ...", flush=True) | |
| depth_processor = AutoImageProcessor.from_pretrained( | |
| DEPTH_MODEL_NAME, | |
| local_files_only=hf_offline(), | |
| ) | |
| depth_model = AutoModelForDepthEstimation.from_pretrained( | |
| DEPTH_MODEL_NAME, | |
| local_files_only=hf_offline(), | |
| ).to(device) | |
| depth_model.eval() | |
| print("Depth model loaded.", flush=True) | |
| return True | |
| def _load_moge_model() -> bool: | |
| global moge_model | |
| if not ENABLE_MOGE_GEOMETRY or GEOMETRY_MODEL != "moge" or device.type != "cuda": | |
| return False | |
| if moge_model is not None: | |
| return True | |
| print(f"Loading MoGe geometry model: {MOGE_MODEL_NAME} ...", flush=True) | |
| from moge.model.v2 import MoGeModel | |
| moge_model = MoGeModel.from_pretrained(MOGE_MODEL_NAME).to(device) | |
| moge_model.eval() | |
| print("MoGe geometry model loaded.", flush=True) | |
| return True | |
| def warm_start_models(): | |
| if not WARM_START_MODELS: | |
| print("Model warm start disabled; models will load on first use.", flush=True) | |
| return | |
| t_start = time.perf_counter() | |
| print(f"Warm-starting visualizer models on {device} ...", flush=True) | |
| _load_segmentation_model() | |
| moge_loaded = False | |
| if ENABLE_MOGE_GEOMETRY and GEOMETRY_MODEL == "moge": | |
| try: | |
| moge_loaded = _load_moge_model() | |
| except Exception as exc: | |
| print(f"MoGe warm start skipped ({exc}); depth fallback may be used.", flush=True) | |
| if device.type == "cuda": | |
| torch.cuda.empty_cache() | |
| if ENABLE_DEPTH_ESTIMATION and not moge_loaded: | |
| try: | |
| _load_depth_model() | |
| except Exception as exc: | |
| print(f"Depth model warm start skipped ({exc}); depth will be retried on first use.", flush=True) | |
| if device.type == "cuda": | |
| torch.cuda.empty_cache() | |
| if ENABLE_SAM_REFINEMENT: | |
| _load_sam2_predictor() | |
| if ENABLE_SHADING_TRANSFER and ENABLE_INTRINSIC_SHADING: | |
| _load_intrinsic_model() | |
| print(f"[TIMING] Model warm start took {time.perf_counter() - t_start:.3f} seconds", flush=True) | |
| app = FastAPI() | |
| app.add_middleware( | |
| CORSMiddleware, | |
| allow_origins=["*"], | |
| allow_methods=["*"], | |
| allow_headers=["*"], | |
| ) | |
| app.add_middleware(GZipMiddleware, minimum_size=1024) | |
| def startup_warm_start_models(): | |
| warm_start_models() | |
| DATA_DIR = Path(VISUALIZER_DATA_DIR).resolve() | |
| UPLOAD_DIR = DATA_DIR / "uploads" | |
| JOB_DIR = DATA_DIR / "jobs" | |
| MATERIAL_DIR = DATA_DIR / "materials" | |
| QA_RUN_DIR = DATA_DIR / "qa-runs" | |
| UPLOAD_DIR.mkdir(parents=True, exist_ok=True) | |
| JOB_DIR.mkdir(parents=True, exist_ok=True) | |
| MATERIAL_DIR.mkdir(parents=True, exist_ok=True) | |
| QA_RUN_DIR.mkdir(parents=True, exist_ok=True) | |
| app.mount("/uploads", StaticFiles(directory=UPLOAD_DIR), name="uploads") | |
| def _debug_resize_rgb(image_rgb: np.ndarray) -> np.ndarray: | |
| height, width = image_rgb.shape[:2] | |
| largest = max(height, width) | |
| if largest <= DEBUG_ARTIFACTS_MAX_DIMENSION: | |
| return image_rgb | |
| scale = DEBUG_ARTIFACTS_MAX_DIMENSION / largest | |
| return cv2.resize( | |
| image_rgb, | |
| (max(1, round(width * scale)), max(1, round(height * scale))), | |
| interpolation=cv2.INTER_AREA, | |
| ) | |
| def _debug_preview_rgb(image_rgb: np.ndarray) -> np.ndarray: | |
| height, width = image_rgb.shape[:2] | |
| largest = max(height, width) | |
| if largest <= DEBUG_ARTIFACTS_PREVIEW_DIMENSION: | |
| return image_rgb | |
| scale = DEBUG_ARTIFACTS_PREVIEW_DIMENSION / largest | |
| return cv2.resize( | |
| image_rgb, | |
| (max(1, round(width * scale)), max(1, round(height * scale))), | |
| interpolation=cv2.INTER_AREA, | |
| ) | |
| def debug_mask_overlay( | |
| image_rgb: np.ndarray, | |
| mask: np.ndarray, | |
| color: tuple[int, int, int] = (45, 180, 255), | |
| ) -> np.ndarray: | |
| """Overlay a binary analysis mask without changing the uploaded image.""" | |
| output = image_rgb.copy() | |
| valid = mask.astype(bool) | |
| if valid.any(): | |
| overlay = np.empty_like(output) | |
| overlay[:] = color | |
| output[valid] = cv2.addWeighted(output, 0.42, overlay, 0.58, 0)[valid] | |
| return output | |
| def debug_segmentation_overlay(image_rgb: np.ndarray, seg_map: np.ndarray) -> np.ndarray: | |
| labels = seg_map.astype(np.int32) | |
| hsv = np.zeros((*labels.shape, 3), dtype=np.uint8) | |
| hsv[..., 0] = (labels * 37) % 180 | |
| hsv[..., 1] = 180 | |
| hsv[..., 2] = 255 | |
| colors = cv2.cvtColor(hsv, cv2.COLOR_HSV2RGB) | |
| return cv2.addWeighted(image_rgb, 0.42, colors, 0.58, 0) | |
| def debug_scalar_map(values: np.ndarray, mask: np.ndarray | None = None) -> np.ndarray: | |
| scalar = values.astype(np.float32) | |
| finite = np.isfinite(scalar) | |
| if mask is not None: | |
| finite &= mask.astype(bool) | |
| valid = scalar[finite] | |
| if valid.size == 0: | |
| return np.zeros((*scalar.shape[:2], 3), dtype=np.uint8) | |
| low, high = np.percentile(valid, [2, 98]) | |
| if high <= low: | |
| high = low + 1.0 | |
| normalized = np.clip((scalar - low) / (high - low), 0.0, 1.0) | |
| normalized[~np.isfinite(normalized)] = 0.0 | |
| output = cv2.applyColorMap(np.round(normalized * 255).astype(np.uint8), cv2.COLORMAP_TURBO) | |
| output = cv2.cvtColor(output, cv2.COLOR_BGR2RGB) | |
| if mask is not None: | |
| output[~mask.astype(bool)] = 0 | |
| return output | |
| def debug_plane_overlay(image_rgb: np.ndarray, plane: dict | None) -> np.ndarray: | |
| output = image_rgb.copy() | |
| if not plane: | |
| return output | |
| raw_quad = plane.get("quad") | |
| if not isinstance(raw_quad, list) or len(raw_quad) != 8: | |
| return output | |
| points = np.asarray(raw_quad, dtype=np.float32).reshape(-1, 2) | |
| if not np.isfinite(points).all(): | |
| return output | |
| points_i = np.round(points).astype(np.int32) | |
| overlay = output.copy() | |
| cv2.fillConvexPoly(overlay, points_i, (45, 180, 255)) | |
| output = cv2.addWeighted(output, 0.64, overlay, 0.36, 0) | |
| cv2.polylines(output, [points_i], isClosed=True, color=(255, 255, 255), thickness=3, lineType=cv2.LINE_AA) | |
| return output | |
| def debug_surface_overlay(image_rgb: np.ndarray, floor_mask: np.ndarray, surface_mask: np.ndarray) -> np.ndarray: | |
| output = image_rgb.copy() | |
| removed = (floor_mask > 0) & (surface_mask == 0) | |
| if removed.any(): | |
| red = np.empty_like(output) | |
| red[:] = (220, 80, 80) | |
| output[removed] = cv2.addWeighted(output, 0.42, red, 0.58, 0)[removed] | |
| return debug_mask_overlay(output, surface_mask, (45, 180, 255)) | |
| def debug_floor_direction_overlay(image_rgb: np.ndarray, floor_direction: dict) -> np.ndarray: | |
| output = image_rgb.copy() | |
| direction_axis = floor_direction.get("directionAxis") | |
| if isinstance(direction_axis, dict): | |
| try: | |
| start = ( | |
| round(float(direction_axis["startX"])), | |
| round(float(direction_axis["startY"])), | |
| ) | |
| end = ( | |
| round(float(direction_axis["endX"])), | |
| round(float(direction_axis["endY"])), | |
| ) | |
| cv2.arrowedLine( | |
| output, | |
| start, | |
| end, | |
| (45, 180, 255), | |
| 4, | |
| cv2.LINE_AA, | |
| tipLength=0.08, | |
| ) | |
| except (KeyError, TypeError, ValueError): | |
| pass | |
| geometry_axis = floor_direction.get("geometryAxis") | |
| if isinstance(geometry_axis, dict): | |
| try: | |
| start = ( | |
| round(float(geometry_axis["nearX"])), | |
| round(float(geometry_axis["nearY"])), | |
| ) | |
| end = ( | |
| round(float(geometry_axis["farX"])), | |
| round(float(geometry_axis["farY"])), | |
| ) | |
| cv2.arrowedLine( | |
| output, | |
| start, | |
| end, | |
| (45, 180, 255), | |
| 4, | |
| cv2.LINE_AA, | |
| tipLength=0.08, | |
| ) | |
| except (KeyError, TypeError, ValueError): | |
| pass | |
| for line in floor_direction.get("dominantLines") or []: | |
| try: | |
| start = (round(float(line["x1"])), round(float(line["y1"]))) | |
| end = (round(float(line["x2"])), round(float(line["y2"]))) | |
| except (KeyError, TypeError, ValueError): | |
| continue | |
| cv2.line(output, start, end, (45, 180, 255), 3, cv2.LINE_AA) | |
| label = floor_direction.get("directionLabel") or "unknown" | |
| angle = floor_direction.get("renderAngleDegrees") | |
| confidence = floor_direction.get("confidence") | |
| detail = f"{label}" | |
| if isinstance(angle, (int, float)): | |
| detail += f" {angle:.1f} deg" | |
| if isinstance(confidence, (int, float)): | |
| detail += f" {confidence:.0%}" | |
| cv2.putText(output, detail, (16, 34), cv2.FONT_HERSHEY_SIMPLEX, 0.72, (255, 255, 255), 3, cv2.LINE_AA) | |
| cv2.putText(output, detail, (16, 34), cv2.FONT_HERSHEY_SIMPLEX, 0.72, (25, 25, 25), 1, cv2.LINE_AA) | |
| return output | |
| def debug_uv_checkerboard( | |
| height: int, | |
| width: int, | |
| surface_indices: np.ndarray, | |
| uv: np.ndarray, | |
| ) -> np.ndarray: | |
| output = np.zeros((height, width, 3), dtype=np.uint8) | |
| valid = np.isfinite(uv).all(axis=1) | |
| if not valid.any(): | |
| return output | |
| coords = uv[valid] | |
| cells = (np.floor(coords[:, 0] * 8) + np.floor(coords[:, 1] * 8)).astype(np.int32) | |
| colors = np.where((cells % 2)[:, None] == 0, np.array([45, 180, 255]), np.array([45, 80, 180])) | |
| output.reshape(-1, 3)[surface_indices[valid]] = colors.astype(np.uint8) | |
| return output | |
| def debug_plane_id_map( | |
| height: int, | |
| width: int, | |
| surface_indices: np.ndarray, | |
| plane_ids: np.ndarray, | |
| ) -> np.ndarray: | |
| output = np.zeros((height, width, 3), dtype=np.uint8) | |
| valid = plane_ids >= 0 | |
| if not valid.any(): | |
| return output | |
| ids = plane_ids[valid].astype(np.int32) | |
| hsv = np.zeros((len(ids), 1, 3), dtype=np.uint8) | |
| hsv[:, 0, 0] = (ids * 47) % 180 | |
| hsv[:, 0, 1] = 190 | |
| hsv[:, 0, 2] = 255 | |
| output.reshape(-1, 3)[surface_indices[valid]] = cv2.cvtColor(hsv, cv2.COLOR_HSV2RGB).reshape(-1, 3) | |
| return output | |
| def _debug_json_default(value): | |
| if isinstance(value, np.generic): | |
| return value.item() | |
| if isinstance(value, np.ndarray): | |
| return value.tolist() | |
| raise TypeError(f"Unsupported debug manifest value: {type(value).__name__}") | |
| class PipelineDebugCapture: | |
| """Writes a generic, versioned inspection manifest alongside a conversion job.""" | |
| def __init__(self, job_id: str): | |
| self.job_id = job_id | |
| self.root = JOB_DIR / f"{job_id}.debug" | |
| self.root.mkdir(parents=True, exist_ok=True) | |
| self.manifest_path = self.root / "manifest.json" | |
| self.manifest = { | |
| "schemaVersion": 2, | |
| "jobId": job_id, | |
| "status": "PROCESSING", | |
| "stages": [], | |
| } | |
| self._write_manifest() | |
| def _write_manifest(self): | |
| self.manifest["updatedAtMs"] = int(time.time() * 1000) | |
| self.manifest_path.write_text( | |
| json.dumps(self.manifest, separators=(",", ":"), default=_debug_json_default), | |
| encoding="utf-8", | |
| ) | |
| def record_images( | |
| self, | |
| stage_id: str, | |
| title: str, | |
| description: str, | |
| images: list[tuple[str, np.ndarray]], | |
| metrics: dict | None = None, | |
| elapsed_ms: float | None = None, | |
| ): | |
| if not DEBUG_ARTIFACTS_ENABLED: | |
| return | |
| try: | |
| artifacts = [] | |
| for index, (label, image_rgb) in enumerate(images): | |
| filename = f"{len(self.manifest['stages']) + 1:02d}-{stage_id}-{index + 1}.jpg" | |
| image_rgb = _debug_resize_rgb(image_rgb) | |
| ok, encoded = cv2.imencode(".jpg", cv2.cvtColor(image_rgb, cv2.COLOR_RGB2BGR), [cv2.IMWRITE_JPEG_QUALITY, 90]) | |
| if not ok: | |
| continue | |
| (self.root / filename).write_bytes(encoded.tobytes()) | |
| preview_filename = f"{filename[:-4]}.preview.jpg" | |
| preview_rgb = _debug_preview_rgb(image_rgb) | |
| preview_ok, preview_encoded = cv2.imencode( | |
| ".jpg", | |
| cv2.cvtColor(preview_rgb, cv2.COLOR_RGB2BGR), | |
| [cv2.IMWRITE_JPEG_QUALITY, 82], | |
| ) | |
| if preview_ok: | |
| (self.root / preview_filename).write_bytes(preview_encoded.tobytes()) | |
| artifacts.append({ | |
| "type": "image", | |
| "label": label, | |
| "url": f"/viz2d/jobs/{self.job_id}/debug/assets/{filename}", | |
| "previewUrl": ( | |
| f"/viz2d/jobs/{self.job_id}/debug/assets/{preview_filename}" | |
| if preview_ok else None | |
| ), | |
| }) | |
| stage = { | |
| "id": stage_id, | |
| "title": title, | |
| "description": description, | |
| "artifacts": artifacts, | |
| "metrics": metrics or {}, | |
| } | |
| if elapsed_ms is not None: | |
| stage["elapsedMs"] = round(float(elapsed_ms), 1) | |
| self.manifest["stages"].append(stage) | |
| self._write_manifest() | |
| except Exception as exc: | |
| print(f"[DEBUG] Could not capture {stage_id} for job {self.job_id}: {exc}", flush=True) | |
| def complete(self, total_elapsed_ms: float): | |
| if not DEBUG_ARTIFACTS_ENABLED: | |
| return | |
| self.manifest["status"] = "COMPLETED" | |
| self.manifest["totalElapsedMs"] = round(float(total_elapsed_ms), 1) | |
| self._write_manifest() | |
| def fail(self, error: Exception): | |
| if not DEBUG_ARTIFACTS_ENABLED: | |
| return | |
| self.manifest["status"] = "FAILED" | |
| self.manifest["error"] = str(error) | |
| self._write_manifest() | |
| PRIMARY_FLOOR_CLASSES = {"floor", "rug", "carpet", "mat"} | |
| FLOOR_SURFACE_CLASSES = { | |
| "floor", "road", "sidewalk", "ground", "field", "grass", "sand", | |
| "runway", "dirt track", "land", "stairs", "step", "rug", "carpet", "mat", | |
| } | |
| REJECT_SURFACE_CLASSES = {"wall", "ceiling", "building", "sky", "window"} | |
| SOFT_FLOOR_COVERING_CLASSES = set() | |
| OCCLUDER_CLASSES = { | |
| "bed", "cabinet", "person", "door", "table", "plant", "curtain", "chair", | |
| "car", "painting", "sofa", "shelf", "mirror", "armchair", "seat", "desk", | |
| "wardrobe", "lamp", "bathtub", "railing", "cushion", "base", "box", | |
| "column", "chest of drawers", "counter", "sink", "fireplace", | |
| "refrigerator", "bookcase", "blind", "coffee table", "toilet", "bench", | |
| "countertop", "stove", "kitchen island", "computer", "swivel chair", | |
| "bar", "ottoman", "bottle", "buffet", "poster", "towel", "television", | |
| "washer", "plaything", "stool", "basket", "bag", "cradle", "oven", | |
| "ball", "food", "microwave", "pot", "dishwasher", "blanket", "sculpture", | |
| "vase", "tray", "fan", "plate", "monitor", "shower", "radiator", "clock", | |
| } | |
| def class_name_for_id(class_id: int) -> str: | |
| return ADE20K_CLASSES[class_id] if class_id < len(ADE20K_CLASSES) else f"class_{class_id}" | |
| def class_ids(names: set[str]) -> list[int]: | |
| return [idx for idx, name in enumerate(ADE20K_CLASSES) if name in names] | |
| def estimate_depth(img: Image.Image, width: int, height: int): | |
| if not ENABLE_DEPTH_ESTIMATION: | |
| return None | |
| try: | |
| if not _load_depth_model(): | |
| return None | |
| inputs = depth_processor(images=img, return_tensors="pt").to(device) | |
| with torch.no_grad(): | |
| outputs = depth_model(**inputs) | |
| depth = torch.nn.functional.interpolate( | |
| outputs.predicted_depth.unsqueeze(1), | |
| size=(height, width), | |
| mode="bicubic", | |
| align_corners=False, | |
| ).squeeze().cpu().numpy() | |
| depth = cv2.GaussianBlur(depth.astype(np.float32), (0, 0), sigmaX=3) | |
| if "metric" in DEPTH_MODEL_NAME.lower(): | |
| return depth | |
| depth_min, depth_max = float(np.min(depth)), float(np.max(depth)) | |
| if depth_max - depth_min < 1e-6: | |
| return None | |
| return (depth - depth_min) / (depth_max - depth_min) | |
| except Exception as exc: | |
| print(f"Depth estimation skipped ({exc}).", flush=True) | |
| return None | |
| def _to_numpy(value): | |
| if value is None: | |
| return None | |
| if isinstance(value, torch.Tensor): | |
| return value.detach().float().cpu().numpy() | |
| return np.asarray(value) | |
| def estimate_moge_geometry(img_np: np.ndarray): | |
| if not ENABLE_MOGE_GEOMETRY or GEOMETRY_MODEL != "moge" or device.type != "cuda": | |
| return None | |
| try: | |
| if not _load_moge_model(): | |
| return None | |
| image_tensor = torch.from_numpy(img_np.astype(np.float32) / 255.0).permute(2, 0, 1).to(device) | |
| with torch.inference_mode(), torch.autocast(device_type="cuda", dtype=torch.float16): | |
| output = moge_model.infer(image_tensor) | |
| points = _to_numpy(output.get("points")) | |
| depth = _to_numpy(output.get("depth")) | |
| normals = _to_numpy(output.get("normal")) | |
| valid_mask = _to_numpy(output.get("mask")) | |
| intrinsics = _to_numpy(output.get("intrinsics")) | |
| if points is None or points.ndim != 3 or points.shape[2] != 3: | |
| raise RuntimeError("MoGe did not return an HxWx3 point map.") | |
| h, w = img_np.shape[:2] | |
| if points.shape[:2] != (h, w): | |
| points = cv2.resize(points, (w, h), interpolation=cv2.INTER_LINEAR) | |
| if depth is not None and depth.shape[:2] != (h, w): | |
| depth = cv2.resize(depth, (w, h), interpolation=cv2.INTER_LINEAR) | |
| if normals is not None and normals.shape[:2] != (h, w): | |
| normals = cv2.resize(normals, (w, h), interpolation=cv2.INTER_LINEAR) | |
| if valid_mask is not None and valid_mask.shape[:2] != (h, w): | |
| valid_mask = cv2.resize( | |
| valid_mask.astype(np.uint8), | |
| (w, h), | |
| interpolation=cv2.INTER_NEAREST, | |
| ).astype(bool) | |
| if depth is None: | |
| depth = points[:, :, 2] | |
| if valid_mask is None: | |
| valid_mask = np.isfinite(depth) & (depth > 0) | |
| print( | |
| "[FLOOR_3D_POINTS] generated raw MoGe HxWx3 point map " | |
| f"shape={points.shape} validPixels={int(valid_mask.sum())} provider=moge-2", | |
| flush=True, | |
| ) | |
| return { | |
| "provider": "moge-2", | |
| "points": points.astype(np.float32), | |
| "depth": depth.astype(np.float32), | |
| "normals": normals.astype(np.float32) if normals is not None else None, | |
| "validMask": valid_mask.astype(bool), | |
| "intrinsics": intrinsics.astype(np.float32) if intrinsics is not None else None, | |
| } | |
| except Exception as exc: | |
| print(f"MoGe geometry estimation failed ({exc}); falling back to depth.", flush=True) | |
| if device.type == "cuda": | |
| torch.cuda.empty_cache() | |
| return None | |
| def estimate_scene_geometry(img: Image.Image, img_np: np.ndarray, width: int, height: int): | |
| geometry = estimate_moge_geometry(img_np) | |
| if geometry is not None: | |
| return geometry | |
| depth = estimate_depth(img, width, height) | |
| depth_provider = "homography" | |
| if depth is not None: | |
| depth_provider = ( | |
| "depth-anything-metric" | |
| if "metric" in DEPTH_MODEL_NAME.lower() | |
| else "relative-depth" | |
| ) | |
| return { | |
| "provider": depth_provider, | |
| "points": None, | |
| "depth": depth, | |
| "normals": None, | |
| "validMask": np.isfinite(depth) & (depth > 0) if depth is not None else None, | |
| "intrinsics": None, | |
| } | |
| def _load_sam2_predictor(): | |
| global sam2_predictor | |
| if not ENABLE_SAM_REFINEMENT: | |
| return None | |
| if sam2_predictor is not None: | |
| return sam2_predictor | |
| try: | |
| from sam2.sam2_image_predictor import SAM2ImagePredictor | |
| if SAM2_MODEL_CFG and SAM2_CHECKPOINT: | |
| from sam2.build_sam import build_sam2 | |
| model = build_sam2(SAM2_MODEL_CFG, SAM2_CHECKPOINT, device=str(device)) | |
| model.eval() | |
| sam2_predictor = SAM2ImagePredictor(model) | |
| else: | |
| sam2_predictor = SAM2ImagePredictor.from_pretrained(SAM2_MODEL_NAME) | |
| if hasattr(sam2_predictor, "model"): | |
| sam2_predictor.model.to(device) | |
| sam2_predictor.model.eval() | |
| print("SAM2 refinement model loaded.", flush=True) | |
| return sam2_predictor | |
| except Exception as exc: | |
| print(f"SAM2 refinement unavailable ({exc}).", flush=True) | |
| if device.type == "cuda": | |
| torch.cuda.empty_cache() | |
| return None | |
| def sample_prompt_points(mask: np.ndarray, count: int, positive: bool) -> list[list[float]]: | |
| ys, xs = np.where(mask > 0) | |
| if len(xs) == 0: | |
| return [] | |
| if positive: | |
| step = max(1, len(xs) // max(count, 1)) | |
| order = np.argsort(np.abs(ys - np.median(ys)) + np.abs(xs - np.median(xs))) | |
| chosen = order[::step][:count] | |
| else: | |
| step = max(1, len(xs) // max(count, 1)) | |
| chosen = np.arange(0, len(xs), step)[:count] | |
| return [[float(xs[i]), float(ys[i])] for i in chosen] | |
| def refine_floor_mask_with_sam2( | |
| img_np: np.ndarray, | |
| floor_mask: np.ndarray, | |
| seg_map: np.ndarray, | |
| ) -> tuple[np.ndarray, dict]: | |
| metadata = { | |
| "samRefinementEnabled": ENABLE_SAM_REFINEMENT, | |
| "samRefinementApplied": False, | |
| "samRefinementReason": "disabled" if not ENABLE_SAM_REFINEMENT else "unavailable", | |
| "samRefinementScore": None, | |
| } | |
| if not ENABLE_SAM_REFINEMENT or int(floor_mask.sum()) < SURFACE_MIN_PLANE_PIXELS: | |
| return floor_mask, metadata | |
| predictor = _load_sam2_predictor() | |
| if predictor is None: | |
| return floor_mask, metadata | |
| ys, xs = np.where(floor_mask > 0) | |
| if len(xs) == 0: | |
| metadata["samRefinementReason"] = "empty-floor-mask" | |
| return floor_mask, metadata | |
| h, w = floor_mask.shape[:2] | |
| pad = max(4, min(h, w) // 80) | |
| box = np.array([ | |
| max(0, int(xs.min()) - pad), | |
| max(0, int(ys.min()) - pad), | |
| min(w - 1, int(xs.max()) + pad), | |
| min(h - 1, int(ys.max()) + pad), | |
| ], dtype=np.float32) | |
| positive_mask = cv2.erode( | |
| floor_mask.astype(np.uint8), | |
| cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (9, 9)), | |
| iterations=1, | |
| ) | |
| reject_mask = np.isin(seg_map, class_ids(REJECT_SURFACE_CLASSES | OCCLUDER_CLASSES)).astype(np.uint8) | |
| negative_mask = cv2.dilate( | |
| reject_mask, | |
| cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (7, 7)), | |
| iterations=1, | |
| ) | |
| positive_points = sample_prompt_points(positive_mask, 4, positive=True) | |
| negative_points = sample_prompt_points(negative_mask, 6, positive=False) | |
| point_coords = np.asarray(positive_points + negative_points, dtype=np.float32) | |
| point_labels = np.asarray( | |
| [1] * len(positive_points) + [0] * len(negative_points), | |
| dtype=np.int32, | |
| ) | |
| try: | |
| predictor.set_image(img_np) | |
| masks, scores, _ = predictor.predict( | |
| point_coords=point_coords if len(point_coords) else None, | |
| point_labels=point_labels if len(point_labels) else None, | |
| box=box, | |
| multimask_output=True, | |
| ) | |
| except Exception as exc: | |
| metadata["samRefinementReason"] = f"predict-failed:{exc}" | |
| return floor_mask, metadata | |
| if masks is None or len(masks) == 0: | |
| metadata["samRefinementReason"] = "no-mask" | |
| return floor_mask, metadata | |
| best_idx = int(np.argmax(scores)) if scores is not None and len(scores) else 0 | |
| sam_mask = masks[best_idx].astype(np.uint8) | |
| if sam_mask.shape[:2] != floor_mask.shape: | |
| sam_mask = cv2.resize(sam_mask, (w, h), interpolation=cv2.INTER_NEAREST) | |
| reject_dilated = cv2.dilate( | |
| reject_mask, | |
| cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (5, 5)), | |
| iterations=1, | |
| ) | |
| floor_dilated = cv2.dilate( | |
| floor_mask.astype(np.uint8), | |
| cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (max(5, min(h, w) // 80) | 1,) * 2), | |
| iterations=2, | |
| ) | |
| refined = (sam_mask & floor_dilated).astype(np.uint8) | |
| refined[reject_dilated > 0] = 0 | |
| refined = clean_floor_mask(refined) | |
| refined_area = int(refined.sum()) | |
| original_area = int(floor_mask.sum()) | |
| if refined_area < SURFACE_MIN_PLANE_PIXELS or refined_area < original_area * 0.35: | |
| metadata["samRefinementReason"] = "refined-mask-too-small" | |
| return floor_mask, metadata | |
| metadata.update({ | |
| "samRefinementApplied": True, | |
| "samRefinementReason": "sam2-mask-refined", | |
| "samRefinementScore": float(scores[best_idx]) if scores is not None and len(scores) else None, | |
| "samRefinementOriginalPixels": original_area, | |
| "samRefinementPixels": refined_area, | |
| }) | |
| return refined, metadata | |
| def encode_relative_map(values: np.ndarray, minimum: float, maximum: float) -> np.ndarray: | |
| clipped = np.clip(values, minimum, maximum) | |
| scale = 255.0 / max(maximum - minimum, 1e-6) | |
| return np.round((clipped - minimum) * scale).clip(0, 255).astype(np.uint8) | |
| def image_luminance(img_np: np.ndarray) -> np.ndarray: | |
| return ( | |
| img_np[:, :, 0].astype(np.float32) * 0.299 | |
| + img_np[:, :, 1].astype(np.float32) * 0.587 | |
| + img_np[:, :, 2].astype(np.float32) * 0.114 | |
| ) | |
| def image_linear_luminance(img_np: np.ndarray) -> np.ndarray: | |
| """Return linear-light luminance in the same 0..255 scale as image_luminance.""" | |
| srgb = np.clip(img_np.astype(np.float32) / 255.0, 0.0, 1.0) | |
| linear = np.where( | |
| srgb <= 0.04045, | |
| srgb / 12.92, | |
| np.power((srgb + 0.055) / 1.055, 2.4), | |
| ) | |
| luminance = ( | |
| linear[:, :, 0] * 0.2126 | |
| + linear[:, :, 1] * 0.7152 | |
| + linear[:, :, 2] * 0.0722 | |
| ) | |
| return luminance * 255.0 | |
| def inpaint_surface_values(values: np.ndarray, mask: np.ndarray, fill_value: float) -> np.ndarray: | |
| h, w = mask.shape[:2] | |
| filled = values.astype(np.float32).copy() | |
| filled[mask == 0] = fill_value | |
| missing = (mask == 0).astype(np.uint8) * 255 | |
| try: | |
| source = np.clip(filled, 0, 255).astype(np.uint8) | |
| filled = cv2.inpaint( | |
| source, | |
| missing, | |
| max(3, min(h, w) // 160), | |
| cv2.INPAINT_TELEA, | |
| ).astype(np.float32) | |
| except cv2.error: | |
| pass | |
| return filled | |
| def build_showroom_shade_map_from_relative( | |
| relative: np.ndarray, | |
| surface_mask: np.ndarray, | |
| ) -> np.ndarray | None: | |
| if not surface_mask.any(): | |
| return None | |
| mask = surface_mask.astype(np.uint8) | |
| h, w = mask.shape[:2] | |
| if relative.shape[:2] != (h, w): | |
| relative = cv2.resize(relative.astype(np.float32), (w, h), interpolation=cv2.INTER_LINEAR) | |
| rel = relative.astype(np.float32).copy() | |
| floor_values = rel[mask > 0] | |
| floor_values = floor_values[np.isfinite(floor_values) & (floor_values > 1e-6)] | |
| if floor_values.size < max(256, int(h * w * 0.002)): | |
| return None | |
| median_rel = float(np.median(floor_values)) | |
| if median_rel < 1e-6: | |
| return None | |
| rel[~np.isfinite(rel)] = median_rel | |
| rel[mask == 0] = median_rel | |
| rel = rel / median_rel | |
| broad_sigma = max(18.0, min(h, w) / 18.0) | |
| broad = cv2.GaussianBlur(rel, (0, 0), sigmaX=broad_sigma, sigmaY=broad_sigma) | |
| broad_values = broad[mask > 0] | |
| broad_values = broad_values[np.isfinite(broad_values) & (broad_values > 1e-6)] | |
| if broad_values.size == 0: | |
| return None | |
| broad_median = float(np.median(broad_values)) | |
| if broad_median < 1e-6: | |
| return None | |
| showroom = broad / broad_median | |
| showroom = 1.0 + (showroom - 1.0) * 0.20 | |
| showroom = np.clip(showroom, SHOWROOM_SHADE_MAP_MIN, SHOWROOM_SHADE_MAP_MAX) | |
| showroom[mask == 0] = 1.0 | |
| return encode_relative_map(showroom, SHOWROOM_SHADE_MAP_MIN, SHOWROOM_SHADE_MAP_MAX) | |
| def build_showroom_shade_map_from_luminance( | |
| luminance: np.ndarray, | |
| surface_mask: np.ndarray, | |
| median_lum: float, | |
| ) -> np.ndarray | None: | |
| if median_lum < 1e-3: | |
| return None | |
| relative = inpaint_surface_values(luminance, surface_mask.astype(np.uint8), median_lum) | |
| relative = relative / max(median_lum, 1e-3) | |
| relative[~np.isfinite(relative)] = 1.0 | |
| return build_showroom_shade_map_from_relative(relative, surface_mask) | |
| def build_reflection_map( | |
| img_np: np.ndarray, | |
| surface_mask: np.ndarray, | |
| relative_shading: np.ndarray | None = None, | |
| ) -> np.ndarray | None: | |
| if not surface_mask.any(): | |
| return None | |
| mask = surface_mask.astype(np.uint8) | |
| h, w = mask.shape[:2] | |
| reflection_candidates: list[np.ndarray] = [] | |
| if relative_shading is not None: | |
| rel = relative_shading.astype(np.float32).copy() | |
| rel[mask == 0] = 1.0 | |
| broad_sigma = max(6.0, min(h, w) / 44.0) | |
| broad = cv2.GaussianBlur(rel, (0, 0), sigmaX=broad_sigma, sigmaY=broad_sigma) | |
| detail = rel / np.maximum(broad, 1e-3) | |
| reflection_candidates.append(np.clip((detail - 1.05) / 0.30, 0.0, 1.0)) | |
| luminance = image_luminance(img_np) / 255.0 | |
| floor_values = luminance[mask > 0] | |
| floor_values = floor_values[np.isfinite(floor_values)] | |
| if floor_values.size >= max(256, int(h * w * 0.002)): | |
| median_lum = float(np.median(floor_values)) | |
| if median_lum > 1e-3: | |
| filled = luminance.copy() | |
| filled[mask == 0] = median_lum | |
| broad_sigma = max(8.0, min(h, w) / 36.0) | |
| broad = cv2.GaussianBlur(filled, (0, 0), sigmaX=broad_sigma, sigmaY=broad_sigma) | |
| detail = filled / np.maximum(broad, 1e-3) | |
| reflection_candidates.append(np.clip((detail - 1.08) / 0.30, 0.0, 1.0)) | |
| if not reflection_candidates: | |
| return None | |
| # A reflection is additive light. Dark source-floor detail belongs to the | |
| # illumination map; transferring it here would turn old material color or | |
| # texture into an artificial dark reflection on the replacement material. | |
| reflection = np.maximum.reduce(reflection_candidates).astype(np.float32) | |
| if min(h, w) > 120: | |
| sigma = max(0.6, min(h, w) / 520.0) | |
| reflection = cv2.GaussianBlur(reflection, (0, 0), sigmaX=sigma, sigmaY=sigma) | |
| reflection = np.clip(reflection, REFLECTION_MAP_MIN, REFLECTION_MAP_MAX) | |
| reflection[mask == 0] = 0.0 | |
| if float(np.max(np.abs(reflection[mask > 0]))) < 0.025: | |
| return None | |
| encoded = (reflection - REFLECTION_MAP_MIN) / (REFLECTION_MAP_MAX - REFLECTION_MAP_MIN) | |
| return np.round(encoded * 255.0).clip(0, 255).astype(np.uint8) | |
| def build_luminance_lighting_maps( | |
| img_np: np.ndarray, | |
| surface_mask: np.ndarray, | |
| ) -> dict[str, np.ndarray | None] | None: | |
| if not surface_mask.any(): | |
| return None | |
| mask = surface_mask.astype(np.uint8) | |
| luminance = image_linear_luminance(img_np) | |
| h, w = mask.shape[:2] | |
| floor_values = luminance[mask > 0] | |
| if floor_values.size < max(256, int(h * w * 0.002)): | |
| return None | |
| median_lum = float(np.median(floor_values)) | |
| if median_lum < 1e-3: | |
| return None | |
| filled = inpaint_surface_values(luminance, mask, median_lum) | |
| relative = filled / median_lum | |
| relative[~np.isfinite(relative)] = 1.0 | |
| shade = smooth_relative_lighting(relative, mask) | |
| if shade is None: | |
| return None | |
| return { | |
| "shadeMap": encode_relative_map(shade, SHADE_MAP_MIN, SHADE_MAP_MAX), | |
| "showroomShadeMap": build_showroom_shade_map_from_luminance(luminance, mask, median_lum), | |
| "reflectionMap": build_reflection_map(img_np, surface_mask, None), | |
| } | |
| def extract_intrinsic_shading(results: dict) -> np.ndarray | None: | |
| shading = None | |
| if "gry_shd" in results: | |
| shading = results["gry_shd"] | |
| elif "dif_shd" in results: | |
| shading = results["dif_shd"] | |
| else: | |
| for key, value in results.items(): | |
| if "shd" in key or "shading" in key: | |
| shading = value | |
| break | |
| if shading is None: | |
| return None | |
| shading = _to_numpy(shading) | |
| if shading is None: | |
| return None | |
| if shading.ndim == 3: | |
| if shading.shape[2] == 1: | |
| shading = shading[:, :, 0] | |
| else: | |
| shading = ( | |
| shading[:, :, 0] * 0.299 | |
| + shading[:, :, 1] * 0.587 | |
| + shading[:, :, 2] * 0.114 | |
| ) | |
| if shading.ndim != 2: | |
| return None | |
| return shading.astype(np.float32) | |
| def normalize_relative_shading( | |
| shading: np.ndarray, | |
| surface_mask: np.ndarray, | |
| ) -> np.ndarray | None: | |
| h, w = surface_mask.shape[:2] | |
| if shading.shape[:2] != (h, w): | |
| shading = cv2.resize(shading, (w, h), interpolation=cv2.INTER_LINEAR) | |
| shading = shading.astype(np.float32) | |
| floor_vals = shading[surface_mask > 0] | |
| floor_vals = floor_vals[np.isfinite(floor_vals) & (floor_vals > 1e-6)] | |
| if floor_vals.size == 0: | |
| return None | |
| median_val = float(np.median(floor_vals)) | |
| if median_val < 1e-6: | |
| return None | |
| relative = shading / median_val | |
| relative[~np.isfinite(relative)] = 1.0 | |
| relative[surface_mask == 0] = 1.0 | |
| return relative.astype(np.float32) | |
| def masked_gaussian_blur( | |
| values: np.ndarray, | |
| surface_mask: np.ndarray, | |
| sigma: float, | |
| ) -> np.ndarray: | |
| mask = surface_mask.astype(np.float32) | |
| numerator = cv2.GaussianBlur(values.astype(np.float32) * mask, (0, 0), sigmaX=sigma, sigmaY=sigma) | |
| denominator = cv2.GaussianBlur(mask, (0, 0), sigmaX=sigma, sigmaY=sigma) | |
| return numerator / np.maximum(denominator, 1e-4) | |
| def normalize_relative_field(values: np.ndarray, surface_mask: np.ndarray) -> np.ndarray | None: | |
| relative = values.astype(np.float32).copy() | |
| valid = relative[surface_mask > 0] | |
| valid = valid[np.isfinite(valid) & (valid > 1e-6)] | |
| if valid.size == 0: | |
| return None | |
| median = float(np.median(valid)) | |
| if median < 1e-6: | |
| return None | |
| relative /= median | |
| relative[~np.isfinite(relative)] = 1.0 | |
| relative[surface_mask == 0] = 1.0 | |
| return relative | |
| def broad_luminance_relative( | |
| img_np: np.ndarray, | |
| surface_mask: np.ndarray, | |
| ) -> np.ndarray | None: | |
| luminance = image_linear_luminance(img_np) / 255.0 | |
| floor_values = luminance[surface_mask > 0] | |
| floor_values = floor_values[np.isfinite(floor_values) & (floor_values > 1e-6)] | |
| if floor_values.size == 0: | |
| return None | |
| sigma = max(12.0, min(surface_mask.shape[:2]) / 32.0) | |
| broad = masked_gaussian_blur(luminance, surface_mask, sigma) | |
| return normalize_relative_field(broad, surface_mask) | |
| def align_intrinsic_shading_to_luminance( | |
| relative_shading: np.ndarray, | |
| img_np: np.ndarray, | |
| surface_mask: np.ndarray, | |
| ) -> tuple[np.ndarray, str]: | |
| """Correct intrinsic maps that encode lightness with the opposite polarity.""" | |
| observed = broad_luminance_relative(img_np, surface_mask) | |
| if observed is None: | |
| return relative_shading, "intrinsic-unchecked" | |
| sigma = max(12.0, min(surface_mask.shape[:2]) / 32.0) | |
| intrinsic_broad = masked_gaussian_blur(relative_shading, surface_mask, sigma) | |
| intrinsic_broad = normalize_relative_field(intrinsic_broad, surface_mask) | |
| if intrinsic_broad is None: | |
| return relative_shading, "intrinsic-unchecked" | |
| valid = ( | |
| (surface_mask > 0) | |
| & np.isfinite(observed) | |
| & np.isfinite(intrinsic_broad) | |
| & (observed > 1e-4) | |
| & (intrinsic_broad > 1e-4) | |
| ) | |
| if int(valid.sum()) < 256: | |
| return relative_shading, "intrinsic-unchecked" | |
| observed_log = np.log(observed[valid]) | |
| intrinsic_log = np.log(intrinsic_broad[valid]) | |
| if float(np.std(observed_log)) < 1e-4 or float(np.std(intrinsic_log)) < 1e-4: | |
| return relative_shading, "intrinsic-unchecked" | |
| correlation = float(np.corrcoef(observed_log, intrinsic_log)[0, 1]) | |
| if correlation < -0.12: | |
| corrected = normalize_relative_field(1.0 / np.maximum(relative_shading, 1e-4), surface_mask) | |
| if corrected is not None: | |
| return corrected, "intrinsic-inverted-corrected" | |
| return relative_shading, "intrinsic-aligned" if correlation >= 0.10 else "intrinsic-low-confidence" | |
| def smooth_relative_lighting(relative_shading: np.ndarray, surface_mask: np.ndarray) -> np.ndarray | None: | |
| """Keep room and cast-shadow illumination while rejecting old material detail.""" | |
| min_side = min(surface_mask.shape[:2]) | |
| broad_sigma = max(12.0, min_side / 32.0) | |
| local_sigma = max(2.5, min_side / 160.0) | |
| bounded = np.clip(relative_shading.astype(np.float32), 0.04, 4.0) | |
| broad = masked_gaussian_blur(bounded, surface_mask, broad_sigma) | |
| local = masked_gaussian_blur(bounded, surface_mask, local_sigma) | |
| broad_relative = normalize_relative_field(broad, surface_mask) | |
| if broad_relative is None: | |
| return None | |
| local_contrast = local / np.maximum(broad, 1e-4) | |
| local_log = np.log(np.maximum(local_contrast, 1e-4)) | |
| local_log = np.clip(local_log, np.log(0.25), np.log(1.45)) | |
| combined = broad_relative * np.exp(local_log * 0.70) | |
| normalized = normalize_relative_field(combined, surface_mask) | |
| if normalized is None: | |
| return None | |
| normalized = np.clip(normalized, SHADE_MAP_MIN, 1.65) | |
| normalized[surface_mask == 0] = 1.0 | |
| return normalized | |
| def lighting_map_statistics(encoded: np.ndarray | None, surface_mask: np.ndarray) -> dict | None: | |
| if encoded is None or encoded.shape[:2] != surface_mask.shape[:2]: | |
| return None | |
| decoded = SHADE_MAP_MIN + (encoded.astype(np.float32) / 255.0) * (SHADE_MAP_MAX - SHADE_MAP_MIN) | |
| values = decoded[surface_mask > 0] | |
| if values.size == 0: | |
| return None | |
| percentiles = np.percentile(values, [1, 5, 50, 95, 99]) | |
| return { | |
| "minimum": round(float(values.min()), 4), | |
| "p01": round(float(percentiles[0]), 4), | |
| "p05": round(float(percentiles[1]), 4), | |
| "median": round(float(percentiles[2]), 4), | |
| "p95": round(float(percentiles[3]), 4), | |
| "p99": round(float(percentiles[4]), 4), | |
| "maximum": round(float(values.max()), 4), | |
| } | |
| def decode_relative_map(encoded: np.ndarray, minimum: float, maximum: float) -> np.ndarray: | |
| return minimum + (encoded.astype(np.float32) / 255.0) * (maximum - minimum) | |
| def preserve_observed_lighting( | |
| intrinsic_relative: np.ndarray, | |
| observed_relative: np.ndarray, | |
| surface_mask: np.ndarray, | |
| ) -> np.ndarray: | |
| """Keep observed room shadows and highlights from being weakened by intrinsic decomposition.""" | |
| fused = intrinsic_relative.astype(np.float32).copy() | |
| shadow_pixels = (surface_mask > 0) & (observed_relative < 1.0) | |
| highlight_pixels = (surface_mask > 0) & (observed_relative > 1.0) | |
| fused[shadow_pixels] = np.minimum( | |
| fused[shadow_pixels], | |
| observed_relative[shadow_pixels], | |
| ) | |
| fused[highlight_pixels] = np.maximum( | |
| fused[highlight_pixels], | |
| observed_relative[highlight_pixels], | |
| ) | |
| fused = np.clip(fused, SHADE_MAP_MIN, 1.65) | |
| fused[surface_mask == 0] = 1.0 | |
| return fused | |
| def build_intrinsic_lighting_maps( | |
| img_np: np.ndarray, | |
| surface_mask: np.ndarray, | |
| ) -> dict[str, np.ndarray | None] | None: | |
| if not surface_mask.any() or intrinsic_models is None: | |
| return None | |
| try: | |
| img_float = img_np.astype(np.float32) / 255.0 | |
| from intrinsic.pipeline import run_pipeline | |
| results = run_pipeline(intrinsic_models, img_float, device=str(device)) | |
| shading = extract_intrinsic_shading(results) | |
| if shading is None: | |
| return None | |
| relative = normalize_relative_shading(shading, surface_mask) | |
| if relative is None: | |
| return None | |
| relative, alignment = align_intrinsic_shading_to_luminance(relative, img_np, surface_mask) | |
| smoothed = smooth_relative_lighting(relative, surface_mask) | |
| if smoothed is None: | |
| return None | |
| luminance_maps = build_luminance_lighting_maps(img_np, surface_mask) | |
| if luminance_maps and luminance_maps.get("shadeMap") is not None: | |
| observed = decode_relative_map( | |
| luminance_maps["shadeMap"], | |
| SHADE_MAP_MIN, | |
| SHADE_MAP_MAX, | |
| ) | |
| smoothed = preserve_observed_lighting(smoothed, observed, surface_mask) | |
| alignment = f"{alignment}+observed-lighting" | |
| print(f"[LIGHTING] intrinsic shading alignment={alignment}", flush=True) | |
| return { | |
| "shadeMap": encode_relative_map(smoothed, SHADE_MAP_MIN, SHADE_MAP_MAX), | |
| "showroomShadeMap": build_showroom_shade_map_from_relative(smoothed, surface_mask), | |
| "reflectionMap": build_reflection_map(img_np, surface_mask, smoothed), | |
| } | |
| except Exception as exc: | |
| print(f"Intrinsic shading decomposition failed: {exc}. Falling back to default luminance shading.", flush=True) | |
| return None | |
| def build_shade_map(img_np: np.ndarray, surface_mask: np.ndarray) -> np.ndarray | None: | |
| lighting_maps = build_luminance_lighting_maps(img_np, surface_mask) | |
| return lighting_maps.get("shadeMap") if lighting_maps else None | |
| def build_intrinsic_shade_map(img_np: np.ndarray, surface_mask: np.ndarray) -> np.ndarray | None: | |
| lighting_maps = build_intrinsic_lighting_maps(img_np, surface_mask) | |
| return lighting_maps.get("shadeMap") if lighting_maps else None | |
| def clean_floor_mask(mask: np.ndarray) -> np.ndarray: | |
| if not ENABLE_FLOOR_MASK_CLEANUP: | |
| return mask.astype(np.uint8, copy=False) | |
| if mask.dtype != np.uint8: | |
| mask = mask.astype(np.uint8) | |
| h, w = mask.shape[:2] | |
| min_side = max(3, min(h, w)) | |
| close_size = max(5, int(round(min_side * 0.018))) | 1 | |
| open_size = max(3, int(round(min_side * 0.006))) | 1 | |
| closed = cv2.morphologyEx( | |
| mask, | |
| cv2.MORPH_CLOSE, | |
| cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (close_size, close_size)), | |
| ) | |
| cleaned = cv2.morphologyEx( | |
| closed, | |
| cv2.MORPH_OPEN, | |
| cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (open_size, open_size)), | |
| ) | |
| count, labels, stats, _ = cv2.connectedComponentsWithStats(cleaned, connectivity=8) | |
| if count <= 1: | |
| return cleaned | |
| gravity_threshold = int(h * 0.60) | |
| min_area = max(1000, int(h * w * 0.01)) | |
| result = np.zeros_like(cleaned) | |
| for component_id in range(1, count): | |
| area = stats[component_id, cv2.CC_STAT_AREA] | |
| if area < min_area: | |
| continue | |
| comp_bottom = stats[component_id, cv2.CC_STAT_TOP] + stats[component_id, cv2.CC_STAT_HEIGHT] | |
| if comp_bottom <= gravity_threshold: | |
| continue | |
| result[labels == component_id] = 1 | |
| if result.any(): | |
| return result | |
| largest = 1 + int(np.argmax(stats[1:, cv2.CC_STAT_AREA])) | |
| return (labels == largest).astype(np.uint8) | |
| def wall_subtract(mask: np.ndarray, seg_map: np.ndarray, dilation: int = 1) -> np.ndarray: | |
| reject_raw = np.isin(seg_map, class_ids(REJECT_SURFACE_CLASSES)).astype(np.uint8) | |
| if dilation > 0: | |
| kern = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (5, 5)) | |
| reject_raw = cv2.dilate(reject_raw, kern, iterations=dilation) | |
| result = mask.copy() | |
| result[reject_raw > 0] = 0 | |
| return result | |
| def scaled_odd_kernel_size( | |
| height: int, | |
| width: int, | |
| ratio: float, | |
| minimum: int = 3, | |
| maximum: int = 31, | |
| ) -> int: | |
| min_side = max(3, min(height, width)) | |
| size = max(minimum, int(round(min_side * ratio))) | |
| size = min(maximum, size) | |
| return size | 1 | |
| def improve_surface_edge_coverage( | |
| surface: np.ndarray, | |
| protected_mask: np.ndarray, | |
| ) -> tuple[np.ndarray, dict]: | |
| h, w = surface.shape[:2] | |
| before = int(surface.sum()) | |
| repaired = surface.astype(np.uint8).copy() | |
| protected = protected_mask.astype(bool) | |
| close_size = scaled_odd_kernel_size(h, w, SURFACE_EDGE_CLOSE_RATIO, minimum=3) | |
| edge_size = scaled_odd_kernel_size(h, w, SURFACE_EDGE_KERNEL_RATIO, minimum=3) | |
| if close_size > 1: | |
| close_kern = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (close_size, close_size)) | |
| repaired = cv2.morphologyEx(repaired, cv2.MORPH_CLOSE, close_kern) | |
| if SURFACE_EDGE_DILATION_ITERATIONS > 0: | |
| edge_kern = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (edge_size, edge_size)) | |
| repaired = cv2.dilate(repaired, edge_kern, iterations=SURFACE_EDGE_DILATION_ITERATIONS) | |
| repaired[protected] = 0 | |
| repaired = (repaired > 0).astype(np.uint8) | |
| after = int(repaired.sum()) | |
| return repaired, { | |
| "surfacePixelsBeforeEdgeFix": before, | |
| "surfacePixelsAfterEdgeFix": after, | |
| "surfaceEdgeCloseKernel": close_size, | |
| "surfaceEdgeKernel": edge_size, | |
| "surfaceEdgeDilationIterations": SURFACE_EDGE_DILATION_ITERATIONS, | |
| } | |
| def expand_surface_over_soft_floor_coverings( | |
| surface: np.ndarray, | |
| soft_covering_mask: np.ndarray, | |
| protected_mask: np.ndarray, | |
| ) -> tuple[np.ndarray, dict]: | |
| h, w = surface.shape[:2] | |
| before = int(surface.sum()) | |
| soft_pixels = int(soft_covering_mask.sum()) | |
| metadata = { | |
| "softFloorCoveringExpansionEnabled": ENABLE_SOFT_FLOOR_COVERING_EXPANSION, | |
| "softFloorCoveringExpansionApplied": False, | |
| "softFloorCoveringExpansionReason": ( | |
| "disabled" if not ENABLE_SOFT_FLOOR_COVERING_EXPANSION else "unavailable" | |
| ), | |
| "softFloorCoveringPixels": soft_pixels, | |
| "softFloorCoveringPixelsBefore": before, | |
| "softFloorCoveringPixelsAfter": before, | |
| "softFloorCoveringKernel": 0, | |
| "softFloorCoveringDilationIterations": SOFT_FLOOR_COVERING_DILATION_ITERATIONS, | |
| } | |
| if not ENABLE_SOFT_FLOOR_COVERING_EXPANSION: | |
| return surface, metadata | |
| if before == 0: | |
| metadata["softFloorCoveringExpansionReason"] = "empty-surface" | |
| return surface, metadata | |
| if soft_pixels == 0: | |
| metadata["softFloorCoveringExpansionReason"] = "no-soft-floor-covering" | |
| return surface, metadata | |
| if SOFT_FLOOR_COVERING_DILATION_ITERATIONS <= 0: | |
| metadata["softFloorCoveringExpansionReason"] = "zero-dilation" | |
| return surface, metadata | |
| kernel_size = scaled_odd_kernel_size( | |
| h, | |
| w, | |
| SOFT_FLOOR_COVERING_KERNEL_RATIO, | |
| minimum=3, | |
| ) | |
| metadata["softFloorCoveringKernel"] = kernel_size | |
| kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (kernel_size, kernel_size)) | |
| surface_reach = cv2.dilate( | |
| surface.astype(np.uint8), | |
| kernel, | |
| iterations=SOFT_FLOOR_COVERING_DILATION_ITERATIONS, | |
| ) | |
| protected = protected_mask.astype(bool) | |
| candidates = (surface_reach > 0) & (soft_covering_mask > 0) & (~protected) | |
| expanded = surface.astype(np.uint8).copy() | |
| expanded[candidates] = 1 | |
| expanded[protected] = 0 | |
| expanded = (expanded > 0).astype(np.uint8) | |
| after = int(expanded.sum()) | |
| metadata["softFloorCoveringPixelsAfter"] = after | |
| metadata["softFloorCoveringExpansionApplied"] = after > before | |
| metadata["softFloorCoveringExpansionReason"] = ( | |
| "expanded-near-soft-floor-covering" if after > before else "no-adjacent-candidates" | |
| ) | |
| return expanded, metadata | |
| def filter_surface_by_floor_geometry( | |
| surface: np.ndarray, | |
| geometry: dict | None, | |
| plane: dict | None, | |
| ) -> tuple[np.ndarray, dict]: | |
| before = int(surface.sum()) | |
| metadata = { | |
| "geometrySurfaceFilterEnabled": ENABLE_GEOMETRY_SURFACE_FILTER, | |
| "geometrySurfaceFilterApplied": False, | |
| "geometrySurfaceFilterReason": "disabled" if not ENABLE_GEOMETRY_SURFACE_FILTER else "unavailable", | |
| "geometrySurfacePixelsBefore": before, | |
| "geometrySurfacePixelsAfter": before, | |
| "geometrySurfacePlaneDistanceThresholdMeters": SURFACE_PLANE_DISTANCE_METERS, | |
| "geometrySurfacePlaneNormalMinCos": SURFACE_PLANE_NORMAL_MIN_COS, | |
| } | |
| if not ENABLE_GEOMETRY_SURFACE_FILTER or before == 0: | |
| return surface, metadata | |
| if geometry is None or plane is None: | |
| return surface, metadata | |
| if not plane.get("planeNormal") or not plane.get("planeOrigin"): | |
| return surface, metadata | |
| maps = geometry_point_maps(surface, geometry) | |
| if maps is None: | |
| metadata["geometrySurfaceFilterReason"] = "point-map-unavailable" | |
| return surface, metadata | |
| points = maps["points"] | |
| normals = maps["normals"] | |
| valid = maps["validMask"] & (surface > 0) | |
| if int(valid.sum()) < SURFACE_MIN_PLANE_PIXELS: | |
| metadata["geometrySurfaceFilterReason"] = "too-few-valid-pixels" | |
| return surface, metadata | |
| plane_normal = np.asarray(plane["planeNormal"], dtype=np.float32) | |
| plane_origin = np.asarray(plane["planeOrigin"], dtype=np.float32) | |
| normal_len = float(np.linalg.norm(plane_normal)) | |
| if normal_len < 1e-6 or not np.isfinite(plane_normal).all() or not np.isfinite(plane_origin).all(): | |
| metadata["geometrySurfaceFilterReason"] = "invalid-plane" | |
| return surface, metadata | |
| plane_normal /= normal_len | |
| distances = np.full(surface.shape, np.inf, dtype=np.float32) | |
| distances[valid] = np.abs((points[valid] - plane_origin) @ plane_normal) | |
| keep = valid & (distances <= SURFACE_PLANE_DISTANCE_METERS) | |
| if normals is not None and normals.shape[:2] == surface.shape: | |
| normal_lengths = np.linalg.norm(normals, axis=2) | |
| normal_valid = valid & np.isfinite(normals).all(axis=2) & (normal_lengths > 1e-4) | |
| if int(normal_valid.sum()) >= SURFACE_MIN_PLANE_PIXELS: | |
| normalized_normals = np.zeros_like(normals, dtype=np.float32) | |
| normalized_normals[normal_valid] = normals[normal_valid] / normal_lengths[normal_valid, None] | |
| alignment = np.zeros(surface.shape, dtype=np.float32) | |
| alignment[normal_valid] = np.abs(normalized_normals[normal_valid] @ plane_normal) | |
| keep &= (~normal_valid) | (alignment >= SURFACE_PLANE_NORMAL_MIN_COS) | |
| metadata["geometrySurfaceNormalPixels"] = int(normal_valid.sum()) | |
| after = int(keep.sum()) | |
| keep_ratio = after / max(before, 1) | |
| metadata["geometrySurfacePixelsAfter"] = after | |
| metadata["geometrySurfaceFilterKeepRatio"] = float(keep_ratio) | |
| if after < SURFACE_MIN_PLANE_PIXELS or keep_ratio < SURFACE_PLANE_FILTER_MIN_KEEP_RATIO: | |
| metadata["geometrySurfaceFilterReason"] = "filter-too-aggressive" | |
| return surface, metadata | |
| filtered = keep.astype(np.uint8) | |
| close_size = scaled_odd_kernel_size( | |
| surface.shape[0], | |
| surface.shape[1], | |
| SURFACE_EDGE_CLOSE_RATIO, | |
| minimum=3, | |
| ) | |
| if close_size > 1: | |
| close_kern = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (close_size, close_size)) | |
| filtered = cv2.morphologyEx(filtered, cv2.MORPH_CLOSE, close_kern) | |
| filtered = (filtered > 0).astype(np.uint8) | |
| metadata["geometrySurfaceFilterApplied"] = True | |
| metadata["geometrySurfaceFilterReason"] = "floor-plane-consistent" | |
| metadata["geometrySurfacePixelsAfter"] = int(filtered.sum()) | |
| return filtered, metadata | |
| def fit_floor_edges(mask: np.ndarray): | |
| h, w = mask.shape[:2] | |
| row_ys, lefts, rights = [], [], [] | |
| step = max(1, h // 260) | |
| for y in range(0, h, step): | |
| row_xs = np.where(mask[y] > 0)[0] | |
| if len(row_xs) < max(8, w * 0.01): | |
| continue | |
| row_ys.append(float(y)) | |
| lefts.append(float(np.percentile(row_xs, 3))) | |
| rights.append(float(np.percentile(row_xs, 97))) | |
| if len(row_ys) < 8: | |
| return None | |
| row_ys_np = np.asarray(row_ys, dtype=np.float32) | |
| return np.polyfit(row_ys_np, np.asarray(lefts, dtype=np.float32), 1), np.polyfit( | |
| row_ys_np, | |
| np.asarray(rights, dtype=np.float32), | |
| 1, | |
| ) | |
| def detect_vanishing_point(img_np: np.ndarray, floor_mask: np.ndarray): | |
| gray = cv2.cvtColor(img_np, cv2.COLOR_RGB2GRAY) | |
| gray = cv2.GaussianBlur(gray, (5, 5), 0) | |
| edges = cv2.Canny(gray, 60, 160) | |
| edges[floor_mask == 0] = 0 | |
| lines = cv2.HoughLinesP( | |
| edges, | |
| rho=1, | |
| theta=np.pi / 180, | |
| threshold=60, | |
| minLineLength=max(40, min(img_np.shape[:2]) // 16), | |
| maxLineGap=24, | |
| ) | |
| if lines is None: | |
| return None | |
| h, w = img_np.shape[:2] | |
| candidates = [] | |
| for line in lines.reshape(-1, 4): | |
| x1, y1, x2, y2 = [float(v) for v in line] | |
| dx, dy = x2 - x1, y2 - y1 | |
| length = float(np.hypot(dx, dy)) | |
| if length < 40 or abs(dx) < 1: | |
| continue | |
| slope = dy / dx | |
| if abs(slope) >= 0.18: | |
| candidates.append((x1, y1, x2, y2, slope, length)) | |
| intersections = [] | |
| for i, (x1, y1, _, _, s1, l1) in enumerate(candidates): | |
| a1 = y1 - s1 * x1 | |
| for x3, y3, _, _, s2, l2 in candidates[i + 1:]: | |
| if s1 * s2 > 0 or abs(s1 - s2) < 0.12: | |
| continue | |
| a2 = y3 - s2 * x3 | |
| x = (a2 - a1) / (s1 - s2) | |
| y = s1 * x + a1 | |
| if -w * 0.5 <= x <= w * 1.5 and -h <= y <= h * 0.95: | |
| intersections.append((x, y, min(l1, l2))) | |
| if not intersections: | |
| return None | |
| pts = np.asarray([[p[0], p[1]] for p in intersections], dtype=np.float32) | |
| weights = np.asarray([p[2] for p in intersections], dtype=np.float32) | |
| center = np.average(pts, axis=0, weights=weights) | |
| dist = np.linalg.norm(pts - center, axis=1) | |
| keep = dist <= np.percentile(dist, 70) | |
| if keep.sum() >= 3: | |
| center = np.average(pts[keep], axis=0, weights=weights[keep]) | |
| return {"x": float(center[0]), "y": float(center[1])} | |
| def default_camera_intrinsics(width: int, height: int) -> np.ndarray: | |
| focal = float(max(height, width) * 0.95) | |
| cx, cy = (width - 1) * 0.5, (height - 1) * 0.5 | |
| return np.array([[focal, 0, cx], [0, focal, cy], [0, 0, 1]], dtype=np.float32) | |
| def pixel_intrinsics_from_model( | |
| intrinsics: np.ndarray | None, | |
| width: int, | |
| height: int, | |
| ) -> np.ndarray: | |
| if intrinsics is None: | |
| return default_camera_intrinsics(width, height) | |
| pixel_intrinsics = intrinsics.astype(np.float32).copy() | |
| if abs(float(pixel_intrinsics[0, 0])) < 10: | |
| pixel_intrinsics[0, 0] *= width | |
| pixel_intrinsics[0, 2] *= width | |
| pixel_intrinsics[1, 1] *= height | |
| pixel_intrinsics[1, 2] *= height | |
| return pixel_intrinsics | |
| def prepare_depth_for_backprojection( | |
| depth: np.ndarray, | |
| mask: np.ndarray, | |
| is_metric_depth: bool, | |
| ) -> tuple[np.ndarray, str]: | |
| depth_float = depth.astype(np.float32).copy() | |
| if is_metric_depth: | |
| return depth_float, "metric" | |
| valid = (mask > 0) & np.isfinite(depth_float) | |
| if valid.sum() < 1000: | |
| return depth_float, "relative-raw" | |
| ys, xs = np.where(valid) | |
| floor_depth = depth_float[ys, xs] | |
| if floor_depth.size >= 1000: | |
| corr = np.corrcoef(floor_depth, ys.astype(np.float32))[0, 1] | |
| if np.isfinite(corr) and corr > 0: | |
| depth_float = 1.0 - depth_float | |
| floor_depth = depth_float[ys, xs] | |
| lo, hi = np.percentile(floor_depth[np.isfinite(floor_depth)], [2, 98]) | |
| if hi - lo < 1e-6: | |
| return depth_float, "relative-raw" | |
| normalized = np.clip((depth_float - lo) / (hi - lo), 0.0, 1.0) | |
| return (0.8 + normalized * 3.2).astype(np.float32), "relative-normalized" | |
| def backproject_depth( | |
| depth: np.ndarray, | |
| intrinsics: np.ndarray, | |
| ) -> np.ndarray: | |
| h, w = depth.shape[:2] | |
| yy, xx = np.mgrid[0:h, 0:w] | |
| fx = max(float(intrinsics[0, 0]), 1e-6) | |
| fy = max(float(intrinsics[1, 1]), 1e-6) | |
| cx = float(intrinsics[0, 2]) | |
| cy = float(intrinsics[1, 2]) | |
| return np.dstack(( | |
| (xx - cx) * depth / fx, | |
| (yy - cy) * depth / fy, | |
| depth, | |
| )).astype(np.float32) | |
| def ransac_plane_fit(points: np.ndarray): | |
| if len(points) < 1000: | |
| return None | |
| rng = np.random.default_rng(17) | |
| span = np.percentile(points, 95, axis=0) - np.percentile(points, 5, axis=0) | |
| threshold = max(0.025, float(np.linalg.norm(span)) * 0.012) | |
| best_keep = None | |
| best_count = 0 | |
| best_normal = None | |
| best_centroid = None | |
| iterations = min(160, max(48, len(points) // 120)) | |
| for _ in range(iterations): | |
| sample_idx = rng.choice(len(points), size=3, replace=False) | |
| p0, p1, p2 = points[sample_idx] | |
| normal = np.cross(p1 - p0, p2 - p0).astype(np.float32) | |
| length = float(np.linalg.norm(normal)) | |
| if length < 1e-6: | |
| continue | |
| normal /= length | |
| distances = np.abs((points - p0) @ normal) | |
| keep = distances <= threshold | |
| count = int(keep.sum()) | |
| if count > best_count: | |
| best_keep = keep | |
| best_count = count | |
| best_normal = normal | |
| best_centroid = p0.astype(np.float32) | |
| if best_keep is None or best_count < 1000: | |
| return None | |
| return best_centroid, best_normal, best_keep | |
| def project_points_to_plane(points: np.ndarray, origin: np.ndarray, normal: np.ndarray): | |
| normal = normal.astype(np.float32) | |
| normal /= max(float(np.linalg.norm(normal)), 1e-6) | |
| signed_distance = (points - origin) @ normal | |
| projected = points - signed_distance[:, None] * normal | |
| return projected.astype(np.float32), signed_distance.astype(np.float32) | |
| def fit_metric_floor_transform( | |
| mask: np.ndarray, | |
| points_map: np.ndarray, | |
| valid_mask: np.ndarray | None = None, | |
| normals_map: np.ndarray | None = None, | |
| intrinsics: np.ndarray | None = None, | |
| provider: str = "metric-depth", | |
| ): | |
| h, w = mask.shape[:2] | |
| valid = (mask > 0) & np.isfinite(points_map).all(axis=2) & (points_map[:, :, 2] > 0.05) | |
| if valid_mask is not None: | |
| valid &= valid_mask.astype(bool) | |
| ys, xs = np.where(valid) | |
| if len(xs) < 2000: | |
| return None | |
| stride = max(1, len(xs) // 18000) | |
| xs = xs[::stride] | |
| ys = ys[::stride] | |
| points = points_map[ys, xs].astype(np.float32) | |
| normal_consistency = 1.0 | |
| if normals_map is not None and normals_map.shape[:2] == mask.shape: | |
| sample_normals = normals_map[ys, xs].astype(np.float32) | |
| lengths = np.linalg.norm(sample_normals, axis=1) | |
| good_normals = np.isfinite(sample_normals).all(axis=1) & (lengths > 1e-4) | |
| sample_normals[good_normals] /= lengths[good_normals, None] | |
| if good_normals.sum() >= 1000: | |
| reference = np.median(sample_normals[good_normals], axis=0) | |
| reference /= max(float(np.linalg.norm(reference)), 1e-6) | |
| aligned = np.abs(sample_normals @ reference) | |
| normal_keep = good_normals & (aligned >= np.cos(np.deg2rad(30))) | |
| normal_consistency = float(normal_keep.mean()) | |
| if normal_keep.sum() >= 1000: | |
| points = points[normal_keep] | |
| xs = xs[normal_keep] | |
| ys = ys[normal_keep] | |
| ransac_fit = ransac_plane_fit(points) | |
| if ransac_fit is not None: | |
| centroid, normal, keep = ransac_fit | |
| else: | |
| keep = np.ones(len(points), dtype=bool) | |
| normal = np.array([0.0, 1.0, 0.0], dtype=np.float32) | |
| centroid = np.median(points, axis=0) | |
| for _ in range(4): | |
| fit_points = points[keep] | |
| if len(fit_points) < 1000: | |
| return None | |
| centroid = np.median(fit_points, axis=0) | |
| _, _, vh = np.linalg.svd(fit_points - centroid, full_matrices=False) | |
| normal = vh[-1].astype(np.float32) | |
| distances = np.abs((points - centroid) @ normal) | |
| threshold = max(0.025, float(np.percentile(distances[keep], 70)) * 1.8) | |
| keep = distances <= threshold | |
| if normal[1] < 0: | |
| normal = -normal | |
| inlier_ratio = float(keep.mean()) | |
| residual = float(np.median(np.abs((points[keep] - centroid) @ normal))) | |
| if inlier_ratio < 0.45 or residual > 0.18: | |
| return None | |
| # Build stable floor axes: U runs left/right; V runs away from the camera. | |
| u_axis = np.cross(np.array([0.0, 0.0, 1.0], dtype=np.float32), normal) | |
| if np.linalg.norm(u_axis) < 1e-4: | |
| u_axis = np.array([1.0, 0.0, 0.0], dtype=np.float32) | |
| u_axis /= np.linalg.norm(u_axis) | |
| v_axis = np.cross(normal, u_axis) | |
| v_axis /= np.linalg.norm(v_axis) | |
| projected_points, _ = project_points_to_plane(points[keep], centroid, normal) | |
| floor_points = projected_points - centroid | |
| uv = np.column_stack((floor_points @ u_axis, floor_points @ v_axis)).astype(np.float32) | |
| image_points = np.column_stack((xs[keep], ys[keep])).astype(np.float32) | |
| transform, ransac_mask = cv2.findHomography( | |
| image_points, | |
| uv, | |
| method=cv2.RANSAC, | |
| ransacReprojThreshold=0.08, | |
| maxIters=3000, | |
| confidence=0.995, | |
| ) | |
| if transform is None: | |
| return None | |
| homography_inliers = float(ransac_mask.mean()) if ransac_mask is not None else 0.0 | |
| confidence = float(np.clip( | |
| 0.55 * inlier_ratio | |
| + 0.30 * homography_inliers | |
| + 0.15 * normal_consistency | |
| - min(residual / 0.18, 1.0) * 0.25, | |
| 0.0, | |
| 1.0, | |
| )) | |
| pixel_intrinsics = ( | |
| pixel_intrinsics_from_model(intrinsics, w, h) | |
| if intrinsics is not None | |
| else None | |
| ) | |
| return { | |
| "floorTransform": transform.flatten().tolist(), | |
| "metersPerUnit": 1.0, | |
| "geometryConfidence": confidence, | |
| "geometryProvider": provider, | |
| "cameraIntrinsics": pixel_intrinsics.flatten().tolist() if pixel_intrinsics is not None else None, | |
| "camera": { | |
| "focalLength": ( | |
| float(pixel_intrinsics[0, 0]) | |
| if pixel_intrinsics is not None | |
| else float(max(h, w) * 0.95) | |
| ), | |
| "confidence": confidence, | |
| }, | |
| "planeNormal": normal.tolist(), | |
| "planeOrigin": centroid.tolist(), | |
| "fitResidualMeters": residual, | |
| "fitInlierRatio": inlier_ratio, | |
| "normalConsistency": normal_consistency, | |
| "planeFit": "ransac", | |
| } | |
| def estimate_metric_floor_transform( | |
| mask: np.ndarray, | |
| depth: np.ndarray | None, | |
| provider: str = "depth-anything", | |
| intrinsics: np.ndarray | None = None, | |
| ): | |
| if depth is None or not ENABLE_PLANE_FIT: | |
| return None | |
| h, w = mask.shape[:2] | |
| is_metric_depth = "metric" in DEPTH_MODEL_NAME.lower() or "metric" in provider.lower() | |
| depth_for_points, depth_scale = prepare_depth_for_backprojection(depth, mask, is_metric_depth) | |
| pixel_intrinsics = pixel_intrinsics_from_model(intrinsics, w, h) | |
| points = backproject_depth(depth_for_points, pixel_intrinsics) | |
| result = fit_metric_floor_transform(mask, points, intrinsics=pixel_intrinsics, provider=provider) | |
| if result is not None: | |
| result["depthScale"] = depth_scale | |
| return result | |
| def normalize_vectors(vectors: np.ndarray) -> np.ndarray: | |
| out = np.zeros_like(vectors, dtype=np.float32) | |
| lengths = np.linalg.norm(vectors, axis=-1) | |
| good = np.isfinite(vectors).all(axis=-1) & (lengths > 1e-6) | |
| out[good] = vectors[good] / lengths[good, None] | |
| return out | |
| def smooth_normal_map(normals: np.ndarray, kernel_size: int) -> np.ndarray: | |
| kernel_size = max(1, int(kernel_size)) | 1 | |
| smoothed = normals.astype(np.float32) | |
| if kernel_size > 1: | |
| smoothed = cv2.GaussianBlur(smoothed, (kernel_size, kernel_size), 0) | |
| return normalize_vectors(smoothed) | |
| def compute_normals_from_points( | |
| points_map: np.ndarray, | |
| valid_mask: np.ndarray, | |
| smoothing_kernel: int, | |
| ) -> np.ndarray: | |
| points = points_map.astype(np.float32).copy() | |
| valid = valid_mask.astype(bool) & np.isfinite(points).all(axis=2) | |
| if not valid.any(): | |
| return np.zeros_like(points, dtype=np.float32) | |
| filled = points.copy() | |
| for channel_idx in range(3): | |
| channel = filled[:, :, channel_idx] | |
| fill_value = float(np.median(channel[valid])) | |
| channel[~valid] = fill_value | |
| filled[:, :, channel_idx] = channel | |
| kernel_size = max(1, int(smoothing_kernel)) | 1 | |
| if kernel_size > 1: | |
| filled = cv2.GaussianBlur(filled, (kernel_size, kernel_size), 0) | |
| dx = np.zeros_like(filled, dtype=np.float32) | |
| dy = np.zeros_like(filled, dtype=np.float32) | |
| dx[:, 1:-1] = filled[:, 2:] - filled[:, :-2] | |
| dx[:, 0] = filled[:, 1] - filled[:, 0] | |
| dx[:, -1] = filled[:, -1] - filled[:, -2] | |
| dy[1:-1, :] = filled[2:, :] - filled[:-2, :] | |
| dy[0, :] = filled[1, :] - filled[0, :] | |
| dy[-1, :] = filled[-1, :] - filled[-2, :] | |
| normals = normalize_vectors(np.cross(dx, dy)) | |
| normals[~valid] = 0.0 | |
| return smooth_normal_map(normals, smoothing_kernel) | |
| def geometry_point_maps(mask: np.ndarray, geometry: dict | None): | |
| if not ENABLE_MULTI_PLANE_SURFACES or geometry is None: | |
| return None | |
| h, w = mask.shape[:2] | |
| points = geometry.get("points") | |
| point_source = "raw-point-map" | |
| valid_mask = geometry.get("validMask") | |
| intrinsics = geometry.get("intrinsics") | |
| depth_scale = None | |
| if points is None: | |
| depth = geometry.get("depth") | |
| if depth is None: | |
| return None | |
| provider = geometry.get("provider", "") | |
| is_metric_depth = "metric" in DEPTH_MODEL_NAME.lower() or "metric" in provider.lower() | |
| depth_for_points, depth_scale = prepare_depth_for_backprojection(depth, mask, is_metric_depth) | |
| pixel_intrinsics = pixel_intrinsics_from_model(intrinsics, w, h) | |
| points = backproject_depth(depth_for_points, pixel_intrinsics) | |
| valid_mask = np.isfinite(depth_for_points) & (depth_for_points > 0.05) | |
| point_source = f"{provider or 'depth'}-backprojection" | |
| else: | |
| points = points.astype(np.float32) | |
| pixel_intrinsics = pixel_intrinsics_from_model(intrinsics, w, h) | |
| if valid_mask is None: | |
| valid_mask = np.isfinite(points).all(axis=2) & (points[:, :, 2] > 0.05) | |
| valid_mask = valid_mask.astype(bool) & np.isfinite(points).all(axis=2) & (points[:, :, 2] > 0.05) | |
| print( | |
| "[FLOOR_3D_POINTS] point map ready for floor mapping " | |
| f"source={point_source} provider={geometry.get('provider')} shape={points.shape} " | |
| f"validPixels={int(valid_mask.sum())} floorMaskPixels={int(mask.sum())}", | |
| flush=True, | |
| ) | |
| normals = geometry.get("normals") | |
| if normals is None or normals.shape[:2] != mask.shape: | |
| normals = compute_normals_from_points(points, valid_mask, SURFACE_NORMAL_SMOOTHING_KERNEL) | |
| else: | |
| normals = smooth_normal_map(normals.astype(np.float32), SURFACE_NORMAL_SMOOTHING_KERNEL) | |
| return { | |
| "points": points.astype(np.float32), | |
| "normals": normals.astype(np.float32), | |
| "validMask": valid_mask, | |
| "intrinsics": pixel_intrinsics.astype(np.float32), | |
| "depthScale": depth_scale, | |
| } | |
| def refine_plane(points: np.ndarray, normals: np.ndarray | None = None): | |
| centroid = np.median(points, axis=0).astype(np.float32) | |
| _, _, vh = np.linalg.svd(points - centroid, full_matrices=False) | |
| normal = vh[-1].astype(np.float32) | |
| normal /= max(float(np.linalg.norm(normal)), 1e-6) | |
| if normals is not None and len(normals): | |
| reference = np.median(normals, axis=0).astype(np.float32) | |
| if float(np.linalg.norm(reference)) > 1e-6 and float(normal @ reference) < 0: | |
| normal = -normal | |
| elif normal[1] < 0: | |
| normal = -normal | |
| return centroid, normal | |
| def ransac_surface_plane( | |
| points: np.ndarray, | |
| normals: np.ndarray, | |
| candidate_indices: np.ndarray, | |
| remaining: np.ndarray, | |
| distance_threshold: float, | |
| normal_cos: float, | |
| ): | |
| if len(candidate_indices) < 3: | |
| return None | |
| rng = np.random.default_rng(41) | |
| best_normal = None | |
| best_origin = None | |
| best_count = 0 | |
| candidate_points = points[candidate_indices] | |
| candidate_normals = normals[candidate_indices] | |
| relaxed_cos = max(0.0, normal_cos - 0.15) | |
| for _ in range(220): | |
| local = rng.choice(len(candidate_indices), size=3, replace=False) | |
| p0, p1, p2 = candidate_points[local] | |
| normal = np.cross(p1 - p0, p2 - p0).astype(np.float32) | |
| length = float(np.linalg.norm(normal)) | |
| if length < 1e-6: | |
| continue | |
| normal /= length | |
| align = np.abs(candidate_normals @ normal) | |
| distances = np.abs((candidate_points - p0) @ normal) | |
| count = int(((distances <= distance_threshold) & (align >= relaxed_cos)).sum()) | |
| if count > best_count: | |
| best_count = count | |
| best_origin = p0 | |
| best_normal = normal | |
| if best_normal is None: | |
| return None | |
| distances = np.abs((points - best_origin) @ best_normal) | |
| align = np.abs(normals @ best_normal) | |
| inliers = remaining & (distances <= distance_threshold) & (align >= relaxed_cos) | |
| if int(inliers.sum()) < SURFACE_MIN_PLANE_PIXELS: | |
| return None | |
| refine_idx = np.flatnonzero(inliers) | |
| if len(refine_idx) > 50000: | |
| refine_idx = refine_idx[:: max(1, len(refine_idx) // 50000)] | |
| origin, normal = refine_plane(points[refine_idx], normals[refine_idx]) | |
| distances = np.abs((points - origin) @ normal) | |
| align = np.abs(normals @ normal) | |
| inliers = remaining & (distances <= distance_threshold) & (align >= normal_cos) | |
| if int(inliers.sum()) < SURFACE_MIN_PLANE_PIXELS: | |
| return None | |
| final_idx = np.flatnonzero(inliers) | |
| if len(final_idx) > 50000: | |
| final_idx = final_idx[:: max(1, len(final_idx) // 50000)] | |
| origin, normal = refine_plane(points[final_idx], normals[final_idx]) | |
| return origin, normal, inliers | |
| def ransac_surface_plane_by_distance( | |
| points: np.ndarray, | |
| candidate_indices: np.ndarray, | |
| remaining: np.ndarray, | |
| distance_threshold: float, | |
| ): | |
| if len(candidate_indices) < 3: | |
| return None | |
| rng = np.random.default_rng(53) | |
| best_normal = None | |
| best_origin = None | |
| best_count = 0 | |
| candidate_points = points[candidate_indices] | |
| for _ in range(260): | |
| local = rng.choice(len(candidate_indices), size=3, replace=False) | |
| p0, p1, p2 = candidate_points[local] | |
| normal = np.cross(p1 - p0, p2 - p0).astype(np.float32) | |
| length = float(np.linalg.norm(normal)) | |
| if length < 1e-6: | |
| continue | |
| normal /= length | |
| distances = np.abs((candidate_points - p0) @ normal) | |
| count = int((distances <= distance_threshold).sum()) | |
| if count > best_count: | |
| best_count = count | |
| best_origin = p0 | |
| best_normal = normal | |
| if best_normal is None: | |
| return None | |
| distances = np.abs((points - best_origin) @ best_normal) | |
| inliers = remaining & (distances <= distance_threshold) | |
| if int(inliers.sum()) < SURFACE_MIN_PLANE_PIXELS: | |
| return None | |
| refine_idx = np.flatnonzero(inliers) | |
| if len(refine_idx) > 50000: | |
| refine_idx = refine_idx[:: max(1, len(refine_idx) // 50000)] | |
| origin, normal = refine_plane(points[refine_idx], None) | |
| distances = np.abs((points - origin) @ normal) | |
| inliers = remaining & (distances <= distance_threshold * 1.25) | |
| if int(inliers.sum()) < SURFACE_MIN_PLANE_PIXELS: | |
| return None | |
| final_idx = np.flatnonzero(inliers) | |
| if len(final_idx) > 50000: | |
| final_idx = final_idx[:: max(1, len(final_idx) // 50000)] | |
| origin, normal = refine_plane(points[final_idx], None) | |
| return origin, normal, inliers | |
| def plane_basis(normal: np.ndarray, is_floor: bool): | |
| normal = normal.astype(np.float32) | |
| normal /= max(float(np.linalg.norm(normal)), 1e-6) | |
| if is_floor: | |
| ref = np.array([1.0, 0.0, 0.0], dtype=np.float32) | |
| u_axis = ref - normal * float(ref @ normal) | |
| if float(np.linalg.norm(u_axis)) < 1e-5: | |
| ref = np.array([0.0, 0.0, 1.0], dtype=np.float32) | |
| u_axis = ref - normal * float(ref @ normal) | |
| u_axis /= max(float(np.linalg.norm(u_axis)), 1e-6) | |
| v_axis = np.cross(normal, u_axis) | |
| else: | |
| vertical_ref = np.array([0.0, 1.0, 0.0], dtype=np.float32) | |
| v_axis = vertical_ref - normal * float(vertical_ref @ normal) | |
| if float(np.linalg.norm(v_axis)) < 1e-5: | |
| vertical_ref = np.array([0.0, 0.0, 1.0], dtype=np.float32) | |
| v_axis = vertical_ref - normal * float(vertical_ref @ normal) | |
| v_axis /= max(float(np.linalg.norm(v_axis)), 1e-6) | |
| u_axis = np.cross(v_axis, normal) | |
| u_axis /= max(float(np.linalg.norm(u_axis)), 1e-6) | |
| v_axis /= max(float(np.linalg.norm(v_axis)), 1e-6) | |
| return u_axis.astype(np.float32), v_axis.astype(np.float32) | |
| def surface_distance_threshold(points: np.ndarray) -> float: | |
| span = np.percentile(points, 95, axis=0) - np.percentile(points, 5, axis=0) | |
| adaptive = max(0.025, float(np.linalg.norm(span)) * 0.012) | |
| return float(np.clip(max(float(SURFACE_RANSAC_DISTANCE), adaptive), 0.025, 0.20)) | |
| def fit_multi_surface_planes(points: np.ndarray, normals: np.ndarray, normals_reliable: bool = True): | |
| if len(points) < SURFACE_MIN_PLANE_PIXELS: | |
| return [] | |
| distance_threshold = surface_distance_threshold(points) | |
| normal_cos = float(np.cos(np.deg2rad(SURFACE_NORMAL_ANGLE_DEGREES))) | |
| rng = np.random.default_rng(29) | |
| sample_count = min(len(points), 26000) | |
| sample_indices = rng.choice(len(points), size=sample_count, replace=False) | |
| remaining = np.ones(len(points), dtype=bool) | |
| planes = [] | |
| for plane_id in range(SURFACE_MAX_PLANES): | |
| candidates = sample_indices[remaining[sample_indices]] | |
| if len(candidates) < max(300, SURFACE_MIN_PLANE_PIXELS // 6): | |
| break | |
| fit_mode = "normal-aware" | |
| result = None | |
| if normals_reliable: | |
| result = ransac_surface_plane( | |
| points, | |
| normals, | |
| candidates, | |
| remaining, | |
| distance_threshold, | |
| normal_cos, | |
| ) | |
| if result is None: | |
| fit_mode = "distance-only" | |
| result = ransac_surface_plane_by_distance( | |
| points, | |
| candidates, | |
| remaining, | |
| distance_threshold, | |
| ) | |
| if result is None: | |
| break | |
| origin, normal, inliers = result | |
| inlier_count = int(inliers.sum()) | |
| if inlier_count < SURFACE_MIN_PLANE_PIXELS: | |
| break | |
| planes.append({ | |
| "id": plane_id, | |
| "origin": origin.astype(np.float32), | |
| "normal": normal.astype(np.float32), | |
| "offset": float(-origin @ normal), | |
| "inlierCount": inlier_count, | |
| "fitMode": fit_mode, | |
| }) | |
| remaining[inliers] = False | |
| if int(remaining.sum()) < SURFACE_MIN_PLANE_PIXELS: | |
| break | |
| if not planes: | |
| return [] | |
| floor_id = int(np.argmax([ | |
| plane["inlierCount"] * max(0.1, abs(float(plane["normal"][1]))) | |
| for plane in planes | |
| ])) | |
| for idx, plane in enumerate(planes): | |
| plane["isFloor"] = idx == floor_id | |
| u_axis, v_axis = plane_basis(plane["normal"], plane["isFloor"]) | |
| plane["uAxis"] = u_axis | |
| plane["vAxis"] = v_axis | |
| return planes | |
| def assign_points_to_surface_planes( | |
| points: np.ndarray, | |
| normals: np.ndarray, | |
| planes: list[dict], | |
| normals_reliable: bool = True, | |
| ): | |
| plane_normals = np.asarray([plane["normal"] for plane in planes], dtype=np.float32) | |
| plane_origins = np.asarray([plane["origin"] for plane in planes], dtype=np.float32) | |
| offsets = -np.sum(plane_normals * plane_origins, axis=1) | |
| distances = np.abs(points @ plane_normals.T + offsets[None, :]) | |
| if not normals_reliable: | |
| return np.argmin(distances, axis=1).astype(np.int32) | |
| alignments = np.abs(normals @ plane_normals.T) | |
| normal_cos = float(np.cos(np.deg2rad(SURFACE_NORMAL_ANGLE_DEGREES))) | |
| distance_threshold = float(SURFACE_RANSAC_DISTANCE) * 1.5 | |
| denom = max(1.0 - normal_cos, 1e-3) | |
| score = distances / max(distance_threshold, 1e-6) + (1.0 - alignments) / denom | |
| ok = (distances <= distance_threshold) & (alignments >= normal_cos) | |
| score[~ok] = np.inf | |
| assignment = np.argmin(score, axis=1).astype(np.int32) | |
| has_match = np.isfinite(score[np.arange(len(points)), assignment]) | |
| if not has_match.all(): | |
| assignment[~has_match] = np.argmin(distances[~has_match], axis=1).astype(np.int32) | |
| return assignment | |
| def analyze_surface_uv_quality( | |
| surface_mask: np.ndarray, | |
| surface_indices: np.ndarray, | |
| uv_by_mask: np.ndarray, | |
| plane_ids: np.ndarray, | |
| ) -> dict: | |
| total_pixels = int(len(surface_indices)) | |
| finite_uv = np.isfinite(uv_by_mask).all(axis=1) if total_pixels else np.zeros(0, dtype=bool) | |
| finite_pixels = int(finite_uv.sum()) | |
| coverage = float(finite_pixels / max(total_pixels, 1)) | |
| active_plane_ids = sorted({ | |
| int(plane_id) | |
| for plane_id in np.unique(plane_ids[finite_uv]) | |
| if int(plane_id) != 255 | |
| }) | |
| quality = { | |
| "surfaceUvEnabled": False, | |
| "textureMappingMode": "regularized-floor-plane", | |
| "textureMappingReason": "surface-uv-unavailable", | |
| "surfaceUvCoverage": coverage, | |
| "surfaceUvFinitePixels": finite_pixels, | |
| "surfaceUvTotalPixels": total_pixels, | |
| "surfaceUvNeighborCount": 0, | |
| "surfaceUvMedianNeighborDelta": None, | |
| "surfaceUvP95NeighborDelta": None, | |
| "surfaceUvP99NeighborDelta": None, | |
| "surfaceUvJumpLimit": None, | |
| "surfaceUvBadNeighborRatio": None, | |
| "surfaceUvPlaneBoundaryRatio": None, | |
| "surfaceUvAssignedPlaneCount": len(active_plane_ids), | |
| } | |
| if total_pixels == 0: | |
| quality["textureMappingReason"] = "empty-surface" | |
| return quality | |
| if coverage < SURFACE_UV_MIN_COVERAGE: | |
| quality["textureMappingReason"] = "incomplete-surface-uv" | |
| return quality | |
| if len(active_plane_ids) != 1: | |
| quality["textureMappingReason"] = "multiple-surface-planes" | |
| return quality | |
| h, w = surface_mask.shape[:2] | |
| uv_grid = np.full((h * w, 2), np.nan, dtype=np.float32) | |
| plane_grid = np.full(h * w, 255, dtype=np.uint8) | |
| uv_grid[surface_indices[finite_uv]] = uv_by_mask[finite_uv] | |
| plane_grid[surface_indices[finite_uv]] = plane_ids[finite_uv] | |
| uv_grid = uv_grid.reshape(h, w, 2) | |
| plane_grid = plane_grid.reshape(h, w) | |
| finite_grid = np.isfinite(uv_grid).all(axis=2) | |
| surface_bool = surface_mask.astype(bool) | |
| right = surface_bool[:, :-1] & surface_bool[:, 1:] & finite_grid[:, :-1] & finite_grid[:, 1:] | |
| down = surface_bool[:-1, :] & surface_bool[1:, :] & finite_grid[:-1, :] & finite_grid[1:, :] | |
| right_plane_boundary = right & (plane_grid[:, :-1] != plane_grid[:, 1:]) | |
| down_plane_boundary = down & (plane_grid[:-1, :] != plane_grid[1:, :]) | |
| same_right = right & ~right_plane_boundary | |
| same_down = down & ~down_plane_boundary | |
| deltas = [] | |
| if same_right.any(): | |
| deltas.append(np.linalg.norm(uv_grid[:, 1:][same_right] - uv_grid[:, :-1][same_right], axis=1)) | |
| if same_down.any(): | |
| deltas.append(np.linalg.norm(uv_grid[1:, :][same_down] - uv_grid[:-1, :][same_down], axis=1)) | |
| neighbor_count = int(right.sum() + down.sum()) | |
| plane_boundary_count = int(right_plane_boundary.sum() + down_plane_boundary.sum()) | |
| quality["surfaceUvNeighborCount"] = neighbor_count | |
| quality["surfaceUvPlaneBoundaryRatio"] = float(plane_boundary_count / max(neighbor_count, 1)) | |
| if not deltas: | |
| quality["textureMappingReason"] = "insufficient-uv-neighbors" | |
| return quality | |
| all_deltas = np.concatenate(deltas).astype(np.float32) | |
| median_delta = float(np.median(all_deltas)) | |
| p95_delta = float(np.percentile(all_deltas, 95)) | |
| p99_delta = float(np.percentile(all_deltas, 99)) | |
| jump_limit = max(SURFACE_UV_ABSOLUTE_JUMP_METERS, median_delta * SURFACE_UV_MAX_JUMP_RATIO) | |
| bad_ratio = float((all_deltas > jump_limit).mean()) | |
| quality.update({ | |
| "surfaceUvMedianNeighborDelta": median_delta, | |
| "surfaceUvP95NeighborDelta": p95_delta, | |
| "surfaceUvP99NeighborDelta": p99_delta, | |
| "surfaceUvJumpLimit": float(jump_limit), | |
| "surfaceUvBadNeighborRatio": bad_ratio, | |
| }) | |
| if bad_ratio > SURFACE_UV_MAX_BAD_NEIGHBOR_RATIO: | |
| quality["textureMappingReason"] = "noisy-surface-uv" | |
| return quality | |
| if quality["surfaceUvPlaneBoundaryRatio"] > 0: | |
| quality["textureMappingReason"] = "surface-plane-boundaries" | |
| return quality | |
| quality["surfaceUvEnabled"] = True | |
| quality["textureMappingMode"] = "surface-uv" | |
| quality["textureMappingReason"] = "surface-uv-quality-ok" | |
| return quality | |
| def build_surface_uv_mapping( | |
| surface_mask: np.ndarray, | |
| surface_indices: np.ndarray, | |
| geometry: dict | None, | |
| ): | |
| maps = geometry_point_maps(surface_mask, geometry) | |
| if maps is None: | |
| print( | |
| "[SURFACE_MAPPING] skipped: no point maps " | |
| f"enabled={ENABLE_MULTI_PLANE_SURFACES} provider={geometry.get('provider') if geometry else None}", | |
| flush=True, | |
| ) | |
| return None | |
| points_map = maps["points"] | |
| normals_map = maps["normals"] | |
| valid_mask = maps["validMask"] & (surface_mask > 0) | |
| ys, xs = np.where(valid_mask) | |
| if len(xs) < SURFACE_MIN_PLANE_PIXELS: | |
| print( | |
| "[SURFACE_MAPPING] skipped: too few valid surface pixels " | |
| f"valid={len(xs)} min={SURFACE_MIN_PLANE_PIXELS}", | |
| flush=True, | |
| ) | |
| return None | |
| valid_flat = (ys * surface_mask.shape[1] + xs).astype(np.uint32) | |
| points = points_map[ys, xs].astype(np.float32) | |
| normals_raw = normals_map[ys, xs].astype(np.float32) | |
| normal_lengths = np.linalg.norm(normals_raw, axis=1) | |
| good_points = np.isfinite(points).all(axis=1) | |
| if int(good_points.sum()) < SURFACE_MIN_PLANE_PIXELS: | |
| print( | |
| "[SURFACE_MAPPING] skipped: too few finite 3D points " | |
| f"finitePoints={int(good_points.sum())} min={SURFACE_MIN_PLANE_PIXELS}", | |
| flush=True, | |
| ) | |
| return None | |
| valid_flat = valid_flat[good_points] | |
| points = points[good_points] | |
| normals_raw = normals_raw[good_points] | |
| normal_lengths = normal_lengths[good_points] | |
| normal_good = np.isfinite(normals_raw).all(axis=1) & (normal_lengths > 1e-4) | |
| normals = np.zeros_like(points, dtype=np.float32) | |
| if normal_good.any(): | |
| normals[normal_good] = normals_raw[normal_good] / normal_lengths[normal_good, None] | |
| normals_reliable = int(normal_good.sum()) >= SURFACE_MIN_PLANE_PIXELS | |
| print( | |
| "[SURFACE_MAPPING] candidates " | |
| f"surfacePixels={int(surface_mask.sum())} validPixels={len(points)} " | |
| f"normalPixels={int(normal_good.sum())} normalsReliable={normals_reliable} " | |
| f"provider={geometry.get('provider') if geometry else None}", | |
| flush=True, | |
| ) | |
| planes = fit_multi_surface_planes(points, normals, normals_reliable) | |
| if not planes: | |
| print( | |
| "[SURFACE_MAPPING] skipped: no dominant planes fitted " | |
| f"validPixels={len(points)} threshold={surface_distance_threshold(points):.4f} " | |
| f"normalsReliable={normals_reliable}", | |
| flush=True, | |
| ) | |
| return None | |
| assignment = assign_points_to_surface_planes(points, normals, planes, normals_reliable) | |
| uv_by_mask = np.full((len(surface_indices), 2), np.nan, dtype=np.float32) | |
| plane_ids = np.full(len(surface_indices), 255, dtype=np.uint8) | |
| ordinal = np.full(surface_mask.size, -1, dtype=np.int32) | |
| ordinal[surface_indices] = np.arange(len(surface_indices), dtype=np.int32) | |
| positions = ordinal[valid_flat] | |
| assignment_distance_limit = max( | |
| float(SURFACE_RANSAC_DISTANCE) * 1.5, | |
| min(float(SURFACE_PLANE_DISTANCE_METERS), surface_distance_threshold(points) * 1.35), | |
| ) | |
| rejected_by_distance = 0 | |
| for plane_idx, plane in enumerate(planes): | |
| assigned = assignment == plane_idx | |
| if not assigned.any(): | |
| plane["assignedCount"] = 0 | |
| continue | |
| projected_points, signed_distances = project_points_to_plane( | |
| points[assigned], | |
| plane["origin"], | |
| plane["normal"], | |
| ) | |
| near_plane = np.abs(signed_distances) <= assignment_distance_limit | |
| rejected_by_distance += int((~near_plane).sum()) | |
| if not near_plane.any(): | |
| plane["assignedCount"] = 0 | |
| continue | |
| delta = projected_points[near_plane] - plane["origin"] | |
| uv = np.column_stack((delta @ plane["uAxis"], delta @ plane["vAxis"])).astype(np.float32) | |
| assigned_positions = positions[assigned][near_plane] | |
| assigned_keep = assigned_positions >= 0 | |
| assigned_positions = assigned_positions[assigned_keep] | |
| uv_by_mask[assigned_positions] = uv[assigned_keep] | |
| plane_ids[assigned_positions] = plane_idx | |
| plane["assignedCount"] = int(assigned_keep.sum()) | |
| serializable_planes = [] | |
| for plane in planes: | |
| serializable_planes.append({ | |
| "id": int(plane["id"]), | |
| "normal": plane["normal"].astype(float).tolist(), | |
| "origin": plane["origin"].astype(float).tolist(), | |
| "offset": float(plane["offset"]), | |
| "uAxis": plane["uAxis"].astype(float).tolist(), | |
| "vAxis": plane["vAxis"].astype(float).tolist(), | |
| "inlierCount": int(plane["inlierCount"]), | |
| "assignedCount": int(plane.get("assignedCount", 0)), | |
| "isFloor": bool(plane["isFloor"]), | |
| "fitMode": str(plane.get("fitMode", "normal-aware")), | |
| }) | |
| finite_uv = np.isfinite(uv_by_mask).all(axis=1) | |
| uv_quality = analyze_surface_uv_quality(surface_mask, surface_indices, uv_by_mask, plane_ids) | |
| uv_quality["surfaceUvPlaneDistanceLimitMeters"] = float(assignment_distance_limit) | |
| uv_quality["surfaceUvRejectedByPlaneDistance"] = int(rejected_by_distance) | |
| print( | |
| "[SURFACE_MAPPING] success " | |
| f"planes={len(serializable_planes)} assignedPixels={int(finite_uv.sum())}/{len(surface_indices)} " | |
| f"fitModes={[plane['fitMode'] for plane in serializable_planes]} " | |
| f"mode={uv_quality['textureMappingMode']} reason={uv_quality['textureMappingReason']}", | |
| flush=True, | |
| ) | |
| return { | |
| "uv": uv_by_mask, | |
| "planeIds": plane_ids, | |
| "planes": serializable_planes, | |
| "normalMap": normals_map, | |
| "validMask": valid_mask, | |
| "intrinsics": maps["intrinsics"], | |
| "depthScale": maps["depthScale"], | |
| "normalsReliable": normals_reliable, | |
| "useSurfaceUv": bool(uv_quality["surfaceUvEnabled"]), | |
| "metadata": uv_quality, | |
| } | |
| def estimate_floor_plane(mask: np.ndarray, img_np: np.ndarray, geometry: dict | None = None): | |
| ys, xs = np.where(mask > 0) | |
| if len(xs) < 1000: | |
| return None, None | |
| xs_f, ys_f = xs.astype(np.float32), ys.astype(np.float32) | |
| x1, x2 = float(np.percentile(xs_f, 1)), float(np.percentile(xs_f, 99)) | |
| y1, y2 = float(np.percentile(ys_f, 1)), float(np.percentile(ys_f, 99)) | |
| width, height = x2 - x1, y2 - y1 | |
| if width < 20 or height < 20: | |
| return None, None | |
| top_y = float(np.percentile(ys_f, 8)) | |
| bottom_y = float(np.percentile(ys_f, 97)) | |
| edge_fits = fit_floor_edges(mask) | |
| if edge_fits is None: | |
| return None, None | |
| left_fit, right_fit = edge_fits | |
| top_left = float(np.polyval(left_fit, top_y)) | |
| top_right = float(np.polyval(right_fit, top_y)) | |
| bottom_left = float(np.polyval(left_fit, bottom_y)) | |
| bottom_right = float(np.polyval(right_fit, bottom_y)) | |
| lower_xs = xs_f[ys_f >= np.percentile(ys_f, 80)] | |
| bottom_left = min(bottom_left, float(np.percentile(lower_xs, 4))) | |
| bottom_right = max(bottom_right, float(np.percentile(lower_xs, 96))) | |
| min_top_width = max(24.0, width * 0.18) | |
| top_center = (top_left + top_right) * 0.5 | |
| if top_right - top_left < min_top_width: | |
| top_left = top_center - min_top_width * 0.5 | |
| top_right = top_center + min_top_width * 0.5 | |
| min_bottom_width = max(min_top_width * 1.25, width * 0.45) | |
| bottom_center = (bottom_left + bottom_right) * 0.5 | |
| if bottom_right - bottom_left < min_bottom_width: | |
| bottom_left = bottom_center - min_bottom_width * 0.5 | |
| bottom_right = bottom_center + min_bottom_width * 0.5 | |
| h, w = mask.shape[:2] | |
| src = np.float32([ | |
| [np.clip(bottom_left, 0, w - 1), np.clip(bottom_y, 0, h - 1)], | |
| [np.clip(bottom_right, 0, w - 1), np.clip(bottom_y, 0, h - 1)], | |
| [np.clip(top_right, 0, w - 1), np.clip(top_y, 0, h - 1)], | |
| [np.clip(top_left, 0, w - 1), np.clip(top_y, 0, h - 1)], | |
| ]) | |
| vanishing_point = detect_vanishing_point(img_np, mask) | |
| if vanishing_point is not None and vanishing_point["y"] < bottom_y: | |
| vp_x = float(np.clip(vanishing_point["x"], -w * 0.25, w * 1.25)) | |
| top_width = max(src[2][0] - src[3][0], width * 0.16) | |
| horizon_gap = max(bottom_y - top_y, 1.0) | |
| convergence = np.clip((top_y - vanishing_point["y"]) / horizon_gap, 0.12, 0.75) | |
| top_center = top_center * (1 - convergence * 0.35) + vp_x * (convergence * 0.35) | |
| src[3][0] = np.clip(top_center - top_width * 0.5, 0, w - 1) | |
| src[2][0] = np.clip(top_center + top_width * 0.5, 0, w - 1) | |
| if cv2.contourArea(src) < 100: | |
| return None, None | |
| dst = np.float32([[x1, y2], [x2, y2], [x2, y1], [x1, y1]]) | |
| homography = cv2.getPerspectiveTransform(src, dst).flatten().tolist() | |
| metric_geometry = None | |
| if ENABLE_PLANE_FIT and geometry and geometry.get("points") is not None: | |
| metric_geometry = fit_metric_floor_transform( | |
| mask, | |
| geometry["points"], | |
| geometry.get("validMask"), | |
| geometry.get("normals"), | |
| geometry.get("intrinsics"), | |
| geometry.get("provider", "moge-2"), | |
| ) | |
| if ENABLE_PLANE_FIT and metric_geometry is None and geometry: | |
| depth_provider = geometry.get("provider", "depth-anything") | |
| if depth_provider == "moge-2": | |
| depth_provider = "moge-2-depth" | |
| metric_geometry = estimate_metric_floor_transform( | |
| mask, | |
| geometry.get("depth"), | |
| depth_provider, | |
| geometry.get("intrinsics"), | |
| ) | |
| plane = { | |
| "x": x1, | |
| "y": y1, | |
| "width": width, | |
| "height": height, | |
| "quad": src.flatten().tolist(), | |
| "vanishingPoint": vanishing_point, | |
| } | |
| if metric_geometry: | |
| plane.update(metric_geometry) | |
| plane["camera"]["horizonY"] = ( | |
| float(vanishing_point["y"]) if vanishing_point is not None else top_y | |
| ) | |
| else: | |
| plane.update({ | |
| "floorTransform": None, | |
| "metersPerUnit": None, | |
| "geometryConfidence": 0.25 if vanishing_point is not None else 0.1, | |
| "geometryProvider": "homography", | |
| "cameraIntrinsics": None, | |
| "camera": { | |
| "focalLength": float(max(h, w) * 0.95), | |
| "horizonY": float(vanishing_point["y"]) if vanishing_point is not None else top_y, | |
| "confidence": 0.25 if vanishing_point is not None else 0.1, | |
| }, | |
| }) | |
| return homography, plane | |
| def build_floor_surface_mask( | |
| floor_mask: np.ndarray, | |
| seg_map: np.ndarray, | |
| quad: np.ndarray | None, | |
| depth: np.ndarray | None, | |
| geometry: dict | None, | |
| plane: dict | None, | |
| ): | |
| h, w = floor_mask.shape[:2] | |
| kern_size = max(5, min(h, w) // 160) | 1 | |
| kern = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (kern_size, kern_size)) | |
| occluder_mask = np.isin(seg_map, class_ids(OCCLUDER_CLASSES)).astype(np.uint8) | |
| soft_covering_mask = np.isin(seg_map, class_ids(SOFT_FLOOR_COVERING_CLASSES)).astype(np.uint8) | |
| reject_mask = np.isin(seg_map, class_ids(REJECT_SURFACE_CLASSES)).astype(np.uint8) | |
| hard_protected_mask = (occluder_mask > 0) | (reject_mask > 0) | |
| normal_edge_protected_mask = hard_protected_mask | (soft_covering_mask > 0) | |
| surface = floor_mask.copy() | |
| surface[reject_mask > 0] = 0 | |
| surface[soft_covering_mask > 0] = 0 | |
| if not surface.any(): | |
| surface = floor_mask.copy() | |
| surface[soft_covering_mask > 0] = 0 | |
| contours, _ = cv2.findContours(surface, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) | |
| if contours: | |
| filled = np.zeros((h, w), dtype=np.uint8) | |
| cv2.drawContours(filled, contours, -1, 1, cv2.FILLED) | |
| filled[reject_mask > 0] = 0 | |
| filled[soft_covering_mask > 0] = 0 | |
| surface = filled | |
| if quad is not None and surface.any(): | |
| plane_mask = np.zeros((h, w), dtype=np.uint8) | |
| cv2.fillConvexPoly(plane_mask, np.round(quad).astype(np.int32), 1) | |
| plane_mask[reject_mask > 0] = 0 | |
| plane_mask[soft_covering_mask > 0] = 0 | |
| near_floor = cv2.dilate(surface, kern, iterations=6) | |
| surface = cv2.bitwise_or(surface, cv2.bitwise_and(plane_mask, near_floor)) | |
| surface[soft_covering_mask > 0] = 0 | |
| surface[occluder_mask > 0] = 0 | |
| surface[soft_covering_mask > 0] = 0 | |
| if depth is not None and floor_mask.any(): | |
| floor_depth = depth[(floor_mask > 0) & np.isfinite(depth)] | |
| if floor_depth.size == 0: | |
| floor_depth = None | |
| else: | |
| floor_depth = None | |
| if floor_depth is not None: | |
| lo, hi = float(np.percentile(floor_depth, 2)), float(np.percentile(floor_depth, 98)) | |
| margin = max(0.08, (hi - lo) * 0.35) | |
| depth_keep = (depth >= lo - margin) & (depth <= hi + margin) | |
| surface = (surface & depth_keep.astype(np.uint8)).astype(np.uint8) | |
| surface[floor_mask > 0] = np.maximum(surface[floor_mask > 0], 1) | |
| surface[occluder_mask > 0] = 0 | |
| surface[reject_mask > 0] = 0 | |
| surface[soft_covering_mask > 0] = 0 | |
| surface, geometry_filter_metadata = filter_surface_by_floor_geometry(surface, geometry, plane) | |
| surface[occluder_mask > 0] = 0 | |
| surface[reject_mask > 0] = 0 | |
| surface[soft_covering_mask > 0] = 0 | |
| surface = clean_floor_mask(surface) | |
| surface[occluder_mask > 0] = 0 | |
| surface[reject_mask > 0] = 0 | |
| surface[soft_covering_mask > 0] = 0 | |
| surface, edge_metadata = improve_surface_edge_coverage(surface, normal_edge_protected_mask) | |
| surface, post_edge_filter_metadata = filter_surface_by_floor_geometry(surface, geometry, plane) | |
| surface[occluder_mask > 0] = 0 | |
| surface[reject_mask > 0] = 0 | |
| surface[soft_covering_mask > 0] = 0 | |
| post_edge_filter_metadata = { | |
| f"postEdge{k[0].upper()}{k[1:]}": value | |
| for k, value in post_edge_filter_metadata.items() | |
| } | |
| surface, soft_covering_metadata = expand_surface_over_soft_floor_coverings( | |
| surface, | |
| soft_covering_mask, | |
| hard_protected_mask, | |
| ) | |
| surface, post_soft_covering_filter_metadata = filter_surface_by_floor_geometry(surface, geometry, plane) | |
| surface[occluder_mask > 0] = 0 | |
| surface[reject_mask > 0] = 0 | |
| post_soft_covering_filter_metadata = { | |
| f"postSoftFloorCovering{k[0].upper()}{k[1:]}": value | |
| for k, value in post_soft_covering_filter_metadata.items() | |
| } | |
| return surface, { | |
| **geometry_filter_metadata, | |
| **edge_metadata, | |
| **post_edge_filter_metadata, | |
| **soft_covering_metadata, | |
| **post_soft_covering_filter_metadata, | |
| } | |
| def run_segmentation(img: Image.Image, img_np: np.ndarray): | |
| global seg_processor, seg_model | |
| if seg_model is None: | |
| _load_segmentation_model() | |
| h, w = img_np.shape[:2] | |
| if segmentation_backend == "oneformer": | |
| inputs = seg_processor( | |
| images=img, | |
| task_inputs=["semantic"], | |
| return_tensors="pt", | |
| ).to(device) | |
| with torch.no_grad(): | |
| outputs = seg_model(**inputs) | |
| result = seg_processor.post_process_semantic_segmentation( | |
| outputs, | |
| target_sizes=[(h, w)], | |
| )[0] | |
| return result.cpu().numpy().astype(np.uint8) | |
| if segmentation_backend == "mask2former": | |
| inputs = seg_processor(images=img, return_tensors="pt").to(device) | |
| with torch.no_grad(): | |
| outputs = seg_model(**inputs) | |
| is_panoptic = "panoptic" in MASK2FORMER_MODEL_NAME | |
| if is_panoptic: | |
| pan_result = seg_processor.post_process_panoptic_segmentation( | |
| outputs, | |
| target_sizes=[(h, w)], | |
| )[0] | |
| seg_map = np.zeros((h, w), dtype=np.uint8) | |
| pan_map = pan_result["segmentation"].cpu().numpy() | |
| for seg_info in pan_result["segments_info"]: | |
| seg_map[pan_map == seg_info["id"]] = min(seg_info["label_id"], 255) | |
| return seg_map | |
| result = seg_processor.post_process_semantic_segmentation( | |
| outputs, | |
| target_sizes=[(h, w)], | |
| )[0] | |
| return result.cpu().numpy().astype(np.uint8) | |
| inputs = seg_processor(images=img, return_tensors="pt").to(device) | |
| with torch.no_grad(): | |
| outputs = seg_model(**inputs) | |
| logits = torch.nn.functional.interpolate( | |
| outputs.logits, | |
| size=(h, w), | |
| mode="bilinear", | |
| align_corners=False, | |
| ) | |
| return logits.argmax(dim=1).squeeze().cpu().numpy().astype(np.uint8) | |
| def segmenter_metadata_name() -> str: | |
| if segmentation_backend == "oneformer": | |
| return "oneformer-ade20k-swin-large" | |
| return segmentation_backend | |
| def active_segmentation_model_name() -> str: | |
| if segmentation_backend == "oneformer": | |
| return ONEFORMER_MODEL_NAME | |
| if segmentation_backend == "mask2former": | |
| return MASK2FORMER_MODEL_NAME | |
| if segmentation_backend == "segformer": | |
| return SEGFORMER_MODEL_NAME | |
| return segmentation_backend | |
| def active_geometry_model_name(provider: str | None) -> str | None: | |
| if provider == "moge-2": | |
| return MOGE_MODEL_NAME | |
| if provider and provider != "homography": | |
| return DEPTH_MODEL_NAME | |
| return None | |
| def active_model_metadata( | |
| geometry: dict | None, | |
| depth: np.ndarray | None, | |
| shade_map: np.ndarray | None, | |
| intrinsic_lighting_used: bool = False, | |
| ) -> dict: | |
| geometry_provider = geometry.get("provider") if geometry else None | |
| intrinsic_enabled = ENABLE_INTRINSIC_SHADING and intrinsic_lighting_used and shade_map is not None | |
| return { | |
| "segmentation": { | |
| "backend": segmentation_backend, | |
| "name": active_segmentation_model_name(), | |
| }, | |
| "geometry": { | |
| "provider": geometry_provider, | |
| "name": active_geometry_model_name(geometry_provider), | |
| }, | |
| "depth": { | |
| "enabled": depth is not None, | |
| "name": DEPTH_MODEL_NAME if depth is not None else None, | |
| }, | |
| "intrinsic": { | |
| "enabled": intrinsic_enabled, | |
| "name": INTRINSIC_MODEL_VERSION if intrinsic_enabled else None, | |
| }, | |
| } | |
| MATERIAL_GENERATOR_VERSION = "opencv-procedural-pbr-v1" | |
| MATERIAL_MAX_DIMENSION = 1024 | |
| def load_tile_image(contents: bytes) -> np.ndarray: | |
| try: | |
| return np.array(Image.open(io.BytesIO(contents)).convert("RGB")) | |
| except Exception as exc: | |
| raise HTTPException(status_code=400, detail="Upload must be a readable tile image.") from exc | |
| def load_uploaded_rgb_image(contents: bytes) -> tuple[Image.Image, np.ndarray]: | |
| try: | |
| image = Image.open(io.BytesIO(contents)).convert("RGB") | |
| except Exception as exc: | |
| raise HTTPException(status_code=400, detail="Upload must be a readable image.") from exc | |
| return image, np.array(image) | |
| def normalize_room_upload(contents: bytes) -> bytes: | |
| try: | |
| image = Image.open(io.BytesIO(contents)) | |
| image.load() | |
| except Exception as exc: | |
| raise HTTPException(status_code=400, detail="Upload must be a readable image.") from exc | |
| source_format = (image.format or "").upper() | |
| has_alpha = image.mode in {"RGBA", "LA"} or (image.mode == "P" and "transparency" in image.info) | |
| width, height = image.size | |
| largest = max(width, height) | |
| needs_resize = largest > ROOM_UPLOAD_MAX_DIMENSION | |
| try: | |
| orientation = image.getexif().get(274) | |
| except Exception: | |
| orientation = None | |
| needs_orientation = orientation not in {None, 1} | |
| if source_format == "JPEG" and not has_alpha and not needs_resize and not needs_orientation: | |
| return contents | |
| image = ImageOps.exif_transpose(image) | |
| if has_alpha: | |
| alpha = image.convert("RGBA") | |
| background = Image.new("RGB", alpha.size, (255, 255, 255)) | |
| background.paste(alpha, mask=alpha.getchannel("A")) | |
| image = background | |
| else: | |
| image = image.convert("RGB") | |
| width, height = image.size | |
| largest = max(width, height) | |
| if largest > ROOM_UPLOAD_MAX_DIMENSION: | |
| scale = ROOM_UPLOAD_MAX_DIMENSION / float(largest) | |
| target_size = ( | |
| max(1, int(round(width * scale))), | |
| max(1, int(round(height * scale))), | |
| ) | |
| image = image.resize(target_size, Image.Resampling.LANCZOS) | |
| buffer = io.BytesIO() | |
| image.save(buffer, format="JPEG", quality=ROOM_UPLOAD_JPEG_QUALITY, optimize=True) | |
| return buffer.getvalue() | |
| def contour_to_polygon(contour: np.ndarray, simplify_epsilon: float) -> np.ndarray: | |
| if simplify_epsilon > 0: | |
| contour = cv2.approxPolyDP(contour, simplify_epsilon, True) | |
| return contour.reshape(-1, 2) | |
| def mask_to_polygons( | |
| mask: np.ndarray, | |
| min_area: float, | |
| simplify_epsilon: float, | |
| ) -> list[dict]: | |
| mask_u8 = (mask > 0).astype(np.uint8) | |
| contours, _ = cv2.findContours(mask_u8, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) | |
| polygons = [] | |
| for contour in contours: | |
| area = float(cv2.contourArea(contour)) | |
| if area < min_area: | |
| continue | |
| polygon = contour_to_polygon(contour, simplify_epsilon) | |
| if len(polygon) < 3: | |
| continue | |
| x, y, width, height = cv2.boundingRect(polygon.astype(np.int32).reshape(-1, 1, 2)) | |
| polygons.append({ | |
| "points": [[int(px), int(py)] for px, py in polygon], | |
| "area": area, | |
| "bbox": [int(x), int(y), int(width), int(height)], | |
| }) | |
| polygons.sort(key=lambda item: item["area"], reverse=True) | |
| return polygons | |
| def limit_material_image_size(img_np: np.ndarray, max_dimension: int = MATERIAL_MAX_DIMENSION) -> np.ndarray: | |
| h, w = img_np.shape[:2] | |
| largest = max(h, w) | |
| if largest <= max_dimension: | |
| return img_np | |
| scale = max_dimension / float(largest) | |
| new_w = max(1, int(round(w * scale))) | |
| new_h = max(1, int(round(h * scale))) | |
| return cv2.resize(img_np, (new_w, new_h), interpolation=cv2.INTER_AREA) | |
| def generate_tile_material_maps(img_np: np.ndarray, base_roughness: float = 0.42) -> tuple[dict[str, np.ndarray], dict]: | |
| albedo = limit_material_image_size(img_np) | |
| rgb = albedo.astype(np.float32) / 255.0 | |
| height = ( | |
| rgb[:, :, 0] * 0.299 + | |
| rgb[:, :, 1] * 0.587 + | |
| rgb[:, :, 2] * 0.114 | |
| ).astype(np.float32) | |
| h_min = float(np.min(height)) | |
| h_max = float(np.max(height)) | |
| height = np.clip((height - h_min) / max(h_max - h_min, 1e-3), 0.0, 1.0) | |
| height_smooth = cv2.GaussianBlur(height, (3, 3), 0) | |
| dx = np.roll(height_smooth, -1, axis=1) - np.roll(height_smooth, 1, axis=1) | |
| dy = np.roll(height_smooth, -1, axis=0) - np.roll(height_smooth, 1, axis=0) | |
| edge = np.clip((np.abs(dx) + np.abs(dy)) * 3.5, 0.0, 1.0) | |
| normal_strength = 3.0 | |
| nx = -dx * normal_strength | |
| ny = -dy * normal_strength | |
| nz = np.ones_like(height_smooth) | |
| norm = np.maximum(np.sqrt(nx * nx + ny * ny + nz * nz), 1e-4) | |
| normal = np.dstack(( | |
| (nx / norm) * 0.5 + 0.5, | |
| (ny / norm) * 0.5 + 0.5, | |
| (nz / norm) * 0.5 + 0.5, | |
| )) | |
| roughness_base = float(np.clip(base_roughness, 0.04, 0.96)) | |
| roughness = np.clip(roughness_base + edge * 0.34 + (0.5 - height_smooth) * 0.16, 0.04, 0.98) | |
| ao = np.clip(0.74 + height_smooth * 0.26 - edge * 0.28, 0.45, 1.0) | |
| specular = np.clip((1.0 - roughness) * 0.72 + height_smooth * 0.22, 0.02, 1.0) | |
| maps = { | |
| "albedoMap": albedo, | |
| "normalMap": float_to_rgb8(normal), | |
| "heightMap": gray_to_rgb8(height_smooth), | |
| "roughnessMap": gray_to_rgb8(roughness), | |
| "aoMap": gray_to_rgb8(ao), | |
| "specularMap": gray_to_rgb8(specular), | |
| } | |
| metadata = { | |
| "generator": MATERIAL_GENERATOR_VERSION, | |
| "generatedKinds": ["normalMap", "heightMap", "roughnessMap", "aoMap", "specularMap"], | |
| "width": int(albedo.shape[1]), | |
| "height": int(albedo.shape[0]), | |
| "baseRoughness": roughness_base, | |
| } | |
| return maps, metadata | |
| def float_to_rgb8(values: np.ndarray) -> np.ndarray: | |
| return np.clip(values * 255.0, 0, 255).round().astype(np.uint8) | |
| def gray_to_rgb8(values: np.ndarray) -> np.ndarray: | |
| gray = np.clip(values * 255.0, 0, 255).round().astype(np.uint8) | |
| return np.dstack((gray, gray, gray)) | |
| def material_cache_key(contents: bytes, base_roughness: float) -> str: | |
| digest = hashlib.sha256() | |
| digest.update(MATERIAL_GENERATOR_VERSION.encode("utf-8")) | |
| digest.update(f":{float(base_roughness):.4f}:".encode("utf-8")) | |
| digest.update(contents) | |
| return digest.hexdigest()[:32] | |
| def material_package_from_cache(cache_key: str) -> dict | None: | |
| metadata_path = MATERIAL_DIR / cache_key / "metadata.json" | |
| if not metadata_path.exists(): | |
| return None | |
| try: | |
| metadata = json.loads(metadata_path.read_text()) | |
| except Exception: | |
| return None | |
| maps = material_map_urls(cache_key) | |
| if not all((MATERIAL_DIR / cache_key / Path(url).name).exists() for url in maps.values()): | |
| return None | |
| return { | |
| "id": cache_key, | |
| "cached": True, | |
| "maps": maps, | |
| "metadata": metadata, | |
| } | |
| def build_tile_material_package(contents: bytes, base_roughness: float = 0.42) -> dict: | |
| cache_key = material_cache_key(contents, base_roughness) | |
| cached = material_package_from_cache(cache_key) | |
| if cached is not None: | |
| return cached | |
| img_np = load_tile_image(contents) | |
| maps, metadata = generate_tile_material_maps(img_np, base_roughness) | |
| package_dir = MATERIAL_DIR / cache_key | |
| package_dir.mkdir(parents=True, exist_ok=True) | |
| filenames = { | |
| "albedoMap": "albedo.png", | |
| "normalMap": "normal.png", | |
| "heightMap": "height.png", | |
| "roughnessMap": "roughness.png", | |
| "aoMap": "ao.png", | |
| "specularMap": "specular.png", | |
| } | |
| for key, filename in filenames.items(): | |
| Image.fromarray(maps[key]).save(package_dir / filename, format="PNG", optimize=True) | |
| metadata_path = package_dir / "metadata.json" | |
| metadata_path.write_text(json.dumps(metadata)) | |
| return { | |
| "id": cache_key, | |
| "cached": False, | |
| "maps": material_map_urls(cache_key), | |
| "metadata": metadata, | |
| } | |
| def material_map_urls(cache_key: str) -> dict[str, str]: | |
| return { | |
| "albedoMap": f"/materials/{cache_key}/albedo.png", | |
| "normalMap": f"/materials/{cache_key}/normal.png", | |
| "heightMap": f"/materials/{cache_key}/height.png", | |
| "roughnessMap": f"/materials/{cache_key}/roughness.png", | |
| "aoMap": f"/materials/{cache_key}/ao.png", | |
| "specularMap": f"/materials/{cache_key}/specular.png", | |
| } | |
| def encode_bundle_image_data_url(img_np: np.ndarray) -> str: | |
| image_format = BUNDLE_IMAGE_FORMAT | |
| if image_format in {"jpg", "jpeg"}: | |
| pil_format = "JPEG" | |
| mime_type = "image/jpeg" | |
| extension = "jpg" | |
| elif image_format == "webp": | |
| pil_format = "WEBP" | |
| mime_type = "image/webp" | |
| extension = "webp" | |
| else: | |
| pil_format = "JPEG" | |
| mime_type = "image/jpeg" | |
| extension = "jpg" | |
| image = Image.fromarray(img_np.astype(np.uint8)) | |
| buffer = io.BytesIO() | |
| try: | |
| save_kwargs = { | |
| "quality": int(np.clip(BUNDLE_IMAGE_QUALITY, 1, 100)), | |
| "optimize": True, | |
| } | |
| if pil_format == "WEBP": | |
| save_kwargs["method"] = 4 | |
| image.save(buffer, format=pil_format, **save_kwargs) | |
| except Exception as exc: | |
| print(f"Bundle {extension} encoding failed ({exc}); falling back to JPEG.", flush=True) | |
| buffer = io.BytesIO() | |
| image.save(buffer, format="JPEG", quality=92, optimize=True) | |
| mime_type = "image/jpeg" | |
| encoded = base64.b64encode(buffer.getvalue()).decode() | |
| return f"data:{mime_type};base64,{encoded}" | |
| def data_url_to_bytes(value: str) -> tuple[str, bytes]: | |
| if value.startswith("data:"): | |
| header, encoded = value.split(",", 1) | |
| mime_type = header[5:].split(";", 1)[0] or "application/octet-stream" | |
| return mime_type, base64.b64decode(encoded) | |
| return "application/octet-stream", base64.b64decode(value) | |
| def bundle_image_extension(mime_type: str) -> str: | |
| if mime_type == "image/webp": | |
| return "webp" | |
| if mime_type == "image/png": | |
| return "png" | |
| return "jpg" | |
| def make_binary_ref(path: str, data: bytes, dtype: str) -> dict: | |
| return { | |
| "path": path, | |
| "dtype": dtype, | |
| "byteLength": len(data), | |
| } | |
| def encode_mask_bitset(mask_bytes: bytes, pixel_count: int) -> bytes: | |
| """Encode uint32 pixel indices as an exact, least-significant-bit-first bitset.""" | |
| indices = np.frombuffer(mask_bytes, dtype=np.uint32) | |
| if indices.size and int(indices.max()) >= pixel_count: | |
| raise ValueError("Mask contains a pixel index outside the source image.") | |
| bits = np.zeros(pixel_count, dtype=np.uint8) | |
| bits[indices] = 1 | |
| return np.packbits(bits, bitorder="little").tobytes() | |
| def foreground_occlusion_mask(surface_mask: np.ndarray, seg_map: np.ndarray) -> np.ndarray: | |
| h, w = surface_mask.shape[:2] | |
| kern_size = max(5, min(h, w) // 180) | 1 | |
| kern = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (kern_size, kern_size)) | |
| occluder_mask = np.isin(seg_map, class_ids(OCCLUDER_CLASSES)).astype(np.uint8) | |
| if not occluder_mask.any() or not surface_mask.any(): | |
| return np.zeros((h, w), dtype=np.uint8) | |
| near_surface = cv2.dilate(surface_mask.astype(np.uint8), kern, iterations=5) | |
| return ((occluder_mask > 0) & (near_surface > 0)).astype(np.uint8) | |
| def build_manifest_segmentation_bundle( | |
| bundle: dict, | |
| job_id: str, | |
| *, | |
| image_url: str | None = None, | |
| include_image: bool = False, | |
| ) -> bytes: | |
| metadata = dict(bundle) | |
| metadata["format"] = "baetes-vizbundle" | |
| metadata["bundleVersion"] = 3 | |
| metadata["encoding"] = "zip-store" | |
| metadata["assetBaseUrl"] = f"/viz2d/jobs/{job_id}/assets/" | |
| metadata["pixels"] = None | |
| if image_url: | |
| metadata["imageUrl"] = image_url | |
| metadata["segments"] = [] | |
| asset_dir = JOB_DIR / f"{job_id}.assets" | |
| asset_dir.mkdir(parents=True, exist_ok=True) | |
| buffer = io.BytesIO() | |
| with zipfile.ZipFile(buffer, "w", compression=zipfile.ZIP_STORED) as archive: | |
| if include_image: | |
| pixels_mime, pixels = data_url_to_bytes(bundle["pixels"]) | |
| image_path = f"image.{bundle_image_extension(pixels_mime)}" | |
| archive.writestr(image_path, pixels) | |
| metadata["pixels"] = { | |
| "path": image_path, | |
| "mimeType": pixels_mime, | |
| "byteLength": len(pixels), | |
| } | |
| lazy_asset_fields = { | |
| "surfaceUv": ("surface-uv.f32", "float32"), | |
| "surfacePlaneIds": ("surface-plane-ids.u8", "uint8"), | |
| "shadeMap": ("shade.u8", "uint8"), | |
| "showroomShadeMap": ("showroom-shade.u8", "uint8"), | |
| "reflectionMap": ("reflection.u8", "uint8"), | |
| } | |
| for segment_index, segment in enumerate(bundle.get("segments", [])): | |
| segment_metadata = dict(segment) | |
| pixel_count = int(metadata["width"]) * int(metadata["height"]) | |
| for mask_field in ("mask", "occlusionMask"): | |
| mask_value = segment.get(mask_field) | |
| if mask_value: | |
| mask_bytes = base64.b64decode(mask_value) | |
| mask_bitset = encode_mask_bitset(mask_bytes, pixel_count) | |
| mask_path = f"segments/{segment_index}/{mask_field}.bitset" | |
| archive.writestr(mask_path, mask_bitset) | |
| segment_metadata[mask_field] = { | |
| **make_binary_ref(mask_path, mask_bitset, "bitset"), | |
| "encoding": "bitset-lsb0", | |
| "pixelCount": pixel_count, | |
| } | |
| print( | |
| f"[BUNDLE] Compressed {mask_field} for job {job_id}: " | |
| f"{len(mask_bytes)} -> {len(mask_bitset)} bytes", | |
| flush=True, | |
| ) | |
| for field, (filename, dtype) in lazy_asset_fields.items(): | |
| value = segment.get(field) | |
| if not value: | |
| continue | |
| data = base64.b64decode(value) | |
| asset_name = f"s{segment_index}-{filename}" | |
| (asset_dir / asset_name).write_bytes(data) | |
| segment_metadata[field] = { | |
| **make_binary_ref(asset_name, data, dtype), | |
| "url": f"/viz2d/jobs/{job_id}/assets/{asset_name}", | |
| "lazy": True, | |
| } | |
| metadata["segments"].append(segment_metadata) | |
| archive.writestr( | |
| "metadata.json", | |
| json.dumps(metadata, separators=(",", ":")).encode("utf-8"), | |
| ) | |
| return buffer.getvalue() | |
| def build_segmentation_bundle(contents: bytes, debug: PipelineDebugCapture | None = None): | |
| t_start = time.perf_counter() | |
| t0 = time.perf_counter() | |
| img = Image.open(io.BytesIO(contents)).convert("RGB") | |
| img_np = np.array(img) | |
| h, w = img_np.shape[:2] | |
| min_floor_area = max(1200, int(w * h * 0.015)) | |
| image_loading_elapsed = time.perf_counter() - t0 | |
| print(f"[TIMING] Image loading/parsing took {image_loading_elapsed:.3f} seconds", flush=True) | |
| if debug is not None: | |
| debug.record_images( | |
| "source-image", | |
| "Uploaded image", | |
| "The RGB image passed to the analysis pipeline.", | |
| [("Source", img_np)], | |
| {"width": int(w), "height": int(h), "minimumFloorPixels": int(min_floor_area)}, | |
| image_loading_elapsed * 1000, | |
| ) | |
| t0 = time.perf_counter() | |
| seg_map = run_segmentation(img, img_np) | |
| segmentation_elapsed = time.perf_counter() - t0 | |
| print(f"[TIMING] Floor segmentation took {segmentation_elapsed:.3f} seconds", flush=True) | |
| if debug is not None: | |
| label_ids, label_counts = np.unique(seg_map, return_counts=True) | |
| class_areas = { | |
| class_name_for_id(int(label_id)): int(pixel_count) | |
| for label_id, pixel_count in zip(label_ids, label_counts) | |
| if int(pixel_count) >= max(128, int(w * h * 0.001)) | |
| } | |
| debug.record_images( | |
| "semantic-segmentation", | |
| "Semantic segmentation", | |
| "Colour overlay of the ADE20K semantic labels used to find the floor and reject room objects.", | |
| [("Class overlay", debug_segmentation_overlay(img_np, seg_map))], | |
| { | |
| "backend": segmentation_backend, | |
| "model": active_segmentation_model_name(), | |
| "classPixels": class_areas, | |
| }, | |
| segmentation_elapsed * 1000, | |
| ) | |
| t0 = time.perf_counter() | |
| pixels_b64 = encode_bundle_image_data_url(img_np) | |
| print(f"[TIMING] Image bundle encoding took {time.perf_counter() - t0:.3f} seconds", flush=True) | |
| t0 = time.perf_counter() | |
| primary_floor_ids = class_ids(PRIMARY_FLOOR_CLASSES) | |
| floor_class_ids = class_ids(FLOOR_SURFACE_CLASSES) | |
| raw_primary_floor_mask = np.isin(seg_map, primary_floor_ids).astype(np.uint8) | |
| floor_mask = raw_primary_floor_mask.copy() | |
| floor_mask = wall_subtract(floor_mask, seg_map, dilation=0) | |
| floor_mask = clean_floor_mask(floor_mask) | |
| floor_mask_source = "primary-floor" | |
| if int(floor_mask.sum()) < min_floor_area: | |
| floor_mask = np.isin(seg_map, floor_class_ids).astype(np.uint8) | |
| floor_mask = wall_subtract(floor_mask, seg_map, dilation=0) | |
| floor_mask = clean_floor_mask(floor_mask) | |
| floor_mask_source = "floor-surface-fallback" | |
| cleaned_floor_mask = floor_mask.copy() | |
| floor_mask, sam_metadata = refine_floor_mask_with_sam2(img_np, floor_mask, seg_map) | |
| floor_mask_elapsed = time.perf_counter() - t0 | |
| print(f"[TIMING] Floor masking/cleanup took {floor_mask_elapsed:.3f} seconds", flush=True) | |
| if debug is not None: | |
| debug.record_images( | |
| "floor-mask", | |
| "Floor mask selection", | |
| "Compare the raw floor prediction with cleanup and optional SAM refinement before geometry is estimated.", | |
| [ | |
| ("Raw floor class", debug_mask_overlay(img_np, raw_primary_floor_mask)), | |
| ("Cleaned candidate", debug_mask_overlay(img_np, cleaned_floor_mask)), | |
| ("Final floor mask", debug_mask_overlay(img_np, floor_mask)), | |
| ], | |
| { | |
| "selection": floor_mask_source, | |
| "rawFloorPixels": int(raw_primary_floor_mask.sum()), | |
| "cleanedFloorPixels": int(cleaned_floor_mask.sum()), | |
| "finalFloorPixels": int(floor_mask.sum()), | |
| **sam_metadata, | |
| }, | |
| floor_mask_elapsed * 1000, | |
| ) | |
| t0 = time.perf_counter() | |
| geometry = estimate_scene_geometry(img, img_np, w, h) | |
| depth = geometry.get("depth") | |
| geometry_elapsed = time.perf_counter() - t0 | |
| print( | |
| f"[TIMING] Geometry estimation ({geometry.get('provider')}) took " | |
| f"{geometry_elapsed:.3f} seconds", | |
| flush=True, | |
| ) | |
| if debug is not None: | |
| geometry_images: list[tuple[str, np.ndarray]] = [] | |
| if depth is not None: | |
| geometry_images.append(("Depth", debug_scalar_map(depth, floor_mask))) | |
| debug.record_images( | |
| "scene-geometry", | |
| "Scene geometry", | |
| "Depth or metric geometry used to fit and validate the floor plane.", | |
| geometry_images, | |
| { | |
| "provider": geometry.get("provider"), | |
| "depthAvailable": depth is not None, | |
| "metric": bool(geometry.get("metric")), | |
| }, | |
| geometry_elapsed * 1000, | |
| ) | |
| t0 = time.perf_counter() | |
| homography, plane = estimate_floor_plane(floor_mask, img_np, geometry) | |
| plane_elapsed = time.perf_counter() - t0 | |
| print(f"[TIMING] Plane fitting / homography calculation took {plane_elapsed:.3f} seconds", flush=True) | |
| if debug is not None: | |
| debug.record_images( | |
| "floor-plane", | |
| "Floor plane and perspective", | |
| "The fitted floor quadrilateral is the source of the perspective transform used for tile placement.", | |
| [("Fitted quadrilateral", debug_plane_overlay(img_np, plane))], | |
| { | |
| "planeFound": plane is not None, | |
| "homographyAvailable": homography is not None, | |
| "geometryProvider": plane.get("geometryProvider") if plane else None, | |
| "geometryConfidence": plane.get("geometryConfidence") if plane else 0.0, | |
| "metricGeometry": bool(plane and plane.get("floorTransform")), | |
| "quad": plane.get("quad") if plane else None, | |
| }, | |
| plane_elapsed * 1000, | |
| ) | |
| t0 = time.perf_counter() | |
| quad = np.asarray(plane["quad"], dtype=np.float32).reshape(4, 2) if plane and plane.get("quad") else None | |
| surface_mask, edge_metadata = build_floor_surface_mask(floor_mask, seg_map, quad, depth, geometry, plane) | |
| surface_indices = np.flatnonzero(surface_mask.ravel()).astype(np.uint32) | |
| occlusion_mask = foreground_occlusion_mask(surface_mask, seg_map) | |
| occlusion_indices = np.flatnonzero(occlusion_mask.ravel()).astype(np.uint32) | |
| surface_mask_elapsed = time.perf_counter() - t0 | |
| print(f"[TIMING] Surface masking took {surface_mask_elapsed:.3f} seconds", flush=True) | |
| if debug is not None: | |
| debug.record_images( | |
| "renderable-surface", | |
| "Renderable floor surface", | |
| "Gold pixels are retained for rendering; red pixels were part of the floor mask but removed by protection, edge, or geometry filters.", | |
| [("Final surface", debug_surface_overlay(img_np, floor_mask, surface_mask))], | |
| { | |
| "floorPixels": int(floor_mask.sum()), | |
| "surfacePixels": int(surface_mask.sum()), | |
| "occlusionPixels": int(occlusion_mask.sum()), | |
| "surfaceCoverage": round(float(surface_mask.sum()) / max(int(floor_mask.sum()), 1), 4), | |
| **edge_metadata, | |
| }, | |
| surface_mask_elapsed * 1000, | |
| ) | |
| t0 = time.perf_counter() | |
| surface_mapping = build_surface_uv_mapping(surface_mask, surface_indices, geometry) | |
| use_surface_uv = bool(surface_mapping and surface_mapping.get("useSurfaceUv")) | |
| mapping_metadata = ( | |
| surface_mapping["metadata"] | |
| if surface_mapping is not None | |
| else { | |
| "surfaceUvEnabled": False, | |
| "textureMappingMode": "regularized-floor-plane", | |
| "textureMappingReason": "surface-uv-unavailable", | |
| "surfaceUvCoverage": 0.0, | |
| "surfaceUvFinitePixels": 0, | |
| "surfaceUvTotalPixels": int(len(surface_indices)), | |
| "surfaceUvNeighborCount": 0, | |
| "surfaceUvMedianNeighborDelta": None, | |
| "surfaceUvP95NeighborDelta": None, | |
| "surfaceUvP99NeighborDelta": None, | |
| "surfaceUvJumpLimit": None, | |
| "surfaceUvBadNeighborRatio": None, | |
| "surfaceUvPlaneBoundaryRatio": None, | |
| "surfaceUvAssignedPlaneCount": 0, | |
| } | |
| ) | |
| surface_mapping_elapsed = time.perf_counter() - t0 | |
| print(f"[TIMING] Multi-plane surface mapping took {surface_mapping_elapsed:.3f} seconds", flush=True) | |
| print( | |
| "[FLOOR_3D_BUNDLE] serialized derived floor geometry " | |
| f"rawPointMapSent=false floorTransformSent={bool(plane and plane.get('floorTransform'))} " | |
| f"surfaceUvSent={use_surface_uv} surfaceUvPixels=" | |
| f"{int(np.isfinite(surface_mapping['uv']).all(axis=1).sum()) if use_surface_uv else 0}/" | |
| f"{int(len(surface_indices))} " | |
| f"surfacePlaneCount={len(surface_mapping['planes']) if surface_mapping is not None else 0} " | |
| f"geometryProvider={plane.get('geometryProvider', geometry.get('provider')) if plane else geometry.get('provider')}", | |
| flush=True, | |
| ) | |
| if debug is not None: | |
| mapping_images: list[tuple[str, np.ndarray]] = [] | |
| if use_surface_uv: | |
| mapping_images.append(( | |
| "UV checkerboard", | |
| debug_uv_checkerboard(h, w, surface_indices, surface_mapping["uv"]), | |
| )) | |
| mapping_images.append(( | |
| "Surface planes", | |
| debug_plane_id_map(h, w, surface_indices, surface_mapping["planeIds"]), | |
| )) | |
| debug.record_images( | |
| "surface-mapping", | |
| "Texture-coordinate mapping", | |
| "The checkerboard shows the coordinates sent to the renderer. Plane colours identify separately fitted surface planes.", | |
| mapping_images, | |
| mapping_metadata, | |
| surface_mapping_elapsed * 1000, | |
| ) | |
| t0 = time.perf_counter() | |
| lighting_maps = None | |
| shade_map = None | |
| showroom_shade_map = None | |
| reflection_map = None | |
| lighting_source = None | |
| intrinsic_lighting_used = False | |
| if ENABLE_SHADING_TRANSFER and ENABLE_INTRINSIC_SHADING: | |
| if intrinsic_models is None: | |
| _load_intrinsic_model() | |
| if intrinsic_models is not None: | |
| lighting_maps = build_intrinsic_lighting_maps(img_np, surface_mask) | |
| if lighting_maps and lighting_maps.get("shadeMap") is not None: | |
| lighting_source = f"intrinsic-{INTRINSIC_MODEL_VERSION}" | |
| intrinsic_lighting_used = True | |
| if ENABLE_SHADING_TRANSFER and (not lighting_maps or lighting_maps.get("shadeMap") is None): | |
| lighting_maps = build_luminance_lighting_maps(img_np, surface_mask) | |
| if lighting_maps and lighting_maps.get("shadeMap") is not None: | |
| lighting_source = "luminance" | |
| if lighting_maps: | |
| shade_map = lighting_maps.get("shadeMap") | |
| showroom_shade_map = lighting_maps.get("showroomShadeMap") | |
| reflection_map = lighting_maps.get("reflectionMap") | |
| shade_statistics = lighting_map_statistics(shade_map, surface_mask) | |
| lighting_elapsed = time.perf_counter() - t0 | |
| print( | |
| "[TIMING] Lighting map construction took " | |
| f"{lighting_elapsed:.3f} seconds " | |
| f"(source={lighting_source}, reflections={reflection_map is not None})", | |
| flush=True, | |
| ) | |
| if debug is not None: | |
| lighting_images: list[tuple[str, np.ndarray]] = [] | |
| if shade_map is not None: | |
| lighting_images.append(("Shade map", debug_scalar_map(shade_map, surface_mask))) | |
| if showroom_shade_map is not None: | |
| lighting_images.append(("Showroom shade map", debug_scalar_map(showroom_shade_map, surface_mask))) | |
| if reflection_map is not None: | |
| lighting_images.append(("Reflection map", debug_scalar_map(reflection_map, surface_mask))) | |
| debug.record_images( | |
| "lighting-transfer", | |
| "Lighting and reflections", | |
| "Maps transferred from the room image to make the replacement material fit its surrounding light.", | |
| lighting_images, | |
| { | |
| "source": lighting_source, | |
| "shadingAvailable": shade_map is not None, | |
| "reflectionsAvailable": reflection_map is not None, | |
| }, | |
| lighting_elapsed * 1000, | |
| ) | |
| t0 = time.perf_counter() | |
| direction_transform = ( | |
| plane.get("floorTransform") | |
| if plane | |
| and plane.get("floorTransform") | |
| and float(plane.get("geometryConfidence") or 0.0) >= 0.35 | |
| else homography | |
| ) | |
| direction_inputs = floor_direction_geometry_inputs( | |
| surface_mask, | |
| geometry, | |
| plane, | |
| surface_mapping, | |
| use_surface_uv=use_surface_uv, | |
| ) | |
| wall_mask = np.isin(seg_map, class_ids({"wall"})).astype(np.uint8) | |
| structure_mask = np.isin( | |
| seg_map, | |
| class_ids({"wall", "ceiling", "door", "window", "column", "base", "railing"}), | |
| ).astype(np.uint8) | |
| structure_kernel_size = scaled_odd_kernel_size(h, w, 0.006, minimum=3) | |
| structure_kernel = cv2.getStructuringElement( | |
| cv2.MORPH_ELLIPSE, | |
| (structure_kernel_size, structure_kernel_size), | |
| ) | |
| structure_mask = cv2.dilate(structure_mask, structure_kernel, iterations=1) | |
| floor_direction = find_floor_direction( | |
| img_np, | |
| surface_mask, | |
| wall_mask=wall_mask, | |
| structure_mask=structure_mask, | |
| normals_map=direction_inputs["normalsMap"], | |
| geometry_valid_mask=direction_inputs["validMask"], | |
| intrinsics=direction_inputs["intrinsics"], | |
| floor_normal=direction_inputs["floorNormal"], | |
| render_u_axis=direction_inputs["renderUAxis"], | |
| render_v_axis=direction_inputs["renderVAxis"], | |
| render_transform=direction_transform, | |
| surface_uv=surface_mapping["uv"] if use_surface_uv else None, | |
| surface_indices=surface_indices if use_surface_uv else None, | |
| surface_plane_ids=surface_mapping["planeIds"] if use_surface_uv else None, | |
| ) | |
| floor_direction_elapsed = time.perf_counter() - t0 | |
| print( | |
| "[TIMING] Floor direction analysis took " | |
| f"{floor_direction_elapsed:.3f} seconds " | |
| f"(method={floor_direction.get('directionMethod')}, " | |
| f"angle={floor_direction.get('angleDegrees')}, " | |
| f"confidence={floor_direction.get('confidence')})", | |
| flush=True, | |
| ) | |
| if debug is not None: | |
| debug.record_images( | |
| "floor-direction", | |
| "Floor direction", | |
| "Independent material, 3D wall-normal, and architectural cues are scored in renderer coordinates. Low-evidence images are marked as ambiguous.", | |
| [("Direction analysis", debug_floor_direction_overlay(img_np, floor_direction))], | |
| floor_direction, | |
| floor_direction_elapsed * 1000, | |
| ) | |
| t0 = time.perf_counter() | |
| segments = [] | |
| models = active_model_metadata(geometry, depth, shade_map, intrinsic_lighting_used) | |
| if len(surface_indices) >= min_floor_area: | |
| segments.append({ | |
| "id": 0, | |
| "className": "floor", | |
| "mask": base64.b64encode(surface_indices.tobytes()).decode(), | |
| "occlusionMask": ( | |
| base64.b64encode(occlusion_indices.tobytes()).decode() | |
| if len(occlusion_indices) else None | |
| ), | |
| "homography": homography, | |
| "floorTransform": plane.get("floorTransform") if plane else None, | |
| "plane": plane, | |
| "surfaceUv": ( | |
| base64.b64encode(surface_mapping["uv"].astype(np.float32).tobytes()).decode() | |
| if use_surface_uv else None | |
| ), | |
| "surfacePlaneIds": ( | |
| base64.b64encode(surface_mapping["planeIds"].astype(np.uint8).tobytes()).decode() | |
| if use_surface_uv else None | |
| ), | |
| "surfacePlanes": surface_mapping["planes"] if surface_mapping is not None else None, | |
| "shadeMap": base64.b64encode(shade_map.tobytes()).decode() if shade_map is not None else None, | |
| "showroomShadeMap": ( | |
| base64.b64encode(showroom_shade_map.tobytes()).decode() | |
| if showroom_shade_map is not None else None | |
| ), | |
| "reflectionMap": ( | |
| base64.b64encode(reflection_map.tobytes()).decode() | |
| if reflection_map is not None else None | |
| ), | |
| "metadata": { | |
| "segmenter": segmenter_metadata_name(), | |
| "segmentationModel": models["segmentation"]["name"], | |
| "floorPixels": int(floor_mask.sum()), | |
| "surfacePixels": int(surface_mask.sum()), | |
| "occlusionPixels": int(occlusion_mask.sum()), | |
| "floorDirection": floor_direction, | |
| **sam_metadata, | |
| **edge_metadata, | |
| **mapping_metadata, | |
| "depthEnabled": depth is not None, | |
| "shadingEnabled": shade_map is not None, | |
| "planeFitEnabled": ENABLE_PLANE_FIT, | |
| "multiPlaneSurfacesEnabled": ENABLE_MULTI_PLANE_SURFACES, | |
| "multiPlaneSurfaceCount": len(surface_mapping["planes"]) if surface_mapping is not None else 0, | |
| "multiPlaneNormalsReliable": ( | |
| bool(surface_mapping.get("normalsReliable")) | |
| if surface_mapping is not None else False | |
| ), | |
| "multiPlaneFitModes": ( | |
| [plane.get("fitMode") for plane in surface_mapping["planes"]] | |
| if surface_mapping is not None else [] | |
| ), | |
| "shadingTransferEnabled": ENABLE_SHADING_TRANSFER, | |
| "lightingSource": lighting_source, | |
| "lightingTransferVersion": LIGHTING_TRANSFER_VERSION, | |
| "shadeMapEncoding": SHADE_MAP_ENCODING, | |
| "shadeMapStatistics": shade_statistics, | |
| "shadeMapRange": [SHADE_MAP_MIN, SHADE_MAP_MAX], | |
| "showroomShadeMapRange": [SHOWROOM_SHADE_MAP_MIN, SHOWROOM_SHADE_MAP_MAX], | |
| "renderPresetDefaults": { | |
| "preset": "realistic", | |
| "shadingStrength": 1.0, | |
| "reflectionTransfer": True, | |
| "edgeFeatherPixels": 2, | |
| }, | |
| "reflectionTransferEnabled": reflection_map is not None, | |
| "reflectionMapRange": [REFLECTION_MAP_MIN, REFLECTION_MAP_MAX], | |
| "geometryConfidence": plane.get("geometryConfidence", 0.0) if plane else 0.0, | |
| "metricGeometry": bool(plane and plane.get("floorTransform")), | |
| "geometryProvider": plane.get("geometryProvider", geometry.get("provider")) if plane else geometry.get("provider"), | |
| "geometryModel": models["geometry"]["name"], | |
| }, | |
| }) | |
| if not segments: | |
| flat_seg = seg_map.ravel() | |
| for seg_id, class_id in enumerate(np.unique(flat_seg)): | |
| indices = np.where(flat_seg == class_id)[0].astype(np.uint32) | |
| if len(indices) < 1000: | |
| continue | |
| segments.append({ | |
| "id": int(seg_id), | |
| "className": class_name_for_id(int(class_id)), | |
| "mask": base64.b64encode(indices.tobytes()).decode(), | |
| "homography": None, | |
| "plane": None, | |
| "shadeMap": None, | |
| "showroomShadeMap": None, | |
| "reflectionMap": None, | |
| "metadata": { | |
| "segmenter": segmenter_metadata_name(), | |
| "segmentationModel": models["segmentation"]["name"], | |
| "depthEnabled": depth is not None, | |
| "shadingEnabled": False, | |
| "planeFitEnabled": ENABLE_PLANE_FIT, | |
| "samRefinementEnabled": ENABLE_SAM_REFINEMENT, | |
| "shadingTransferEnabled": ENABLE_SHADING_TRANSFER, | |
| "reflectionTransferEnabled": False, | |
| }, | |
| }) | |
| total_elapsed = time.perf_counter() - t_start | |
| print(f"[TIMING] Total bundle processing completed in {total_elapsed:.3f} seconds", flush=True) | |
| if debug is not None: | |
| debug.complete(total_elapsed * 1000) | |
| return { | |
| "width": w, | |
| "height": h, | |
| "pixels": pixels_b64, | |
| "segments": segments, | |
| "modelName": models["segmentation"]["name"], | |
| "models": models, | |
| } | |
| def job_path(job_id: str) -> Path: | |
| return JOB_DIR / f"{job_id}.json" | |
| def read_job(job_id: str): | |
| path = job_path(job_id) | |
| if not path.exists(): | |
| raise HTTPException(status_code=404, detail="Job not found.") | |
| return json.loads(path.read_text()) | |
| def write_job(job: dict): | |
| job_path(job["id"]).write_text(json.dumps(job)) | |
| def spaces_upload_enabled() -> bool: | |
| return bool(SPACES_BUCKET and SPACES_REGION and SPACES_ACCESS_KEY_ID and SPACES_SECRET_ACCESS_KEY) | |
| def spaces_endpoint_url() -> str: | |
| return SPACES_ENDPOINT_URL or f"https://{SPACES_REGION}.digitaloceanspaces.com" | |
| def spaces_upload_key(job_id: str, upload_path: Path) -> str: | |
| suffix = upload_path.suffix.lower() | |
| if suffix not in {".jpg", ".jpeg", ".png", ".webp"}: | |
| suffix = ".jpg" | |
| date_prefix = time.strftime("%Y/%m/%d", time.gmtime()) | |
| filename = f"{job_id}{suffix}" | |
| return "/".join(part for part in [SPACES_UPLOAD_PREFIX, date_prefix, filename] if part) | |
| def upload_input_image_to_spaces( | |
| job_id: str, | |
| upload_path: Path, | |
| original_filename: str | None, | |
| content_type: str | None, | |
| ): | |
| if not spaces_upload_enabled(): | |
| return | |
| try: | |
| import boto3 | |
| from botocore.config import Config | |
| except ImportError: | |
| print("DigitalOcean Spaces upload skipped; install boto3 to enable object storage.", flush=True) | |
| return | |
| try: | |
| key = spaces_upload_key(job_id, upload_path) | |
| metadata = {"job-id": job_id} | |
| if original_filename: | |
| metadata["original-filename"] = urllib.parse.quote(original_filename)[:900] | |
| extra_args = { | |
| "ContentType": content_type or "application/octet-stream", | |
| "Metadata": metadata, | |
| } | |
| if SPACES_ACL: | |
| extra_args["ACL"] = SPACES_ACL | |
| client = boto3.client( | |
| "s3", | |
| region_name=SPACES_REGION, | |
| endpoint_url=spaces_endpoint_url(), | |
| aws_access_key_id=SPACES_ACCESS_KEY_ID, | |
| aws_secret_access_key=SPACES_SECRET_ACCESS_KEY, | |
| config=Config(signature_version="s3v4"), | |
| ) | |
| client.upload_file(str(upload_path), SPACES_BUCKET, key, ExtraArgs=extra_args) | |
| job = read_job(job_id) | |
| job["inputObjectStorage"] = { | |
| "provider": "digitalocean-spaces", | |
| "bucket": SPACES_BUCKET, | |
| "region": SPACES_REGION, | |
| "key": key, | |
| "endpointUrl": spaces_endpoint_url(), | |
| } | |
| write_job(job) | |
| print(f"Uploaded input image for job {job_id} to DigitalOcean Spaces: {SPACES_BUCKET}/{key}", flush=True) | |
| except Exception as exc: | |
| print(f"DigitalOcean Spaces upload failed for job {job_id}: {exc}", flush=True) | |
| QA_IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".webp"} | |
| QA_TILE_DESCRIPTIONS = { | |
| "White Wood Flooring": "light plank flooring", | |
| "Black White Marble Checkered": "black and white marble checkerboard tile", | |
| "Grey Checkerboard": "grey checkerboard floor tile", | |
| "Lilac 20x20 Ceramic": "lilac 20x20 ceramic tile", | |
| } | |
| QA_TILE_ALIASES = { | |
| "Black White Marble Checker": "Black White Marble Checkered", | |
| } | |
| def qa_utc_now() -> datetime: | |
| return datetime.now(timezone.utc) | |
| def qa_iso_now() -> str: | |
| return qa_utc_now().isoformat().replace("+00:00", "Z") | |
| def qa_run_root(run_id: str) -> Path: | |
| if Path(run_id).name != run_id or not run_id: | |
| raise HTTPException(status_code=400, detail="Invalid QA run id.") | |
| return QA_RUN_DIR / run_id | |
| def qa_run_file(run_id: str) -> Path: | |
| return qa_run_root(run_id) / "run.json" | |
| def read_qa_run(run_id: str) -> dict: | |
| path = qa_run_file(run_id) | |
| if not path.exists(): | |
| raise HTTPException(status_code=404, detail="QA run not found.") | |
| return json.loads(path.read_text()) | |
| def write_qa_run(run: dict): | |
| run["updatedAt"] = qa_iso_now() | |
| root = qa_run_root(run["id"]) | |
| root.mkdir(parents=True, exist_ok=True) | |
| qa_run_file(run["id"]).write_text(json.dumps(run, indent=2)) | |
| def qa_sse_event(event_name: str, payload: dict) -> str: | |
| data = json.dumps(payload, separators=(",", ":")) | |
| return f"event: {event_name}\ndata: {data}\n\n" | |
| def spaces_client(): | |
| if not spaces_upload_enabled(): | |
| raise HTTPException(status_code=503, detail="DigitalOcean Spaces is not configured.") | |
| try: | |
| import boto3 | |
| from botocore.config import Config | |
| except ImportError as exc: | |
| raise HTTPException(status_code=503, detail="Install boto3 to use DigitalOcean Spaces QA images.") from exc | |
| return boto3.client( | |
| "s3", | |
| region_name=SPACES_REGION, | |
| endpoint_url=spaces_endpoint_url(), | |
| aws_access_key_id=SPACES_ACCESS_KEY_ID, | |
| aws_secret_access_key=SPACES_SECRET_ACCESS_KEY, | |
| config=Config(signature_version="s3v4"), | |
| ) | |
| def qa_parse_date(value: str | None, name: str): | |
| if not value: | |
| return None | |
| try: | |
| return datetime.strptime(value, "%Y-%m-%d").date() | |
| except ValueError as exc: | |
| raise HTTPException(status_code=400, detail=f"{name} must use YYYY-MM-DD.") from exc | |
| def qa_date_prefixes(range_name: str, from_value: str | None, to_value: str | None): | |
| today = qa_utc_now().date() | |
| range_key = (range_name or "week").lower() | |
| if range_key == "all": | |
| return [SPACES_UPLOAD_PREFIX] if SPACES_UPLOAD_PREFIX else [""] | |
| if range_key == "today": | |
| start = today | |
| end = today | |
| elif range_key == "week": | |
| start = today - timedelta(days=6) | |
| end = today | |
| elif range_key == "month": | |
| start = today.replace(day=1) | |
| end = today | |
| elif range_key == "custom": | |
| start = qa_parse_date(from_value, "from") or today | |
| end = qa_parse_date(to_value, "to") or start | |
| else: | |
| raise HTTPException(status_code=400, detail="range must be today, week, month, all, or custom.") | |
| if start > end: | |
| raise HTTPException(status_code=400, detail="from must be before or equal to to.") | |
| if (end - start).days > 370: | |
| raise HTTPException(status_code=400, detail="Date range is too large.") | |
| prefixes = [] | |
| cursor = start | |
| while cursor <= end: | |
| prefixes.append("/".join(part for part in [SPACES_UPLOAD_PREFIX, cursor.strftime("%Y/%m/%d")] if part)) | |
| cursor += timedelta(days=1) | |
| return prefixes | |
| def qa_clean_etag(value) -> str: | |
| return str(value or "").strip().strip('"') | |
| def qa_object_fingerprint(etag, size) -> str | None: | |
| clean_etag = qa_clean_etag(etag) | |
| if not clean_etag: | |
| return None | |
| return f"{clean_etag}:{size or 0}" | |
| def dedupe_qa_image_keys(image_keys: list[str]) -> list[str]: | |
| client = spaces_client() | |
| unique_keys: list[str] = [] | |
| seen_keys: set[str] = set() | |
| seen_fingerprints: set[str] = set() | |
| for key in image_keys: | |
| if key in seen_keys: | |
| continue | |
| seen_keys.add(key) | |
| try: | |
| metadata = client.head_object(Bucket=SPACES_BUCKET, Key=key) | |
| except Exception as exc: | |
| raise HTTPException(status_code=400, detail=f"Could not read DigitalOcean image metadata for {key}.") from exc | |
| fingerprint = qa_object_fingerprint(metadata.get("ETag"), metadata.get("ContentLength")) | |
| if fingerprint and fingerprint in seen_fingerprints: | |
| continue | |
| if fingerprint: | |
| seen_fingerprints.add(fingerprint) | |
| unique_keys.append(key) | |
| return unique_keys | |
| def list_qa_spaces_images(range_name: str, from_value: str | None, to_value: str | None, max_items: int): | |
| client = spaces_client() | |
| paginator = client.get_paginator("list_objects_v2") | |
| images: list[dict] = [] | |
| seen: set[str] = set() | |
| seen_fingerprints: set[str] = set() | |
| for prefix in qa_date_prefixes(range_name, from_value, to_value): | |
| kwargs = {"Bucket": SPACES_BUCKET, "Prefix": prefix} | |
| for page in paginator.paginate(**kwargs): | |
| for item in page.get("Contents", []): | |
| key = item.get("Key", "") | |
| if not key or key in seen or Path(key).suffix.lower() not in QA_IMAGE_EXTENSIONS: | |
| continue | |
| fingerprint = qa_object_fingerprint(item.get("ETag"), item.get("Size")) | |
| if fingerprint and fingerprint in seen_fingerprints: | |
| continue | |
| seen.add(key) | |
| if fingerprint: | |
| seen_fingerprints.add(fingerprint) | |
| url = client.generate_presigned_url( | |
| "get_object", | |
| Params={"Bucket": SPACES_BUCKET, "Key": key}, | |
| ExpiresIn=QA_PRESIGNED_URL_EXPIRES_SECONDS, | |
| ) | |
| last_modified = item.get("LastModified") | |
| images.append({ | |
| "key": key, | |
| "fileName": urllib.parse.unquote(Path(key).name), | |
| "url": url, | |
| "thumbnailUrl": url, | |
| "uploadedAt": last_modified.isoformat().replace("+00:00", "Z") if last_modified else None, | |
| "sizeBytes": item.get("Size"), | |
| "contentType": mimetypes.guess_type(key)[0] or "application/octet-stream", | |
| "etag": qa_clean_etag(item.get("ETag")), | |
| }) | |
| if len(images) >= max_items: | |
| return sorted(images, key=lambda image: image.get("uploadedAt") or "", reverse=True) | |
| return sorted(images, key=lambda image: image.get("uploadedAt") or "", reverse=True) | |
| def qa_safe_filename(value: str, fallback: str = "image") -> str: | |
| stem = "".join(char.lower() if char.isalnum() else "-" for char in value).strip("-") | |
| while "--" in stem: | |
| stem = stem.replace("--", "-") | |
| return (stem or fallback)[:90] | |
| def normalize_qa_tile(tile_name: str) -> str: | |
| name = str(tile_name).strip() | |
| return QA_TILE_ALIASES.get(name, name) | |
| def qa_tile_description(tile_name: str, supplied_descriptions: dict[str, str]) -> str: | |
| return supplied_descriptions.get(tile_name) or QA_TILE_DESCRIPTIONS.get(tile_name) or f"{tile_name} flooring material" | |
| def parse_tile_descriptions(values) -> dict[str, str]: | |
| descriptions: dict[str, str] = {} | |
| if not isinstance(values, list): | |
| return descriptions | |
| for value in values: | |
| if not isinstance(value, str) or ":" not in value: | |
| continue | |
| name, description = value.split(":", 1) | |
| normalized = normalize_qa_tile(name) | |
| if normalized and description.strip(): | |
| descriptions[normalized] = description.strip() | |
| return descriptions | |
| def qa_status_from_playwright(result_status: str | None, test_status: str | None) -> str: | |
| status = result_status or "" | |
| if status == "passed": | |
| return "passed" | |
| if status == "skipped": | |
| return "skipped" | |
| if status: | |
| return "failed" | |
| if test_status == "expected": | |
| return "passed" | |
| if test_status == "skipped": | |
| return "skipped" | |
| return "failed" | |
| def qa_find_attachment(result: dict, name: str): | |
| for attachment in result.get("attachments", []) or []: | |
| if attachment.get("name") == name: | |
| return attachment | |
| return None | |
| def qa_attachment_path(attachment: dict | None, report_path: Path): | |
| if not attachment or not attachment.get("path"): | |
| return None | |
| raw_path = Path(str(attachment["path"])) | |
| candidates = [ | |
| raw_path if raw_path.is_absolute() else (QA_FRONTEND_DIR / raw_path), | |
| raw_path if raw_path.is_absolute() else (report_path.parent / raw_path), | |
| ] | |
| for candidate in candidates: | |
| if candidate.exists(): | |
| return candidate.resolve() | |
| return None | |
| def qa_asset_url(run_id: str, source_path: Path | None): | |
| if source_path is None: | |
| return None | |
| root = qa_run_root(run_id).resolve() | |
| try: | |
| relative = source_path.resolve().relative_to(root) | |
| except ValueError: | |
| assets_dir = root / "assets" | |
| assets_dir.mkdir(parents=True, exist_ok=True) | |
| destination = assets_dir / source_path.name | |
| if not destination.exists(): | |
| shutil.copy2(source_path, destination) | |
| relative = destination.relative_to(root) | |
| return f"/qa/runs/{run_id}/assets/{urllib.parse.quote(relative.as_posix())}" | |
| def qa_read_attachment_text(attachment: dict | None, report_path: Path): | |
| if not attachment: | |
| return "" | |
| source_path = qa_attachment_path(attachment, report_path) | |
| if source_path is not None: | |
| return source_path.read_text() | |
| body = attachment.get("body") | |
| if not isinstance(body, str): | |
| return "" | |
| stripped = body.strip() | |
| if stripped.startswith("{") or stripped.startswith("["): | |
| return body | |
| try: | |
| return base64.b64decode(body).decode("utf-8") | |
| except Exception: | |
| return "" | |
| def qa_parse_json_text(text: str): | |
| if not text.strip(): | |
| return None | |
| try: | |
| return json.loads(text) | |
| except Exception: | |
| start = text.find("{") | |
| end = text.rfind("}") | |
| if start >= 0 and end > start: | |
| try: | |
| return json.loads(text[start:end + 1]) | |
| except Exception: | |
| return None | |
| return None | |
| def qa_title_parts(title: str): | |
| if ": apply " not in title: | |
| return title, "Unknown tile" | |
| image_name, rest = title.split(": apply ", 1) | |
| tile_name = rest.rsplit(" to ", 1)[0] if " to " in rest else rest | |
| return image_name, tile_name | |
| def qa_grouped_case_metas(result: dict, report_path: Path): | |
| metas = [] | |
| for attachment in result.get("attachments", []) or []: | |
| name = str(attachment.get("name") or "") | |
| if not name.startswith("qa-case-") or not name.endswith("-meta"): | |
| continue | |
| meta = qa_parse_json_text(qa_read_attachment_text(attachment, report_path)) | |
| if not isinstance(meta, dict): | |
| continue | |
| prefix = str(meta.get("attachmentPrefix") or name.removesuffix("-meta")) | |
| metas.append((prefix, meta)) | |
| def sort_key(item): | |
| _, meta = item | |
| return qa_int(meta.get("caseIndex"), 0) | |
| return sorted(metas, key=sort_key) | |
| def qa_case_status(meta: dict, fallback_status: str) -> str: | |
| status = str(meta.get("status") or "").lower() | |
| if status in {"passed", "failed", "skipped"}: | |
| return status | |
| return fallback_status | |
| def collect_playwright_specs(suite: dict, entries: list[tuple[str, dict]]): | |
| for spec in suite.get("specs", []) or []: | |
| title = spec.get("title") or "Untitled QA case" | |
| for test in spec.get("tests", []) or []: | |
| entries.append((title, test)) | |
| for child in suite.get("suites", []) or []: | |
| collect_playwright_specs(child, entries) | |
| def parse_qa_results(run_id: str, report_path: Path): | |
| if not report_path.exists(): | |
| return [] | |
| report = json.loads(report_path.read_text()) | |
| entries: list[tuple[str, dict]] = [] | |
| for suite in report.get("suites", []) or []: | |
| collect_playwright_specs(suite, entries) | |
| results = [] | |
| for index, (title, test) in enumerate(entries): | |
| playwright_results = test.get("results", []) or [{}] | |
| final_result = playwright_results[-1] | |
| fallback_status = qa_status_from_playwright(final_result.get("status"), test.get("status")) | |
| grouped_cases = qa_grouped_case_metas(final_result, report_path) | |
| if grouped_cases: | |
| has_case_failure = any(qa_case_status(meta, fallback_status) == "failed" for _, meta in grouped_cases) | |
| global_error = final_result.get("error", {}).get("message") if isinstance(final_result.get("error"), dict) else None | |
| for case_index, (prefix, meta) in enumerate(grouped_cases, start=1): | |
| before_attachment = qa_find_attachment(final_result, f"{prefix}-before-canvas") | |
| after_attachment = qa_find_attachment(final_result, f"{prefix}-after-texture-canvas") | |
| gemini_attachment = qa_find_attachment(final_result, f"{prefix}-gemini-judge") | |
| status = qa_case_status(meta, fallback_status) | |
| if fallback_status == "failed" and not has_case_failure: | |
| status = "failed" | |
| case_id = qa_int(meta.get("caseIndex"), case_index) | |
| results.append({ | |
| "id": f"{run_id}-{case_id}", | |
| "imageKey": meta.get("imageName") or meta.get("roomName") or title, | |
| "imageName": meta.get("imageName") or meta.get("roomName") or title, | |
| "tileName": meta.get("tileName") or "Unknown tile", | |
| "status": status, | |
| "beforeImageUrl": qa_asset_url(run_id, qa_attachment_path(before_attachment, report_path)), | |
| "afterImageUrl": qa_asset_url(run_id, qa_attachment_path(after_attachment, report_path)), | |
| "gemini": qa_parse_json_text(qa_read_attachment_text(gemini_attachment, report_path)), | |
| "error": meta.get("error") if isinstance(meta.get("error"), str) else global_error, | |
| }) | |
| continue | |
| image_name, tile_name = qa_title_parts(title) | |
| before_attachment = qa_find_attachment(final_result, "before-canvas") | |
| after_attachment = qa_find_attachment(final_result, "after-texture-canvas") | |
| gemini_attachment = qa_find_attachment(final_result, "gemini-judge") | |
| error = final_result.get("error", {}).get("message") if isinstance(final_result.get("error"), dict) else None | |
| results.append({ | |
| "id": f"{run_id}-{index + 1}", | |
| "imageKey": image_name, | |
| "imageName": image_name, | |
| "tileName": tile_name, | |
| "status": fallback_status, | |
| "beforeImageUrl": qa_asset_url(run_id, qa_attachment_path(before_attachment, report_path)), | |
| "afterImageUrl": qa_asset_url(run_id, qa_attachment_path(after_attachment, report_path)), | |
| "gemini": qa_parse_json_text(qa_read_attachment_text(gemini_attachment, report_path)), | |
| "error": error, | |
| }) | |
| return results | |
| def build_qa_scenarios(downloaded_images: list[dict], tiles: list[str], descriptions: dict[str, str]): | |
| scenarios = [] | |
| for image in downloaded_images: | |
| image_name = Path(image["key"]).stem | |
| for tile in tiles: | |
| scenarios.append({ | |
| "imagePath": str(image["path"]), | |
| "keyword": "floor", | |
| "roomName": f"{image_name}", | |
| "surfaceClass": "floor", | |
| "textureName": tile, | |
| "textureDescription": qa_tile_description(tile, descriptions), | |
| "expectedDirectionDescription": ( | |
| "The applied material direction should follow the floor perspective and should not look arbitrarily rotated." | |
| ), | |
| }) | |
| return scenarios | |
| def download_qa_images(run_id: str, image_keys: list[str]): | |
| client = spaces_client() | |
| images_dir = qa_run_root(run_id) / "images" | |
| images_dir.mkdir(parents=True, exist_ok=True) | |
| downloaded = [] | |
| for index, key in enumerate(image_keys, start=1): | |
| extension = Path(key).suffix.lower() | |
| if extension not in QA_IMAGE_EXTENSIONS: | |
| extension = ".jpg" | |
| destination = images_dir / f"{index:03d}-{qa_safe_filename(Path(key).stem)}{extension}" | |
| client.download_file(SPACES_BUCKET, key, str(destination)) | |
| downloaded.append({"key": key, "path": destination}) | |
| return downloaded | |
| def qa_runner_env(run_root: Path, scenarios: list[dict], use_gemini: bool, backend_url: str): | |
| env = os.environ.copy() | |
| env.update({ | |
| "API_URL": backend_url, | |
| "PLAYWRIGHT_BASE_URL": QA_FRONTEND_URL, | |
| "QA_HTML_REPORT_DIR": str(run_root / "html-report"), | |
| "QA_PLAYWRIGHT_OUTPUT_DIR": str(run_root / "artifacts"), | |
| "QA_PROGRESS_FILE": str(run_root / "progress.json"), | |
| "QA_PROGRESS_MODE": "cases", | |
| "QA_RESULTS_DIR": str(run_root / "results"), | |
| "QA_SCENARIOS": json.dumps(scenarios), | |
| "QA_SKIP_GEMINI": "0" if use_gemini else "1", | |
| "QA_TEST_IMAGES_DIR": str(run_root / "images"), | |
| "QA_WORKERS": "1", | |
| "VITE_API_URL": backend_url, | |
| }) | |
| if use_gemini: | |
| if not GEMINI_API_KEY: | |
| raise RuntimeError("Gemini is enabled for this QA run, but GEMINI_API_KEY is not configured.") | |
| env["GEMINI_API_KEY"] = GEMINI_API_KEY | |
| env.setdefault("GEMINI_MODEL", "gemini-3.1-flash-lite") | |
| return env | |
| def finalize_qa_run_from_report(run: dict, report_path: Path): | |
| results = parse_qa_results(run["id"], report_path) | |
| passed = sum(1 for result in results if result["status"] == "passed") | |
| failed = sum(1 for result in results if result["status"] == "failed") | |
| completed = sum(1 for result in results if result["status"] in {"passed", "failed", "skipped"}) | |
| run.update({ | |
| "completed": completed, | |
| "failed": failed, | |
| "passed": passed, | |
| "results": results, | |
| "status": "completed" if report_path.exists() else "failed", | |
| }) | |
| if failed: | |
| run["status"] = "failed" | |
| def qa_int(value, default: int = 0) -> int: | |
| try: | |
| return int(value) | |
| except (TypeError, ValueError): | |
| return default | |
| def sync_qa_run_progress(run_id: str, progress_path: Path): | |
| if not progress_path.exists(): | |
| return | |
| try: | |
| progress = json.loads(progress_path.read_text()) | |
| except Exception: | |
| return | |
| run = read_qa_run(run_id) | |
| updates = { | |
| "completed": qa_int(progress.get("completed"), qa_int(run.get("completed"))), | |
| "failed": qa_int(progress.get("failed"), qa_int(run.get("failed"))), | |
| "passed": qa_int(progress.get("passed"), qa_int(run.get("passed"))), | |
| "progressUpdatedAt": progress.get("updatedAt"), | |
| "skipped": qa_int(progress.get("skipped"), qa_int(run.get("skipped"))), | |
| "total": qa_int(progress.get("total"), qa_int(run.get("total"))), | |
| } | |
| if run.get("status") == "queued": | |
| updates["status"] = "running" | |
| if all(run.get(key) == value for key, value in updates.items()): | |
| return | |
| run.update(updates) | |
| write_qa_run(run) | |
| def run_playwright_qa(run_id: str, run_root: Path, env: dict, log_file): | |
| command = ["npm", "run", "qa:e2e"] | |
| progress_path = run_root / "progress.json" | |
| process = subprocess.Popen( | |
| command, | |
| cwd=QA_FRONTEND_DIR, | |
| env=env, | |
| stderr=subprocess.STDOUT, | |
| stdout=log_file, | |
| ) | |
| deadline = time.monotonic() + QA_RUN_TIMEOUT_SECONDS | |
| while True: | |
| return_code = process.poll() | |
| sync_qa_run_progress(run_id, progress_path) | |
| if return_code is not None: | |
| return return_code | |
| if time.monotonic() >= deadline: | |
| process.terminate() | |
| try: | |
| process.wait(timeout=10) | |
| except subprocess.TimeoutExpired: | |
| process.kill() | |
| process.wait(timeout=10) | |
| sync_qa_run_progress(run_id, progress_path) | |
| raise subprocess.TimeoutExpired(command, QA_RUN_TIMEOUT_SECONDS) | |
| time.sleep(1) | |
| def qa_html(value) -> str: | |
| if value is None: | |
| return "" | |
| return html.escape(str(value), quote=True) | |
| def qa_status_label(status) -> str: | |
| value = str(status or "unknown").lower() | |
| if value in {"queued", "running", "completed", "failed", "cancelled", "passed", "skipped"}: | |
| return value | |
| return "unknown" | |
| def qa_yes_no(value) -> str: | |
| if isinstance(value, bool): | |
| return "yes" if value else "no" | |
| return "n/a" | |
| def qa_score_label(value) -> str: | |
| if isinstance(value, (int, float)) and np.isfinite(value): | |
| return f"{float(value):.2f}" | |
| return "n/a" | |
| def qa_list_label(value) -> str: | |
| if isinstance(value, list): | |
| items = [str(item) for item in value if isinstance(item, str) and item.strip()] | |
| if items: | |
| return "; ".join(items) | |
| return "none" | |
| def qa_metric_html(label: str, value: str) -> str: | |
| return f"<span><strong>{qa_html(value)}</strong>{qa_html(label)}</span>" | |
| def qa_image_html(label: str, url) -> str: | |
| if not isinstance(url, str) or not url.strip(): | |
| return f""" | |
| <div class="image-panel missing"> | |
| <span>{qa_html(label)}</span> | |
| <p>Missing screenshot</p> | |
| </div> | |
| """ | |
| return f""" | |
| <div class="image-panel"> | |
| <span>{qa_html(label)}</span> | |
| <img src="{qa_html(url)}" alt="{qa_html(label)} QA screenshot" loading="lazy" /> | |
| </div> | |
| """ | |
| def qa_gemini_report_html(gemini) -> str: | |
| if not isinstance(gemini, dict): | |
| return '<p class="muted">Gemini was not used or did not return a judgment for this result.</p>' | |
| issue_rows = [] | |
| if gemini.get("coverageGapDetected"): | |
| issue_rows.append(f"<li>Coverage gaps: {qa_html(qa_list_label(gemini.get('missedFloorAreas')))}</li>") | |
| if gemini.get("directionMismatchDetected"): | |
| issue_rows.append(f"<li>Direction issues: {qa_html(qa_list_label(gemini.get('directionIssues')))}</li>") | |
| if gemini.get("bleedDetected"): | |
| issue_rows.append("<li>Bleed detected on non-target surfaces.</li>") | |
| if isinstance(gemini.get("failureReasons"), list) and gemini.get("failureReasons"): | |
| issue_rows.append(f"<li>Failure reasons: {qa_html(qa_list_label(gemini.get('failureReasons')))}</li>") | |
| issues_html = f"<ul>{''.join(issue_rows)}</ul>" if issue_rows else "" | |
| metrics_html = "".join([ | |
| qa_metric_html("overall", qa_yes_no(gemini.get("overallPass"))), | |
| qa_metric_html("tile", qa_yes_no(gemini.get("tileApplied"))), | |
| qa_metric_html("coverage gap", qa_yes_no(gemini.get("coverageGapDetected"))), | |
| qa_metric_html("direction", qa_score_label(gemini.get("directionQuality"))), | |
| qa_metric_html("surface", qa_score_label(gemini.get("surfaceCorrectness"))), | |
| qa_metric_html("edge", qa_score_label(gemini.get("edgeCompleteness"))), | |
| qa_metric_html("occlusion", qa_score_label(gemini.get("occlusionBoundaryQuality"))), | |
| qa_metric_html("perspective", qa_score_label(gemini.get("perspectiveQuality"))), | |
| qa_metric_html("lighting", qa_score_label(gemini.get("lightingPreserved"))), | |
| qa_metric_html("realism", qa_score_label(gemini.get("realism"))), | |
| ]) | |
| return f""" | |
| <div class="gemini"> | |
| <div class="metrics">{metrics_html}</div> | |
| <p>{qa_html(gemini.get("notes") or "No notes returned.")}</p> | |
| {issues_html} | |
| </div> | |
| """ | |
| def qa_result_report_html(result: dict, index: int) -> str: | |
| status = qa_status_label(result.get("status")) | |
| image_name = result.get("imageName") or result.get("imageKey") or "Unknown image" | |
| tile_name = result.get("tileName") or "Unknown tile" | |
| error_html = f'<p class="error">{qa_html(result.get("error"))}</p>' if result.get("error") else "" | |
| return f""" | |
| <article class="result-card status-{status}"> | |
| <header> | |
| <div> | |
| <p class="eyebrow">{qa_html(tile_name)}</p> | |
| <h2>{index}. {qa_html(image_name)}</h2> | |
| </div> | |
| <span class="badge status-{status}">{qa_html(status)}</span> | |
| </header> | |
| <div class="image-stack"> | |
| {qa_image_html("Before", result.get("beforeImageUrl"))} | |
| {qa_image_html("After Tile Applied", result.get("afterImageUrl"))} | |
| </div> | |
| {error_html} | |
| <section> | |
| <h3>Gemini Judgment</h3> | |
| {qa_gemini_report_html(result.get("gemini"))} | |
| </section> | |
| </article> | |
| """ | |
| def build_qa_report_html(run: dict) -> str: | |
| status = qa_status_label(run.get("status")) | |
| results = run.get("results") if isinstance(run.get("results"), list) else [] | |
| result_cards = "\n".join( | |
| qa_result_report_html(result, index) | |
| for index, result in enumerate(results, start=1) | |
| if isinstance(result, dict) | |
| ) | |
| if not result_cards: | |
| result_cards = """ | |
| <section class="empty"> | |
| <h2>No screenshots yet</h2> | |
| <p>The report will include screenshots and Gemini judgment after Playwright writes the final report.</p> | |
| </section> | |
| """ | |
| error_html = f'<p class="error">{qa_html(run.get("error"))}</p>' if run.get("error") else "" | |
| title = f"QA Report {run.get('id', '')}".strip() | |
| return f"""<!doctype html> | |
| <html lang="en"> | |
| <head> | |
| <meta charset="utf-8" /> | |
| <meta name="viewport" content="width=device-width, initial-scale=1" /> | |
| <title>{qa_html(title)}</title> | |
| <style> | |
| :root {{ | |
| color-scheme: light; | |
| --bg: #f6f7fb; | |
| --panel: #ffffff; | |
| --ink: #172033; | |
| --muted: #667085; | |
| --line: #d8deea; | |
| --passed: #177245; | |
| --failed: #b4233a; | |
| --running: #1d5fbf; | |
| }} | |
| * {{ box-sizing: border-box; }} | |
| body {{ | |
| margin: 0; | |
| background: var(--bg); | |
| color: var(--ink); | |
| font-family: Inter, ui-sans-serif, system-ui, -apple-system, BlinkMacSystemFont, "Segoe UI", sans-serif; | |
| line-height: 1.5; | |
| }} | |
| main {{ width: min(1120px, calc(100% - 32px)); margin: 0 auto; padding: 32px 0 56px; }} | |
| .topbar {{ display: flex; justify-content: space-between; gap: 16px; align-items: flex-start; margin-bottom: 20px; }} | |
| h1, h2, h3, p {{ margin-top: 0; }} | |
| h1 {{ font-size: clamp(28px, 4vw, 42px); margin-bottom: 8px; }} | |
| h2 {{ font-size: 20px; margin-bottom: 4px; }} | |
| h3 {{ font-size: 14px; margin-bottom: 10px; text-transform: uppercase; color: var(--muted); letter-spacing: .04em; }} | |
| .eyebrow {{ color: var(--muted); font-size: 12px; font-weight: 800; letter-spacing: .06em; text-transform: uppercase; margin-bottom: 6px; }} | |
| .summary, .metrics {{ display: flex; flex-wrap: wrap; gap: 8px; }} | |
| .summary span, .metrics span {{ | |
| border: 1px solid var(--line); | |
| border-radius: 8px; | |
| background: #fff; | |
| color: var(--muted); | |
| display: inline-flex; | |
| gap: 6px; | |
| padding: 7px 10px; | |
| font-size: 13px; | |
| align-items: baseline; | |
| }} | |
| .summary strong, .metrics strong {{ color: var(--ink); }} | |
| .badge {{ | |
| border: 1px solid var(--line); | |
| border-radius: 999px; | |
| background: #fff; | |
| color: var(--muted); | |
| padding: 6px 10px; | |
| font-size: 12px; | |
| font-weight: 800; | |
| text-transform: uppercase; | |
| white-space: nowrap; | |
| }} | |
| .status-passed, .status-completed {{ color: var(--passed); border-color: #9bd4b5; }} | |
| .status-failed {{ color: var(--failed); border-color: #f0a4b3; }} | |
| .status-running, .status-queued {{ color: var(--running); border-color: #a9c7ff; }} | |
| .result-card, .empty {{ | |
| background: var(--panel); | |
| border: 1px solid var(--line); | |
| border-radius: 10px; | |
| padding: 18px; | |
| margin-top: 18px; | |
| box-shadow: 0 10px 30px rgba(15, 23, 42, .06); | |
| }} | |
| .result-card > header {{ display: flex; justify-content: space-between; gap: 16px; align-items: flex-start; margin-bottom: 16px; }} | |
| .image-stack {{ display: grid; gap: 16px; }} | |
| .image-panel {{ border: 1px solid var(--line); border-radius: 8px; overflow: hidden; background: #f8fafc; }} | |
| .image-panel span {{ display: block; padding: 8px 10px; color: var(--muted); font-size: 12px; font-weight: 800; text-transform: uppercase; }} | |
| .image-panel img {{ display: block; width: 100%; height: auto; }} | |
| .image-panel.missing {{ padding-bottom: 16px; }} | |
| .image-panel.missing p {{ margin: 0; padding: 0 10px; color: var(--muted); }} | |
| .gemini {{ margin-top: 8px; }} | |
| .gemini p {{ color: var(--ink); margin: 12px 0 0; }} | |
| .gemini ul {{ margin: 12px 0 0; padding-left: 20px; color: var(--failed); }} | |
| .muted {{ color: var(--muted); }} | |
| .error {{ | |
| border: 1px solid #f0a4b3; | |
| border-radius: 8px; | |
| background: #fff1f3; | |
| color: var(--failed); | |
| padding: 10px 12px; | |
| white-space: pre-wrap; | |
| }} | |
| @media (max-width: 720px) {{ | |
| main {{ width: min(100% - 20px, 1120px); padding-top: 20px; }} | |
| .topbar, .result-card > header {{ display: block; }} | |
| .badge {{ display: inline-block; margin-top: 8px; }} | |
| }} | |
| </style> | |
| </head> | |
| <body> | |
| <main> | |
| <header class="topbar"> | |
| <div> | |
| <p class="eyebrow">Real Browser QA Report</p> | |
| <h1>{qa_html(run.get("id") or "QA run")}</h1> | |
| <p class="muted">Generated from backend QA run data. Screenshots are served as report assets, not embedded base64.</p> | |
| </div> | |
| <span class="badge status-{status}">{qa_html(status)}</span> | |
| </header> | |
| <section class="summary"> | |
| {qa_metric_html("done", f"{qa_int(run.get('completed'))}/{qa_int(run.get('total'))}")} | |
| {qa_metric_html("passed", str(qa_int(run.get("passed"))))} | |
| {qa_metric_html("failed", str(qa_int(run.get("failed"))))} | |
| {qa_metric_html("updated", str(run.get("updatedAt") or "unknown"))} | |
| </section> | |
| {error_html} | |
| {result_cards} | |
| </main> | |
| </body> | |
| </html> | |
| """ | |
| def run_qa_task(run_id: str, image_keys: list[str], tiles: list[str], descriptions: dict[str, str], use_gemini: bool, backend_url: str): | |
| run = read_qa_run(run_id) | |
| run_root = qa_run_root(run_id) | |
| log_path = run_root / "runner.log" | |
| report_path = run_root / "results" / "real-qa.json" | |
| summary_path = run_root / "qa-summary.md" | |
| try: | |
| if not QA_FRONTEND_URL: | |
| raise RuntimeError("QA_FRONTEND_URL must point to the hosted frontend.") | |
| if not QA_FRONTEND_DIR.exists(): | |
| raise RuntimeError(f"QA_FRONTEND_DIR does not exist: {QA_FRONTEND_DIR}") | |
| if not (QA_FRONTEND_DIR / "package.json").exists(): | |
| raise RuntimeError(f"QA_FRONTEND_DIR must point to frontend/viz2d-demo: {QA_FRONTEND_DIR}") | |
| run.update({"status": "running", "startedAt": qa_iso_now()}) | |
| write_qa_run(run) | |
| downloaded_images = download_qa_images(run_id, image_keys) | |
| scenarios = build_qa_scenarios(downloaded_images, tiles, descriptions) | |
| (run_root / "scenarios.json").write_text(json.dumps(scenarios, indent=2)) | |
| env = qa_runner_env(run_root, scenarios, use_gemini, backend_url) | |
| with log_path.open("w") as log_file: | |
| exit_code = run_playwright_qa(run_id, run_root, env, log_file) | |
| log_file.write(f"\nqa:e2e exit code: {exit_code}\n") | |
| subprocess.run( | |
| ["npm", "run", "qa:report", "--", str(report_path), str(summary_path)], | |
| cwd=QA_FRONTEND_DIR, | |
| env=env, | |
| stderr=subprocess.STDOUT, | |
| stdout=log_file, | |
| timeout=300, | |
| ) | |
| run = read_qa_run(run_id) | |
| finalize_qa_run_from_report(run, report_path) | |
| if not report_path.exists(): | |
| run["error"] = "Playwright did not create a JSON report. Check runner.log." | |
| run["reportUrl"] = f"/qa/runs/{run_id}/report" | |
| write_qa_run(run) | |
| except subprocess.TimeoutExpired: | |
| run = read_qa_run(run_id) | |
| run.update({"status": "failed", "error": "QA run timed out.", "reportUrl": f"/qa/runs/{run_id}/report"}) | |
| write_qa_run(run) | |
| except Exception as exc: | |
| run = read_qa_run(run_id) | |
| run.update({"status": "failed", "error": str(exc), "reportUrl": f"/qa/runs/{run_id}/report"}) | |
| write_qa_run(run) | |
| def run_conversion_task(job_id: str, upload_path: Path): | |
| debug = PipelineDebugCapture(job_id) if DEBUG_ARTIFACTS_ENABLED else None | |
| try: | |
| t_start = time.perf_counter() | |
| image_bytes = upload_path.read_bytes() | |
| bundle = build_segmentation_bundle(image_bytes, debug=debug) | |
| binary_bundle = build_manifest_segmentation_bundle( | |
| bundle, | |
| job_id, | |
| image_url=f"/uploads/{upload_path.name}", | |
| include_image=False, | |
| ) | |
| (JOB_DIR / f"{job_id}.vizbundle").write_bytes(binary_bundle) | |
| job = read_job(job_id) | |
| job["status"] = "COMPLETED" | |
| job["modelName"] = bundle.get("modelName") | |
| job["models"] = bundle.get("models") | |
| job["outputFormat"] = "split-vizbundle" | |
| write_job(job) | |
| print(f"[TIMING] Background conversion task for job {job_id} took {time.perf_counter() - t_start:.3f} seconds", flush=True) | |
| except Exception as exc: | |
| print(f"Background conversion failed: {exc}", flush=True) | |
| if debug is not None: | |
| debug.fail(exc) | |
| try: | |
| job = read_job(job_id) | |
| job["status"] = "FAILED" | |
| job["error"] = str(exc) | |
| write_job(job) | |
| except Exception: | |
| pass | |
| def build_gemini_tile_refinement_prompt(tile_name: str | None = None) -> str: | |
| tile_hint = f'\nTile/material label hint: "{tile_name}".' if tile_name else "" | |
| return f"""You are performing a conservative photorealistic finishing pass on an interior tile visualizer render. | |
| Use the supplied image as the source of truth. The tile has already been selected, laid out, scaled, rotated, grouted, and composited by the renderer. | |
| Primary goal: | |
| Make the already-applied tile surface look more naturally integrated into the room by refining missing realism: local reflections, gloss response, tiny contact shadows, ambient occlusion, edge blending, perspective-light consistency, color bleed, highlight rolloff, and surface micro-detail. | |
| Hard preservation rules: | |
| - Do not replace, recreate, reinterpret, or redesign the tile. | |
| - Do not change the tile pattern, print, veining, motif, color palette, layout, scale, grout spacing, grout color, rotation, perspective, or tile boundaries. | |
| - Do not invent a different material, add new decorative features, or remove existing tile details. | |
| - Do not alter furniture, walls, fixtures, objects, room architecture, camera angle, crop, image size, or composition. | |
| - Do not smooth away the tile's existing texture or make repeated tiles look like a different product. | |
| - If an improvement would require changing the tile design, leave that area unchanged. | |
| Allowed refinements only: | |
| - Subtle physically plausible reflections and specular highlights that follow the existing room lighting. | |
| - More natural shadowing where the tiled surface meets walls, cabinets, furniture legs, baseboards, or other occluding objects. | |
| - Better blending at tile-mask edges while preserving the existing boundary. | |
| - Gentle local contrast, roughness, and micro-surface detail so the tile feels photographed in the room. | |
| - Correct small rendering artifacts, halos, missing reflection cues, or flat-looking areas without changing the design. | |
| Return only the refined image. Keep the same aspect ratio and composition. The result should look like the same rendered tile installation, just more realistic and finished.{tile_hint}""" | |
| def decode_inline_image_payload(image_value: str) -> tuple[bytes, str]: | |
| if not image_value: | |
| raise HTTPException(status_code=400, detail="Image is required.") | |
| mime_type = "image/png" | |
| encoded = image_value | |
| if image_value.startswith("data:"): | |
| try: | |
| header, encoded = image_value.split(",", 1) | |
| except ValueError as exc: | |
| raise HTTPException(status_code=400, detail="Image data URL is malformed.") from exc | |
| mime_type = header[5:].split(";", 1)[0] or mime_type | |
| try: | |
| image_bytes = base64.b64decode(encoded, validate=True) | |
| except Exception as exc: | |
| raise HTTPException(status_code=400, detail="Image must be base64 encoded.") from exc | |
| if not image_bytes: | |
| raise HTTPException(status_code=400, detail="Image must not be empty.") | |
| try: | |
| with Image.open(io.BytesIO(image_bytes)) as uploaded: | |
| uploaded.verify() | |
| detected_mime = Image.MIME.get(uploaded.format) | |
| if detected_mime: | |
| mime_type = detected_mime | |
| except Exception as exc: | |
| raise HTTPException(status_code=400, detail="Image must be a readable PNG, JPEG, or WebP.") from exc | |
| if mime_type not in {"image/png", "image/jpeg", "image/webp"}: | |
| raise HTTPException(status_code=400, detail="Image must be a PNG, JPEG, or WebP.") | |
| return image_bytes, mime_type | |
| def call_gemini_image_refinement(image_bytes: bytes, mime_type: str, prompt: str) -> tuple[str, str]: | |
| if not GEMINI_API_KEY: | |
| raise HTTPException(status_code=503, detail="Gemini refinement is not configured. Set GEMINI_API_KEY.") | |
| if not GEMINI_IMAGE_MODEL: | |
| raise HTTPException(status_code=503, detail="Gemini image model is not configured.") | |
| request_body = { | |
| "contents": [{ | |
| "parts": [ | |
| {"text": prompt}, | |
| { | |
| "inline_data": { | |
| "mime_type": mime_type, | |
| "data": base64.b64encode(image_bytes).decode("ascii"), | |
| } | |
| }, | |
| ] | |
| }] | |
| } | |
| encoded_model = urllib.parse.quote(GEMINI_IMAGE_MODEL, safe="") | |
| url = f"{GEMINI_API_BASE_URL}/{encoded_model}:generateContent" | |
| request = urllib.request.Request( | |
| url, | |
| data=json.dumps(request_body).encode("utf-8"), | |
| headers={ | |
| "Content-Type": "application/json", | |
| "x-goog-api-key": GEMINI_API_KEY, | |
| }, | |
| method="POST", | |
| ) | |
| try: | |
| with urllib.request.urlopen(request, timeout=GEMINI_TIMEOUT_SECONDS) as response: | |
| response_body = response.read().decode("utf-8") | |
| except urllib.error.HTTPError as exc: | |
| detail = exc.read().decode("utf-8", errors="replace")[:1000] | |
| raise HTTPException(status_code=502, detail=f"Gemini refinement failed ({exc.code}): {detail}") from exc | |
| except urllib.error.URLError as exc: | |
| raise HTTPException(status_code=502, detail=f"Gemini refinement request failed: {exc.reason}") from exc | |
| try: | |
| data = json.loads(response_body) | |
| except json.JSONDecodeError as exc: | |
| raise HTTPException(status_code=502, detail="Gemini returned invalid JSON.") from exc | |
| for candidate in data.get("candidates", []): | |
| content = candidate.get("content") or {} | |
| for part in content.get("parts", []): | |
| inline_data = part.get("inlineData") or part.get("inline_data") | |
| if not inline_data: | |
| continue | |
| output_b64 = inline_data.get("data") | |
| if output_b64: | |
| output_mime = inline_data.get("mimeType") or inline_data.get("mime_type") or "image/png" | |
| return output_b64, output_mime | |
| raise HTTPException(status_code=502, detail="Gemini did not return a refined image.") | |
| async def refine_tile_render_with_gemini(payload: dict): | |
| image_value = payload.get("image") if isinstance(payload, dict) else None | |
| if not isinstance(image_value, str): | |
| raise HTTPException(status_code=400, detail="Image is required.") | |
| tile_name = payload.get("tileName") if isinstance(payload, dict) else None | |
| if not isinstance(tile_name, str): | |
| tile_name = None | |
| image_bytes, mime_type = decode_inline_image_payload(image_value) | |
| prompt = build_gemini_tile_refinement_prompt(tile_name) | |
| output_b64, output_mime = await asyncio.to_thread( | |
| call_gemini_image_refinement, | |
| image_bytes, | |
| mime_type, | |
| prompt, | |
| ) | |
| return { | |
| "image": f"data:{output_mime};base64,{output_b64}", | |
| "model": GEMINI_IMAGE_MODEL, | |
| } | |
| async def prepare_material_maps( | |
| file: UploadFile = File(...), | |
| base_roughness: float = Form(0.42), | |
| ): | |
| if file.content_type and not file.content_type.startswith("image/"): | |
| raise HTTPException(status_code=400, detail="Upload must be a tile image.") | |
| contents = await file.read() | |
| if not contents: | |
| raise HTTPException(status_code=400, detail="Upload must not be empty.") | |
| return build_tile_material_package(contents, base_roughness) | |
| async def qa_images( | |
| range: str = Query("week"), | |
| from_value: str | None = Query(None, alias="from"), | |
| to: str | None = Query(None), | |
| max_items: int = Query(100, ge=1, le=500), | |
| ): | |
| return {"images": list_qa_spaces_images(range, from_value, to, min(max_items, QA_MAX_IMAGES))} | |
| async def create_qa_run( | |
| request: Request, | |
| background_tasks: BackgroundTasks, | |
| payload: dict = Body(...), | |
| ): | |
| raw_image_keys = payload.get("imageKeys") | |
| if not isinstance(raw_image_keys, list): | |
| raise HTTPException(status_code=400, detail="imageKeys must be a non-empty array.") | |
| image_keys = [str(key).strip() for key in raw_image_keys if isinstance(key, str) and str(key).strip()] | |
| if not image_keys: | |
| raise HTTPException(status_code=400, detail="Select at least one image.") | |
| image_keys = dedupe_qa_image_keys(image_keys) | |
| if not image_keys: | |
| raise HTTPException(status_code=400, detail="All selected images were duplicates.") | |
| if len(image_keys) > QA_MAX_IMAGES: | |
| raise HTTPException(status_code=400, detail=f"Select at most {QA_MAX_IMAGES} images.") | |
| raw_tiles = payload.get("tiles") | |
| if raw_tiles is None: | |
| raw_tiles = list(QA_TILE_DESCRIPTIONS) | |
| if not isinstance(raw_tiles, list): | |
| raise HTTPException(status_code=400, detail="tiles must be an array.") | |
| tiles = [] | |
| for tile in raw_tiles: | |
| if not isinstance(tile, str) or not tile.strip(): | |
| continue | |
| normalized = normalize_qa_tile(tile) | |
| if normalized not in tiles: | |
| tiles.append(normalized) | |
| if not tiles: | |
| raise HTTPException(status_code=400, detail="Select at least one tile.") | |
| total = len(image_keys) * len(tiles) | |
| if total > QA_MAX_TESTS: | |
| raise HTTPException(status_code=400, detail=f"QA matrix is too large. Max tests: {QA_MAX_TESTS}.") | |
| run_id = f"qa_{uuid.uuid4().hex}" | |
| use_gemini = bool(payload.get("useGemini", True)) | |
| descriptions = parse_tile_descriptions(payload.get("tileDescriptions")) | |
| backend_url = QA_BACKEND_URL or str(request.base_url).rstrip("/") | |
| run = { | |
| "id": run_id, | |
| "status": "queued", | |
| "createdAt": qa_iso_now(), | |
| "updatedAt": qa_iso_now(), | |
| "total": total, | |
| "completed": 0, | |
| "passed": 0, | |
| "failed": 0, | |
| "imageKeys": image_keys, | |
| "tiles": tiles, | |
| "useGemini": use_gemini, | |
| "reportUrl": f"/qa/runs/{run_id}/report", | |
| "results": [], | |
| } | |
| write_qa_run(run) | |
| background_tasks.add_task(run_qa_task, run_id, image_keys, tiles, descriptions, use_gemini, backend_url) | |
| return run | |
| async def qa_run_status(run_id: str): | |
| return read_qa_run(run_id) | |
| async def qa_run_events(run_id: str, request: Request): | |
| read_qa_run(run_id) | |
| async def event_stream(): | |
| last_payload = "" | |
| last_keepalive = time.monotonic() | |
| while True: | |
| if await request.is_disconnected(): | |
| break | |
| run = read_qa_run(run_id) | |
| payload = json.dumps(run, sort_keys=True) | |
| if payload != last_payload: | |
| last_payload = payload | |
| yield qa_sse_event("run", run) | |
| if run.get("status") not in {"queued", "running"}: | |
| break | |
| elif time.monotonic() - last_keepalive >= 15: | |
| last_keepalive = time.monotonic() | |
| yield ": keepalive\n\n" | |
| await asyncio.sleep(1) | |
| return StreamingResponse( | |
| event_stream(), | |
| media_type="text/event-stream", | |
| headers={ | |
| "Cache-Control": "no-cache, no-transform", | |
| "Connection": "keep-alive", | |
| "X-Accel-Buffering": "no", | |
| }, | |
| ) | |
| async def qa_run_report(run_id: str): | |
| run = read_qa_run(run_id) | |
| return Response(content=build_qa_report_html(run), media_type="text/html", headers={"Cache-Control": "no-store"}) | |
| async def qa_run_markdown_report(run_id: str): | |
| root = qa_run_root(run_id) | |
| report_path = root / "qa-summary.md" | |
| if not report_path.exists(): | |
| run = read_qa_run(run_id) | |
| content = "\n".join([ | |
| "# Real Browser QA Summary", | |
| "", | |
| f"Status: {run.get('status', 'unknown')}", | |
| "", | |
| run.get("error") or "The QA summary has not been generated yet.", | |
| "", | |
| ]) | |
| return Response(content=content, media_type="text/markdown", headers={"Cache-Control": "no-store"}) | |
| return Response(content=report_path.read_text(), media_type="text/markdown", headers={"Cache-Control": "no-store"}) | |
| async def qa_run_asset(run_id: str, asset_path: str): | |
| root = qa_run_root(run_id).resolve() | |
| target = (root / asset_path).resolve() | |
| if root != target and root not in target.parents: | |
| raise HTTPException(status_code=400, detail="Invalid QA asset path.") | |
| if not target.exists() or not target.is_file(): | |
| raise HTTPException(status_code=404, detail="QA asset not found.") | |
| media_type = mimetypes.guess_type(target.name)[0] or "application/octet-stream" | |
| return Response(content=target.read_bytes(), media_type=media_type, headers={"Cache-Control": "private, max-age=3600"}) | |
| async def convert_to_viz2d(background_tasks: BackgroundTasks, file: UploadFile = File(...)): | |
| if file.content_type and not file.content_type.startswith("image/"): | |
| raise HTTPException(status_code=400, detail="Upload must be a JPG or PNG image.") | |
| contents = await file.read() | |
| if not contents: | |
| raise HTTPException(status_code=400, detail="Upload must not be empty.") | |
| normalized_contents = normalize_room_upload(contents) | |
| job_id = uuid.uuid4().hex | |
| original_filename = file.filename | |
| content_type = "image/jpeg" | |
| ext = ".jpg" | |
| upload_path = UPLOAD_DIR / f"{job_id}{ext}" | |
| upload_path.write_bytes(normalized_contents) | |
| job = { | |
| "id": job_id, | |
| "status": "PROCESSING", | |
| "inputUrl": f"/uploads/{upload_path.name}", | |
| "outputUrl": f"/viz2d/jobs/{job_id}/file", | |
| "debugUrl": f"/viz2d/jobs/{job_id}/debug" if DEBUG_ARTIFACTS_ENABLED else None, | |
| } | |
| write_job(job) | |
| background_tasks.add_task(run_conversion_task, job_id, upload_path) | |
| background_tasks.add_task(upload_input_image_to_spaces, job_id, upload_path, original_filename, content_type) | |
| return job | |
| async def viz2d_job_status(job_id: str): | |
| return read_job(job_id) | |
| async def viz2d_job_debug(job_id: str): | |
| read_job(job_id) | |
| if not DEBUG_ARTIFACTS_ENABLED: | |
| raise HTTPException(status_code=404, detail="Debug artifacts are disabled on this server.") | |
| manifest_path = JOB_DIR / f"{job_id}.debug" / "manifest.json" | |
| if not manifest_path.exists(): | |
| raise HTTPException(status_code=404, detail="Debug artifacts are not available yet.") | |
| return Response( | |
| content=manifest_path.read_bytes(), | |
| media_type="application/json", | |
| headers={"Cache-Control": "no-store"}, | |
| ) | |
| async def viz2d_job_debug_asset(job_id: str, asset_name: str): | |
| read_job(job_id) | |
| if Path(asset_name).name != asset_name: | |
| raise HTTPException(status_code=400, detail="Invalid debug asset name.") | |
| asset_path = JOB_DIR / f"{job_id}.debug" / asset_name | |
| if not asset_path.exists() or not asset_path.is_file(): | |
| raise HTTPException(status_code=404, detail="Debug asset not found.") | |
| media_type = "image/jpeg" if asset_path.suffix.lower() in {".jpg", ".jpeg"} else "application/octet-stream" | |
| return Response( | |
| content=asset_path.read_bytes(), | |
| media_type=media_type, | |
| headers={"Cache-Control": "private, max-age=3600"}, | |
| ) | |
| async def viz2d_job_file(job_id: str): | |
| job = read_job(job_id) | |
| if job.get("status") != "COMPLETED": | |
| raise HTTPException(status_code=409, detail="Job is not completed yet.") | |
| binary_bundle_path = JOB_DIR / f"{job_id}.vizbundle" | |
| if not binary_bundle_path.exists(): | |
| raise HTTPException(status_code=404, detail="Job output not found.") | |
| return Response( | |
| content=binary_bundle_path.read_bytes(), | |
| media_type="application/zip", | |
| headers={"Content-Disposition": 'attachment; filename="visualizer.vizbundle"'}, | |
| ) | |
| async def viz2d_job_asset(job_id: str, asset_name: str): | |
| read_job(job_id) | |
| if Path(asset_name).name != asset_name: | |
| raise HTTPException(status_code=400, detail="Invalid asset name.") | |
| asset_path = JOB_DIR / f"{job_id}.assets" / asset_name | |
| if not asset_path.exists() or not asset_path.is_file(): | |
| raise HTTPException(status_code=404, detail="Job asset not found.") | |
| return Response( | |
| content=asset_path.read_bytes(), | |
| media_type="application/octet-stream", | |
| headers={"Cache-Control": "private, max-age=3600"}, | |
| ) | |
| def build_oneformer_polygon_response( | |
| contents: bytes, | |
| *, | |
| threshold: float = 0.5, | |
| mask_threshold: float = 0.5, | |
| min_area: float = 32.0, | |
| simplify_epsilon: float = 2.0, | |
| ) -> dict: | |
| img, img_np = load_uploaded_rgb_image(contents) | |
| h, w = img_np.shape[:2] | |
| threshold = float(np.clip(threshold, 0.0, 1.0)) | |
| mask_threshold = float(np.clip(mask_threshold, 0.0, 1.0)) | |
| min_area = max(0.0, float(min_area)) | |
| simplify_epsilon = max(0.0, float(simplify_epsilon)) | |
| try: | |
| processor, model = _load_oneformer_polygon_model() | |
| except Exception as exc: | |
| raise HTTPException(status_code=503, detail=f"OneFormer model is unavailable: {exc}") from exc | |
| inputs = processor( | |
| images=img, | |
| task_inputs=["panoptic"], | |
| return_tensors="pt", | |
| ).to(device) | |
| with torch.no_grad(): | |
| outputs = model(**inputs) | |
| result = processor.post_process_panoptic_segmentation( | |
| outputs, | |
| threshold=threshold, | |
| mask_threshold=mask_threshold, | |
| target_sizes=[(h, w)], | |
| )[0] | |
| segmentation = result.get("segmentation") | |
| if segmentation is None: | |
| return { | |
| "model": ONEFORMER_MODEL_NAME, | |
| "task": "panoptic", | |
| "width": int(w), | |
| "height": int(h), | |
| "segments": [], | |
| } | |
| if isinstance(segmentation, torch.Tensor): | |
| segment_map = segmentation.cpu().numpy() | |
| else: | |
| segment_map = np.asarray(segmentation) | |
| segments = [] | |
| for segment_info in result.get("segments_info", []): | |
| segment_id = int(segment_info["id"]) | |
| label_id = int(segment_info["label_id"]) | |
| mask = segment_map == segment_id | |
| polygons = mask_to_polygons(mask, min_area, simplify_epsilon) | |
| if not polygons: | |
| continue | |
| score = segment_info.get("score") | |
| segments.append({ | |
| "id": segment_id, | |
| "labelId": label_id, | |
| "label": class_name_for_id(label_id), | |
| "confidence": float(score) if score is not None else None, | |
| "wasFused": bool(segment_info.get("was_fused", False)), | |
| "pixelArea": int(mask.sum()), | |
| "polygons": polygons, | |
| }) | |
| segments.sort(key=lambda item: item["pixelArea"], reverse=True) | |
| return { | |
| "model": ONEFORMER_MODEL_NAME, | |
| "task": "panoptic", | |
| "width": int(w), | |
| "height": int(h), | |
| "threshold": threshold, | |
| "maskThreshold": mask_threshold, | |
| "minArea": min_area, | |
| "simplifyEpsilon": simplify_epsilon, | |
| "segmentCount": len(segments), | |
| "segments": segments, | |
| } | |
| def encode_png_data_url(image_bgr: np.ndarray) -> str: | |
| ok, encoded = cv2.imencode(".png", image_bgr) | |
| if not ok: | |
| raise HTTPException(status_code=500, detail="Could not encode overlay image.") | |
| payload = base64.b64encode(encoded.tobytes()).decode("ascii") | |
| return f"data:image/png;base64,{payload}" | |
| def floor_direction_response(result, segmentation) -> dict: | |
| overlay_image_base64 = None | |
| if result.overlay_bgr is not None: | |
| overlay_image_base64 = encode_png_data_url(result.overlay_bgr) | |
| return { | |
| "angle_degrees": result.angle_degrees, | |
| "render_angle_degrees": result.render_angle_degrees, | |
| "secondary_angle_degrees": result.secondary_angle_degrees, | |
| "direction_label": result.direction_label, | |
| "confidence": result.confidence, | |
| "segmentation": { | |
| "label": segmentation.label, | |
| "confidence": segmentation.confidence, | |
| "polygon_count": len(segmentation.polygons), | |
| "floor_area_ratio": result.floor_area_ratio, | |
| "source_path": segmentation.source_path, | |
| "raw_keys": segmentation.raw_keys or [], | |
| }, | |
| "line_count": result.line_count, | |
| "dominant_line_count": result.dominant_line_count, | |
| "grid_pattern_detected": result.grid_pattern_detected, | |
| "dominant_lines": [ | |
| { | |
| "x1": line.x1, | |
| "y1": line.y1, | |
| "x2": line.x2, | |
| "y2": line.y2, | |
| "angle_degrees": line.angle_degrees, | |
| "length": line.length, | |
| } | |
| for line in result.dominant_lines | |
| ], | |
| "warnings": result.warnings, | |
| "overlay_image_base64": overlay_image_base64, | |
| } | |
| def floor_direction_metadata(result, segmentation) -> dict: | |
| return { | |
| "source": "backend-floor-direction", | |
| "angleDegrees": result.angle_degrees, | |
| "renderAngleDegrees": result.render_angle_degrees, | |
| "secondaryAngleDegrees": result.secondary_angle_degrees, | |
| "directionLabel": result.direction_label, | |
| "confidence": result.confidence, | |
| "floorAreaRatio": result.floor_area_ratio, | |
| "lineCount": result.line_count, | |
| "dominantLineCount": result.dominant_line_count, | |
| "gridPatternDetected": result.grid_pattern_detected, | |
| "warnings": result.warnings, | |
| "dominantLines": [ | |
| { | |
| "x1": line.x1, | |
| "y1": line.y1, | |
| "x2": line.x2, | |
| "y2": line.y2, | |
| "angleDegrees": line.angle_degrees, | |
| "length": line.length, | |
| } | |
| for line in result.dominant_lines | |
| ], | |
| "segmentation": { | |
| "label": segmentation.label, | |
| "confidence": segmentation.confidence, | |
| "polygonCount": len(segmentation.polygons), | |
| "sourcePath": segmentation.source_path, | |
| }, | |
| } | |
| def floor_direction_geometry_inputs( | |
| surface_mask: np.ndarray, | |
| geometry: dict | None, | |
| plane: dict | None, | |
| surface_mapping: dict | None, | |
| *, | |
| use_surface_uv: bool, | |
| ) -> dict: | |
| direction_maps = surface_mapping | |
| if direction_maps is None: | |
| raw_maps = geometry_point_maps(surface_mask, geometry) | |
| if raw_maps is not None: | |
| direction_maps = { | |
| "normalMap": raw_maps["normals"], | |
| "validMask": raw_maps["validMask"], | |
| "intrinsics": raw_maps["intrinsics"], | |
| "planes": [], | |
| } | |
| normal_map = direction_maps.get("normalMap") if direction_maps else None | |
| valid_mask = direction_maps.get("validMask") if direction_maps else None | |
| intrinsics = direction_maps.get("intrinsics") if direction_maps else None | |
| surface_planes = direction_maps.get("planes") or [] if direction_maps else [] | |
| dominant_floor_plane = None | |
| if surface_planes: | |
| floor_planes = [item for item in surface_planes if item.get("isFloor")] | |
| candidates = floor_planes or surface_planes | |
| dominant_floor_plane = max( | |
| candidates, | |
| key=lambda item: int(item.get("assignedCount") or item.get("inlierCount") or 0), | |
| ) | |
| floor_normal = None | |
| render_u_axis = None | |
| render_v_axis = None | |
| if use_surface_uv and dominant_floor_plane is not None: | |
| floor_normal = dominant_floor_plane.get("normal") | |
| render_u_axis = dominant_floor_plane.get("uAxis") | |
| render_v_axis = dominant_floor_plane.get("vAxis") | |
| elif plane and plane.get("planeNormal"): | |
| floor_normal = plane.get("planeNormal") | |
| elif dominant_floor_plane is not None: | |
| floor_normal = dominant_floor_plane.get("normal") | |
| if floor_normal is None and normal_map is not None and normal_map.shape[:2] == surface_mask.shape: | |
| lengths = np.linalg.norm(normal_map, axis=2) | |
| valid_floor = ( | |
| (surface_mask > 0) | |
| & np.isfinite(normal_map).all(axis=2) | |
| & (lengths > 1e-4) | |
| ) | |
| if valid_mask is not None and valid_mask.shape == surface_mask.shape: | |
| valid_floor &= valid_mask.astype(bool) | |
| floor_normals = normal_map[valid_floor] | |
| if len(floor_normals) >= 200: | |
| floor_normals = floor_normals / np.linalg.norm(floor_normals, axis=1, keepdims=True) | |
| reference = np.median(floor_normals, axis=0) | |
| if reference[1] < 0: | |
| reference = -reference | |
| floor_normal = reference.tolist() | |
| if intrinsics is None and plane and plane.get("cameraIntrinsics"): | |
| intrinsics = np.asarray(plane["cameraIntrinsics"], dtype=np.float32).reshape(3, 3) | |
| if intrinsics is None: | |
| height, width = surface_mask.shape[:2] | |
| intrinsics = default_camera_intrinsics(width, height) | |
| return { | |
| "normalsMap": normal_map, | |
| "validMask": valid_mask, | |
| "intrinsics": intrinsics, | |
| "floorNormal": floor_normal, | |
| "renderUAxis": render_u_axis, | |
| "renderVAxis": render_v_axis, | |
| } | |
| def analyze_floor_direction_for_surface_mask( | |
| img_np: np.ndarray, | |
| surface_mask: np.ndarray, | |
| *, | |
| render_transform: list[float] | np.ndarray | None = None, | |
| surface_uv: np.ndarray | None = None, | |
| surface_indices: np.ndarray | None = None, | |
| surface_plane_ids: np.ndarray | None = None, | |
| ) -> dict: | |
| h, w = img_np.shape[:2] | |
| polygons = mask_to_polygons( | |
| surface_mask, | |
| min_area=max(32.0, float(w * h) * 0.001), | |
| simplify_epsilon=2.0, | |
| ) | |
| floor_segmentation = parse_floor_segmentation({ | |
| "label": "floor", | |
| "confidence": 1.0, | |
| "width": int(w), | |
| "height": int(h), | |
| "polygons": polygons, | |
| }) | |
| result = analyze_floor_direction_from_segmentation( | |
| cv2.cvtColor(img_np, cv2.COLOR_RGB2BGR), | |
| floor_segmentation, | |
| include_overlay=False, | |
| render_transform=render_transform, | |
| surface_uv=surface_uv, | |
| surface_indices=surface_indices, | |
| surface_plane_ids=surface_plane_ids, | |
| ) | |
| return floor_direction_metadata(result, floor_segmentation) | |
| async def oneformer_polygons( | |
| file: UploadFile = File(...), | |
| threshold: float = Form(0.5), | |
| mask_threshold: float = Form(0.5), | |
| min_area: float = Form(32.0), | |
| simplify_epsilon: float = Form(2.0), | |
| ): | |
| contents = await file.read() | |
| return build_oneformer_polygon_response( | |
| contents, | |
| threshold=threshold, | |
| mask_threshold=mask_threshold, | |
| min_area=min_area, | |
| simplify_epsilon=simplify_epsilon, | |
| ) | |
| async def analyze_floor_direction_endpoint( | |
| file: UploadFile = File(...), | |
| include_overlay: bool = Form(False), | |
| threshold: float = Form(0.5), | |
| mask_threshold: float = Form(0.5), | |
| min_area: float = Form(32.0), | |
| simplify_epsilon: float = Form(2.0), | |
| ): | |
| contents = await file.read() | |
| if not contents: | |
| raise HTTPException(status_code=400, detail="Upload a non-empty image file in the 'file' field.") | |
| _img, img_np = load_uploaded_rgb_image(contents) | |
| image_bgr = cv2.cvtColor(img_np, cv2.COLOR_RGB2BGR) | |
| segmentation_payload = build_oneformer_polygon_response( | |
| contents, | |
| threshold=threshold, | |
| mask_threshold=mask_threshold, | |
| min_area=min_area, | |
| simplify_epsilon=simplify_epsilon, | |
| ) | |
| floor_segmentation = parse_floor_segmentation(segmentation_payload) | |
| result = analyze_floor_direction_from_segmentation( | |
| image_bgr, | |
| floor_segmentation, | |
| include_overlay=include_overlay, | |
| ) | |
| return floor_direction_response(result, floor_segmentation) | |
| async def segment(file: UploadFile = File(...)): | |
| contents = await file.read() | |
| return build_segmentation_bundle(contents) | |
| app.mount("/materials", StaticFiles(directory=MATERIAL_DIR), name="materials") | |
| if __name__ == "__main__": | |
| import uvicorn | |
| uvicorn.run(app, host="0.0.0.0", port=8002) | |