""" Validation layer. Every raw output from a model (VLM classification, caption, LLM JSON) passes through here before it is allowed to become a schemas.py object. This is where "never blindly trust malformed model output" (section 8) is enforced in one place instead of scattered ad-hoc checks through the agent code. """ from __future__ import annotations import json import re class ValidationError(Exception): pass def safe_parse_llm_json(raw_text: str, required_keys: list[str]) -> dict: """ LLMs (especially small ones) sometimes wrap JSON in prose or markdown fences, or emit near-JSON with trailing commas. This extracts the first plausible JSON object and validates required keys exist, raising ValidationError (never a raw crash) on failure so callers can fall back. """ if not raw_text or not raw_text.strip(): raise ValidationError("empty model output") # Strip markdown code fences if present text = re.sub(r"```(?:json)?", "", raw_text).strip() # Find the first {...} block match = re.search(r"\{.*\}", text, re.DOTALL) if not match: raise ValidationError(f"no JSON object found in output: {raw_text[:200]!r}") candidate = match.group(0) try: data = json.loads(candidate) except json.JSONDecodeError as e: raise ValidationError(f"malformed JSON: {e}") from e missing = [k for k in required_keys if k not in data] if missing: raise ValidationError(f"missing required keys {missing} in {data}") return data def clamp(value: float, lo: float, hi: float) -> float: return max(lo, min(hi, value)) def coerce_int(value, default: int = 0, lo: int = 0, hi: int = 999) -> int: try: v = int(round(float(value))) except (TypeError, ValueError): return default return max(lo, min(hi, v)) def coerce_float(value, default: float = 0.0, lo: float = 0.0, hi: float = 10.0) -> float: try: v = float(value) except (TypeError, ValueError): return default return max(lo, min(hi, v)) def coerce_need_types(value) -> list[str]: """Normalizes need-type lists from LLM output against the fixed vocabulary.""" from core.schemas import VALID_NEED_TYPES if isinstance(value, str): value = [value] if not isinstance(value, list): return [] out = [] for v in value: v_norm = str(v).strip().lower() if v_norm in VALID_NEED_TYPES: out.append(v_norm) return list(dict.fromkeys(out)) # dedupe, preserve order def validate_image_file(path: str, max_bytes: int = 15 * 1024 * 1024) -> tuple[bool, str]: """Returns (is_valid, error_message). Never raises — callers check the bool.""" import os if not path or not os.path.exists(path): return False, f"image file not found: {path}" size = os.path.getsize(path) if size == 0: return False, "image file is empty (0 bytes)" if size > max_bytes: return False, f"image file too large ({size} bytes > {max_bytes} limit)" try: from PIL import Image with Image.open(path) as img: img.verify() except Exception as e: return False, f"invalid or corrupt image: {e}" return True, "" def validate_report_text(text: str, max_chars: int = 4000) -> tuple[str, str]: """ Returns (cleaned_text, warning). Handles empty and very-long reports per section 25's failure-handling requirement instead of crashing or silently truncating without telling anyone. """ if text is None: return "", "no report text provided" text = text.strip() if not text: return "", "empty report text" if len(text) > max_chars: return text[:max_chars], f"report truncated from {len(text)} to {max_chars} characters" return text, ""