Spaces:
Running on Zero
Running on Zero
| """ | |
| Validation layer. | |
| Every raw output from a model (VLM classification, caption, LLM JSON) passes | |
| through here before it is allowed to become a schemas.py object. This is | |
| where "never blindly trust malformed model output" (section 8) is enforced | |
| in one place instead of scattered ad-hoc checks through the agent code. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import re | |
| class ValidationError(Exception): | |
| pass | |
| def safe_parse_llm_json(raw_text: str, required_keys: list[str]) -> dict: | |
| """ | |
| LLMs (especially small ones) sometimes wrap JSON in prose or markdown | |
| fences, or emit near-JSON with trailing commas. This extracts the first | |
| plausible JSON object and validates required keys exist, raising | |
| ValidationError (never a raw crash) on failure so callers can fall back. | |
| """ | |
| if not raw_text or not raw_text.strip(): | |
| raise ValidationError("empty model output") | |
| # Strip markdown code fences if present | |
| text = re.sub(r"```(?:json)?", "", raw_text).strip() | |
| # Find the first {...} block | |
| match = re.search(r"\{.*\}", text, re.DOTALL) | |
| if not match: | |
| raise ValidationError(f"no JSON object found in output: {raw_text[:200]!r}") | |
| candidate = match.group(0) | |
| try: | |
| data = json.loads(candidate) | |
| except json.JSONDecodeError as e: | |
| raise ValidationError(f"malformed JSON: {e}") from e | |
| missing = [k for k in required_keys if k not in data] | |
| if missing: | |
| raise ValidationError(f"missing required keys {missing} in {data}") | |
| return data | |
| def clamp(value: float, lo: float, hi: float) -> float: | |
| return max(lo, min(hi, value)) | |
| def coerce_int(value, default: int = 0, lo: int = 0, hi: int = 999) -> int: | |
| try: | |
| v = int(round(float(value))) | |
| except (TypeError, ValueError): | |
| return default | |
| return max(lo, min(hi, v)) | |
| def coerce_float(value, default: float = 0.0, lo: float = 0.0, hi: float = 10.0) -> float: | |
| try: | |
| v = float(value) | |
| except (TypeError, ValueError): | |
| return default | |
| return max(lo, min(hi, v)) | |
| def coerce_need_types(value) -> list[str]: | |
| """Normalizes need-type lists from LLM output against the fixed vocabulary.""" | |
| from core.schemas import VALID_NEED_TYPES | |
| if isinstance(value, str): | |
| value = [value] | |
| if not isinstance(value, list): | |
| return [] | |
| out = [] | |
| for v in value: | |
| v_norm = str(v).strip().lower() | |
| if v_norm in VALID_NEED_TYPES: | |
| out.append(v_norm) | |
| return list(dict.fromkeys(out)) # dedupe, preserve order | |
| def validate_image_file(path: str, max_bytes: int = 15 * 1024 * 1024) -> tuple[bool, str]: | |
| """Returns (is_valid, error_message). Never raises — callers check the bool.""" | |
| import os | |
| if not path or not os.path.exists(path): | |
| return False, f"image file not found: {path}" | |
| size = os.path.getsize(path) | |
| if size == 0: | |
| return False, "image file is empty (0 bytes)" | |
| if size > max_bytes: | |
| return False, f"image file too large ({size} bytes > {max_bytes} limit)" | |
| try: | |
| from PIL import Image | |
| with Image.open(path) as img: | |
| img.verify() | |
| except Exception as e: | |
| return False, f"invalid or corrupt image: {e}" | |
| return True, "" | |
| def validate_report_text(text: str, max_chars: int = 4000) -> tuple[str, str]: | |
| """ | |
| Returns (cleaned_text, warning). Handles empty and very-long reports | |
| per section 25's failure-handling requirement instead of crashing | |
| or silently truncating without telling anyone. | |
| """ | |
| if text is None: | |
| return "", "no report text provided" | |
| text = text.strip() | |
| if not text: | |
| return "", "empty report text" | |
| if len(text) > max_chars: | |
| return text[:max_chars], f"report truncated from {len(text)} to {max_chars} characters" | |
| return text, "" | |