"""TurboVision public-track miner for manak0/Detect-worksite-ppe-compliance-detection-probe-set. Contract: miner.py at the HF repo root exposes class Miner(path_hf_repo) with predict_batch(batch_images, offset, n_keypoints) -> list[TVFrameResult]. Images arrive as BGR uint8. Pipeline per frame: letterbox to the ONNX input size (grey 114) -> RGB/255 NCHW float32 -> YOLO26 ONNX on the CPU execution provider (the conformity checker re-runs this file on CPU with onnxruntime 1.26.0 and requires matching boxes, so the live chute runs on CPU too) -> per-class confidence threshold -> boxes in frame pixels. Both export heads are handled: end-to-end [1, K, 6] (x1, y1, x2, y2, score, cls) and raw [1, 4 + nc, N] (cx, cy, w, h, class scores; every class thresholded on its own + per-class NMS). cls_id is the index into the manifest `objects` list. """ import json from pathlib import Path import cv2 import numpy as np import onnxruntime as ort from numpy import ndarray from pydantic import BaseModel CLASSES = ["safety helmet", "high-visibility safety vest", "protective gloves", "protective eyewear"] class BoundingBox(BaseModel): x1: int y1: int x2: int y2: int cls_id: int conf: float class TVFrameResult(BaseModel): frame_id: int boxes: list[BoundingBox] | None = None keypoints: list[tuple[int, int]] | None = None class Miner: def __init__(self, path_hf_repo: Path) -> None: repo = Path(path_hf_repo) cfg = json.loads((repo / "config.json").read_text()) nc = len(CLASSES) self.thr = np.asarray(cfg.get("thr", [0.25] * nc), dtype=np.float32) self.nms_iou = float(cfg.get("nms_iou", 0.6)) self.max_det = int(cfg.get("max_det", 300)) self.min_side = float(cfg.get("min_side", 2.0)) so = ort.SessionOptions() so.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL so.intra_op_num_threads = int(cfg.get("threads", 2)) so.inter_op_num_threads = 1 so.execution_mode = ort.ExecutionMode.ORT_SEQUENTIAL self.session = ort.InferenceSession(str(repo / "weights.onnx"), sess_options=so, providers=["CPUExecutionProvider"]) inp = self.session.get_inputs()[0] self.input_name = inp.name self.size = int(inp.shape[2]) dummy = np.zeros((1, 3, self.size, self.size), np.float32) for _ in range(3): self.session.run(None, {self.input_name: dummy}) def __repr__(self) -> str: return f"PPE ONNX detector input={self.size} thr={self.thr.tolist()} providers={self.session.get_providers()}" def _prep(self, img: ndarray): h, w = img.shape[:2] s = self.size r = min(s / max(1, h), s / max(1, w)) nw, nh = max(1, int(round(w * r))), max(1, int(round(h * r))) canvas = np.full((s, s, 3), 114, np.uint8) px, py = (s - nw) // 2, (s - nh) // 2 canvas[py:py + nh, px:px + nw] = cv2.resize(img, (nw, nh), interpolation=cv2.INTER_LINEAR) x = cv2.cvtColor(canvas, cv2.COLOR_BGR2RGB).transpose(2, 0, 1)[None].astype(np.float32) / 255.0 return np.ascontiguousarray(x), r, px, py def _raw_head(self, y: ndarray) -> ndarray: y = y[0].T # [N, 4 + nc] sc = y[:, 4:] b = np.stack([y[:, 0] - y[:, 2] / 2, y[:, 1] - y[:, 3] / 2, y[:, 0] + y[:, 2] / 2, y[:, 1] + y[:, 3] / 2], 1) out = [] for c in range(min(sc.shape[1], len(CLASSES))): m = sc[:, c] >= self.thr[c] if not m.any(): continue bb, ss = b[m], sc[m, c] xywh = np.concatenate([bb[:, :2], bb[:, 2:] - bb[:, :2]], 1).astype(np.float64) k = np.asarray(cv2.dnn.NMSBoxes(xywh.tolist(), ss.astype(np.float32).tolist(), 0.0, self.nms_iou), dtype=int).reshape(-1) if k.size: out.append(np.concatenate([bb[k], ss[k, None], np.full((k.size, 1), c, np.float32)], 1)) return np.concatenate(out, 0) if out else np.zeros((0, 6), np.float32) def _end2end(self, y: ndarray) -> ndarray: d = y[0].astype(np.float32) # [K, 6] cls = d[:, 5].astype(int) ok = (cls >= 0) & (cls < len(CLASSES)) d, cls = d[ok], cls[ok] return d[d[:, 4] >= self.thr[cls]] def _detect(self, img: ndarray) -> ndarray: """-> [n, 6] float32 (x1, y1, x2, y2, score, cls) in frame pixels, sorted by score.""" x, r, px, py = self._prep(img) y = self.session.run(None, {self.input_name: x})[0] d = self._end2end(y) if y.ndim == 3 and y.shape[2] == 6 else self._raw_head(y) if not len(d): return np.zeros((0, 6), np.float32) h, w = img.shape[:2] d = d.copy() d[:, [0, 2]] = ((d[:, [0, 2]] - px) / max(r, 1e-9)).clip(0, w) d[:, [1, 3]] = ((d[:, [1, 3]] - py) / max(r, 1e-9)).clip(0, h) keep = ((d[:, 2] - d[:, 0]) >= self.min_side) & ((d[:, 3] - d[:, 1]) >= self.min_side) d = d[keep] return d[np.argsort(-d[:, 4], kind="stable")][: self.max_det].astype(np.float32) def predict_batch(self, batch_images: list[ndarray], offset: int, n_keypoints: int) -> list[TVFrameResult]: results = [] for i, img in enumerate(batch_images): boxes = [] if img is not None and img.size: for x1, y1, x2, y2, s, c in self._detect(img): boxes.append(BoundingBox(x1=int(round(float(x1))), y1=int(round(float(y1))), x2=int(round(float(x2))), y2=int(round(float(y2))), cls_id=int(c), conf=round(float(s), 5))) results.append(TVFrameResult(frame_id=offset + i, boxes=boxes, keypoints=[(0, 0)] * n_keypoints if n_keypoints else [])) return results