ScoreVision / miner.py
VALOR0316's picture
scorevision: push artifact
87d6b9f verified
Raw History Blame Contribute Delete
5.93 kB
"""TurboVision public-track miner for manak0/Detect-worksite-ppe-compliance-detection-probe-set.
Contract: miner.py at the HF repo root exposes class Miner(path_hf_repo) with
predict_batch(batch_images, offset, n_keypoints) -> list[TVFrameResult]. Images arrive as BGR uint8.
Pipeline per frame: letterbox to the ONNX input size (grey 114) -> RGB/255 NCHW float32 -> YOLO26 ONNX on the
CPU execution provider (the conformity checker re-runs this file on CPU with onnxruntime 1.26.0 and requires
matching boxes, so the live chute runs on CPU too) -> per-class confidence threshold -> boxes in frame pixels.
Both export heads are handled: end-to-end [1, K, 6] (x1, y1, x2, y2, score, cls) and raw [1, 4 + nc, N]
(cx, cy, w, h, class scores; every class thresholded on its own + per-class NMS).
cls_id is the index into the manifest `objects` list.
"""
import json
from pathlib import Path
import cv2
import numpy as np
import onnxruntime as ort
from numpy import ndarray
from pydantic import BaseModel
CLASSES = ["safety helmet", "high-visibility safety vest", "protective gloves", "protective eyewear"]
class BoundingBox(BaseModel):
x1: int
y1: int
x2: int
y2: int
cls_id: int
conf: float
class TVFrameResult(BaseModel):
frame_id: int
boxes: list[BoundingBox] | None = None
keypoints: list[tuple[int, int]] | None = None
class Miner:
def __init__(self, path_hf_repo: Path) -> None:
repo = Path(path_hf_repo)
cfg = json.loads((repo / "config.json").read_text())
nc = len(CLASSES)
self.thr = np.asarray(cfg.get("thr", [0.25] * nc), dtype=np.float32)
self.nms_iou = float(cfg.get("nms_iou", 0.6))
self.max_det = int(cfg.get("max_det", 300))
self.min_side = float(cfg.get("min_side", 2.0))
so = ort.SessionOptions()
so.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL
so.intra_op_num_threads = int(cfg.get("threads", 2))
so.inter_op_num_threads = 1
so.execution_mode = ort.ExecutionMode.ORT_SEQUENTIAL
self.session = ort.InferenceSession(str(repo / "weights.onnx"), sess_options=so,
providers=["CPUExecutionProvider"])
inp = self.session.get_inputs()[0]
self.input_name = inp.name
self.size = int(inp.shape[2])
dummy = np.zeros((1, 3, self.size, self.size), np.float32)
for _ in range(3):
self.session.run(None, {self.input_name: dummy})
def __repr__(self) -> str:
return f"PPE ONNX detector input={self.size} thr={self.thr.tolist()} providers={self.session.get_providers()}"
def _prep(self, img: ndarray):
h, w = img.shape[:2]
s = self.size
r = min(s / max(1, h), s / max(1, w))
nw, nh = max(1, int(round(w * r))), max(1, int(round(h * r)))
canvas = np.full((s, s, 3), 114, np.uint8)
px, py = (s - nw) // 2, (s - nh) // 2
canvas[py:py + nh, px:px + nw] = cv2.resize(img, (nw, nh), interpolation=cv2.INTER_LINEAR)
x = cv2.cvtColor(canvas, cv2.COLOR_BGR2RGB).transpose(2, 0, 1)[None].astype(np.float32) / 255.0
return np.ascontiguousarray(x), r, px, py
def _raw_head(self, y: ndarray) -> ndarray:
y = y[0].T # [N, 4 + nc]
sc = y[:, 4:]
b = np.stack([y[:, 0] - y[:, 2] / 2, y[:, 1] - y[:, 3] / 2, y[:, 0] + y[:, 2] / 2, y[:, 1] + y[:, 3] / 2], 1)
out = []
for c in range(min(sc.shape[1], len(CLASSES))):
m = sc[:, c] >= self.thr[c]
if not m.any():
continue
bb, ss = b[m], sc[m, c]
xywh = np.concatenate([bb[:, :2], bb[:, 2:] - bb[:, :2]], 1).astype(np.float64)
k = np.asarray(cv2.dnn.NMSBoxes(xywh.tolist(), ss.astype(np.float32).tolist(), 0.0, self.nms_iou),
dtype=int).reshape(-1)
if k.size:
out.append(np.concatenate([bb[k], ss[k, None], np.full((k.size, 1), c, np.float32)], 1))
return np.concatenate(out, 0) if out else np.zeros((0, 6), np.float32)
def _end2end(self, y: ndarray) -> ndarray:
d = y[0].astype(np.float32) # [K, 6]
cls = d[:, 5].astype(int)
ok = (cls >= 0) & (cls < len(CLASSES))
d, cls = d[ok], cls[ok]
return d[d[:, 4] >= self.thr[cls]]
def _detect(self, img: ndarray) -> ndarray:
"""-> [n, 6] float32 (x1, y1, x2, y2, score, cls) in frame pixels, sorted by score."""
x, r, px, py = self._prep(img)
y = self.session.run(None, {self.input_name: x})[0]
d = self._end2end(y) if y.ndim == 3 and y.shape[2] == 6 else self._raw_head(y)
if not len(d):
return np.zeros((0, 6), np.float32)
h, w = img.shape[:2]
d = d.copy()
d[:, [0, 2]] = ((d[:, [0, 2]] - px) / max(r, 1e-9)).clip(0, w)
d[:, [1, 3]] = ((d[:, [1, 3]] - py) / max(r, 1e-9)).clip(0, h)
keep = ((d[:, 2] - d[:, 0]) >= self.min_side) & ((d[:, 3] - d[:, 1]) >= self.min_side)
d = d[keep]
return d[np.argsort(-d[:, 4], kind="stable")][: self.max_det].astype(np.float32)
def predict_batch(self, batch_images: list[ndarray], offset: int, n_keypoints: int) -> list[TVFrameResult]:
results = []
for i, img in enumerate(batch_images):
boxes = []
if img is not None and img.size:
for x1, y1, x2, y2, s, c in self._detect(img):
boxes.append(BoundingBox(x1=int(round(float(x1))), y1=int(round(float(y1))),
x2=int(round(float(x2))), y2=int(round(float(y2))),
cls_id=int(c), conf=round(float(s), 5)))
results.append(TVFrameResult(frame_id=offset + i, boxes=boxes,
keypoints=[(0, 0)] * n_keypoints if n_keypoints else []))
return results