scorevision: push artifact
Browse files
miner.py
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""TurboVision public-track miner for manak0/Detect-worksite-ppe-compliance-detection-probe-set.
|
| 2 |
+
|
| 3 |
+
Contract: miner.py at the HF repo root exposes class Miner(path_hf_repo) with
|
| 4 |
+
predict_batch(batch_images, offset, n_keypoints) -> list[TVFrameResult]. Images arrive as BGR uint8.
|
| 5 |
+
|
| 6 |
+
Pipeline per frame: letterbox to the ONNX input size (grey 114) -> RGB/255 NCHW float32 -> YOLO26 ONNX on the
|
| 7 |
+
CPU execution provider (the conformity checker re-runs this file on CPU with onnxruntime 1.26.0 and requires
|
| 8 |
+
matching boxes, so the live chute runs on CPU too) -> per-class confidence threshold -> boxes in frame pixels.
|
| 9 |
+
Both export heads are handled: end-to-end [1, K, 6] (x1, y1, x2, y2, score, cls) and raw [1, 4 + nc, N]
|
| 10 |
+
(cx, cy, w, h, class scores; every class thresholded on its own + per-class NMS).
|
| 11 |
+
cls_id is the index into the manifest `objects` list.
|
| 12 |
+
"""
|
| 13 |
+
|
| 14 |
+
import json
|
| 15 |
+
from pathlib import Path
|
| 16 |
+
|
| 17 |
+
import cv2
|
| 18 |
+
import numpy as np
|
| 19 |
+
import onnxruntime as ort
|
| 20 |
+
from numpy import ndarray
|
| 21 |
+
from pydantic import BaseModel
|
| 22 |
+
|
| 23 |
+
CLASSES = ["safety helmet", "high-visibility safety vest", "protective gloves", "protective eyewear"]
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
class BoundingBox(BaseModel):
|
| 27 |
+
x1: int
|
| 28 |
+
y1: int
|
| 29 |
+
x2: int
|
| 30 |
+
y2: int
|
| 31 |
+
cls_id: int
|
| 32 |
+
conf: float
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
class TVFrameResult(BaseModel):
|
| 36 |
+
frame_id: int
|
| 37 |
+
boxes: list[BoundingBox] | None = None
|
| 38 |
+
keypoints: list[tuple[int, int]] | None = None
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
class Miner:
|
| 42 |
+
def __init__(self, path_hf_repo: Path) -> None:
|
| 43 |
+
repo = Path(path_hf_repo)
|
| 44 |
+
cfg = json.loads((repo / "config.json").read_text())
|
| 45 |
+
nc = len(CLASSES)
|
| 46 |
+
self.thr = np.asarray(cfg.get("thr", [0.25] * nc), dtype=np.float32)
|
| 47 |
+
self.nms_iou = float(cfg.get("nms_iou", 0.6))
|
| 48 |
+
self.max_det = int(cfg.get("max_det", 300))
|
| 49 |
+
self.min_side = float(cfg.get("min_side", 2.0))
|
| 50 |
+
so = ort.SessionOptions()
|
| 51 |
+
so.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL
|
| 52 |
+
so.intra_op_num_threads = int(cfg.get("threads", 2))
|
| 53 |
+
so.inter_op_num_threads = 1
|
| 54 |
+
so.execution_mode = ort.ExecutionMode.ORT_SEQUENTIAL
|
| 55 |
+
self.session = ort.InferenceSession(str(repo / "weights.onnx"), sess_options=so,
|
| 56 |
+
providers=["CPUExecutionProvider"])
|
| 57 |
+
inp = self.session.get_inputs()[0]
|
| 58 |
+
self.input_name = inp.name
|
| 59 |
+
self.size = int(inp.shape[2])
|
| 60 |
+
dummy = np.zeros((1, 3, self.size, self.size), np.float32)
|
| 61 |
+
for _ in range(3):
|
| 62 |
+
self.session.run(None, {self.input_name: dummy})
|
| 63 |
+
|
| 64 |
+
def __repr__(self) -> str:
|
| 65 |
+
return f"PPE ONNX detector input={self.size} thr={self.thr.tolist()} providers={self.session.get_providers()}"
|
| 66 |
+
|
| 67 |
+
def _prep(self, img: ndarray):
|
| 68 |
+
h, w = img.shape[:2]
|
| 69 |
+
s = self.size
|
| 70 |
+
r = min(s / max(1, h), s / max(1, w))
|
| 71 |
+
nw, nh = max(1, int(round(w * r))), max(1, int(round(h * r)))
|
| 72 |
+
canvas = np.full((s, s, 3), 114, np.uint8)
|
| 73 |
+
px, py = (s - nw) // 2, (s - nh) // 2
|
| 74 |
+
canvas[py:py + nh, px:px + nw] = cv2.resize(img, (nw, nh), interpolation=cv2.INTER_LINEAR)
|
| 75 |
+
x = cv2.cvtColor(canvas, cv2.COLOR_BGR2RGB).transpose(2, 0, 1)[None].astype(np.float32) / 255.0
|
| 76 |
+
return np.ascontiguousarray(x), r, px, py
|
| 77 |
+
|
| 78 |
+
def _raw_head(self, y: ndarray) -> ndarray:
|
| 79 |
+
y = y[0].T # [N, 4 + nc]
|
| 80 |
+
sc = y[:, 4:]
|
| 81 |
+
b = np.stack([y[:, 0] - y[:, 2] / 2, y[:, 1] - y[:, 3] / 2, y[:, 0] + y[:, 2] / 2, y[:, 1] + y[:, 3] / 2], 1)
|
| 82 |
+
out = []
|
| 83 |
+
for c in range(min(sc.shape[1], len(CLASSES))):
|
| 84 |
+
m = sc[:, c] >= self.thr[c]
|
| 85 |
+
if not m.any():
|
| 86 |
+
continue
|
| 87 |
+
bb, ss = b[m], sc[m, c]
|
| 88 |
+
xywh = np.concatenate([bb[:, :2], bb[:, 2:] - bb[:, :2]], 1).astype(np.float64)
|
| 89 |
+
k = np.asarray(cv2.dnn.NMSBoxes(xywh.tolist(), ss.astype(np.float32).tolist(), 0.0, self.nms_iou),
|
| 90 |
+
dtype=int).reshape(-1)
|
| 91 |
+
if k.size:
|
| 92 |
+
out.append(np.concatenate([bb[k], ss[k, None], np.full((k.size, 1), c, np.float32)], 1))
|
| 93 |
+
return np.concatenate(out, 0) if out else np.zeros((0, 6), np.float32)
|
| 94 |
+
|
| 95 |
+
def _end2end(self, y: ndarray) -> ndarray:
|
| 96 |
+
d = y[0].astype(np.float32) # [K, 6]
|
| 97 |
+
cls = d[:, 5].astype(int)
|
| 98 |
+
ok = (cls >= 0) & (cls < len(CLASSES))
|
| 99 |
+
d, cls = d[ok], cls[ok]
|
| 100 |
+
return d[d[:, 4] >= self.thr[cls]]
|
| 101 |
+
|
| 102 |
+
def _detect(self, img: ndarray) -> ndarray:
|
| 103 |
+
"""-> [n, 6] float32 (x1, y1, x2, y2, score, cls) in frame pixels, sorted by score."""
|
| 104 |
+
x, r, px, py = self._prep(img)
|
| 105 |
+
y = self.session.run(None, {self.input_name: x})[0]
|
| 106 |
+
d = self._end2end(y) if y.ndim == 3 and y.shape[2] == 6 else self._raw_head(y)
|
| 107 |
+
if not len(d):
|
| 108 |
+
return np.zeros((0, 6), np.float32)
|
| 109 |
+
h, w = img.shape[:2]
|
| 110 |
+
d = d.copy()
|
| 111 |
+
d[:, [0, 2]] = ((d[:, [0, 2]] - px) / max(r, 1e-9)).clip(0, w)
|
| 112 |
+
d[:, [1, 3]] = ((d[:, [1, 3]] - py) / max(r, 1e-9)).clip(0, h)
|
| 113 |
+
keep = ((d[:, 2] - d[:, 0]) >= self.min_side) & ((d[:, 3] - d[:, 1]) >= self.min_side)
|
| 114 |
+
d = d[keep]
|
| 115 |
+
return d[np.argsort(-d[:, 4], kind="stable")][: self.max_det].astype(np.float32)
|
| 116 |
+
|
| 117 |
+
def predict_batch(self, batch_images: list[ndarray], offset: int, n_keypoints: int) -> list[TVFrameResult]:
|
| 118 |
+
results = []
|
| 119 |
+
for i, img in enumerate(batch_images):
|
| 120 |
+
boxes = []
|
| 121 |
+
if img is not None and img.size:
|
| 122 |
+
for x1, y1, x2, y2, s, c in self._detect(img):
|
| 123 |
+
boxes.append(BoundingBox(x1=int(round(float(x1))), y1=int(round(float(y1))),
|
| 124 |
+
x2=int(round(float(x2))), y2=int(round(float(y2))),
|
| 125 |
+
cls_id=int(c), conf=round(float(s), 5)))
|
| 126 |
+
results.append(TVFrameResult(frame_id=offset + i, boxes=boxes,
|
| 127 |
+
keypoints=[(0, 0)] * n_keypoints if n_keypoints else []))
|
| 128 |
+
return results
|