VALOR0316 commited on
Commit
87d6b9f
·
verified ·
1 Parent(s): f0bfb73

scorevision: push artifact

Browse files
Files changed (1) hide show
  1. miner.py +128 -0
miner.py ADDED
@@ -0,0 +1,128 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """TurboVision public-track miner for manak0/Detect-worksite-ppe-compliance-detection-probe-set.
2
+
3
+ Contract: miner.py at the HF repo root exposes class Miner(path_hf_repo) with
4
+ predict_batch(batch_images, offset, n_keypoints) -> list[TVFrameResult]. Images arrive as BGR uint8.
5
+
6
+ Pipeline per frame: letterbox to the ONNX input size (grey 114) -> RGB/255 NCHW float32 -> YOLO26 ONNX on the
7
+ CPU execution provider (the conformity checker re-runs this file on CPU with onnxruntime 1.26.0 and requires
8
+ matching boxes, so the live chute runs on CPU too) -> per-class confidence threshold -> boxes in frame pixels.
9
+ Both export heads are handled: end-to-end [1, K, 6] (x1, y1, x2, y2, score, cls) and raw [1, 4 + nc, N]
10
+ (cx, cy, w, h, class scores; every class thresholded on its own + per-class NMS).
11
+ cls_id is the index into the manifest `objects` list.
12
+ """
13
+
14
+ import json
15
+ from pathlib import Path
16
+
17
+ import cv2
18
+ import numpy as np
19
+ import onnxruntime as ort
20
+ from numpy import ndarray
21
+ from pydantic import BaseModel
22
+
23
+ CLASSES = ["safety helmet", "high-visibility safety vest", "protective gloves", "protective eyewear"]
24
+
25
+
26
+ class BoundingBox(BaseModel):
27
+ x1: int
28
+ y1: int
29
+ x2: int
30
+ y2: int
31
+ cls_id: int
32
+ conf: float
33
+
34
+
35
+ class TVFrameResult(BaseModel):
36
+ frame_id: int
37
+ boxes: list[BoundingBox] | None = None
38
+ keypoints: list[tuple[int, int]] | None = None
39
+
40
+
41
+ class Miner:
42
+ def __init__(self, path_hf_repo: Path) -> None:
43
+ repo = Path(path_hf_repo)
44
+ cfg = json.loads((repo / "config.json").read_text())
45
+ nc = len(CLASSES)
46
+ self.thr = np.asarray(cfg.get("thr", [0.25] * nc), dtype=np.float32)
47
+ self.nms_iou = float(cfg.get("nms_iou", 0.6))
48
+ self.max_det = int(cfg.get("max_det", 300))
49
+ self.min_side = float(cfg.get("min_side", 2.0))
50
+ so = ort.SessionOptions()
51
+ so.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL
52
+ so.intra_op_num_threads = int(cfg.get("threads", 2))
53
+ so.inter_op_num_threads = 1
54
+ so.execution_mode = ort.ExecutionMode.ORT_SEQUENTIAL
55
+ self.session = ort.InferenceSession(str(repo / "weights.onnx"), sess_options=so,
56
+ providers=["CPUExecutionProvider"])
57
+ inp = self.session.get_inputs()[0]
58
+ self.input_name = inp.name
59
+ self.size = int(inp.shape[2])
60
+ dummy = np.zeros((1, 3, self.size, self.size), np.float32)
61
+ for _ in range(3):
62
+ self.session.run(None, {self.input_name: dummy})
63
+
64
+ def __repr__(self) -> str:
65
+ return f"PPE ONNX detector input={self.size} thr={self.thr.tolist()} providers={self.session.get_providers()}"
66
+
67
+ def _prep(self, img: ndarray):
68
+ h, w = img.shape[:2]
69
+ s = self.size
70
+ r = min(s / max(1, h), s / max(1, w))
71
+ nw, nh = max(1, int(round(w * r))), max(1, int(round(h * r)))
72
+ canvas = np.full((s, s, 3), 114, np.uint8)
73
+ px, py = (s - nw) // 2, (s - nh) // 2
74
+ canvas[py:py + nh, px:px + nw] = cv2.resize(img, (nw, nh), interpolation=cv2.INTER_LINEAR)
75
+ x = cv2.cvtColor(canvas, cv2.COLOR_BGR2RGB).transpose(2, 0, 1)[None].astype(np.float32) / 255.0
76
+ return np.ascontiguousarray(x), r, px, py
77
+
78
+ def _raw_head(self, y: ndarray) -> ndarray:
79
+ y = y[0].T # [N, 4 + nc]
80
+ sc = y[:, 4:]
81
+ b = np.stack([y[:, 0] - y[:, 2] / 2, y[:, 1] - y[:, 3] / 2, y[:, 0] + y[:, 2] / 2, y[:, 1] + y[:, 3] / 2], 1)
82
+ out = []
83
+ for c in range(min(sc.shape[1], len(CLASSES))):
84
+ m = sc[:, c] >= self.thr[c]
85
+ if not m.any():
86
+ continue
87
+ bb, ss = b[m], sc[m, c]
88
+ xywh = np.concatenate([bb[:, :2], bb[:, 2:] - bb[:, :2]], 1).astype(np.float64)
89
+ k = np.asarray(cv2.dnn.NMSBoxes(xywh.tolist(), ss.astype(np.float32).tolist(), 0.0, self.nms_iou),
90
+ dtype=int).reshape(-1)
91
+ if k.size:
92
+ out.append(np.concatenate([bb[k], ss[k, None], np.full((k.size, 1), c, np.float32)], 1))
93
+ return np.concatenate(out, 0) if out else np.zeros((0, 6), np.float32)
94
+
95
+ def _end2end(self, y: ndarray) -> ndarray:
96
+ d = y[0].astype(np.float32) # [K, 6]
97
+ cls = d[:, 5].astype(int)
98
+ ok = (cls >= 0) & (cls < len(CLASSES))
99
+ d, cls = d[ok], cls[ok]
100
+ return d[d[:, 4] >= self.thr[cls]]
101
+
102
+ def _detect(self, img: ndarray) -> ndarray:
103
+ """-> [n, 6] float32 (x1, y1, x2, y2, score, cls) in frame pixels, sorted by score."""
104
+ x, r, px, py = self._prep(img)
105
+ y = self.session.run(None, {self.input_name: x})[0]
106
+ d = self._end2end(y) if y.ndim == 3 and y.shape[2] == 6 else self._raw_head(y)
107
+ if not len(d):
108
+ return np.zeros((0, 6), np.float32)
109
+ h, w = img.shape[:2]
110
+ d = d.copy()
111
+ d[:, [0, 2]] = ((d[:, [0, 2]] - px) / max(r, 1e-9)).clip(0, w)
112
+ d[:, [1, 3]] = ((d[:, [1, 3]] - py) / max(r, 1e-9)).clip(0, h)
113
+ keep = ((d[:, 2] - d[:, 0]) >= self.min_side) & ((d[:, 3] - d[:, 1]) >= self.min_side)
114
+ d = d[keep]
115
+ return d[np.argsort(-d[:, 4], kind="stable")][: self.max_det].astype(np.float32)
116
+
117
+ def predict_batch(self, batch_images: list[ndarray], offset: int, n_keypoints: int) -> list[TVFrameResult]:
118
+ results = []
119
+ for i, img in enumerate(batch_images):
120
+ boxes = []
121
+ if img is not None and img.size:
122
+ for x1, y1, x2, y2, s, c in self._detect(img):
123
+ boxes.append(BoundingBox(x1=int(round(float(x1))), y1=int(round(float(y1))),
124
+ x2=int(round(float(x2))), y2=int(round(float(y2))),
125
+ cls_id=int(c), conf=round(float(s), 5)))
126
+ results.append(TVFrameResult(frame_id=offset + i, boxes=boxes,
127
+ keypoints=[(0, 0)] * n_keypoints if n_keypoints else []))
128
+ return results