File size: 5,929 Bytes
87d6b9f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
"""TurboVision public-track miner for manak0/Detect-worksite-ppe-compliance-detection-probe-set.

Contract: miner.py at the HF repo root exposes class Miner(path_hf_repo) with
predict_batch(batch_images, offset, n_keypoints) -> list[TVFrameResult]. Images arrive as BGR uint8.

Pipeline per frame: letterbox to the ONNX input size (grey 114) -> RGB/255 NCHW float32 -> YOLO26 ONNX on the
CPU execution provider (the conformity checker re-runs this file on CPU with onnxruntime 1.26.0 and requires
matching boxes, so the live chute runs on CPU too) -> per-class confidence threshold -> boxes in frame pixels.
Both export heads are handled: end-to-end [1, K, 6] (x1, y1, x2, y2, score, cls) and raw [1, 4 + nc, N]
(cx, cy, w, h, class scores; every class thresholded on its own + per-class NMS).
cls_id is the index into the manifest `objects` list.
"""

import json
from pathlib import Path

import cv2
import numpy as np
import onnxruntime as ort
from numpy import ndarray
from pydantic import BaseModel

CLASSES = ["safety helmet", "high-visibility safety vest", "protective gloves", "protective eyewear"]


class BoundingBox(BaseModel):
    x1: int
    y1: int
    x2: int
    y2: int
    cls_id: int
    conf: float


class TVFrameResult(BaseModel):
    frame_id: int
    boxes: list[BoundingBox] | None = None
    keypoints: list[tuple[int, int]] | None = None


class Miner:
    def __init__(self, path_hf_repo: Path) -> None:
        repo = Path(path_hf_repo)
        cfg = json.loads((repo / "config.json").read_text())
        nc = len(CLASSES)
        self.thr = np.asarray(cfg.get("thr", [0.25] * nc), dtype=np.float32)
        self.nms_iou = float(cfg.get("nms_iou", 0.6))
        self.max_det = int(cfg.get("max_det", 300))
        self.min_side = float(cfg.get("min_side", 2.0))
        so = ort.SessionOptions()
        so.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL
        so.intra_op_num_threads = int(cfg.get("threads", 2))
        so.inter_op_num_threads = 1
        so.execution_mode = ort.ExecutionMode.ORT_SEQUENTIAL
        self.session = ort.InferenceSession(str(repo / "weights.onnx"), sess_options=so,
                                            providers=["CPUExecutionProvider"])
        inp = self.session.get_inputs()[0]
        self.input_name = inp.name
        self.size = int(inp.shape[2])
        dummy = np.zeros((1, 3, self.size, self.size), np.float32)
        for _ in range(3):
            self.session.run(None, {self.input_name: dummy})

    def __repr__(self) -> str:
        return f"PPE ONNX detector input={self.size} thr={self.thr.tolist()} providers={self.session.get_providers()}"

    def _prep(self, img: ndarray):
        h, w = img.shape[:2]
        s = self.size
        r = min(s / max(1, h), s / max(1, w))
        nw, nh = max(1, int(round(w * r))), max(1, int(round(h * r)))
        canvas = np.full((s, s, 3), 114, np.uint8)
        px, py = (s - nw) // 2, (s - nh) // 2
        canvas[py:py + nh, px:px + nw] = cv2.resize(img, (nw, nh), interpolation=cv2.INTER_LINEAR)
        x = cv2.cvtColor(canvas, cv2.COLOR_BGR2RGB).transpose(2, 0, 1)[None].astype(np.float32) / 255.0
        return np.ascontiguousarray(x), r, px, py

    def _raw_head(self, y: ndarray) -> ndarray:
        y = y[0].T  # [N, 4 + nc]
        sc = y[:, 4:]
        b = np.stack([y[:, 0] - y[:, 2] / 2, y[:, 1] - y[:, 3] / 2, y[:, 0] + y[:, 2] / 2, y[:, 1] + y[:, 3] / 2], 1)
        out = []
        for c in range(min(sc.shape[1], len(CLASSES))):
            m = sc[:, c] >= self.thr[c]
            if not m.any():
                continue
            bb, ss = b[m], sc[m, c]
            xywh = np.concatenate([bb[:, :2], bb[:, 2:] - bb[:, :2]], 1).astype(np.float64)
            k = np.asarray(cv2.dnn.NMSBoxes(xywh.tolist(), ss.astype(np.float32).tolist(), 0.0, self.nms_iou),
                           dtype=int).reshape(-1)
            if k.size:
                out.append(np.concatenate([bb[k], ss[k, None], np.full((k.size, 1), c, np.float32)], 1))
        return np.concatenate(out, 0) if out else np.zeros((0, 6), np.float32)

    def _end2end(self, y: ndarray) -> ndarray:
        d = y[0].astype(np.float32)  # [K, 6]
        cls = d[:, 5].astype(int)
        ok = (cls >= 0) & (cls < len(CLASSES))
        d, cls = d[ok], cls[ok]
        return d[d[:, 4] >= self.thr[cls]]

    def _detect(self, img: ndarray) -> ndarray:
        """-> [n, 6] float32 (x1, y1, x2, y2, score, cls) in frame pixels, sorted by score."""
        x, r, px, py = self._prep(img)
        y = self.session.run(None, {self.input_name: x})[0]
        d = self._end2end(y) if y.ndim == 3 and y.shape[2] == 6 else self._raw_head(y)
        if not len(d):
            return np.zeros((0, 6), np.float32)
        h, w = img.shape[:2]
        d = d.copy()
        d[:, [0, 2]] = ((d[:, [0, 2]] - px) / max(r, 1e-9)).clip(0, w)
        d[:, [1, 3]] = ((d[:, [1, 3]] - py) / max(r, 1e-9)).clip(0, h)
        keep = ((d[:, 2] - d[:, 0]) >= self.min_side) & ((d[:, 3] - d[:, 1]) >= self.min_side)
        d = d[keep]
        return d[np.argsort(-d[:, 4], kind="stable")][: self.max_det].astype(np.float32)

    def predict_batch(self, batch_images: list[ndarray], offset: int, n_keypoints: int) -> list[TVFrameResult]:
        results = []
        for i, img in enumerate(batch_images):
            boxes = []
            if img is not None and img.size:
                for x1, y1, x2, y2, s, c in self._detect(img):
                    boxes.append(BoundingBox(x1=int(round(float(x1))), y1=int(round(float(y1))),
                                             x2=int(round(float(x2))), y2=int(round(float(y2))),
                                             cls_id=int(c), conf=round(float(s), 5)))
            results.append(TVFrameResult(frame_id=offset + i, boxes=boxes,
                                         keypoints=[(0, 0)] * n_keypoints if n_keypoints else []))
        return results