firenet-v8d / miner.py
maple-matrix's picture
scorevision: push artifact
32f5e20 verified
Raw
History Blame Contribute Delete
8.57 kB
# v8: yolo11s trained on validator-aligned SAM3 labels.
# Pool val backtest: F1=0.862 vs v32 F1=0.742 (+0.125 absolute, smoke recall 89% vs 43%).
from pathlib import Path
import cv2
import numpy as np
import onnxruntime as ort
from numpy import ndarray
from pydantic import BaseModel
class BoundingBox(BaseModel):
x1: int
y1: int
x2: int
y2: int
cls_id: int
conf: float
class TVFrameResult(BaseModel):
frame_id: int
boxes: list[BoundingBox]
keypoints: list[tuple[int, int]]
class Miner:
"""v8: ONNX with built-in NMS → light post-processing.
Pipeline:
1. Letterbox to 1280x1280
2. ONNX inference (returns [1, 300, 6] post-NMS)
3. Conf filter
4. Extra per-class dedup (IoU > 0.3 OR ≥80% containment) — catches nested
duplicates the model inherits from SAM3 training labels that default
NMS@0.5 doesn't suppress
5. Top-1 fallback if everything got filtered — empty predictions are heavily
penalized; even a low-conf best guess scores better than nothing
6. Un-letterbox coords back to original size + clip
"""
# validator-visible class output order (what the runner expects in cls_id)
class_names = ["fire", "smoke", "fire extinguisher"]
# order the v8 ONNX emits classes (training CLASS_ORDER in v8_build_dataset.py)
_model_class_order = ["fire", "fire extinguisher", "smoke"]
input_size = 1280
conf_thresh = 0.25
nms_iou_thresh = 0.3
contain_thresh = 0.80
fallback_min_conf = 0.05 # top-1 fallback floors at this; never return total junk
def __init__(self, path_hf_repo: Path) -> None:
model_path = path_hf_repo / "weights.onnx"
self.cls_remap = np.array(
[self.class_names.index(n) for n in self._model_class_order],
dtype=np.int32,
)
try:
ort.preload_dlls()
except Exception as e:
print(f"preload_dlls: {e}")
sess_options = ort.SessionOptions()
sess_options.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL
try:
self.session = ort.InferenceSession(
str(model_path),
sess_options=sess_options,
providers=["CUDAExecutionProvider", "CPUExecutionProvider"],
)
except Exception as e:
print(f"CUDA failed, CPU: {e}")
self.session = ort.InferenceSession(
str(model_path),
sess_options=sess_options,
providers=["CPUExecutionProvider"],
)
self.input_name = self.session.get_inputs()[0].name
print(f"v8 ONNX loaded, providers={self.session.get_providers()}")
def __repr__(self) -> str:
return f"v8 Miner (providers={self.session.get_providers()})"
def _letterbox(self, image: ndarray):
h, w = image.shape[:2]
s = self.input_size / max(h, w)
nw, nh = int(round(w * s)), int(round(h * s))
if (nw, nh) != (w, h):
interp = cv2.INTER_CUBIC if s > 1.0 else cv2.INTER_LINEAR
image = cv2.resize(image, (nw, nh), interpolation=interp)
canvas = np.full((self.input_size, self.input_size, 3), 114, dtype=np.uint8)
dx = (self.input_size - nw) // 2
dy = (self.input_size - nh) // 2
canvas[dy:dy + nh, dx:dx + nw] = image
return canvas, s, (dx, dy)
def _preprocess(self, image: ndarray):
H, W = image.shape[:2]
padded, scale, (dx, dy) = self._letterbox(image)
x = padded[:, :, ::-1].astype(np.float32) / 255.0 # BGR→RGB, /255
x = np.ascontiguousarray(x.transpose(2, 0, 1)[None], dtype=np.float32)
return x, scale, (dx, dy), (W, H)
@staticmethod
def _iou(a, b):
ix1 = max(a[0], b[0]); iy1 = max(a[1], b[1])
ix2 = min(a[2], b[2]); iy2 = min(a[3], b[3])
iw = max(0.0, ix2 - ix1); ih = max(0.0, iy2 - iy1)
inter = iw * ih
ua = (a[2]-a[0])*(a[3]-a[1]) + (b[2]-b[0])*(b[3]-b[1]) - inter
return inter / ua if ua > 0 else 0.0
@staticmethod
def _containment(inner, outer):
ix1 = max(inner[0], outer[0]); iy1 = max(inner[1], outer[1])
ix2 = min(inner[2], outer[2]); iy2 = min(inner[3], outer[3])
iw = max(0.0, ix2 - ix1); ih = max(0.0, iy2 - iy1)
inter = iw * ih
a_in = (inner[2]-inner[0]) * (inner[3]-inner[1])
return inter / a_in if a_in > 0 else 0.0
def _dedup(self, boxes_xyxy, scores, cls_ids):
"""Per-class dedup: drop a box if same-class IoU>nms_iou OR ≥contain_thresh
contained in a larger same-class box. Keep larger box on ties."""
n = len(boxes_xyxy)
if n <= 1:
return np.arange(n, dtype=np.intp)
# Sort by area desc (so we always test smaller against larger we already kept)
areas = (boxes_xyxy[:, 2] - boxes_xyxy[:, 0]) * (boxes_xyxy[:, 3] - boxes_xyxy[:, 1])
order = np.argsort(-areas)
keep = []
suppressed = np.zeros(n, dtype=bool)
for i in order:
if suppressed[i]: continue
keep.append(int(i))
for j in order:
if j == i or suppressed[j]: continue
if cls_ids[i] != cls_ids[j]: continue
if self._iou(boxes_xyxy[i], boxes_xyxy[j]) > self.nms_iou_thresh:
suppressed[j] = True; continue
if self._containment(boxes_xyxy[j], boxes_xyxy[i]) >= self.contain_thresh:
suppressed[j] = True
keep.sort()
return np.array(keep, dtype=np.intp)
def _predict_one(self, frame: ndarray) -> list[BoundingBox]:
x, scale, (dx, dy), (W, H) = self._preprocess(frame)
out = self.session.run(None, {self.input_name: x})[0]
# output shape: [1, 300, 6] — (x1, y1, x2, y2, conf, cls_id)
raw = out[0]
if raw.shape[0] == 0:
return []
# Apply conf filter (keep raw for fallback)
primary = raw[raw[:, 4] >= self.conf_thresh]
# Per-class dedup on the conf-filtered set
final_dets = []
if len(primary) > 0:
xyxy = primary[:, :4].astype(np.float32)
scores = primary[:, 4].astype(np.float32)
cls_ids = primary[:, 5].astype(np.int32)
keep_idx = self._dedup(xyxy, scores, cls_ids)
primary = primary[keep_idx]
for det in primary:
final_dets.append(det)
# Fallback: nothing left → return single highest-conf raw box (above floor)
if not final_dets and raw.shape[0] > 0:
top = raw[np.argmax(raw[:, 4])]
if top[4] >= self.fallback_min_conf:
final_dets.append(top)
# Build BoundingBox list with un-letterbox + cls remap
boxes_out: list[BoundingBox] = []
for det in final_dets:
x1, y1, x2, y2, conf, model_cls_id = det
x1 = (x1 - dx) / scale; x2 = (x2 - dx) / scale
y1 = (y1 - dy) / scale; y2 = (y2 - dy) / scale
x1 = max(0.0, min(W - 1.0, x1)); x2 = max(0.0, min(W - 1.0, x2))
y1 = max(0.0, min(H - 1.0, y1)); y2 = max(0.0, min(H - 1.0, y2))
if x2 <= x1 or y2 <= y1:
continue
mapped_cls = int(self.cls_remap[int(model_cls_id)])
boxes_out.append(BoundingBox(
x1=int(x1), y1=int(y1), x2=int(x2), y2=int(y2),
cls_id=mapped_cls, conf=float(conf),
))
return boxes_out
def predict_batch(
self,
batch_images: list[ndarray],
offset: int,
n_keypoints: int,
) -> list[TVFrameResult]:
"""Required interface for chute template (sv_chutes_*.py)."""
results: list[TVFrameResult] = []
for frame_number_in_batch, image in enumerate(batch_images):
try:
boxes = self._predict_one(image)
except Exception as e:
print(f"⚠️ Inference failed for frame "
f"{offset + frame_number_in_batch}: {e}")
boxes = []
results.append(TVFrameResult(
frame_id=offset + frame_number_in_batch,
boxes=boxes,
keypoints=[(0, 0) for _ in range(max(0, int(n_keypoints)))],
))
return results
# Back-compat alias for local sanity testing
def run(self, frames: list[ndarray]) -> list[TVFrameResult]:
return self.predict_batch(frames, offset=0, n_keypoints=0)