#!/usr/bin/env python3 """Face identity helper: SCRFD landmarks + ArcFace (w600k_r50) embeddings via plain onnxruntime. No `insightface` package needed (it would pull a CPU onnxruntime over onnxruntime-gpu); only the two ONNX files from InsightFace's buffalo_l pack in models/. NOTE: those pretrained models are licensed for non-commercial research use only. Used by blur_video.py --keep to leave one person's face visible while blurring everyone else. """ from pathlib import Path import cv2 import numpy as np import onnxruntime as ort MODELS = Path(__file__).resolve().parent / "models" # ArcFace 112x112 alignment template (left eye, right eye, nose, left mouth, right mouth) ARC_TEMPLATE = np.array([[38.2946, 51.6963], [73.5318, 51.5014], [56.0252, 71.7366], [41.5493, 92.3655], [70.7299, 92.2041]], np.float32) def _session(path, device): prov = ["CUDAExecutionProvider", "CPUExecutionProvider"] if device != "cpu" else ["CPUExecutionProvider"] so = ort.SessionOptions() so.log_severity_level = 3 return ort.InferenceSession(str(path), sess_options=so, providers=prov) def _nms(boxes, scores, thr=0.4): idx = cv2.dnn.NMSBoxes([[float(b[0]), float(b[1]), float(b[2] - b[0]), float(b[3] - b[1])] for b in boxes], [float(s) for s in scores], 0.0, thr) return np.array(idx, dtype=int).reshape(-1) class FaceID: def __init__(self, device="cpu"): self.det = _session(MODELS / "det_10g.onnx", device) self.rec = _session(MODELS / "w600k_r50.onnx", device) self.det_in = self.det.get_inputs()[0].name self.rec_in = self.rec.get_inputs()[0].name def detect(self, img, size=192, thr=0.4): """SCRFD on `img` letterboxed (top-left) into a size x size canvas. Returns boxes (N,4) xyxy, scores (N,), landmarks (N,5,2), all in `img` pixel coordinates.""" h, w = img.shape[:2] s = size / max(h, w) nw, nh = max(1, int(round(w * s))), max(1, int(round(h * s))) canvas = np.zeros((size, size, 3), np.uint8) canvas[:nh, :nw] = cv2.resize(img, (nw, nh)) blob = cv2.dnn.blobFromImage(canvas, 1 / 128.0, (size, size), (127.5, 127.5, 127.5), swapRB=True) outs = self.det.run(None, {self.det_in: blob}) B, S, K = [], [], [] for i, stride in enumerate((8, 16, 32)): sc, bb, kp = outs[i][:, 0], outs[i + 3] * stride, outs[i + 6] * stride g = size // stride ys, xs = np.mgrid[:g, :g] ctr = np.repeat(np.stack([xs, ys], -1).reshape(-1, 2) * stride, 2, axis=0).astype(np.float32) m = sc >= thr if not m.any(): continue c, bb, kp = ctr[m], bb[m], kp[m] B.append(np.stack([c[:, 0] - bb[:, 0], c[:, 1] - bb[:, 1], c[:, 0] + bb[:, 2], c[:, 1] + bb[:, 3]], 1)) K.append(np.stack([c[:, 0:1] + kp[:, 0::2], c[:, 1:2] + kp[:, 1::2]], -1)) S.append(sc[m]) if not B: return np.zeros((0, 4)), np.zeros(0), np.zeros((0, 5, 2)) B, S, K = np.concatenate(B) / s, np.concatenate(S), np.concatenate(K) / s keep = _nms(B, S) return B[keep], S[keep], K[keep] def embed_aligned(self, img, lmk): M, _ = cv2.estimateAffinePartial2D(lmk.astype(np.float32), ARC_TEMPLATE, method=cv2.LMEDS) if M is None: return None face = cv2.warpAffine(img, M, (112, 112), borderValue=0) blob = cv2.dnn.blobFromImage(face, 1 / 127.5, (112, 112), (127.5, 127.5, 127.5), swapRB=True) e = self.rec.run(None, {self.rec_in: blob})[0][0] return e / (np.linalg.norm(e) + 1e-9) def embed_box(self, frame, box, expand=0.5, min_size=40): """Embedding of the face inside `box` (xyxy) of a full frame, or None when it is too small / not found. Crops the box expanded by `expand`, finds landmarks of the face nearest the box centre, aligns, embeds.""" H, W = frame.shape[:2] x1, y1, x2, y2 = [float(v) for v in box[:4]] bw, bh = x2 - x1, y2 - y1 if min(bw, bh) < min_size: return None cx1, cy1 = max(0, int(x1 - bw * expand)), max(0, int(y1 - bh * expand)) cx2, cy2 = min(W, int(x2 + bw * expand)), min(H, int(y2 + bh * expand)) crop = frame[cy1:cy2, cx1:cx2] if crop.size == 0: return None B, S, K = self.detect(crop) if len(B) == 0: return None tx, ty = (x1 + x2) / 2 - cx1, (y1 + y2) / 2 - cy1 d = [np.hypot((b[0] + b[2]) / 2 - tx, (b[1] + b[3]) / 2 - ty) / max(bw, 1) for b in B] j = int(np.argmin(d)) if d[j] > 0.35: # the found face is not the one in `box` return None return self.embed_aligned(crop, K[j]) def reference_embeddings(self, paths): """One embedding per reference photo (largest face in each). Raises if a photo has no usable face.""" embs = [] for p in paths: img = cv2.imread(str(p)) if img is None: raise SystemExit(f"cannot read reference image {p}") B, S, K = self.detect(img, size=640, thr=0.5) if len(B) == 0: raise SystemExit(f"no face found in reference image {p}") if len(B) > 1: print(f"warning: {len(B)} faces in {p}; using the largest") j = int(np.argmax((B[:, 2] - B[:, 0]) * (B[:, 3] - B[:, 1]))) e = self.embed_aligned(img, K[j]) if e is None: raise SystemExit(f"could not align face in {p}") embs.append(e) return np.stack(embs) @staticmethod def similarity(emb, refs): """Best cosine similarity of `emb` against the reference embeddings.""" return float((refs @ emb).max())