import gradio as gr import numpy as np import csv from datetime import datetime import cv2 _CASCADES_AVAILABLE = hasattr(cv2, "CascadeClassifier") from PIL import Image, ImageDraw, ImageFont # ── PyTorch patch ───────────────────────────────────────────── import torch _orig_load = torch.load def _safe_load(f, *a, **kw): kw.setdefault("weights_only", False) return _orig_load(f, *a, **kw) torch.load = _safe_load from transformers import ViTImageProcessor, AutoModelForImageClassification from PIL import Image as PILImage emotion_map = {"angry":0,"disgust":1,"fear":2,"happy":3,"sad":4,"surprise":5,"neutral":6} EMOTION_LIST = ["angry","disgust","fear","happy","sad","surprise","neutral"] CHART_LIST = ["angry","disgust","fear","happy","sad","surprise","neutral"] COLORS_RGB = { "angry": (255, 68, 68), "disgust": (170, 68, 255), "fear": (255,170, 0), "happy": ( 0,255, 153), "sad": ( 68,136, 255), "surprise":(255,102, 204), "neutral": (136,136, 136), } # ViT model ("HardlyHumans/Facial-expression-detection") was fine-tuned on # FER2013 + AffectNet with 8 classes; map its labels to our app's vocabulary. VIT_MODEL_NAME = "HardlyHumans/Facial-expression-detection" VIT_TO_APP = {"anger":"angry","contempt":"disgust","disgust":"disgust","fear":"fear", "happy":"happy","neutral":"neutral","sad":"sad","surprise":"surprise"} log = [] recent_emotions = [] SMOOTH_WINDOW = 10 PENALTIES = {"fear":0.65,"sad":0.75,"angry":0.75,"disgust":0.75,"surprise":0.70} BONUS = {"neutral":1.1} # Multi-face state _show_face_boxes = False # toggled by sidebar button _face_smoothers = {} # per-face smoothed display scores, keyed by face index _face_emotions = {} # per-face current emotion _max_faces_seen = 0 # for summary analysis FACE_SMOOTH = 0.18 # same as DISPLAY_SMOOTH_FACTOR # ── Eye tracking via OpenCV (gracefully degrades if cascades unavailable) ── if _CASCADES_AVAILABLE: _face_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + "haarcascade_frontalface_default.xml") _face_cascade_alt = cv2.CascadeClassifier(cv2.data.haarcascades + "haarcascade_frontalface_alt.xml") _face_cascade_alt2 = cv2.CascadeClassifier(cv2.data.haarcascades + "haarcascade_frontalface_alt2.xml") _eye_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + "haarcascade_eye.xml") _clahe = cv2.createCLAHE(clipLimit=3.0, tileGridSize=(8,8)) else: _face_cascade = _face_cascade_alt = _face_cascade_alt2 = _eye_cascade = None _clahe = None def detect_face_robust(gray): """ Try multiple cascades + CLAHE-enhanced contrast to handle backlit or low-angle shots where the plain default cascade misses the face. Returns the first non-empty detection result. Falls back gracefully when cv2 cascades are unavailable (headless env). """ if not _CASCADES_AVAILABLE or _clahe is None: # Cascades not available — pretend face is found so the ViT model # still runs on the full frame. Gaze-based attention tracking is # skipped in this mode. H, W = gray.shape fake_face = np.array([[W//4, H//4, W//2, H//2]], dtype=int) return fake_face clahe_img = _clahe.apply(gray) for casc in (_face_cascade, _face_cascade_alt, _face_cascade_alt2): faces = casc.detectMultiScale(clahe_img, 1.05, 4, minSize=(50,50)) if len(faces) > 0: return faces # Fallback to plain grayscale if CLAHE-enhanced version found nothing for casc in (_face_cascade, _face_cascade_alt, _face_cascade_alt2): faces = casc.detectMultiScale(gray, 1.05, 4, minSize=(50,50)) if len(faces) > 0: return faces return [] attention_score = 100.0 # 0-100, how focused on camera (100=fully attentive) _away_since = None ATTENTION_DROP_DELAY = 1.0 # seconds before attention starts dropping ATTENTION_DROP_RAMP = 2.0 # seconds to fall to minimum once dropping ATTENTION_RECOVER_RATE = 25.0 # how fast attention climbs back per frame (fast recovery) _prev_gaze = [] # history for darting detection DART_WINDOW = 8 # Smoothed face-center history for stable head-pose estimate _face_center_hist = [] FACE_HIST_SIZE = 5 def analyze_eyes(frame_rgb): """ Head-pose based gaze proxy (robust, no eye cascade dependency). """ gray = cv2.cvtColor(frame_rgb, cv2.COLOR_RGB2GRAY) H, W = gray.shape scale = 0.5 small = cv2.resize(gray, (int(W*scale), int(H*scale))) faces = detect_face_robust(small) if len(faces) == 0: return {"away": True, "down": True, "up": False, "staring": False, "darting": False, "wide_open": False, "squinting": False, "h": 0.5, "v": 0.9, "openness": 0.2, "eyes_found": 0, "face_cx": None, "face_cy": None, "head_turn": 1.0, "face_reliable": False, "no_face": True} fx, fy, fw, fh = [int(v/scale) for v in faces[0]] face_cx = fx + fw/2 face_cy = fy + fh/2 norm_cx = face_cx / W norm_cy = face_cy / H _face_center_hist.append((norm_cx, norm_cy)) if len(_face_center_hist) > FACE_HIST_SIZE: _face_center_hist.pop(0) face_roi = gray[fy:fy+fh, fx:fx+fw] if _CASCADES_AVAILABLE and _eye_cascade is not None: eyes = _eye_cascade.detectMultiScale(face_roi, 1.05, 3, minSize=(15,15)) else: eyes = [] eyes_top = [(ex,ey,ew,eh) for ex,ey,ew,eh in eyes if ey < fh*0.55] eyes_found = len(eyes_top) openness = 0.35 if eyes_found >= 1: openness = float(np.mean([eh/max(ew,1) for ex,ey,ew,eh in eyes_top[:2]])) avg_h, avg_v = 0.5, 0.5 if eyes_found >= 2: eyes_top = sorted(eyes_top, key=lambda e: e[0]) h_vals, v_vals = [], [] for (ex,ey,ew,eh) in eyes_top[:2]: roi = face_roi[ey:ey+eh, ex:ex+ew] blurred = cv2.GaussianBlur(roi, (5,5), 0) mn = int(blurred.min()) _, thresh = cv2.threshold(blurred, min(mn+30,80), 255, cv2.THRESH_BINARY_INV) cnts,_ = cv2.findContours(thresh, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) if cnts: c = max(cnts, key=cv2.contourArea) M = cv2.moments(c) if M["m00"] > 0: h_vals.append((M["m10"]/M["m00"])/max(ew,1)) v_vals.append((M["m01"]/M["m00"])/max(eh,1)) if h_vals: avg_h = float(np.mean(h_vals)) avg_v = float(np.mean(v_vals)) _prev_gaze.append((avg_h, avg_v)) if len(_prev_gaze) > DART_WINDOW: _prev_gaze.pop(0) darting = False if len(_prev_gaze) >= DART_WINDOW: darting = (np.std([g[0] for g in _prev_gaze]) > 0.12 or np.std([g[1] for g in _prev_gaze]) > 0.10) head_turned_h = norm_cx < 0.32 or norm_cx > 0.68 head_turned_v = norm_cy < 0.25 or norm_cy > 0.70 eyes_missing = eyes_found < 2 pupil_away = avg_h < 0.30 or avg_h > 0.70 or avg_v < 0.25 or avg_v > 0.72 pupil_down = avg_v > 0.68 looking_away = head_turned_h or head_turned_v or eyes_missing or pupil_away looking_down = eyes_missing or pupil_down or norm_cy > 0.68 looking_up = (eyes_found >= 2 and avg_v < 0.25) and not eyes_missing staring = (eyes_found >= 2 and 0.35 0.48 and eyes_found >= 2 h_dev = max(0.0, max(0.32-norm_cx, norm_cx-0.68) / 0.32) v_dev = max(0.0, max(0.25-norm_cy, norm_cy-0.70) / 0.30) eyes_penalty = 1.0 if eyes_found < 2 else 0.0 head_turn_severity = min(1.0, max(h_dev, v_dev, eyes_penalty*0.7)) face_reliable = head_turn_severity < 0.35 and eyes_found >= 2 return {"away": looking_away, "down": looking_down, "up": looking_up, "staring": staring, "darting": darting, "wide_open": wide_open, "squinting": squinting, "h": avg_h, "v": avg_v, "openness": openness, "eyes_found": eyes_found, "face_cx": round(norm_cx,2), "face_cy": round(norm_cy,2), "head_turn": round(head_turn_severity,2), "face_reliable": face_reliable, "no_face": False} def apply_gaze_modifiers(emotions, gaze): if gaze is None: return emotions adj = dict(emotions) if gaze["down"]: adj["sad"] = adj.get("sad",0) * 1.6 adj["neutral"] = adj.get("neutral",0) * 0.8 if gaze["staring"] and gaze["squinting"]: adj["angry"] = adj.get("angry",0) * 1.7 if gaze.get("darting"): adj["fear"] = adj.get("fear",0) * 1.5 if gaze["wide_open"]: adj["surprise"]= adj.get("surprise",0)* 1.5 adj["fear"] = adj.get("fear",0) * 1.2 if gaze["away"] and not gaze["down"]: adj["neutral"] = adj.get("neutral",0) * 0.85 return adj _attn_gaze_buf = [] ATTN_BUF_SIZE = 4 last_gaze_debug = "no data" def update_attention(gaze): """ Tracks how focused the person is on the camera, independent of emotion (unlike the old 'bored' state, which only applied when emotion == neutral). Attention drops when gaze leaves the camera and recovers quickly once it returns. """ global attention_score, _away_since, _attn_gaze_buf, last_gaze_debug if gaze is None: attention_score = max(0.0, attention_score - 8) last_gaze_debug = "NO FACE" return round(attention_score, 1) head_turn = gaze.get("head_turn", 0.0) looking_away = gaze["away"] or gaze["down"] or gaze["up"] or head_turn > 0.3 _attn_gaze_buf.append(looking_away) if len(_attn_gaze_buf) > 8: _attn_gaze_buf.pop(0) sustained_away = sum(_attn_gaze_buf) >= ATTN_BUF_SIZE last_gaze_debug = (f"away={gaze['away']} turn={head_turn:.2f} eyes={gaze.get('eyes_found','?')} " f"fcx={gaze.get('face_cx','?')} fcy={gaze.get('face_cy','?')} " f"buf={sum(_attn_gaze_buf)}/8") if sustained_away: if _away_since is None: _away_since = datetime.now() t = max(0.0, (datetime.now()-_away_since).total_seconds() - ATTENTION_DROP_DELAY) ramp = ATTENTION_DROP_RAMP * (1.0 - 0.4*head_turn) drop_pct = min(100.0, (t / max(ramp,0.5)) * 100) attention_score = max(0.0, 100.0 - drop_pct) else: _away_since = None attention_score = min(100.0, attention_score + ATTENTION_RECOVER_RATE) return round(attention_score, 1) _vit_model = None _vit_processor = None def get_rec(): """Lazily load the ViT facial-expression model + its image processor.""" global _vit_model, _vit_processor if _vit_model is None: try: _vit_processor = ViTImageProcessor.from_pretrained(VIT_MODEL_NAME) except Exception: # This repo doesn't ship its own preprocessor_config.json — # fall back to the base ViT model's standard preprocessing # (224x224, ImageNet normalization), which matches what the # model was fine-tuned with (google/vit-base-patch16-224-in21k). _vit_processor = ViTImageProcessor.from_pretrained( "google/vit-base-patch16-224-in21k" ) _vit_model = AutoModelForImageClassification.from_pretrained(VIT_MODEL_NAME) _vit_model.eval() return _vit_model, _vit_processor def predict_emotions_vit(frame_rgb): """ Runs the ViT model on a cropped face region. Returns a dict of emotion -> percentage (0-100). """ model, processor = get_rec() pil_img = PILImage.fromarray(frame_rgb) inputs = processor(images=pil_img, return_tensors="pt") with torch.no_grad(): outputs = model(**inputs) probs = torch.nn.functional.softmax(outputs.logits, dim=-1)[0] id2label = model.config.id2label raw_emotions = {} for idx, prob in enumerate(probs): label = id2label[idx].lower() app_label = VIT_TO_APP.get(label) if app_label: raw_emotions[app_label] = raw_emotions.get(app_label, 0.0) + float(prob) * 100 return raw_emotions def detect_all_faces(frame_rgb): """ Detect ALL faces in the frame (not just the first one). Returns list of (x, y, w, h) in original frame coordinates. """ gray = cv2.cvtColor(frame_rgb, cv2.COLOR_RGB2GRAY) H, W = gray.shape scale = 0.5 small = cv2.resize(gray, (int(W*scale), int(H*scale))) faces_raw = detect_face_robust(small) if len(faces_raw) == 0: return [] # Scale back to original coordinates and sort left-to-right faces = [(int(x/scale), int(y/scale), int(w/scale), int(h/scale)) for (x,y,w,h) in faces_raw] return sorted(faces, key=lambda f: f[0]) def draw_face_boxes(frame_rgb, face_boxes, face_emotions): """ Draw a thin coloured rectangle + emotion label above each face. Returns a new RGB image with the overlays applied. """ img = PILImage.fromarray(frame_rgb.copy()) d = ImageDraw.Draw(img) _, _, _, ti = fonts() for i, (fx, fy, fw, fh) in enumerate(face_boxes): emo = face_emotions.get(i, "") color = COLORS_RGB.get(emo, (0,255,153)) # Thin rectangle d.rectangle([fx, fy, fx+fw, fy+fh], outline=color, width=2) # Label above the box label = f"#{i+1} {emo.upper()}" if emo else f"#{i+1}" d.text((fx+4, max(0, fy-14)), label, font=ti, fill=color) return np.array(img) def adjust(emotions): adj = dict(emotions) for e,f in PENALTIES.items(): if e in adj: adj[e] *= f for e,f in BONUS.items(): if e in adj: adj[e] *= f return adj def smooth(raw): recent_emotions.append(raw) if len(recent_emotions) > SMOOTH_WINDOW: recent_emotions.pop(0) return max(set(recent_emotions), key=recent_emotions.count) def fonts(): try: r = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf",11) b = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf",13) lg = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf",28) ti = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf",9) return r,b,lg,ti except: f = ImageFont.load_default(); return f,f,f,f def rr(d,xy,r,fill,outline=None,ow=1): d.rounded_rectangle(xy,radius=r,fill=fill,outline=outline,width=ow) def draw_bars(emotions=None, emotion="", confidence=0.0, error="", debug=""): W = 600 H = 88 + 10 + 30 + 7*34 + 20 + 24 BG,PNL,BDR = (13,13,13),(22,22,22),(37,37,37) img = Image.new("RGB",(W,H),BG); d = ImageDraw.Draw(img) reg,bold,lg,ti = fonts() rr(d,(0,0,W,86),12,PNL,BDR) d.text((16,10),"EMOTION AI",font=ti,fill=(0,255,153)) if emotion == "no_face": d.text((16,24),"FACE NOT DETECTED",font=lg,fill=(255,140,40)) d.text((16,64),"Лицо не обнаружено в кадре",font=reg,fill=(160,110,40)) else: label = emotion.upper() if emotion else "—" lc = COLORS_RGB.get(emotion,(0,255,153)) d.text((16,24),label,font=lg,fill=lc) txt = f"ERR: {error[:60]}" if error else (f"Confidence: {confidence:.1f}%" if emotion else "Waiting...") d.text((16,64),txt,font=reg,fill=(255,80,80) if error else (100,100,100)) y0=96; rr(d,(0,y0,W,H),12,PNL,BDR) d.text((16,y0+10),"EMOTION SCORES",font=ti,fill=(0,255,153)) if emotion == "no_face" or not emotions: msg = "Лицо не обнаружено — нет данных" if emotion == "no_face" else "No data yet" d.text((16,y0+28),msg,font=reg,fill=(60,60,60)) return np.array(img) BX,BW,BH = 16,W-32,8; y=y0+28 for name in EMOTION_LIST: pct=emotions.get(name,0.0); c=COLORS_RGB.get(name,(0,255,153)) d.text((BX,y),name,font=reg,fill=(187,187,187)) d.text((W-BX,y),f"{pct:.1f}%",font=reg,fill=(140,140,140),anchor="ra") y+=15; rr(d,(BX,y,BX+BW,y+BH),4,(40,40,40)) fw=max(0,int(BW*min(pct,100)/100)) if fw>5: rr(d,(BX,y,BX+fw,y+BH),4,c) y+=19 if debug: y+=10 d.text((BX,y),f"DEBUG: {debug}",font=ti,fill=(0,200,255)) return np.array(img) def attention_color(score): """Green when attentive, fading through yellow to red when distracted.""" if score >= 70: return (0,255,153) elif score >= 40: return (255,200,0) else: return (255,80,80) def draw_attention_gauge(score=100.0): """ Vertical attention gauge — a tall rounded bar that fills from the bottom, green when focused on the camera, sliding to red when attention drops (gaze away / head turned / no face). """ W, H = 140, 480 BG, PNL, BDR = (13,13,13), (22,22,22), (37,37,37) img = Image.new("RGB", (W,H), BG) d = ImageDraw.Draw(img) reg,bold,lg,ti = fonts() rr(d,(0,0,W,H),12,PNL,BDR) d.text((W//2,16),"ВНИМАНИЕ",font=ti,fill=(0,255,153),anchor="mt") # Big percentage number near the top color = attention_color(score) d.text((W//2,36),f"{score:.0f}%",font=lg,fill=color,anchor="mt") # Vertical track track_x0, track_x1 = W//2 - 22, W//2 + 22 track_y0, track_y1 = 92, H - 36 track_h = track_y1 - track_y0 rr(d,(track_x0,track_y0,track_x1,track_y1),18,(38,38,38),BDR,1) # Filled portion, from the bottom up fill_h = int(track_h * min(max(score,0),100) / 100) if fill_h > 4: fy0 = track_y1 - fill_h rr(d,(track_x0+3,fy0,track_x1-3,track_y1-3),16,color) # Tick labels down the side for pct in (100,75,50,25,0): ty = track_y1 - int(track_h * pct/100) d.text((track_x0-8,ty),f"{pct}",font=ti,fill=(90,90,90),anchor="rm") # Caption at the bottom if score >= 70: caption = "Сфокусирован" elif score >= 40: caption = "Отвлекается" else: caption = "Не смотрит" d.text((W//2,H-22),caption,font=reg,fill=color,anchor="mt") return np.array(img) def draw_chart(): W,H=600,260; BG,PNL,BDR=(13,13,13),(22,22,22),(37,37,37) PL,PR,PT,PB=70,14,44,30; cw,ch=W-PL-PR,H-PT-PB n_levels = len(CHART_LIST)-1 img=Image.new("RGB",(W,H),BG); d=ImageDraw.Draw(img) reg,bold,lg,ti=fonts() rr(d,(0,0,W,H),12,PNL,BDR) d.text((16,12),"EMOTION TIMELINE",font=ti,fill=(0,255,153)) for i,name in enumerate(CHART_LIST): py=int(PT+(1-i/n_levels)*ch) d.line([(PL,py),(PL+cw,py)],fill=(30,30,30),width=1) d.text((PL-5,py),name,font=ti,fill=(90,90,90),anchor="rm") d.line([(PL,PT),(PL,PT+ch)],fill=(55,55,55),width=1) d.line([(PL,PT+ch),(PL+cw,PT+ch)],fill=(55,55,55),width=1) history=log[-40:]; n=len(history) if n>=2: pts=[(int(PL+(i/(n-1))*cw), int(PT+(1-x.get("chart_value",x["value"])/n_levels)*ch), x.get("chart_emotion",x["emotion"])) for i,x in enumerate(history)] # Draw line segments, but skip any segment touching a no_face point # (leaves a visible gap instead of a misleading flat/false line). for i in range(len(pts)-1): (x0,y0,e0),(x1,y1,e1) = pts[i], pts[i+1] if e0 == "no_face" or e1 == "no_face": continue d.line([(x0,y0),(x1,y1)],fill=(0,210,120),width=2) # Dots — skip no_face entirely (no point drawn = visibly empty) for px,py,emo in pts: if emo == "no_face": continue c=COLORS_RGB.get(emo,(0,255,153)) d.ellipse([(px-4,py-4),(px+4,py+4)],fill=c,outline=(13,13,13),width=1) return np.array(img) def draw_thin_meters(emotions=None, attention=100.0, emotion="", debug=""): """ A single slim horizontal strip with thin progress bars for every emotion plus the attention level, MorphCast-style: small label, tiny percentage, a hairline track, and a bright fill. Meant to sit along the bottom of the page as a persistent always-visible readout, rather than the bigger boxed panel. """ emotions = emotions or {} items = [(name, emotions.get(name, 0.0), COLORS_RGB.get(name,(0,255,153))) for name in EMOTION_LIST] items.append(("attention", attention, attention_color(attention))) W = 760 PAD_X, PAD_Y = 18, 14 ROW_H = 28 H = PAD_Y*2 + ROW_H*len(items) + (14 if debug else 0) BG, BDR = (15,15,15), (34,34,34) img = Image.new("RGB", (W,H), BG) d = ImageDraw.Draw(img) reg, bold, lg, ti = fonts() rr(d, (0,0,W,H), 10, BG, BDR, 1) LABEL_W = 92 PCT_W = 46 TRACK_X0 = PAD_X + LABEL_W TRACK_X1 = W - PAD_X - PCT_W TRACK_W = TRACK_X1 - TRACK_X0 TRACK_H = 5 y = PAD_Y for name, pct, color in items: label = "attention" if name == "attention" else name cy = y + ROW_H//2 # Label, left-aligned d.text((PAD_X, cy), label, font=reg, fill=(190,190,190), anchor="lm") # Thin hairline track track_y0 = cy - TRACK_H//2 track_y1 = cy + TRACK_H//2 rr(d, (TRACK_X0, track_y0, TRACK_X1, track_y1), TRACK_H//2, (40,40,40)) # Fill — width proportional to value, capped at 100 fw = int(TRACK_W * min(max(pct,0),100) / 100) if fw > TRACK_H: rr(d, (TRACK_X0, track_y0, TRACK_X0+fw, track_y1), TRACK_H//2, color) # Small bright dot at the leading edge, like a slider handle hx = TRACK_X0 + fw r = 5 d.ellipse((hx-r, cy-r, hx+r, cy+r), fill=color) # Percentage, right-aligned d.text((W-PAD_X, cy), f"{pct:.0f}%", font=reg, fill=(150,150,150), anchor="rm") y += ROW_H if debug: d.text((PAD_X, y+2), f"DEBUG: {debug}", font=ti, fill=(0,200,255)) return np.array(img) _display_smooth = {e: 0.0 for e in EMOTION_LIST} DISPLAY_SMOOTH_FACTOR = 0.18 _last_reliable_emotion = "neutral" _last_reliable_adj_scores = {e: 0.0 for e in EMOTION_LIST} _no_face_buf = [] NO_FACE_BUF_SIZE = 5 # need 4/5 recent frames to agree face is missing def format_log_time(video_time_sec=None): """ Returns the timestamp string to store in a log entry. - Live camera: real wall-clock time (HH:MM:SS) — useful for live sessions. - Video analysis: time relative to the start of the video (MM:SS) — far more useful than the wall-clock time the video happened to be uploaded at. """ if video_time_sec is not None: m = int(video_time_sec) // 60 s = int(video_time_sec) % 60 return f"{m:02d}:{s:02d}" return datetime.now().strftime("%H:%M:%S") def analyze_frame(frame, video_time_sec=None): global log, _display_smooth, _last_reliable_emotion, _last_reliable_adj_scores if frame is None: return draw_bars(),draw_chart(),draw_attention_gauge(attention_score),np.zeros((10,10,3),dtype=np.uint8),frame if frame is not None else np.zeros((10,10,3),dtype=np.uint8) try: gaze = analyze_eyes(frame) raw_no_face = gaze.get("no_face", True) if gaze else True # ── Smooth the no_face signal: a single missed detection # (motion blur, brief occlusion, bad lighting on one frame) # shouldn't instantly flip to "face not detected". Require a # majority of recent frames to agree before declaring it. _no_face_buf.append(raw_no_face) if len(_no_face_buf) > NO_FACE_BUF_SIZE: _no_face_buf.pop(0) no_face = sum(_no_face_buf) >= 4 if len(_no_face_buf) >= NO_FACE_BUF_SIZE else raw_no_face and sum(_no_face_buf) == len(_no_face_buf) # ── No face at all: show explicit "face not detected" state ── if no_face: recent_emotions.clear() # don't let old emotion bleed back in once face returns attn = update_attention(gaze) for k in EMOTION_LIST: _display_smooth[k] = 0.0 # clear bars while face is absent log.append({ "time": format_log_time(video_time_sec), "emotion": "no_face", "value": 0, "confidence": 0.0, "chart_emotion": "no_face", "chart_value": 0, "attention": round(attn,1), "all_scores": {}, }) return (draw_bars(emotion="no_face", confidence=0.0, debug=last_gaze_debug, error="Лицо не обнаружено"), draw_chart(), draw_attention_gauge(attn), frame) face_reliable = gaze.get("face_reliable", True) # ── Detect ALL faces in the frame ─────────────────────────────── all_face_boxes = detect_all_faces(frame) n_faces = max(1, len(all_face_boxes)) global _max_faces_seen _max_faces_seen = max(_max_faces_seen, n_faces) # Analyse each face separately, then average scores for overall bars. per_face_emotions = [] new_face_emotions = {} if face_reliable and all_face_boxes: H, W = frame.shape[:2] for fi, (fx, fy, fw, fh) in enumerate(all_face_boxes): # Crop face with small padding for better ViT accuracy pad = int(min(fw, fh) * 0.1) x0 = max(0, fx-pad); y0 = max(0, fy-pad) x1 = min(W, fx+fw+pad); y1 = min(H, fy+fh+pad) face_crop = frame[y0:y1, x0:x1] if face_crop.size == 0: continue raw_f = predict_emotions_vit(face_crop) adj_f = apply_gaze_modifiers(dict(raw_f), gaze) adj_f = adjust(adj_f) per_face_emotions.append(adj_f) # Per-face smoother if fi not in _face_smoothers: _face_smoothers[fi] = {e: 0.0 for e in EMOTION_LIST} for k in EMOTION_LIST: _face_smoothers[fi][k] = (_face_smoothers[fi][k]*(1-FACE_SMOOTH) + adj_f.get(k,0)*FACE_SMOOTH) new_face_emotions[fi] = max(adj_f, key=adj_f.get) elif face_reliable: # No boxes found from detector but gaze says reliable → full frame raw_emotions = predict_emotions_vit(frame) adj_emotions_single = apply_gaze_modifiers(dict(raw_emotions), gaze) adj_emotions_single = adjust(adj_emotions_single) per_face_emotions.append(adj_emotions_single) # Keep _face_emotions pruned to active faces _face_emotions.clear() _face_emotions.update(new_face_emotions) # Remove smoothers for faces no longer visible for k in list(_face_smoothers.keys()): if k >= n_faces: del _face_smoothers[k] # Aggregate: average across all detected faces if per_face_emotions: adj_emotions = {k: float(np.mean([f.get(k,0) for f in per_face_emotions])) for k in EMOTION_LIST} elif face_reliable: adj_emotions = _last_reliable_adj_scores else: adj_emotions = _last_reliable_adj_scores if face_reliable and per_face_emotions: emotion = smooth(max(adj_emotions, key=adj_emotions.get)) confidence = float(adj_emotions.get(emotion, 0.0)) _last_reliable_emotion = emotion _last_reliable_adj_scores = dict(adj_emotions) else: emotion = _last_reliable_emotion confidence = float(_last_reliable_adj_scores.get(emotion, 0.0)) recent_emotions.append(emotion) if len(recent_emotions) > SMOOTH_WINDOW: recent_emotions.pop(0) for k in EMOTION_LIST: target = adj_emotions.get(k, 0.0) _display_smooth[k] = (_display_smooth[k]*(1-DISPLAY_SMOOTH_FACTOR) + target*DISPLAY_SMOOTH_FACTOR) attn = update_attention(gaze) log.append({ "time": format_log_time(video_time_sec), "emotion": emotion, "value": emotion_map[emotion], "confidence": round(confidence,1), "chart_emotion": emotion, "chart_value": emotion_map[emotion], "attention": round(attn,1), "n_faces": n_faces, "all_scores": {k: round(v,1) for k,v in adj_emotions.items()}, }) smoothed = {k:round(v,1) for k,v in _display_smooth.items()} # Optionally overlay face boxes on the camera image display_frame = frame if _show_face_boxes and all_face_boxes: display_frame = draw_face_boxes(frame, all_face_boxes, _face_emotions) return (draw_bars(smoothed, emotion, confidence, debug=f"{n_faces} face(s) | {last_gaze_debug}"), draw_chart(), draw_attention_gauge(attn), display_frame) except Exception as e: return draw_bars(error=str(e)),draw_chart(),draw_attention_gauge(attention_score),frame def reset_log(): global log,attention_score,_away_since log=[]; recent_emotions.clear() attention_score=100.0; _away_since=None _no_face_buf.clear() _face_smoothers.clear(); _face_emotions.clear() global _max_faces_seen; _max_faces_seen = 0 return draw_bars(),draw_chart(),draw_attention_gauge(attention_score),np.zeros((10,10,3),dtype=np.uint8) def save_csv(): if not log: return None path="/tmp/emotion_log.csv" fields = ["time","emotion","confidence","value","chart_emotion","attention"] with open(path,"w",newline="") as f: w=csv.DictWriter(f,fieldnames=fields,extrasaction="ignore") w.writeheader() for e in log: w.writerow(e) return path MUSIC_HTML = """
🎵 Music Player
no file selected
""" def process_video(video_path, progress=gr.Progress()): """ Process an uploaded video file frame by frame. Samples ~2 frames per second, runs the same emotion + gaze pipeline as live camera, returns final bars/chart/attention + CSV. """ global log if video_path is None: return draw_bars(), draw_chart(), draw_attention_gauge(attention_score), None # gr.Video can hand us a plain path string, or (in some versions) a # dict/object with a "path"/"name" attribute. Normalize to a string. if isinstance(video_path, str): path_str = video_path elif isinstance(video_path, dict): path_str = video_path.get("path") or video_path.get("name") else: path_str = getattr(video_path, "path", None) or getattr(video_path, "name", None) if not path_str: return draw_bars(error="no valid video path"), draw_chart(), draw_attention_gauge(attention_score), None try: log = [] recent_emotions.clear() cap = cv2.VideoCapture(path_str) if not cap.isOpened(): return draw_bars(error=f"cannot open video: {path_str}"), draw_chart(), draw_attention_gauge(attention_score), None fps = cap.get(cv2.CAP_PROP_FPS) or 25 total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) or 1 sample_every = max(1, int(fps / 2)) # ~2 samples per second frame_idx = 0 last_bars, last_chart, last_attn = draw_bars(), draw_chart(), draw_attention_gauge(attention_score) while True: ret, frame_bgr = cap.read() if not ret: break if frame_idx % sample_every == 0: frame_rgb = cv2.cvtColor(frame_bgr, cv2.COLOR_BGR2RGB) video_time_sec = frame_idx / fps try: last_bars, last_chart, last_attn = analyze_frame(frame_rgb, video_time_sec=video_time_sec) except Exception as e: last_bars = draw_bars(error=str(e)) pct = frame_idx / max(total_frames, 1) progress(pct, desc=f"Analyzing frame {frame_idx}/{total_frames}") frame_idx += 1 cap.release() csv_path = save_csv() return last_bars, last_chart, last_attn, csv_path except Exception as e: return (draw_bars(error=f"video processing failed: {e}"), draw_chart(), draw_attention_gauge(attention_score), None) import random # ── Phrase banks for natural-sounding local summary generation ────── _OPENERS = { "neutral": [ "За это время лицо в основном оставалось спокойным — без резких эмоциональных всплесков.", "Сессия прошла довольно ровно: преобладало нейтральное, расслабленное выражение лица.", "Большую часть времени эмоции были сдержанными — ни напряжения, ни выраженной радости не было.", "На протяжении сессии лицо оставалось преимущественно спокойным, без явных перепадов настроения.", "Заметна общая безмятежность — выражение лица почти не менялось, держась в нейтральном диапазоне.", ], "happy": [ "Настроение было хорошим — на лице большую часть времени читалась радость или лёгкая улыбка.", "Сессия прошла на позитивной волне: радость заметно преобладала над остальными эмоциями.", "В основном фиксировалось хорошее настроение — улыбка или удовлетворённое выражение лица.", "Похоже, было что-то приятное — позитивный настрой держался почти всё время.", "Лицо чаще выражало радость, чем что-либо ещё — общий тон сессии получился светлым.", ], "sad": [ "Чаще проявлялась грусть или задумчивость — выражение лица было приглушённым.", "Заметно преобладало подавленное настроение — взгляд часто был опущен, лицо выглядело уставшим.", "На протяжении сессии чаще читалась печаль, чем что-то более позитивное.", "Тон сессии получился скорее меланхоличным — выражение лица большую часть времени было невесёлым.", ], "angry": [ "Чаще всего фиксировалось напряжённое, раздражённое выражение лица.", "Заметна была некоторая напряжённость — мимика чаще указывала на раздражение, чем на спокойствие.", "Лицо большую часть времени выглядело недовольным или напряжённым.", "Преобладало выражение лёгкой злости или досады — что-то явно вызывало раздражение.", ], "surprise": [ "Удивление было самой частой реакцией — похоже, происходило что-то неожиданное.", "На лице часто читалось удивление — возможно, что-то привлекало внимание или озадачивало.", "Заметна постоянная реакция удивления — будто что-то регулярно заставало врасплох.", ], "fear": [ "Чаще всего фиксировалась настороженность или лёгкое беспокойство.", "Заметна была некоторая тревожность в выражении лица большую часть сессии.", "Преобладало напряжённое, слегка встревоженное выражение лица.", ], "disgust": [ "Чаще проявлялось недовольное или брезгливое выражение лица.", "Заметна была лёгкая гримаса неприязни, повторявшаяся на протяжении сессии.", ], } # Independent connective phrases used to stitch the percentage clause onto # the opener in different ways, so two sessions with the same top emotion # don't always read identically. EMOTION_RU = { "neutral": "нейтральное состояние", "happy": "радость", "sad": "грусть", "angry": "раздражение", "surprise": "удивление", "fear": "тревожность", "disgust": "недовольство", } _PCT_TEMPLATES = [ "{emo_ru_cap} занимало примерно {pct:.0f}% времени — около {sec:.0f} секунд из {total} отслеженных моментов.", "В цифрах это выглядит так: {emo_ru} приходится на {pct:.0f}% сессии (~{sec:.0f} сек из {total} замеров).", "По времени {emo_ru} набрало около {pct:.0f}% — то есть порядка {sec:.0f} секунд из {total} зафиксированных кадров.", "Если считать в процентах — {emo_ru} занимало {pct:.0f}% всей сессии, около {sec:.0f} секунд из {total} наблюдений.", ] _PEAK_INTROS = [ "Среди прочего выделяются такие моменты:", "Самые яркие точки сессии пришлись на:", "Стоит особо отметить такие всплески:", "Вот несколько моментов, которые выбивались из общей картины:", "Заметные пики эмоций зафиксированы здесь:", ] # Attention-drop phrasing is split by BOTH count and total duration so that # "1 раз" never gets paired with a "часто отвлекался" template by mistake. _ATTN_DROP_MANY = [ "Внимание заметно гуляло: взгляд уходил от камеры {count} раз, в сумме это заняло около {sec:.0f} секунд.", "Стоит отметить частые потери концентрации — взгляд отрывался от камеры {count} раз, суммарно почти {sec:.0f} секунд.", "Фокус на камере держался не очень стабильно: {count} заметных провалов внимания общей продолжительностью {sec:.0f} секунд.", "Внимание прыгало туда-сюда — {count} раз взгляд уходил в сторону, на это ушло около {sec:.0f} секунд в сумме.", ] _ATTN_DROP_ONE_LONG = [ "Был один заметный провал внимания — взгляд ушёл от камеры почти на {sec:.0f} секунд.", "Зафиксирован один продолжительный момент, когда внимание явно рассеялось (~{sec:.0f} сек), в остальном всё было стабильно.", "Случился один длинный отрыв внимания на {sec:.0f} секунд, но за пределами этого момента взгляд держался уверенно.", ] _ATTN_DROP_LOW = [ "Взгляд лишь пару раз отходил от камеры (всего {sec:.0f} сек) — в целом внимание оставалось сосредоточенным.", "Было короткое отвлечение (~{sec:.0f} сек), но в остальном фокус держался стабильно.", "Пара коротких моментов потери концентрации (суммарно {sec:.0f} сек) не сильно повлияли на общую картину.", ] _ATTN_DROP_NONE = [ "Внимание всё время оставалось сосредоточенным на камере — провалов концентрации не зафиксировано.", "Взгляд стабильно был направлен в камеру на протяжении всей сессии.", "Концентрация держалась на высоком уровне без единого заметного отвлечения.", ] _NOFACE_PHRASES_HIGH = [ "Лицо пропадало из кадра {count} раз, суммарно почти на {sec:.0f} секунд — похоже, ты периодически выходил из зоны видимости камеры.", "Камера {count} раз теряла лицо из виду на общую сумму около {sec:.0f} секунд — вероятно, были перерывы или движение в сторону.", ] _NOFACE_PHRASES_LOW = [ "Был короткий момент ({sec:.0f} сек), когда лицо выходило из кадра, но это не сильно повлияло на общую картину.", "Лицо ненадолго пропадало из вида (~{sec:.0f} сек), но в остальном оставалось в кадре.", ] _NOFACE_PHRASES_NONE = [ "Лицо оставалось в кадре на протяжении всей сессии.", "Камера непрерывно видела лицо — выходов из кадра не было.", ] _CLOSERS = [ "В целом картина выглядит достаточно ровной и предсказуемой.", "Общая динамика эмоций за сессию получилась довольно цельной.", "Сессия в целом получилась показательной с точки зрения эмоционального фона.", "По совокупности данных сессия выглядит вполне стабильной.", "Если смотреть на всё вместе, картина получилась достаточно последовательной.", "Эмоциональный фон сессии в целом не преподнёс особых сюрпризов.", ] _DYNAMICS_STABLE = [ "Настроение при этом держалось довольно равномерно — заметных скачков от начала и до конца сессии не было.", "Если смотреть на динамику, эмоциональный фон не менялся резко — всё развивалось плавно.", "Стоит отметить и стабильность во времени: эмоция в начале и в конце сессии практически совпадает.", ] _DYNAMICS_SHIFT = [ "Интересно, что к концу сессии настроение заметно изменилось по сравнению с началом — от {start} к {end}.", "Динамика получилась не совсем равномерной: началось с {start}, а закончилось уже на {end}.", "Стоит отдельно отметить смену настроения во времени — сессия началась с {start}, а завершилась состоянием {end}.", ] _INTERPRETATIONS = { "neutral": [ "Такой фон часто говорит о спокойной, рутинной обстановке без сильных раздражителей.", "Возможно, ситуация была привычной или не требующей эмоционального вовлечения.", ], "happy": [ "Это может говорить о комфортной обстановке или приятном занятии в этот момент.", "Похоже, происходящее вызывало искренний интерес или удовольствие.", ], "sad": [ "Это может быть признаком усталости, или просто задумчивого состояния в моменте.", "Стоит иметь в виду, что устойчивая грусть иногда сигнализирует об усталости или сниженном настроении.", ], "angry": [ "Возможно, что-то конкретное вызывало раздражение — стоит обратить внимание, что именно.", "Такой паттерн часто указывает на источник стресса в моменте записи.", ], "surprise": [ "Частое удивление может говорить о новой, нестандартной ситуации или неожиданных событиях.", ], "fear": [ "Повышенная настороженность иногда указывает на дискомфорт в обстановке или неопределённость.", ], "disgust": [ "Такая реакция обычно появляется при явном несогласии с чем-то происходящим.", ], } def generate_local_summary(stats: dict, log_data: list) -> str: """ Builds a natural-sounding multi-sentence summary purely from local Python logic — no external API calls, works instantly and offline. Variety comes from combining several INDEPENDENT phrase banks (opener, percentage clause, dynamics clause, interpretation, attention-drop clause, no-face clause, closer) — each picked with its own random draw — rather than one fixed sentence per topic. The combinatorics across all banks gives thousands of distinct readings, while a given session always reproduces the same text (seeded by its own data). """ seed_val = (stats["total"] * 7 + int(stats["top_pct"]) * 3 + stats["attn_drop_count"] * 11 + stats["noface_count"] * 13) rng = random.Random(seed_val) top_emotion = stats["top_emotion"] opener_bank = _OPENERS.get(top_emotion, _OPENERS["neutral"]) opener = rng.choice(opener_bank) emo_ru = EMOTION_RU.get(top_emotion, top_emotion) pct_line = rng.choice(_PCT_TEMPLATES).format( pct=stats["top_pct"], sec=stats["top_seconds"], total=stats["total"], emo_ru=emo_ru, emo_ru_cap=emo_ru[0].upper()+emo_ru[1:] ) # ── Dynamics: compare emotion at the start vs the end of the session ── real_entries = [e for e in log_data if e["emotion"] != "no_face"] dynamics_line = "" if len(real_entries) >= 6: first_chunk = real_entries[:max(1, len(real_entries)//4)] last_chunk = real_entries[-max(1, len(real_entries)//4):] start_counts = {} for e in first_chunk: start_counts[e["emotion"]] = start_counts.get(e["emotion"],0)+1 end_counts = {} for e in last_chunk: end_counts[e["emotion"]] = end_counts.get(e["emotion"],0)+1 start_emo = max(start_counts, key=start_counts.get) end_emo = max(end_counts, key=end_counts.get) if start_emo != end_emo: dynamics_line = rng.choice(_DYNAMICS_SHIFT).format(start=start_emo, end=end_emo) else: dynamics_line = rng.choice(_DYNAMICS_STABLE) # ── Interpretive layer: a short "what this might mean" remark ── interp_bank = _INTERPRETATIONS.get(top_emotion, []) interp_line = rng.choice(interp_bank) if interp_bank else "" # Peak moments — only mention there WAS variety, without listing every # timestamp (those now live only in the History buttons, not here). peaks = stats["peaks_list"] peak_block = "" unique_peak_emotions = {p["emotion"] for p in peaks} if len(unique_peak_emotions) >= 2: other_emotions = [p["emotion"] for p in peaks if p["emotion"] != top_emotion] if other_emotions: peak_intro = rng.choice(_PEAK_INTROS) names = ", ".join(other_emotions[:2]) peak_block = f"{peak_intro} помимо основного фона также проявлялись моменты {names}." # Attention drops ac, asec = stats["attn_drop_count"], stats["attn_drop_sec"] if ac == 0: attn_line = rng.choice(_ATTN_DROP_NONE) elif ac == 1 and asec >= 10: attn_line = rng.choice(_ATTN_DROP_ONE_LONG).format(sec=asec) elif ac >= 2 and asec >= 10: attn_line = rng.choice(_ATTN_DROP_MANY).format(count=ac, sec=asec) else: attn_line = rng.choice(_ATTN_DROP_LOW).format(sec=asec) # No face nc, ns = stats["noface_count"], stats["noface_sec"] if nc == 0: noface_line = rng.choice(_NOFACE_PHRASES_NONE) elif ns >= 8: noface_line = rng.choice(_NOFACE_PHRASES_HIGH).format(count=nc, sec=ns) else: noface_line = rng.choice(_NOFACE_PHRASES_LOW).format(count=nc, sec=ns) closer = rng.choice(_CLOSERS) # ── Paragraph 1: what happened, in numbers and in plain language ── para1_parts = [opener, pct_line] if dynamics_line: para1_parts.append(dynamics_line) if peak_block: para1_parts.append(peak_block) # Multi-face mention in first paragraph max_faces = stats.get("max_faces", 1) if max_faces > 1: faces_line = rng.choice([ f"При этом в кадре одновременно находилось до {max_faces} человек — анализ усреднён по всем лицам.", f"Стоит отметить, что в кадре было до {max_faces} человек, поэтому данные отражают общую картину группы.", ]) para1_parts.append(faces_line) paragraph1 = " ".join(para1_parts) # ── Paragraph 2: interpretation + attention/face tracking + closer ── para2_parts = [] if interp_line: para2_parts.append(interp_line) para2_parts.append(attn_line) para2_parts.append(noface_line) para2_parts.append(closer) paragraph2 = " ".join(para2_parts) return paragraph1 + "\n\n" + paragraph2 def parse_log_time_to_seconds(time_str): """ Parses a log timestamp into total seconds, supporting both formats: - "HH:MM:SS" (live camera, wall-clock time) - "MM:SS" (video analysis, relative to video start) """ parts = time_str.split(":") if len(parts) == 3: h, m, s = parts return int(h)*3600 + int(m)*60 + int(s) elif len(parts) == 2: m, s = parts return int(m)*60 + int(s) return 0 def render_summary(): """ Build an 8-line-max textual summary of the whole session: - most common emotion + how long it lasted - top 3 peak (most intense) moments and when they happened """ if not log: return ('
' 'Нет данных для анализа — запусти камеру или проанализируй видео.
') from collections import Counter # ── Time spent per emotion ────────────────────────────── # Exclude "no_face" frames from the emotion breakdown — those aren't # an emotion, they're an absence, and would otherwise dilute the # percentages of real emotions. real_entries = [e for e in log if e["emotion"] != "no_face"] counts = Counter(e["emotion"] for e in real_entries) total = len(real_entries) if total == 0: return ('
' 'Лицо не было обнаружено за всё время сессии.
') top_emotion, top_count = counts.most_common(1)[0] top_pct = (top_count / total) * 100 # Rough seconds estimate: count * average sampling interval avg_interval = 0.8 if len(log) >= 2: try: t0 = parse_log_time_to_seconds(log[0]["time"]) t1 = parse_log_time_to_seconds(log[-1]["time"]) span = t1 - t0 if span > 0: avg_interval = span / max(len(log) - 1, 1) except Exception: pass top_seconds = top_count * avg_interval # ── Peak moments — best moment PER UNIQUE EMOTION, not top-3 by raw # confidence. This avoids showing "neutral, neutral, neutral" when # one emotion dominates; instead shows variety across what actually # happened (e.g. the single best surprise moment, best happy moment). best_per_emotion = {} for e in real_entries: emo = e["emotion"] if emo not in best_per_emotion or e["confidence"] > best_per_emotion[emo]["confidence"]: best_per_emotion[emo] = e peaks = sorted(best_per_emotion.values(), key=lambda e: e["confidence"], reverse=True)[:3] peak_lines = "" for p in peaks: c = COLORS_RGB.get(p["emotion"], (0,255,153)) hex_c = "#%02x%02x%02x" % c peak_lines += (f'
' f'{p["time"]} — ' f'{p["emotion"].upper()} ' f'({p["confidence"]:.1f}%)
') top_hex = "#%02x%02x%02x" % COLORS_RGB.get(top_emotion, (0,255,153)) # ── Distraction episodes — track "low attention" and "no_face" SEPARATELY ── def count_episodes(predicate): episodes = [] run = 0 for e in log: if predicate(e): run += 1 else: if run > 0: episodes.append(run) run = 0 if run > 0: episodes.append(run) return episodes attn_drop_episodes = count_episodes(lambda e: e.get("attention", 100) < 40) noface_episodes = count_episodes(lambda e: e["emotion"] == "no_face") attn_drop_count = len(attn_drop_episodes) attn_drop_sec = sum(attn_drop_episodes) * avg_interval attn_drop_avg = (attn_drop_sec / attn_drop_count) if attn_drop_count else 0.0 noface_count = len(noface_episodes) noface_sec = sum(noface_episodes) * avg_interval noface_avg = (noface_sec / noface_count) if noface_count else 0.0 attn_drop_line = ( f'
' f'Падений внимания (смотрел в сторону): {attn_drop_count} раз, ' f'суммарно ~{attn_drop_sec:.0f} сек (в среднем {attn_drop_avg:.0f} сек)
' ) if attn_drop_count else ( '
Падений внимания не обнаружено
' ) noface_line = ( f'
' f'Лицо не было видно: {noface_count} раз, ' f'суммарно ~{noface_sec:.0f} сек (в среднем {noface_avg:.0f} сек)
' ) if noface_count else ( '
Лицо было видно всё время
' ) # ── Generate natural-sounding summary locally (no API needed) ── stats = { "top_emotion": top_emotion, "top_pct": top_pct, "top_seconds": top_seconds, "total": total, "peaks_list": peaks, "attn_drop_count": attn_drop_count, "attn_drop_sec": attn_drop_sec, "noface_count": noface_count, "noface_sec": noface_sec, "max_faces": _max_faces_seen, } narrative = generate_local_summary(stats, log) # Multi-face note for the detail section if _max_faces_seen > 1: faces_note = (f'
' f'Максимум лиц в кадре одновременно: ' f'{_max_faces_seen}
') else: faces_note = "" body_html = (f'
{narrative}
' f'{faces_note}') return f"""
📊 Итоговый анализ
{body_html}
""" def render_chart_history(): if not log: return ('
' 'История графика пуста — запусти камеру или проанализируй видео.
') rows = "" for entry in reversed(log[-200:]): chart_emo = entry.get("chart_emotion", entry["emotion"]) color = COLORS_RGB.get(chart_emo, (0,255,153)) hex_c = "#%02x%02x%02x" % color conf = entry["confidence"] rows += f""" {entry['time']} {chart_emo.upper()} {conf:.1f}% """ return f"""
История графика — пиковые эмоции ({len(log)} точек)
{rows}
Время Эмоция Сила
""" def render_full_history(): if not log: return ('
' 'Полная история пуста — запусти камеру или проанализируй видео.
') blocks = "" for entry in reversed(log[-100:]): scores = dict(entry.get("all_scores", {})) bars = "" for name in CHART_LIST: pct = scores.get(name, 0.0) c = COLORS_RGB.get(name, (0,255,153)) hex_c = "#%02x%02x%02x" % c bars += f"""
{name}
{pct:.1f}%
""" winner_color = COLORS_RGB.get(entry["emotion"], (0,255,153)) winner_hex = "#%02x%02x%02x" % winner_color blocks += f"""
{entry['time']} {entry['emotion'].upper()}
{bars}
""" return f"""
Полная история — все эмоции по секундам ({len(log)} записей)
{blocks}
""" css = """ .gradio-container{background:#0d0d0d!important} footer{display:none!important} [data-testid="image"]{border:none!important;background:transparent!important;padding:0!important;box-shadow:none!important} [data-testid="image"]>div{border:none!important;background:transparent!important;padding:0!important} [data-testid="image"] img{border:none!important;width:100%!important;display:block;border-radius:0!important} """ with gr.Blocks(title="Emotion AI") as demo: with gr.Sidebar(label="Controls", open=True): gr.HTML('
⚙️ Controls
') with gr.Row(): reset_btn = gr.Button("🔄 Reset", variant="stop") csv_btn = gr.Button("💾 Save CSV", variant="primary") with gr.Row(): chart_hist_btn = gr.Button("📈 История графика", variant="secondary") full_hist_btn = gr.Button("📜 Полная история", variant="secondary") summary_btn = gr.Button("📊 Итоговый анализ", variant="secondary") face_box_btn = gr.Button("🔲 Показать лица: OFF", variant="secondary") csv_out = gr.File(label="Download CSV") with gr.Row(): with gr.Column(scale=1): with gr.Tabs() as source_tabs: with gr.Tab("📷 Live Camera") as live_tab: cam = gr.Image(sources=["webcam"], streaming=True, type="numpy", label="Camera") # Overlayed camera view (shows face boxes when enabled) face_cam_out = gr.Image(label="", show_label=False, container=False, visible=False) with gr.Tab("🎬 Upload Video") as video_tab: video_in = gr.Video(label="Upload a video", sources=["upload"]) video_btn = gr.Button("▶️ Analyze Video", variant="primary") video_status = gr.Markdown( "После анализа видео графики скрываются — " "результаты смотри через кнопки **«История графика»**, " "**«Полная история»** или **«Итоговый анализ»** выше.", visible=False ) with gr.Column(scale=2) as visuals_col: bars_out = gr.Image(label="", show_label=False, container=False) chart_out = gr.Image(label="", show_label=False, container=False) history_out = gr.HTML(visible=False) with gr.Column(scale=0, min_width=160) as attention_col: attention_out = gr.Image(label="", show_label=False, container=False, value=draw_attention_gauge(100.0)) cam.stream(fn=analyze_frame, inputs=[cam], outputs=[bars_out, chart_out, attention_out, face_cam_out], stream_every=0.3, time_limit=300) # Toggle face box overlay _face_box_state = gr.State(False) def toggle_face_boxes(state): global _show_face_boxes new_state = not state _show_face_boxes = new_state label = "🔲 Показать лица: ON" if new_state else "🔲 Показать лица: OFF" return (gr.update(label=label), gr.update(visible=new_state), new_state) face_box_btn.click(fn=toggle_face_boxes, inputs=[_face_box_state], outputs=[face_box_btn, face_cam_out, _face_box_state]) def process_video_and_hide(video_path, progress=gr.Progress()): """Run the normal video pipeline, then hide the live gauges — only the history/summary buttons are meaningful for video mode.""" bars, chart, attn, _frame, csv_path = process_video(video_path, progress) return (gr.update(value=bars, visible=False), gr.update(value=chart, visible=False), gr.update(value=attn, visible=False), csv_path, gr.update(visible=True)) video_btn.click(fn=process_video_and_hide, inputs=[video_in], outputs=[bars_out, chart_out, attention_out, csv_out, video_status]) def on_live_tab_select(): """Switching back to the live camera tab restores the visual gauges.""" return (gr.update(visible=True), gr.update(visible=True), gr.update(visible=True)) live_tab.select(fn=on_live_tab_select, inputs=[], outputs=[bars_out, chart_out, attention_out]) reset_btn.click(fn=reset_log, inputs=[], outputs=[bars_out,chart_out,attention_out]) csv_btn.click(fn=save_csv, inputs=[], outputs=[csv_out]) _active_panel = gr.State("none") def show_chart_history(active): if active == "chart": return gr.update(visible=False), "none" return gr.update(value=render_chart_history(), visible=True), "chart" def show_full_history(active): if active == "full": return gr.update(visible=False), "none" return gr.update(value=render_full_history(), visible=True), "full" chart_hist_btn.click(fn=show_chart_history, inputs=[_active_panel], outputs=[history_out, _active_panel]) full_hist_btn.click(fn=show_full_history, inputs=[_active_panel], outputs=[history_out, _active_panel]) def show_summary(active): if active == "summary": return gr.update(visible=False), "none" return gr.update(value=render_summary(), visible=True), "summary" summary_btn.click(fn=show_summary, inputs=[_active_panel], outputs=[history_out, _active_panel]) if __name__=="__main__": demo.launch(css=css)