import json import math import numpy as np import pytest from look_whos_talking.config import EngineConfig from look_whos_talking.engine.engine import GazeEngine from look_whos_talking.engine.types import AudioState, RawFace, VisionFrame SIZE = (1920, 1080) def frame(centers, ts, patches=None): faces = [] for i, cx in enumerate(centers): patch = None if patches is None else patches[i] faces.append(RawFace(bbox=(cx - 100, 400, 200, 200), right_eye=(cx - 40, 470), left_eye=(cx + 40, 470), nose=(cx, 520), mouth_patch=patch)) return VisionFrame(faces=tuple(faces), ts=ts, frame_size=SIZE) def frame_with_patch(cx, ts, patch): """One face at `cx` carrying `patch` as its mouth patch.""" return frame([cx], ts, [patch]) def talking_patch(rng): return rng.integers(0, 255, (24, 32)).astype(np.uint8) def still_patch(): return (np.ones((24, 32)) * 128).astype(np.uint8) def test_solo_person_pixel_aim_with_bias(): cfg = EngineConfig.from_dict({"servo": {"aim_bias_v_frac": 0.1}}) eng = GazeEngine(cfg) out = eng.step(frame([960], ts=0.0), None, now=0.0) assert out.aim.kind == "pixel" assert abs(out.aim.u - 960) < 1e-6 assert abs(out.aim.v - (520 + 0.1 * 200)) < 1e-6 # nose_v + bias*bbox_h def test_aim_bias_scales_with_face_height_not_width(): # Wide-but-short box: pins the bias to bbox h, catching a bbox[2]/bbox[3] slip # that a square test face cannot see. cfg = EngineConfig.from_dict({"servo": {"aim_bias_v_frac": 0.5}}) eng = GazeEngine(cfg) wide = RawFace(bbox=(800, 400, 400, 100), right_eye=(920, 470), left_eye=(1000, 470), nose=(960, 520)) out = eng.step(VisionFrame(faces=(wide,), ts=0.0, frame_size=SIZE), None, now=0.0) assert abs(out.aim.v - (520 + 0.5 * 100)) < 1e-6 def test_two_people_switches_to_talker(): rng = np.random.default_rng(1) eng = GazeEngine() t = 0.0 out = None while t < 6.0: # face 2 talks, face 1 still; audio says speech f = frame([400, 1500], ts=t, patches=[still_patch(), talking_patch(rng)]) out = eng.step(f, AudioState(speech=True, doa_angle=None, ts=t), now=t) t += 0.1 snap = eng.snapshot() assert snap["target_id"] == 2 assert snap["state"] == "GROUP" assert [f["is_target"] for f in snap["faces"]] == [False, True] assert out.aim.kind == "pixel" and out.aim.u > 1000 def test_none_vision_scans(): eng = GazeEngine() out = None for i in range(60): out = eng.step(None, None, now=i * 0.1) assert out.aim.kind == "pose" def test_snapshot_json_safe(): eng = GazeEngine() eng.step(frame([960], ts=0.0), AudioState(True, 1.57, 0.0), now=0.0) snap = eng.snapshot() s = json.dumps(snap) assert "target_id" in s assert snap["audio"] == {"speech": True, "doa_angle": 1.57, "stale": False} assert snap["faces"][0]["is_target"] is True # snapshot() hands out a copy: a dashboard mutating it cannot corrupt or # tear the engine's own state. snap["faces"].append("junk") assert len(eng.snapshot()["faces"]) == 1 def test_snapshot_json_safe_under_numpy_typed_inputs(): # The vision and audio workers hand back numpy scalars (cv2 boxes, DoA # maths). json.dumps rejects every one of them, so the snapshot has to # coerce at the boundary or the dashboard 500s on live hardware. eng = GazeEngine() proto = frame([960], ts=0.0).faces[0] face = RawFace(bbox=tuple(np.float32(v) for v in proto.bbox), right_eye=proto.right_eye, left_eye=proto.left_eye, nose=proto.nose) vision = VisionFrame(faces=(face,), ts=0.0, frame_size=SIZE) audio = AudioState(speech=np.bool_(True), doa_angle=np.float32(1.57), ts=0.0) out = eng.step(vision, audio, now=np.float32(0.0)) snap = eng.snapshot() json.dumps(snap) assert type(snap["audio"]["doa_angle"]) is float assert type(out.aim.v) is float # Antennas are derived from the fused scores, so numpy bboxes/DoA leak # np.float32 all the way to the driver unless the boundary coerces them. assert [type(a) for a in out.antennas] == [float, float] def test_snapshot_is_well_formed_before_the_first_step(): snap = GazeEngine().snapshot() json.dumps(snap) assert snap["audio"] == {"speech": False, "doa_angle": None, "stale": False} assert snap["target_id"] is None assert snap["faces"] == [] and snap["events"] == [] def test_stale_audio_is_never_shown_in_the_snapshot(): # The selector discards audio older than max(vad_hold_s, 0.5); the snapshot # must not display a reading the engine refused to act on. eng = GazeEngine() eng.step(frame([960], ts=10.0), AudioState(speech=True, doa_angle=1.57, ts=0.0), now=10.0) assert eng.snapshot()["audio"] == {"speech": False, "doa_angle": None, "stale": True} def test_camera_death_degrades_to_scan_not_ghost_face(): # A stalled vision worker keeps handing back its LAST frame forever. Without # a staleness verdict the engine reads that fossil as a live face and stays # ENGAGED on a pixel that no longer exists. eng = GazeEngine() for i in range(20): # live face for 2 s ts = i * 0.1 eng.step(frame([960], ts=ts), None, now=ts) assert eng.snapshot()["state"] == "ENGAGED" last = frame([960], ts=1.9) # Camera dies: same last frame re-delivered, ts frozen at 1.9 out = None for i in range(60): # 6 s of frozen mailbox now = 2.0 + i * 0.1 out = eng.step(last, None, now=now) snap = eng.snapshot() assert snap["state"] == "ALONE" # tracks aged out; no ghost ENGAGED assert snap["faces"] == [] or all(f["misses"] > 0 for f in snap["faces"]) assert out.aim.kind == "pose" # scanning, not aiming at a fossil pixel def test_stale_audio_never_fires_sound_glance(): # The selector drops stale audio on its own, but behavior used to receive the # raw reading -- so a dead mic worker's last bearing latched the gaze while # the snapshot simultaneously reported it stale. eng = GazeEngine() fossil = AudioState(speech=True, doa_angle=math.pi, ts=0.0) events = [] for i in range(80): # 8 s alone with a 10-s-stale reading replayed now = 10.0 + i * 0.1 events.extend(eng.step(None, fossil, now=now).events) assert "sound_glance" not in events # fossil bearing must not latch the gaze snap = eng.snapshot() assert snap["audio"]["stale"] is True assert snap["audio"]["speech"] is False def test_hold_aim_passes_through_untouched(): # Target lost for less than lost_hold_s: behavior returns Aim("hold"), # meaning "keep the previous pose target". The engine must forward it # verbatim -- resolution belongs to the driver, not here. cfg = EngineConfig.from_dict({"servo": {"aim_bias_v_frac": 0.25}}) eng = GazeEngine(cfg) eng.step(frame([960], ts=0.0), None, now=0.0) out = eng.step(None, None, now=0.1) assert out.aim.kind == "hold" assert out.aim.u == 0.0 and out.aim.v == 0.0 and out.aim.pose is None def test_stale_vision_ts_reuses_tracks(): # A re-delivered camera frame (same ts) must not age tracks or re-score lips. eng = GazeEngine() eng.step(frame([400], ts=0.0), None, now=0.0) eng.step(frame([1500], ts=0.0), None, now=0.1) snap = eng.snapshot() assert [f["id"] for f in snap["faces"]] == [1] assert snap["faces"][0]["bbox"][0] == 300.0 # still the first frame's box def test_events_are_per_step_not_sticky(): eng = GazeEngine() out1 = eng.step(None, AudioState(True, 1.57, 0.0), now=0.0) assert "sound_glance" in out1.events assert eng.snapshot()["events"] == ["sound_glance"] out2 = eng.step(None, AudioState(True, 1.57, 0.1), now=0.1) assert out2.events == () assert eng.snapshot()["events"] == [] def test_hot_config_swap(): eng = GazeEngine() eng.apply_config(EngineConfig.preset("owl")) assert eng.config.speaker.min_dwell_s == 1.2 def test_apply_config_reaches_the_behavior_policy(): eng = GazeEngine() assert eng.step(frame([960], ts=0.0), None, now=0.0).antennas is not None eng.apply_config(EngineConfig.from_dict({"behavior": {"antennas_enabled": False}})) assert eng.step(frame([960], ts=0.1), None, now=0.1).antennas is None def test_apply_config_preserves_selection_unless_speaker_changed(): # SpeakerSelector has no set_config, so a *speaker-section* swap rebuilds it # and drops both the VAD hold and the current target -- accepted v1 reset, # pinned below. A servo/behavior-only edit (a dashboard slider) must NOT # trigger it: that would hand the gaze to the wrong face for a full # min_dwell_s every time an unrelated knob moves. eng = GazeEngine() right = math.pi # DoA hard right: scores face 2 above face 1 on position eng.step(frame([400], ts=0.0), AudioState(True, right, 0.0), now=0.0) assert eng.snapshot()["target_id"] == 1 # A better-scoring face 2 arrives but cannot win yet (min_dwell_s = 2.0). for i in range(1, 10): t = i * 0.1 eng.step(frame([400, 1500], ts=t), AudioState(True, right, t), now=t) snap = eng.snapshot() assert snap["target_id"] == 1 assert snap["faces"][1]["score"] > snap["faces"][0]["score"] # 2 would win a reset # Servo-only edit: dwell, target and VAD hold all survive. Ordering is # load-bearing: step with audio=None IMMEDIATELY after the swap — any # intervening speech frame would re-arm the hold in a rebuilt selector # too, making the speaking assertion pass under the bug. eng.apply_config(EngineConfig.from_dict({"servo": {"aim_bias_v_frac": 0.1}})) eng.step(frame([400, 1500], ts=1.0), None, now=1.0) assert eng.snapshot()["speaking"] is True # VAD hold carried across the swap assert eng.snapshot()["target_id"] == 1 # a rebuilt selector would adopt 2 # Speaker-section edit: the documented reset does happen. eng.apply_config(EngineConfig.from_dict({"speaker": {"min_dwell_s": 1.5}})) eng.step(frame([400, 1500], ts=1.2), AudioState(True, right, 1.2), now=1.2) assert eng.snapshot()["target_id"] == 2 # rebuilt selector re-adopts the best def test_none_vision_gap_does_not_spike_lips_on_resume(): rng = np.random.default_rng(31) eng = GazeEngine() base = (np.ones((24, 32)) * 120).astype(np.uint8) # steady still face: near-zero lips for i in range(30): ts = i * 0.1 noisy = (base.astype(np.int16) + rng.integers(-3, 4, (24, 32))).clip(0, 255).astype(np.uint8) eng.step(frame_with_patch(960, ts, noisy), None, now=ts) # camera stalls 0.4 s (within tracker coast window) for i in range(4): eng.step(None, None, now=3.0 + i * 0.1) # resumes with the subject having moved a lot during the gap moved = (np.ones((24, 32)) * 200).astype(np.uint8) eng.step(frame_with_patch(960, 3.5, moved), None, now=3.5) snap = eng.snapshot() assert snap["faces"][0]["lip"] < 0.05 # no cross-gap spike def test_none_vision_keeps_last_frame_geometry(): # A coasting track is still scored, and DoA consistency normalizes the face's # x against the frame width -- so falling back to the 1920x1080 default while # the camera is down flips a right-of-center face to left-of-center. eng = GazeEngine() right = AudioState(speech=True, doa_angle=math.pi, ts=0.0) # DoA hard right small = VisionFrame(faces=(RawFace(bbox=(500, 200, 100, 100), right_eye=(530, 230), left_eye=(570, 230), nose=(550, 260)),), ts=0.0, frame_size=(640, 480)) eng.step(small, right, now=0.0) s1 = eng.snapshot()["faces"][0]["score"] # coasting: must score in 640x480 geometry eng.step(None, AudioState(speech=True, doa_angle=math.pi, ts=0.1), now=0.1) s2 = eng.snapshot()["faces"][0]["score"] assert s2 == pytest.approx(s1, abs=0.02) # no geometry flip