speednet / modeling_speednet.py
x2q's picture
SpeedNet v5: corrected 790 m indoor supervision, multi-seed selection, per-mode evaluation, four GPS-verified demo videos
9285f34 verified
Raw
History Blame Contribute Delete
6.54 kB
"""SpeedNet v4 — ego-speed estimation from onboard/POV video.
Self-contained inference module. No project dependencies: just
torch, opencv-python, numpy (and optionally pandas for smoothing).
Pipeline
--------
1. Decode the video, resize frames to 320x180.
2. Dense optical flow (Farneback) between consecutive frames, resized to
128x72. Flow is normalized to "pixels per 1/15 s" so any source frame
rate works (multiply raw flow by fps/15).
3. A low-resolution RGB context frame (128x72, sampled ~1 Hz) feeds a small
context branch that lets the model infer scene scale (indoor kart track
vs. open highway look identical in flow magnitude but differ in scale).
4. Flow features + context embedding run through a GRU. Two inference
modes exist (sequence length acts as a scale cue for this model):
short windows (default; roads/trails/open environments) and
long_context=True (closed-course/track footage such as indoor karting).
Typical usage
-------------
from modeling_speednet import SpeedNet, predict_video
import torch
model = SpeedNet()
model.load_state_dict(torch.load("speednet_v5.pt", map_location="cuda"))
result = predict_video(model, "my_onboard_clip.mp4", device="cuda")
# result["t"] -> timestamps in seconds
# result["speed_mps"]-> speed in m/s (multiply by 3.6 for km/h)
"""
import numpy as np
import torch
import torch.nn as nn
FLOW_SCALE = 8.0 # flow normalization divisor (px per 1/15 s)
SPEED_SCALE = 40.0 # model output * SPEED_SCALE = m/s
BASE_FPS = 15.0 # flow convention: pixels per 1/15 s
class SpeedNet(nn.Module):
"""Flow CNN + RGB context branch -> GRU -> speed (v4 architecture)."""
def __init__(self, hidden=192, ctx_dim=64):
super().__init__()
def blk(i, o):
return [nn.Conv2d(i, o, 3, 2, 1), nn.BatchNorm2d(o), nn.SiLU()]
self.cnn = nn.Sequential(*blk(2, 32), *blk(32, 64), *blk(64, 128),
*blk(128, 256), nn.AdaptiveAvgPool2d(1),
nn.Flatten())
self.ctx_cnn = nn.Sequential(*blk(3, 16), *blk(16, 32), *blk(32, 64),
nn.AdaptiveAvgPool2d(1), nn.Flatten(),
nn.Linear(64, ctx_dim), nn.SiLU())
self.gru = nn.GRU(256 + ctx_dim, hidden, num_layers=2,
batch_first=True)
self.head = nn.Linear(hidden, 1)
def forward(self, flow, ctx, h0=None):
"""flow: [B, T, 2, 72, 128] (normalized), ctx: [B, 3, 72, 128]
(RGB/255 - 0.5). Returns (speed [B, T] normalized, hidden)."""
B, T = flow.shape[:2]
f = self.cnn(flow.flatten(0, 1)).view(B, T, -1)
c = self.ctx_cnn(ctx)[:, None, :].expand(B, T, -1)
y, h = self.gru(torch.cat([f, c], -1), h0)
return self.head(y).squeeze(-1), h
def extract_flow_and_ctx(video_path, max_seconds=None):
"""Decode video -> (flow [N,72,128,2] float32 normalized to 15 fps
convention, ctx [M,72,128,3] uint8 at ~1 Hz, fps)."""
import cv2
cap = cv2.VideoCapture(video_path)
fps = cap.get(cv2.CAP_PROP_FPS) or BASE_FPS
fps_scale = fps / BASE_FPS
ctx_step = max(1, int(round(fps))) # ~1 Hz
prev, flows, ctxs = None, [], []
i = 0
while True:
ok, frame = cap.read()
if not ok:
break
if max_seconds and i / fps > max_seconds:
break
if frame.shape[:2] != (180, 320):
frame = cv2.resize(frame, (320, 180))
if i % ctx_step == 0:
ctxs.append(cv2.cvtColor(cv2.resize(frame, (128, 72)),
cv2.COLOR_BGR2RGB))
g = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
if prev is not None:
fl = cv2.calcOpticalFlowFarneback(prev, g, None,
0.5, 3, 15, 3, 5, 1.2, 0)
fl *= fps_scale
flows.append(cv2.resize(fl, (128, 72)))
prev = g
i += 1
cap.release()
return (np.stack(flows).astype(np.float32),
np.stack(ctxs).astype(np.uint8), fps)
@torch.no_grad()
def predict_speed(model, flow, ctx, fps, device="cuda", long_context=False):
"""Inference in one of two modes (see README "Inference modes"):
- short (default): hidden state reset every 16 frames. Matches the
published GPS-domain metrics (roads, trails, open environments).
- long_context=True: hidden state carried across the clip. Use for
closed-course/track footage (e.g. indoor karting) where absolute
scale must be inferred from long temporal context.
The model was trained with two window lengths, so sequence length acts
as a scale cue — pick the mode matching your footage.
"""
model.eval().to(device)
ctx_hz = max(1, int(round(fps)))
reset = 10**9 if long_context else 16
out = np.empty(len(flow), np.float32)
for r0 in range(0, len(flow), reset):
seg = flow[r0:r0 + reset]
h = None
for s in range(0, len(seg), 240):
x = torch.from_numpy(seg[s:s + 240]
.transpose(0, 3, 1, 2))[None].to(device) \
/ FLOW_SCALE
ci = ctx[min((r0 + s) // ctx_hz, len(ctx) - 1)] \
.astype(np.float32) / 255. - .5
c = torch.from_numpy(ci.transpose(2, 0, 1))[None].to(device)
p, h = model(x, c, h)
out[r0 + s:r0 + s + x.shape[1]] = p[0].cpu().numpy() * SPEED_SCALE
return np.clip(out, 0, None)
predict_streaming = predict_speed # backwards-compatible alias
def smooth(speed_mps, fps, seconds=1.0):
"""Rolling-median smoothing (recommended post-processing)."""
try:
import pandas as pd
k = max(3, int(round(seconds * fps)) | 1)
return pd.Series(speed_mps).rolling(k, center=True, min_periods=1) \
.median().to_numpy()
except ImportError:
return speed_mps
def predict_video(model, video_path, device="cuda", max_seconds=None,
long_context=False):
"""End-to-end: video file -> dict(t, speed_mps, speed_smooth_mps, fps).
Set long_context=True for closed-course/track footage (indoor karting)."""
flow, ctx, fps = extract_flow_and_ctx(video_path, max_seconds)
speed = predict_speed(model, flow, ctx, fps, device,
long_context=long_context)
t = (np.arange(len(speed)) + 1) / fps
return dict(t=t, speed_mps=speed,
speed_smooth_mps=smooth(speed, fps), fps=fps)