File size: 6,543 Bytes
2162853 9285f34 2162853 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 | """SpeedNet v4 — ego-speed estimation from onboard/POV video.
Self-contained inference module. No project dependencies: just
torch, opencv-python, numpy (and optionally pandas for smoothing).
Pipeline
--------
1. Decode the video, resize frames to 320x180.
2. Dense optical flow (Farneback) between consecutive frames, resized to
128x72. Flow is normalized to "pixels per 1/15 s" so any source frame
rate works (multiply raw flow by fps/15).
3. A low-resolution RGB context frame (128x72, sampled ~1 Hz) feeds a small
context branch that lets the model infer scene scale (indoor kart track
vs. open highway look identical in flow magnitude but differ in scale).
4. Flow features + context embedding run through a GRU. Two inference
modes exist (sequence length acts as a scale cue for this model):
short windows (default; roads/trails/open environments) and
long_context=True (closed-course/track footage such as indoor karting).
Typical usage
-------------
from modeling_speednet import SpeedNet, predict_video
import torch
model = SpeedNet()
model.load_state_dict(torch.load("speednet_v5.pt", map_location="cuda"))
result = predict_video(model, "my_onboard_clip.mp4", device="cuda")
# result["t"] -> timestamps in seconds
# result["speed_mps"]-> speed in m/s (multiply by 3.6 for km/h)
"""
import numpy as np
import torch
import torch.nn as nn
FLOW_SCALE = 8.0 # flow normalization divisor (px per 1/15 s)
SPEED_SCALE = 40.0 # model output * SPEED_SCALE = m/s
BASE_FPS = 15.0 # flow convention: pixels per 1/15 s
class SpeedNet(nn.Module):
"""Flow CNN + RGB context branch -> GRU -> speed (v4 architecture)."""
def __init__(self, hidden=192, ctx_dim=64):
super().__init__()
def blk(i, o):
return [nn.Conv2d(i, o, 3, 2, 1), nn.BatchNorm2d(o), nn.SiLU()]
self.cnn = nn.Sequential(*blk(2, 32), *blk(32, 64), *blk(64, 128),
*blk(128, 256), nn.AdaptiveAvgPool2d(1),
nn.Flatten())
self.ctx_cnn = nn.Sequential(*blk(3, 16), *blk(16, 32), *blk(32, 64),
nn.AdaptiveAvgPool2d(1), nn.Flatten(),
nn.Linear(64, ctx_dim), nn.SiLU())
self.gru = nn.GRU(256 + ctx_dim, hidden, num_layers=2,
batch_first=True)
self.head = nn.Linear(hidden, 1)
def forward(self, flow, ctx, h0=None):
"""flow: [B, T, 2, 72, 128] (normalized), ctx: [B, 3, 72, 128]
(RGB/255 - 0.5). Returns (speed [B, T] normalized, hidden)."""
B, T = flow.shape[:2]
f = self.cnn(flow.flatten(0, 1)).view(B, T, -1)
c = self.ctx_cnn(ctx)[:, None, :].expand(B, T, -1)
y, h = self.gru(torch.cat([f, c], -1), h0)
return self.head(y).squeeze(-1), h
def extract_flow_and_ctx(video_path, max_seconds=None):
"""Decode video -> (flow [N,72,128,2] float32 normalized to 15 fps
convention, ctx [M,72,128,3] uint8 at ~1 Hz, fps)."""
import cv2
cap = cv2.VideoCapture(video_path)
fps = cap.get(cv2.CAP_PROP_FPS) or BASE_FPS
fps_scale = fps / BASE_FPS
ctx_step = max(1, int(round(fps))) # ~1 Hz
prev, flows, ctxs = None, [], []
i = 0
while True:
ok, frame = cap.read()
if not ok:
break
if max_seconds and i / fps > max_seconds:
break
if frame.shape[:2] != (180, 320):
frame = cv2.resize(frame, (320, 180))
if i % ctx_step == 0:
ctxs.append(cv2.cvtColor(cv2.resize(frame, (128, 72)),
cv2.COLOR_BGR2RGB))
g = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
if prev is not None:
fl = cv2.calcOpticalFlowFarneback(prev, g, None,
0.5, 3, 15, 3, 5, 1.2, 0)
fl *= fps_scale
flows.append(cv2.resize(fl, (128, 72)))
prev = g
i += 1
cap.release()
return (np.stack(flows).astype(np.float32),
np.stack(ctxs).astype(np.uint8), fps)
@torch.no_grad()
def predict_speed(model, flow, ctx, fps, device="cuda", long_context=False):
"""Inference in one of two modes (see README "Inference modes"):
- short (default): hidden state reset every 16 frames. Matches the
published GPS-domain metrics (roads, trails, open environments).
- long_context=True: hidden state carried across the clip. Use for
closed-course/track footage (e.g. indoor karting) where absolute
scale must be inferred from long temporal context.
The model was trained with two window lengths, so sequence length acts
as a scale cue — pick the mode matching your footage.
"""
model.eval().to(device)
ctx_hz = max(1, int(round(fps)))
reset = 10**9 if long_context else 16
out = np.empty(len(flow), np.float32)
for r0 in range(0, len(flow), reset):
seg = flow[r0:r0 + reset]
h = None
for s in range(0, len(seg), 240):
x = torch.from_numpy(seg[s:s + 240]
.transpose(0, 3, 1, 2))[None].to(device) \
/ FLOW_SCALE
ci = ctx[min((r0 + s) // ctx_hz, len(ctx) - 1)] \
.astype(np.float32) / 255. - .5
c = torch.from_numpy(ci.transpose(2, 0, 1))[None].to(device)
p, h = model(x, c, h)
out[r0 + s:r0 + s + x.shape[1]] = p[0].cpu().numpy() * SPEED_SCALE
return np.clip(out, 0, None)
predict_streaming = predict_speed # backwards-compatible alias
def smooth(speed_mps, fps, seconds=1.0):
"""Rolling-median smoothing (recommended post-processing)."""
try:
import pandas as pd
k = max(3, int(round(seconds * fps)) | 1)
return pd.Series(speed_mps).rolling(k, center=True, min_periods=1) \
.median().to_numpy()
except ImportError:
return speed_mps
def predict_video(model, video_path, device="cuda", max_seconds=None,
long_context=False):
"""End-to-end: video file -> dict(t, speed_mps, speed_smooth_mps, fps).
Set long_context=True for closed-course/track footage (indoor karting)."""
flow, ctx, fps = extract_flow_and_ctx(video_path, max_seconds)
speed = predict_speed(model, flow, ctx, fps, device,
long_context=long_context)
t = (np.arange(len(speed)) + 1) / fps
return dict(t=t, speed_mps=speed,
speed_smooth_mps=smooth(speed, fps), fps=fps)
|