File size: 6,543 Bytes
2162853
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
9285f34
2162853
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
"""SpeedNet v4 — ego-speed estimation from onboard/POV video.

Self-contained inference module. No project dependencies: just
torch, opencv-python, numpy (and optionally pandas for smoothing).

Pipeline
--------
1. Decode the video, resize frames to 320x180.
2. Dense optical flow (Farneback) between consecutive frames, resized to
   128x72. Flow is normalized to "pixels per 1/15 s" so any source frame
   rate works (multiply raw flow by fps/15).
3. A low-resolution RGB context frame (128x72, sampled ~1 Hz) feeds a small
   context branch that lets the model infer scene scale (indoor kart track
   vs. open highway look identical in flow magnitude but differ in scale).
4. Flow features + context embedding run through a GRU. Two inference
   modes exist (sequence length acts as a scale cue for this model):
   short windows (default; roads/trails/open environments) and
   long_context=True (closed-course/track footage such as indoor karting).

Typical usage
-------------
    from modeling_speednet import SpeedNet, predict_video
    import torch

    model = SpeedNet()
    model.load_state_dict(torch.load("speednet_v5.pt", map_location="cuda"))
    result = predict_video(model, "my_onboard_clip.mp4", device="cuda")
    # result["t"]        -> timestamps in seconds
    # result["speed_mps"]-> speed in m/s (multiply by 3.6 for km/h)
"""
import numpy as np
import torch
import torch.nn as nn

FLOW_SCALE = 8.0     # flow normalization divisor (px per 1/15 s)
SPEED_SCALE = 40.0   # model output * SPEED_SCALE = m/s
BASE_FPS = 15.0      # flow convention: pixels per 1/15 s


class SpeedNet(nn.Module):
    """Flow CNN + RGB context branch -> GRU -> speed (v4 architecture)."""

    def __init__(self, hidden=192, ctx_dim=64):
        super().__init__()

        def blk(i, o):
            return [nn.Conv2d(i, o, 3, 2, 1), nn.BatchNorm2d(o), nn.SiLU()]

        self.cnn = nn.Sequential(*blk(2, 32), *blk(32, 64), *blk(64, 128),
                                 *blk(128, 256), nn.AdaptiveAvgPool2d(1),
                                 nn.Flatten())
        self.ctx_cnn = nn.Sequential(*blk(3, 16), *blk(16, 32), *blk(32, 64),
                                     nn.AdaptiveAvgPool2d(1), nn.Flatten(),
                                     nn.Linear(64, ctx_dim), nn.SiLU())
        self.gru = nn.GRU(256 + ctx_dim, hidden, num_layers=2,
                          batch_first=True)
        self.head = nn.Linear(hidden, 1)

    def forward(self, flow, ctx, h0=None):
        """flow: [B, T, 2, 72, 128] (normalized), ctx: [B, 3, 72, 128]
        (RGB/255 - 0.5). Returns (speed [B, T] normalized, hidden)."""
        B, T = flow.shape[:2]
        f = self.cnn(flow.flatten(0, 1)).view(B, T, -1)
        c = self.ctx_cnn(ctx)[:, None, :].expand(B, T, -1)
        y, h = self.gru(torch.cat([f, c], -1), h0)
        return self.head(y).squeeze(-1), h


def extract_flow_and_ctx(video_path, max_seconds=None):
    """Decode video -> (flow [N,72,128,2] float32 normalized to 15 fps
    convention, ctx [M,72,128,3] uint8 at ~1 Hz, fps)."""
    import cv2
    cap = cv2.VideoCapture(video_path)
    fps = cap.get(cv2.CAP_PROP_FPS) or BASE_FPS
    fps_scale = fps / BASE_FPS
    ctx_step = max(1, int(round(fps)))  # ~1 Hz
    prev, flows, ctxs = None, [], []
    i = 0
    while True:
        ok, frame = cap.read()
        if not ok:
            break
        if max_seconds and i / fps > max_seconds:
            break
        if frame.shape[:2] != (180, 320):
            frame = cv2.resize(frame, (320, 180))
        if i % ctx_step == 0:
            ctxs.append(cv2.cvtColor(cv2.resize(frame, (128, 72)),
                                     cv2.COLOR_BGR2RGB))
        g = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
        if prev is not None:
            fl = cv2.calcOpticalFlowFarneback(prev, g, None,
                                              0.5, 3, 15, 3, 5, 1.2, 0)
            fl *= fps_scale
            flows.append(cv2.resize(fl, (128, 72)))
        prev = g
        i += 1
    cap.release()
    return (np.stack(flows).astype(np.float32),
            np.stack(ctxs).astype(np.uint8), fps)


@torch.no_grad()
def predict_speed(model, flow, ctx, fps, device="cuda", long_context=False):
    """Inference in one of two modes (see README "Inference modes"):

    - short (default): hidden state reset every 16 frames. Matches the
      published GPS-domain metrics (roads, trails, open environments).
    - long_context=True: hidden state carried across the clip. Use for
      closed-course/track footage (e.g. indoor karting) where absolute
      scale must be inferred from long temporal context.

    The model was trained with two window lengths, so sequence length acts
    as a scale cue — pick the mode matching your footage.
    """
    model.eval().to(device)
    ctx_hz = max(1, int(round(fps)))
    reset = 10**9 if long_context else 16
    out = np.empty(len(flow), np.float32)
    for r0 in range(0, len(flow), reset):
        seg = flow[r0:r0 + reset]
        h = None
        for s in range(0, len(seg), 240):
            x = torch.from_numpy(seg[s:s + 240]
                                 .transpose(0, 3, 1, 2))[None].to(device) \
                / FLOW_SCALE
            ci = ctx[min((r0 + s) // ctx_hz, len(ctx) - 1)] \
                .astype(np.float32) / 255. - .5
            c = torch.from_numpy(ci.transpose(2, 0, 1))[None].to(device)
            p, h = model(x, c, h)
            out[r0 + s:r0 + s + x.shape[1]] = p[0].cpu().numpy() * SPEED_SCALE
    return np.clip(out, 0, None)


predict_streaming = predict_speed  # backwards-compatible alias


def smooth(speed_mps, fps, seconds=1.0):
    """Rolling-median smoothing (recommended post-processing)."""
    try:
        import pandas as pd
        k = max(3, int(round(seconds * fps)) | 1)
        return pd.Series(speed_mps).rolling(k, center=True, min_periods=1) \
            .median().to_numpy()
    except ImportError:
        return speed_mps


def predict_video(model, video_path, device="cuda", max_seconds=None,
                  long_context=False):
    """End-to-end: video file -> dict(t, speed_mps, speed_smooth_mps, fps).
    Set long_context=True for closed-course/track footage (indoor karting)."""
    flow, ctx, fps = extract_flow_and_ctx(video_path, max_seconds)
    speed = predict_speed(model, flow, ctx, fps, device,
                          long_context=long_context)
    t = (np.arange(len(speed)) + 1) / fps
    return dict(t=t, speed_mps=speed,
                speed_smooth_mps=smooth(speed, fps), fps=fps)