t2m-gpt-vsl-code / prepare_vnhn_data.py
Tri1's picture
T2M-GPT VSL adaptation: Python sources only (82 files, no checkpoints or data)
8e5456b verified
Raw
History Blame Contribute Delete
10.2 kB
#!/usr/bin/env python3
"""Pack the processed_vnhn sentence-level dataset into T2M-GPT memmaps.
Mirrors `prepare_vsl_data.py` (same outputs, same normalization, same layout machinery) but
reads processed_vnhn's per-clip **pickles** instead of Full_TriVis's .npz.
THREE THINGS DIFFER FROM Full_TriVis AND EACH ONE MATTERS
---------------------------------------------------------
1. **Keypoints are raw COCO-WholeBody 133, not the project's 128.** Full_TriVis was
extracted through `easy_dwpose`, which already converts COCO-17 -> OpenPose-18 (it
synthesises a `neck` from the shoulder midpoint) and emits body18+face68+Lhand21+Rhand21.
These pickles are the detector's raw output, so the conversion is done here:
COCO-WholeBody 133 = 0..16 body(COCO-17), 17..22 feet, 23..90 face,
91..111 Lhand, 112..132 Rhand
Verified against the data: hand root 91 sits 0.021 frame-widths from COCO Lwrist 9, and
112 sits 0.032 from Rwrist 10 -- i.e. the hand blocks really do start at the wrist, which
is what makes this index map safe.
2. **`scores` are NOT [0,1] confidences.** They run 0.4..11.3 (median 8.2), roughly 10x a
confidence. Full_TriVis's `--score-thr 0.3` would therefore mark *everything* valid.
Calibrated by matching Full_TriVis's per-group validity (body 82.7 / face 100 / hands
~100): **thr=3.0** gives face 99.7 / hands 98.1 / body 76.3 and correctly kills feet
(1.0% valid). thr=4.0 is too harsh (hands collapse to 78%), thr=2.0 leaks feet (22%).
3. **Legs are out of frame.** This is a broadcast bust shot: knees score ~1.2 and ankles
~0.85, both <0.1% valid at thr=3.0, while hips stay 99.7% valid at median y=0.97 (the
very bottom edge). That is precisely the case the `upper` preset exists for, so the
default here is `upper` (124 kpts: body14 + face68 + hands42) rather than `full`. Using
`full` would hand the decoder four dimensions of pure noise -- the documented
"phantom leg" failure.
TEXT. `text` is the transcript line (Vietnamese, lowercased, unpunctuated). There is no
gloss annotation, so `gloss` and `sentence` are both set to it: stage 2 conditions on
`--text-field`, and with only one text source the two must agree or the field silently
selects nothing.
CAVEAT NOT FIXED HERE. Every clip's end timestamp was padded by +3s before clamping (see
the dataset's own README), so the tail of each clip contains signing the text does not
describe -- ~23% of a 331-frame sample. That is left intact: trimming it changes the
dataset's semantics and should be a deliberate, separately-evaluated choice.
"""
import argparse
import csv
import json
import os
import pickle
import numpy as np
from dataset.layout import Layout, PRESETS
# COCO-17 -> OpenPose-18. Index 1 (neck) is synthesised from the shoulder midpoint, so it
# is encoded as None and handled separately.
# 0 nose 1 neck 2 Rsho 3 Relb 4 Rwri 5 Lsho 6 Lelb 7 Lwri
# 8 Rhip 9 Rknee 10 Rank 11 Lhip 12 Lknee 13 Lank 14 Reye 15 Leye 16 Rear 17 Lear
COCO17_TO_OP18 = [0, None, 6, 8, 10, 5, 7, 9, 12, 14, 16, 11, 13, 15, 2, 1, 4, 3]
FACE = (23, 91)
LHAND = (91, 112)
RHAND = (112, 133)
def wholebody133_to_128(xy, sc):
"""[T,133,2]/[T,133] COCO-WholeBody -> [T,128,2]/[T,128] body18+face68+Lhand21+Rhand21."""
T = xy.shape[0]
b_xy = np.zeros((T, 18, 2), np.float32)
b_sc = np.zeros((T, 18), np.float32)
for j, src in enumerate(COCO17_TO_OP18):
if src is None: # neck = midpoint of the two shoulders
b_xy[:, j] = 0.5 * (xy[:, 5] + xy[:, 6])
b_sc[:, j] = np.minimum(sc[:, 5], sc[:, 6]) # only as good as its worse parent
else:
b_xy[:, j] = xy[:, src]
b_sc[:, j] = sc[:, src]
out_xy = np.concatenate([b_xy, xy[:, FACE[0]:FACE[1]],
xy[:, LHAND[0]:LHAND[1]], xy[:, RHAND[0]:RHAND[1]]], axis=1)
out_sc = np.concatenate([b_sc, sc[:, FACE[0]:FACE[1]],
sc[:, LHAND[0]:LHAND[1]], sc[:, RHAND[0]:RHAND[1]]], axis=1)
return out_xy, out_sc
def load_clip(path, score_thr, layout):
with open(path, "rb") as f:
d = pickle.load(f)
xy = np.asarray(d["keypoints"], np.float32)
sc = np.asarray(d["scores"], np.float32)
if xy.ndim == 4: # [T,1,133,2] -> [T,133,2] (one person per clip)
xy, sc = xy[:, 0], sc[:, 0]
if xy.shape[1] != 133:
raise ValueError(f"expected 133 COCO-WholeBody keypoints, got {xy.shape[1]}")
xy, sc = wholebody133_to_128(xy, sc)
valid = np.isfinite(xy).all(-1) & (sc > score_thr)
keep = layout.keep
xy, valid = xy[:, keep, :], valid[:, keep]
return xy.reshape(len(xy), layout.dim), valid.astype(np.uint8)
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--csv-dir",
default="../processed_vnhn/Cut_video/output/csv",
help="dir holding {train,val,test}_fixed.csv")
ap.add_argument("--root", default="../processed_vnhn/Cut_video/output",
help="what pkl_path in the CSV is relative to")
ap.add_argument("--out-dir", default="./dataset/VNHN")
ap.add_argument("--score-thr", type=float, default=3.0,
help="scores here are ~10x a confidence; 3.0 matches TriVis validity")
ap.add_argument("--min-frames", type=int, default=64)
# Stage 2 uses max_tokens=128 everywhere in this repo, i.e. 512 frames at down_t=2.
# 3.95% of vnhn clips are longer (max 1715 frames = 428 tokens). Dropping them is an
# explicit, countable choice; leaving them in means silent truncation inside the loader,
# which would quietly break the text<->pose correspondence on the longest clips.
ap.add_argument("--max-frames", type=int, default=512,
help="drop clips longer than this (0 = keep all)")
ap.add_argument("--limit", type=int, default=0)
ap.add_argument("--layout", default="upper", choices=list(PRESETS),
help="'upper' by default: knees/ankles are off-frame in this corpus")
args = ap.parse_args()
os.makedirs(args.out_dir, exist_ok=True)
layout = Layout.preset(args.layout)
layout.save(args.out_dir)
print(layout)
for split in ("train", "val", "test"):
path = os.path.join(args.csv_dir, f"{split}_fixed.csv")
rows = list(csv.DictReader(open(path, encoding="utf-8")))
if args.limit:
rows = rows[:args.limit]
# pass 1: lengths, so the memmap can be allocated exactly
keep_rows, lengths = [], []
n_short = n_long = 0
for r in rows:
p = os.path.join(args.root, r["pkl_path"])
try:
with open(p, "rb") as f:
T = len(pickle.load(f)["keypoints"])
except Exception as e: # noqa: BLE001
print(f" skip {r['pkl_path']}: {e}")
continue
if T < args.min_frames:
n_short += 1
continue
if args.max_frames and T > args.max_frames:
n_long += 1
continue
keep_rows.append(r)
lengths.append(T)
total = int(sum(lengths))
print(f"[{split}] kept {len(keep_rows)}/{len(rows)} clips ({total} frames); "
f"dropped {n_short} shorter than {args.min_frames} and "
f"{n_long} longer than {args.max_frames} frames "
f"({100*n_long/max(len(rows),1):.2f}% would have been truncated at "
f"max_tokens=128)")
xy_mm = np.lib.format.open_memmap(
os.path.join(args.out_dir, f"{split}_xy.npy"), mode="w+",
dtype=np.float16, shape=(total, layout.dim))
vd_mm = np.lib.format.open_memmap(
os.path.join(args.out_dir, f"{split}_valid.npy"), mode="w+",
dtype=np.uint8, shape=(total, layout.n_kpts))
s1 = np.zeros(layout.dim, np.float64)
s2 = np.zeros(layout.dim, np.float64)
cnt = np.zeros(layout.dim, np.float64)
index, off = [], 0
for r in keep_rows:
xy, vd = load_clip(os.path.join(args.root, r["pkl_path"]),
args.score_thr, layout)
T = len(xy)
xy_mm[off:off + T] = xy.astype(np.float16)
vd_mm[off:off + T] = vd
if split == "train":
m = np.repeat(vd, 2, axis=1).astype(np.float64)
x64 = xy.astype(np.float64)
s1 += (x64 * m).sum(0)
s2 += (x64 * x64 * m).sum(0)
cnt += m.sum(0)
txt = r["text"].strip()
index.append({
"name": r["sample_id"],
"npz_path": r["pkl_path"], # key name kept for loader compatibility
"start": off,
"length": T,
"gloss": txt, # no gloss annotation exists; both = transcript
"sentence": txt,
"category": r.get("source_id", ""),
"fps": float(r.get("fps", 25) or 25),
"end_clamped": r.get("end_clamped", ""),
})
off += T
xy_mm.flush(); vd_mm.flush(); del xy_mm, vd_mm
with open(os.path.join(args.out_dir, f"{split}_index.json"), "w",
encoding="utf-8") as f:
json.dump(index, f, ensure_ascii=False)
if split == "train":
cnt = np.maximum(cnt, 1.0)
mean = s1 / cnt
std = np.sqrt(np.maximum(s2 / cnt - mean ** 2, 0.0))
std[cnt < 100] = 1.0 # never-observed dim: do not amplify it 100x
std = np.maximum(std, 1e-2)
np.save(os.path.join(args.out_dir, "mean.npy"), mean.astype(np.float32))
np.save(os.path.join(args.out_dir, "std.npy"), std.astype(np.float32))
print(f" mean/std saved; std range {float(std.min()):.4f}..{float(std.max()):.4f}")
L = np.array(lengths)
if len(L):
print(f" frames per clip: min {L.min()} median {int(np.median(L))} "
f"max {L.max()} mean {L.mean():.0f}")
if __name__ == "__main__":
main()