File size: 4,484 Bytes
8e5456b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
#!/usr/bin/env python3
"""Dump generated + GT poses for the hand-weight sweep in the shared 50-joint space,
so `eval_fgd_maej.py` can score FGD/MAEJ on exactly the DTW table's 300 clips.

Joint mapping. NSLP-G's KEEP_50 is expressed in the 124-kpt UPPER layout
(`range(8) + range(82,124)`), but the sweep models are trained on the 128-kpt FULL
layout. The upper layout only drops body 9/10/12/13, so the same physical joints are
`range(8) + range(86,128)` in full indices -- body 0-7, then both 21-joint hands.
Verified equal: upper body 0-7 == full body 0-7 (all drops are >= 9), and the hand
blocks shift by exactly the 4 dropped body joints (82->86, 103->107).

MIND THE AE. `eval_fgd_maej.py` applies KEEP_50 directly to its `--data-dir` store, so
it must be pointed at `dataset/VSL_upper` (124 kpt), NOT `dataset/VSL`. On a 128-kpt
store those indices silently select face-tail + the wrong hand joints and the FGD comes
out plausible but wrong. Same clips either way, so the AE is unaffected.

Clip subset is `RandomState(--seed).choice(...)` sorted -- byte-identical to
`eval_trivis_sent_sw.py`, so FGD and DTW describe the same 300 clips.
"""
import argparse
import os

import numpy as np
import torch

import models.t2m_trans as trans
from dataset import dataset_vsl
from models.text_encoder_vi import ViTextEncoder
from train_t2m_trans_vsl import build_vqvae

KEEP_50_FULL = list(range(8)) + list(range(86, 128))


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument('--data-dir', default='./dataset/VSL')
    ap.add_argument('--vq', required=True)
    ap.add_argument('--gpt', required=True)
    ap.add_argument('--split', default='test')
    ap.add_argument('--n', type=int, default=300)
    ap.add_argument('--seed', type=int, default=0, help='clip-subset seed; keep at 0')
    ap.add_argument('--sample-seed', type=int, default=0, help='generation seed only')
    ap.add_argument('--out', required=True)
    ap.add_argument('--gt-out', default=None, help='also write the GT dump here')
    ap.add_argument('--device', default='cuda')
    args = ap.parse_args()

    device = torch.device(args.device)
    torch.manual_seed(args.sample_seed)

    net, _, _ = build_vqvae(args.vq, device)
    tck = torch.load(args.gpt, map_location='cpu')
    g = argparse.Namespace(**tck['args'])
    text_enc = ViTextEncoder(g.text_model, device=args.device)
    gpt = trans.Text2Motion_Transformer(
        num_vq=g.nb_code, embed_dim=g.embed_dim_gpt, clip_dim=text_enc.dim,
        block_size=g.max_tokens + 1, num_layers=g.num_layers, n_head=g.n_head_gpt,
        drop_out_rate=g.drop_out_rate, fc_rate=g.ff_rate)
    gpt.load_state_dict(tck['trans'], strict=True)
    gpt.eval().to(device)

    st = dataset_vsl.VSLStore(args.data_dir, args.split)
    NK = st.layout.n_kpts
    if NK != 128:
        raise SystemExit(f'expected the 128-kpt full layout, got {NK} -- '
                         f'KEEP_50_FULL would select the wrong joints')
    rs = np.random.RandomState(args.seed)
    items = sorted(rs.choice(len(st.index), size=min(args.n, len(st.index)),
                             replace=False).tolist())

    poses, names, texts = [], [], []
    gt_poses, gt_names = [], []
    with torch.no_grad():
        for k in items:
            c = st.index[k]
            nm, txt = c['name'], c[g.text_field]
            idx = gpt.sample(text_enc([txt]), if_categorial=True)
            if idx is None or idx.numel() == 0:
                print(f'  [skip] {nm}: empty sample')
                continue
            p = net.decode_batch(idx.clamp(max=g.nb_code - 1))[0].cpu().numpy()
            p = (p * st.std + st.mean).reshape(-1, NK, 2)[:, KEEP_50_FULL]
            poses.append(p.astype(np.float32)); names.append(nm); texts.append(txt)
            if args.gt_out:
                m, _ = st.get(k)
                gq = (m * st.std + st.mean).reshape(-1, NK, 2)[:, KEEP_50_FULL]
                gt_poses.append(gq.astype(np.float32)); gt_names.append(nm)

    os.makedirs(os.path.dirname(args.out) or '.', exist_ok=True)
    np.savez(args.out, poses=np.array(poses, dtype=object),
             names=np.array(names), texts=np.array(texts))
    print(f'wrote {args.out}: {len(poses)} clips, 50 joints')
    if args.gt_out:
        np.savez(args.gt_out, poses=np.array(gt_poses, dtype=object),
                 names=np.array(gt_names), texts=np.array(gt_names))
        print(f'wrote {args.gt_out}: {len(gt_poses)} GT clips')


if __name__ == '__main__':
    main()