| |
| """Extract only the front-view (center) Multi-VSL clips from Data.zip. |
| |
| The zip holds 87,817 mp4 across center/left/right plus ~176k stale .lock/.metadata |
| files from an interrupted HuggingFace upload. This pulls out just the clips named in |
| `Multi-VSL_WACV_2025/data/front_{train,val,test}.csv` (28,412 clips, ~23 GB) into a |
| flat directory, and records the per-clip resolution and fps -- both vary across |
| clips because the dataset ships per-clip YOLO crops, and the packing step needs |
| them to normalize. |
| """ |
| import argparse |
| import csv |
| import json |
| import os |
| import subprocess |
| import zipfile |
|
|
| from tqdm import tqdm |
|
|
| REPO = os.path.join(os.path.dirname(os.path.abspath(__file__)), '..') |
|
|
|
|
| def probe(path): |
| out = subprocess.run( |
| ['ffprobe', '-v', 'error', '-select_streams', 'v:0', '-show_entries', |
| 'stream=width,height,r_frame_rate,nb_frames', '-of', 'json', path], |
| capture_output=True, text=True).stdout |
| try: |
| st = json.loads(out)['streams'][0] |
| num, den = st['r_frame_rate'].split('/') |
| return {'width': int(st['width']), 'height': int(st['height']), |
| 'fps': float(num) / float(den), |
| 'nb_frames': int(st.get('nb_frames') or 0)} |
| except Exception: |
| return None |
|
|
|
|
| def main(): |
| ap = argparse.ArgumentParser() |
| ap.add_argument('--zip', default=os.path.join(REPO, 'WACV-Data-SLR', 'Data.zip')) |
| ap.add_argument('--labels-dir', default=os.path.join(REPO, 'Multi-VSL_WACV_2025', 'data')) |
| ap.add_argument('--out-dir', default=os.path.join(REPO, 'Multi-VSL_front')) |
| ap.add_argument('--probe', action='store_true', help='ffprobe each clip (slower)') |
| args = ap.parse_args() |
|
|
| vid_dir = os.path.join(args.out_dir, 'videos') |
| os.makedirs(vid_dir, exist_ok=True) |
|
|
| wanted = {} |
| for split in ('train', 'val', 'test'): |
| with open(os.path.join(args.labels_dir, f'front_{split}.csv'), encoding='utf-8') as f: |
| for r in csv.DictReader(f): |
| wanted[r['name']] = {'split': split, 'label': int(r['label']), |
| 'word': r['word']} |
| print(f'front-view clips wanted: {len(wanted)}') |
|
|
| z = zipfile.ZipFile(args.zip) |
| inzip = {os.path.basename(n): n for n in z.namelist() if n.lower().endswith('.mp4')} |
| print(f'mp4 in zip: {len(inzip)}') |
|
|
| meta, missing, skipped = [], 0, 0 |
| for name, info in tqdm(sorted(wanted.items()), desc='extract'): |
| src = inzip.get(name) |
| if src is None: |
| missing += 1 |
| continue |
| dst = os.path.join(vid_dir, name) |
| if not os.path.exists(dst) or os.path.getsize(dst) == 0: |
| with z.open(src) as fi, open(dst, 'wb') as fo: |
| while True: |
| b = fi.read(1 << 20) |
| if not b: |
| break |
| fo.write(b) |
| else: |
| skipped += 1 |
| rec = {'name': name, **info, 'size': os.path.getsize(dst)} |
| if args.probe: |
| p = probe(dst) |
| if p: |
| rec.update(p) |
| meta.append(rec) |
|
|
| with open(os.path.join(args.out_dir, 'clips.json'), 'w', encoding='utf-8') as f: |
| json.dump(meta, f, ensure_ascii=False) |
| print(f'extracted {len(meta)} clips ({skipped} already present), {missing} missing') |
| print(f'-> {vid_dir}') |
| tot = sum(m['size'] for m in meta) |
| print(f'total {tot/1e9:.1f} GB') |
|
|
|
|
| if __name__ == '__main__': |
| main() |
|
|