Spaces:
Running on Zero
Running on Zero
File size: 11,907 Bytes
4bc559f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 | """
Run this ON THE CLUSTER, where the fully-tokenized example set lives.
Picks N random clips (by their tok_video_rgb@128/<stem>.npy filename) from
--source_dir, and uploads every modality's file for each picked stem to a HF
**dataset** repo, preserving the same per-modality-subfolder layout as the
source folder:
tok_video_rgb@128/<stem>.npy
tok_video_depth@128/<stem>.npy
tok_video_normal@128/<stem>.npy
tok_video_opticalflow@128/<stem>.npy
tok_video_siglipv2@224/<stem>.npy
tok_video_dinov2@224/<stem>.npy
tok_video_vjepa@224/<stem>.npy
det/<stem>.json
caption/<stem>.json
transcription/<stem>.json
crop_settings/<stem>.npy
Keeping the same layout means hf_space_demo/inference.py's any-to-any
loading code can build every other modality's path the same way the
research scripts do: by string-replacing the tok_video_rgb@128 subfolder
name in the anchor path.
Also writes a manifest (examples.json by default, see --manifest_repo_filename)
at the repo root listing the picked stems, so the Space knows what's
available without listing the whole repo.
Pass --stems_file to use an EXACT, hand-picked list of stems instead of
random selection (e.g. clips specifically suited for the Future Prediction
tab) -- combine with a distinct --manifest_repo_filename/--manifest_out so
this second curated set doesn't collide with the default any-to-any one in
the same dataset repo.
This is a two-phase workflow for previews:
1. Run this script once to pick + upload the tokens. This also writes
--manifest_out locally (default examples_manifest.json).
2. Run visualize_multimodal_pretraining_data_13_modalities.py, which reads
that manifest and renders <domain>_detokenized/<basename>.mp4 previews
for exactly these clips into its --output_dir_videos. Point it at a
non-default manifest via the EXAMPLES_MANIFEST_PATH env var.
3. Re-run this script with --detokenized_dir pointed at that output dir
(and --previews_only, so it doesn't re-upload the tokens) to upload the
previews too. Re-running without --force_repick reuses the exact same
stems from --manifest_out rather than picking a new random/explicit set.
Usage:
# Phase 1 -- random selection (any-to-any tab)
python upload_examples_to_hub.py \\
--source_dir /datasets/uzair/weights_from_clariden/test_cvpr_final_set_13_mod \\
--repo_id EPFL-VILAB/Video-4M-examples \\
--num_examples 6 --seed 0
# Phase 1 -- explicit stems (e.g. Future Prediction tab)
python upload_examples_to_hub.py \\
--source_dir /datasets/uzair/weights_from_clariden/test_cvpr_final_set_13_mod \\
--repo_id EPFL-VILAB/Video-4M-examples \\
--stems_file my_future_pred_stems.txt \\
--manifest_out future_examples_manifest.json \\
--manifest_repo_filename future_examples.json
# Phase 3, after running the visualization script (match --manifest_out/
# --manifest_repo_filename to whichever set you're uploading previews for)
python upload_examples_to_hub.py \\
--repo_id EPFL-VILAB/Video-4M-examples \\
--detokenized_dir /datasets/uzair/weights_from_clariden/cvpr_generations/GT_visualizations_post_neurips_opticalflow_fixed \\
--manifest_out future_examples_manifest.json \\
--manifest_repo_filename future_examples.json \\
--previews_only
"""
import argparse
import json
import os
import random
from huggingface_hub import HfApi, create_repo
ANCHOR_SUBFOLDER = "tok_video_rgb@128"
# subfolder -> file extension, relative to --source_dir
MODALITY_SUBFOLDERS = {
"tok_video_rgb@128": ".npy",
"tok_video_depth@128": ".npy",
"tok_video_normal@128": ".npy",
"tok_video_opticalflow@128": ".npy",
"tok_video_siglipv2@224": ".npy",
"tok_video_dinov2@224": ".npy",
"tok_video_vjepa@224": ".npy",
"det": ".json",
"caption": ".json",
"transcription": ".json",
"crop_settings": ".npy",
}
# our short modality key -> visualize_multimodal_pretraining_data_13_modalities.py's
# <domain>_detokenized folder name. Note "opticalflow" maps to "flow@128", a
# naming quirk specific to that script (everything else matches the tokenized
# subfolder names above).
PREVIEW_DOMAIN_FOLDERS = {
"rgb": "tok_video_rgb@128",
"depth": "tok_video_depth@128",
"normal": "tok_video_normal@128",
"opticalflow": "flow@128",
"dinov2": "tok_video_dinov2@224",
"siglip": "tok_video_siglipv2@224",
"vjepa": "tok_video_vjepa@224",
"det": "det",
}
def _find_npy_stems(root_dir):
"""Recursively finds .npy files under root_dir (they may be nested in
per-shard subdirectories), returning stems as paths relative to
root_dir with the .npy extension stripped -- e.g. "vol_00/clip123".
"""
stems = []
for dirpath, _, filenames in os.walk(root_dir):
rel_dir = os.path.relpath(dirpath, root_dir)
for filename in filenames:
if filename.endswith(".npy"):
stem = filename[: -len(".npy")]
stems.append(stem if rel_dir == "." else os.path.join(rel_dir, stem))
return stems
def _pick_stems(args):
if os.path.exists(args.manifest_out) and not args.force_repick:
with open(args.manifest_out) as f:
picked_stems = json.load(f)["examples"]
print(f"Reusing {len(picked_stems)} previously picked examples from {args.manifest_out}")
return picked_stems
if args.stems_file:
with open(args.stems_file) as f:
requested_stems = [line.strip() for line in f if line.strip()]
missing = [
stem for stem in requested_stems
if not all(
os.path.exists(os.path.join(args.source_dir, subfolder, stem + ext))
for subfolder, ext in MODALITY_SUBFOLDERS.items()
)
]
if missing:
raise ValueError(
f"{len(missing)} requested stem(s) are missing one or more modality files under "
f"{args.source_dir}, so they can't be uploaded: {missing}"
)
print(f"Using {len(requested_stems)} explicitly requested examples from {args.stems_file}")
return requested_stems
anchor_dir = os.path.join(args.source_dir, ANCHOR_SUBFOLDER)
all_stems = _find_npy_stems(anchor_dir)
if len(all_stems) < args.num_examples:
raise ValueError(f"Only found {len(all_stems)} candidates in {anchor_dir}, need {args.num_examples}")
# Only keep stems that actually have a file in every modality subfolder --
# skip a candidate rather than fail the whole run if one is incomplete.
complete_stems = []
for stem in all_stems:
if all(
os.path.exists(os.path.join(args.source_dir, subfolder, stem + ext))
for subfolder, ext in MODALITY_SUBFOLDERS.items()
):
complete_stems.append(stem)
if len(complete_stems) < args.num_examples:
raise ValueError(
f"Only {len(complete_stems)}/{len(all_stems)} candidates have every modality file; "
f"need {args.num_examples}. Loosen MODALITY_SUBFOLDERS or check {args.source_dir}."
)
random.Random(args.seed).shuffle(complete_stems)
picked_stems = complete_stems[: args.num_examples]
print(f"Picked {len(picked_stems)} examples: {picked_stems}")
return picked_stems
def main():
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--source_dir", default="/datasets/uzair/weights_from_clariden/test_cvpr_final_set_13_mod")
parser.add_argument("--repo_id", default="EPFL-VILAB/Video-4M-examples")
parser.add_argument("--num_examples", type=int, default=6)
parser.add_argument("--seed", type=int, default=0)
parser.add_argument("--private", action="store_true", default=True)
parser.add_argument("--public", dest="private", action="store_false")
parser.add_argument(
"--manifest_out", default="examples_manifest.json",
help="Local path to save/reuse the picked stems. Feed this into "
"visualize_multimodal_pretraining_data_13_modalities.py so it renders previews for "
"these exact clips, not a different random sample.",
)
parser.add_argument(
"--manifest_repo_filename", default="examples.json",
help="Filename to upload the manifest as, at the repo root. Use a different name (e.g. "
"'future_examples.json') to keep a second curated set (e.g. for the Future Prediction "
"tab) separate from the default any-to-any set in the same dataset repo.",
)
parser.add_argument(
"--stems_file", default=None,
help="Text file with one clip stem per line (e.g. 'vol_12/clip_000123', matching the "
"tok_video_rgb@128 subfolder layout) to use EXACTLY these clips instead of picking "
"randomly -- e.g. a hand-picked set suited for future prediction. Every stem must have "
"a file in every modality subfolder, or the run fails listing what's missing.",
)
parser.add_argument("--force_repick", action="store_true", help="Ignore an existing --manifest_out and pick a fresh random/explicit set")
parser.add_argument(
"--detokenized_dir", default=None,
help="output_dir_videos from visualize_multimodal_pretraining_data_13_modalities.py. "
"If set, also uploads preview mp4s from <detokenized_dir>/<domain>_detokenized/<basename>.mp4.",
)
parser.add_argument("--tokens_only", action="store_true", help="Skip preview upload even if --detokenized_dir is set")
parser.add_argument("--previews_only", action="store_true", help="Skip token upload (use once tokens are already up)")
args = parser.parse_args()
picked_stems = _pick_stems(args)
api = HfApi()
create_repo(args.repo_id, repo_type="dataset", private=args.private, exist_ok=True)
if not args.previews_only:
for stem in picked_stems:
for subfolder, ext in MODALITY_SUBFOLDERS.items():
local_path = os.path.join(args.source_dir, subfolder, stem + ext)
path_in_repo = f"{subfolder}/{stem}{ext}"
print(f"Uploading {local_path} -> {args.repo_id}:{path_in_repo}")
api.upload_file(
path_or_fileobj=local_path,
path_in_repo=path_in_repo,
repo_id=args.repo_id,
repo_type="dataset",
)
if args.detokenized_dir and not args.tokens_only:
for stem in picked_stems:
basename = os.path.basename(stem)
for key, domain_folder in PREVIEW_DOMAIN_FOLDERS.items():
local_path = os.path.join(args.detokenized_dir, f"{domain_folder}_detokenized", basename + ".mp4")
if not os.path.exists(local_path):
print(f"Skipping missing preview: {local_path}")
continue
path_in_repo = f"preview/{key}/{stem}.mp4"
print(f"Uploading {local_path} -> {args.repo_id}:{path_in_repo}")
api.upload_file(
path_or_fileobj=local_path,
path_in_repo=path_in_repo,
repo_id=args.repo_id,
repo_type="dataset",
)
with open(args.manifest_out, "w") as f:
json.dump({"examples": picked_stems}, f, indent=2)
api.upload_file(
path_or_fileobj=args.manifest_out,
path_in_repo=args.manifest_repo_filename,
repo_id=args.repo_id,
repo_type="dataset",
)
print(f"\nDone. {len(picked_stems)} examples processed for {args.repo_id}.")
print(f"Set FOURM_EXAMPLES_REPO={args.repo_id} where the Space/app runs (or rely on the code default).")
if __name__ == "__main__":
main()
|