comparison / eval /run_vbench.py
Cccccz's picture
Add files using upload-large-folder tool
f70ac4f verified
Raw History Blame Contribute Delete
6.56 kB
#!/usr/bin/env python
"""Score one strategy on the eight protocol VBench dimensions.
Standard VBench 0.1.5 metric code is used unmodified. What is built here is only
the evaluation metadata file, because the protocol needs two things the stock
`vbench_standard` mode cannot express:
* `prompt_en` must be the *extended* prompt actually fed to the generator
(section 6.1), not the short canonical one;
* `scene` must keep the canonical `auxiliary_info` keywords.
The file has exactly the shape `build_full_info_json` produces, and each
`compute_<dimension>` selects the rows that list it, so per-suite scoping
(72 / 93 / 86 videos) falls out of the mapping rather than being hard-coded.
python eval/run_vbench.py --strategy sf_ffff --out-root eval_out
"""
import argparse
import json
import os
import sys
import time
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, ROOT)
MAPPING = os.path.join(ROOT, "assets/vbench8_extended_subset_mapping.json")
SUITE_DIMENSIONS = {
"subject_consistency": ["subject_consistency", "motion_smoothness", "dynamic_degree"],
"overall_consistency": ["overall_consistency", "aesthetic_quality", "imaging_quality"],
"scene": ["scene", "background_consistency"],
}
DIMENSIONS = [d for dims in SUITE_DIMENSIONS.values() for d in dims]
def parse_args():
p = argparse.ArgumentParser()
p.add_argument("--strategy", required=True)
p.add_argument("--out-root", default="eval_out")
p.add_argument("--dimensions", default=None,
help="Comma-separated subset (default: all eight)")
p.add_argument("--allow-partial", action="store_true",
help="Score even if fewer than 251 videos exist (smoke tests only)")
p.add_argument("--overwrite", action="store_true",
help="Recompute dimensions whose raw result file already exists")
p.add_argument("--threads", type=int, default=None,
help="torch/OpenMP CPU threads for this process (scoring is CPU-bound; "
"several scorers on one node oversubscribe the cores)")
return p.parse_args()
def main():
args = parse_args()
out_root = (args.out_root if os.path.isabs(args.out_root)
else os.path.join(ROOT, args.out_root))
dims = args.dimensions.split(",") if args.dimensions else list(DIMENSIONS)
with open(MAPPING) as f:
mapping = json.load(f)
video_root = os.path.join(out_root, "generated_videos", args.strategy)
rows, missing = [], []
for r in mapping["rows"]:
vp = os.path.join(video_root, r["prompt_suite"], f"{r['suite_index']:03d}.mp4")
if not os.path.exists(vp):
missing.append(vp)
continue
entry = {
# The extended prompt is what the video was generated from, so it is
# what overall_consistency and scene must be scored against.
"prompt_en": r["extended_prompt"],
"dimension": [d for d in SUITE_DIMENSIONS[r["prompt_suite"]]],
"video_list": [vp],
"global_index": r["global_index"],
"prompt_suite": r["prompt_suite"],
"suite_index": r["suite_index"],
}
if "auxiliary_info" in r:
entry["auxiliary_info"] = r["auxiliary_info"]
rows.append(entry)
if missing and not args.allow_partial:
print(f"ERROR: {len(missing)} of {len(mapping['rows'])} videos missing for "
f"{args.strategy}; refusing to score an incomplete strategy.")
for m in missing[:5]:
print(" ", m)
return 1
print(f"{args.strategy}: {len(rows)} videos, dimensions={dims}", flush=True)
scores_dir = os.path.join(out_root, "vbench", "raw_results", args.strategy)
os.makedirs(scores_dir, exist_ok=True)
info_path = os.path.join(scores_dir, "full_info.json")
# Atomic: several per-dimension processes may start on the same strategy.
with open(info_path + f".{os.getpid()}", "w") as f:
json.dump(rows, f, indent=2)
os.replace(info_path + f".{os.getpid()}", info_path)
import importlib
import torch
if args.threads:
torch.set_num_threads(args.threads)
from vbench.utils import init_submodules
device = torch.device("cuda")
def raw_path(d):
return os.path.join(scores_dir, f"{d}.json")
# A dimension already scored (its raw file exists) is reused, so dimensions can
# be spread over several processes and a killed run resumes where it stopped.
todo = [d for d in dims if args.overwrite or not os.path.exists(raw_path(d))]
submodules = init_submodules(todo, local=True, read_frame=False) if todo else {}
for d in todo:
t0 = time.time()
module = importlib.import_module(f"vbench.{d}")
fn = getattr(module, f"compute_{d}")
score, per_video = fn(info_path, device, submodules[d])
tmp = raw_path(d) + ".tmp"
with open(tmp, "w") as f:
json.dump({"dimension": d, "score": score, "video_results": per_video,
"seconds": time.time() - t0}, f, indent=2)
os.replace(tmp, raw_path(d))
print(f" {d:24s} {score:.6f} ({len(per_video)} videos, "
f"{time.time() - t0:.0f}s)", flush=True)
torch.cuda.empty_cache()
# The strategy's score file is written only once all eight dimensions exist.
missing_dims = [d for d in DIMENSIONS if not os.path.exists(raw_path(d))]
if missing_dims:
print(f"{args.strategy}: {len(DIMENSIONS) - len(missing_dims)}/{len(DIMENSIONS)} "
f"dimensions scored; missing {missing_dims}", flush=True)
return 0
results = {}
for d in DIMENSIONS:
with open(raw_path(d)) as f:
raw = json.load(f)
results[d] = {"score": raw["score"], "num_videos": len(raw["video_results"]),
"seconds": raw.get("seconds")}
out_path = os.path.join(out_root, "vbench", "scores", f"{args.strategy}.json")
os.makedirs(os.path.dirname(out_path), exist_ok=True)
payload = {
"strategy": args.strategy,
"vbench_long": False,
"num_videos": len(rows),
"raw": {d: results[d]["score"] for d in DIMENSIONS},
"detail": results,
"full_info": info_path,
}
tmp = out_path + ".tmp"
with open(tmp, "w") as f:
json.dump(payload, f, indent=2)
os.replace(tmp, out_path)
print(f"wrote {out_path}", flush=True)
return 0
if __name__ == "__main__":
sys.exit(main())