Download scripts/eval_vbench8_extended_subset.py from Cccccz/Self-Forcing: direct link, hf CLI and curl.
- Browser
- Download file 9.27 kB
-
https://huggingface.co/Cccccz/Self-Forcing/resolve/main/scripts/eval_vbench8_extended_subset.py
- Command line
-
hf download hf://Cccccz/Self-Forcing/scripts/eval_vbench8_extended_subset.py
-
curl -L -o eval_vbench8_extended_subset.py https://huggingface.co/Cccccz/Self-Forcing/resolve/main/scripts/eval_vbench8_extended_subset.py
9.27 kB
| #!/usr/bin/env python3 | |
| """Run the standard VBench metrics for one generated strategy. | |
| The official VBench metric modules are used unchanged. Only the metadata | |
| builder is replaced so that the evaluator reads the exact extended prompt used | |
| for generation and the auditable mapping can use suite-indexed filenames. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import hashlib | |
| import json | |
| import os | |
| import sys | |
| import tempfile | |
| from pathlib import Path | |
| from typing import Any | |
| def preparse_gpu() -> str: | |
| parser = argparse.ArgumentParser(add_help=False) | |
| parser.add_argument("--gpu", default="0") | |
| args, _ = parser.parse_known_args() | |
| os.environ["CUDA_VISIBLE_DEVICES"] = str(args.gpu) | |
| os.environ.setdefault("MPLCONFIGDIR", tempfile.mkdtemp(prefix="vbench_mpl_")) | |
| return str(args.gpu) | |
| PHYSICAL_GPU = preparse_gpu() | |
| _vbench_cache = Path( | |
| os.environ.get("VBENCH_CACHE_DIR", "/data3/chenzhuo/.cache/vbench") | |
| ).expanduser() | |
| os.environ.setdefault( | |
| "VBENCH_BERT_MODEL_DIR", str(_vbench_cache / "bert-base-uncased") | |
| ) | |
| from vbench import VBench | |
| REPO_ROOT = Path(__file__).resolve().parents[1] | |
| if str(REPO_ROOT) not in sys.path: | |
| sys.path.insert(0, str(REPO_ROOT)) | |
| from scripts.vbench8_protocol import ( | |
| DIMENSIONS, | |
| PROTOCOL_NAME, | |
| SUITE_COUNTS, | |
| aggregate_selected_score, | |
| ) | |
| def sha256(path: Path) -> str: | |
| digest = hashlib.sha256() | |
| with path.open("rb") as handle: | |
| for block in iter(lambda: handle.read(1024 * 1024), b""): | |
| digest.update(block) | |
| return digest.hexdigest() | |
| def read_mapping(path: Path) -> list[dict[str, Any]]: | |
| value = json.loads(path.read_text(encoding="utf-8")) | |
| if not isinstance(value, list) or len(value) != 251: | |
| raise ValueError(f"Expected a 251-row mapping list: {path}") | |
| counts = {suite: 0 for suite in SUITE_COUNTS} | |
| globals_seen: set[int] = set() | |
| suite_seen: dict[str, set[int]] = {suite: set() for suite in SUITE_COUNTS} | |
| for row in value: | |
| suite = str(row["prompt_suite"]) | |
| global_index = int(row["global_index"]) | |
| suite_index = int(row["suite_index"]) | |
| if suite not in counts: | |
| raise ValueError(f"Unknown suite {suite}") | |
| if global_index in globals_seen: | |
| raise ValueError(f"Duplicate global index {global_index}") | |
| if suite_index in suite_seen[suite]: | |
| raise ValueError(f"Duplicate suite index {suite}/{suite_index}") | |
| globals_seen.add(global_index) | |
| suite_seen[suite].add(suite_index) | |
| counts[suite] += 1 | |
| if counts != SUITE_COUNTS: | |
| raise ValueError(f"Unexpected suite counts: {counts}") | |
| return sorted(value, key=lambda row: int(row["global_index"])) | |
| def build_metadata( | |
| *, | |
| mapping: list[dict[str, Any]], | |
| videos_root: Path, | |
| strategy: str, | |
| metadata_path: Path, | |
| ) -> None: | |
| records: list[dict[str, Any]] = [] | |
| for row in mapping: | |
| suite = str(row["prompt_suite"]) | |
| suite_index = int(row["suite_index"]) | |
| video_path = ( | |
| videos_root / strategy / suite / f"{suite_index:03d}.mp4" | |
| ).resolve() | |
| if not video_path.is_file(): | |
| raise FileNotFoundError(video_path) | |
| official_dimensions = list(row["official_dimensions"]) | |
| if not set(official_dimensions).issubset(set(DIMENSIONS)): | |
| raise ValueError( | |
| f"Unsupported dimensions at global index {row['global_index']}: " | |
| f"{official_dimensions}" | |
| ) | |
| record: dict[str, Any] = { | |
| "prompt_en": str(row["extended_prompt"]), | |
| "dimension": official_dimensions, | |
| "video_list": [str(video_path)], | |
| } | |
| # The standard scene implementation in vbench==0.1.5 needs the | |
| # official scene keyword in auxiliary_info. It is not a replacement | |
| # for prompt_en: overall_consistency still receives the extended text. | |
| if "auxiliary_info" in row: | |
| record["auxiliary_info"] = row["auxiliary_info"] | |
| records.append(record) | |
| if len(records) != 251: | |
| raise ValueError(f"Expected 251 metadata records, got {len(records)}") | |
| metadata_path.parent.mkdir(parents=True, exist_ok=True) | |
| metadata_path.write_text( | |
| json.dumps(records, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" | |
| ) | |
| class FixedMetadataVBench(VBench): | |
| """Use prepared metadata while retaining VBench's official evaluator.""" | |
| def __init__(self, *, device: str, metadata_path: Path, output_path: Path): | |
| super().__init__( | |
| device=device, | |
| full_info_dir=str(metadata_path), | |
| output_path=str(output_path), | |
| ) | |
| self.prepared_metadata_path = metadata_path | |
| def build_full_info_json(self, *args: Any, **kwargs: Any) -> str: | |
| return str(self.prepared_metadata_path) | |
| def parse_result(value: Any) -> tuple[float, Any]: | |
| if isinstance(value, (list, tuple)) and len(value) == 2: | |
| score, details = value | |
| else: | |
| score, details = value, None | |
| score = float(score) | |
| if not 0.0 <= score <= 1.0: | |
| raise ValueError(f"VBench raw score is outside [0, 1]: {score}") | |
| return score, details | |
| def parse_args() -> argparse.Namespace: | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument("--gpu", default=PHYSICAL_GPU) | |
| parser.add_argument("--strategy", required=True) | |
| parser.add_argument("--mapping", type=Path, required=True) | |
| parser.add_argument("--videos-root", type=Path, required=True) | |
| parser.add_argument("--output-root", type=Path, required=True) | |
| parser.add_argument("--vbench-info", type=Path, required=True) | |
| parser.add_argument("--skip-existing", action="store_true") | |
| return parser.parse_args() | |
| def main() -> None: | |
| args = parse_args() | |
| mapping_path = args.mapping.resolve() | |
| videos_root = args.videos_root.resolve() | |
| output_root = args.output_root.resolve() | |
| mapping = read_mapping(mapping_path) | |
| metadata_path = output_root / "vbench/metadata" / f"{args.strategy}_full_info.json" | |
| raw_dir = output_root / "vbench/raw_results" / args.strategy | |
| raw_name = f"vbench8_{args.strategy}" | |
| raw_result_path = raw_dir / f"{raw_name}_eval_results.json" | |
| score_path = output_root / "vbench/scores" / f"{args.strategy}.json" | |
| if args.skip_existing and raw_result_path.is_file() and score_path.is_file(): | |
| print(f"[cached] strategy={args.strategy}", flush=True) | |
| return | |
| build_metadata( | |
| mapping=mapping, | |
| videos_root=videos_root, | |
| strategy=args.strategy, | |
| metadata_path=metadata_path, | |
| ) | |
| raw_dir.mkdir(parents=True, exist_ok=True) | |
| print( | |
| f"[vbench] gpu={args.gpu} strategy={args.strategy} videos=251 " | |
| f"dimensions={','.join(DIMENSIONS)}", | |
| flush=True, | |
| ) | |
| bench = FixedMetadataVBench( | |
| device="cuda", metadata_path=metadata_path, output_path=raw_dir | |
| ) | |
| bench.evaluate( | |
| videos_path=str(videos_root / args.strategy), | |
| name=raw_name, | |
| dimension_list=list(DIMENSIONS), | |
| local=True, | |
| mode="vbench_standard", | |
| ) | |
| if not raw_result_path.is_file(): | |
| raise FileNotFoundError(raw_result_path) | |
| raw_json = json.loads(raw_result_path.read_text(encoding="utf-8")) | |
| raw_scores: dict[str, float] = {} | |
| details: dict[str, Any] = {} | |
| for dimension in DIMENSIONS: | |
| if dimension not in raw_json: | |
| raise KeyError(f"Missing {dimension} in {raw_result_path}") | |
| score, dimension_details = parse_result(raw_json[dimension]) | |
| raw_scores[dimension] = score | |
| if dimension_details is not None: | |
| details[dimension] = dimension_details | |
| aggregate = aggregate_selected_score(raw_scores) | |
| result = { | |
| "protocol": PROTOCOL_NAME, | |
| "strategy": args.strategy, | |
| "physical_gpu": str(args.gpu), | |
| "benchmark": "VBench", | |
| "vbench_version": "0.1.5", | |
| "vbench_long": False, | |
| "vbench_info": str(args.vbench_info.resolve()), | |
| "vbench_info_sha256": sha256(args.vbench_info.resolve()), | |
| "mapping": str(mapping_path), | |
| "mapping_sha256": sha256(mapping_path), | |
| "videos_root": str(videos_root / args.strategy), | |
| "num_videos": 251, | |
| "dimensions": list(DIMENSIONS), | |
| "raw_scores": raw_scores, | |
| **aggregate, | |
| "scene_prompt_note": ( | |
| "prompt_en is the Self-Forcing extended prompt; vbench==0.1.5 " | |
| "scene uses the official auxiliary scene keyword by design." | |
| ), | |
| } | |
| if details: | |
| details_path = output_root / "vbench/details" / f"{args.strategy}.json" | |
| details_path.parent.mkdir(parents=True, exist_ok=True) | |
| details_path.write_text( | |
| json.dumps(details, ensure_ascii=False, indent=2) + "\n", | |
| encoding="utf-8", | |
| ) | |
| result["details"] = str(details_path) | |
| score_path.parent.mkdir(parents=True, exist_ok=True) | |
| score_path.write_text( | |
| json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" | |
| ) | |
| print( | |
| f"[complete] strategy={args.strategy} " | |
| f"selected={result['selected_vbench_percent']:.4f}%", | |
| flush=True, | |
| ) | |
| if __name__ == "__main__": | |
| main() | |