Download hy_import_dev_vbench.py from Cccccz/comparison: direct link, hf CLI and curl.
- Browser
- Download file 2.23 kB
-
https://huggingface.co/Cccccz/comparison/resolve/main/hy_import_dev_vbench.py
- Command line
-
hf download hf://Cccccz/comparison/hy_import_dev_vbench.py
-
curl -L -o hy_import_dev_vbench.py https://huggingface.co/Cccccz/comparison/resolve/main/hy_import_dev_vbench.py
2.23 kB
| #!/usr/bin/env python | |
| """Reuse a DEV-project validation25 VBench run instead of re-scoring. | |
| The DEV run (tools/run_vbench_validation25_groups.sh) scores staged videos named | |
| ``<group>__case_XXXX_action_YY.mp4`` with the same VBench code and custom_input | |
| settings hy_vbench.sh uses, so per-video Core5 scores are identical (verified on | |
| the Stage-1 ATC videos: max |diff| 1e-5). This writes eval_out_hy/vbench/scores/ | |
| <strategy>.json in hy_vbench.sh's format from the DEV raw results. | |
| python hy_import_dev_vbench.py --run-root <DEV vbench run dir> --group rollout_s500_fppf --strategy hy_atc_s2_s500_fppf_c0FFFF | |
| """ | |
| import argparse, glob, json, os | |
| ROOT = os.path.dirname(os.path.abspath(__file__)) | |
| DIMS = ("subject_consistency", "background_consistency", "motion_smoothness", "aesthetic_quality", "imaging_quality") | |
| ap = argparse.ArgumentParser(description=__doc__) | |
| ap.add_argument("--run-root", required=True) | |
| ap.add_argument("--group", required=True) | |
| ap.add_argument("--strategy", required=True) | |
| ap.add_argument("--out-root", default=os.path.join(ROOT, "eval_out_hy")) | |
| a = ap.parse_args() | |
| per_video, raw = {}, {} | |
| for d in DIMS: | |
| files = glob.glob(os.path.join(a.run_root, "raw", d, "*_eval_results.json")) | |
| assert len(files) == 1, (d, files) | |
| records = json.load(open(files[0]))[d][1] | |
| scores = {} | |
| for r in records: | |
| name = os.path.basename(r["video_path"]) | |
| group, _, stem = name.partition("__") | |
| if group != a.group: | |
| continue | |
| v = float(r["video_results"]) | |
| # hy_vbench.sh stores imaging_quality on [0,1]; VBench reports it on [0,100]. | |
| scores[stem] = v / 100.0 if d == "imaging_quality" else v | |
| assert len(scores) == 100, (d, a.group, len(scores)) | |
| per_video[d] = dict(sorted(scores.items())) | |
| raw[d] = sum(scores.values()) / len(scores) | |
| out = {"strategy": a.strategy, "num_videos": 100, "raw": raw, "per_video": per_video, | |
| "source": {"dev_vbench_run": a.run_root, "group": a.group}} | |
| os.makedirs(os.path.join(a.out_root, "vbench", "scores"), exist_ok=True) | |
| path = os.path.join(a.out_root, "vbench", "scores", f"{a.strategy}.json") | |
| json.dump(out, open(path, "w"), indent=2) | |
| print(a.strategy, {k: round(v, 5) for k, v in raw.items()}) | |