File size: 5,113 Bytes
f70ac4f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
#!/usr/bin/env python
"""Build and validate assets/vbench8_extended_subset_mapping.json (Extended-251).

Protocol section 4: take the subject_consistency (72), overall_consistency (93)
and scene (86) suites out of the canonical 946-prompt VBench order, and carry the
Self-Forcing rewritten extended prompt for each.

Selection is strictly by index into VBench_full_info.json -- the short prompt list
contains two duplicate texts, so text matching is not safe, and the protocol
forbids it anyway.

    python eval/build_mapping.py --out assets/vbench8_extended_subset_mapping.json
"""

import argparse
import hashlib
import json
import os
import sys

ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))

SUITES = ["subject_consistency", "overall_consistency", "scene"]
EXPECTED = {"subject_consistency": 72, "overall_consistency": 93, "scene": 86}

DEFAULT_FULL_INFO = "/local/zoubin/cz/projects/VBench/vbench/VBench_full_info.json"
DEFAULT_PROMPT_DIR = "/local/zoubin/cz/projects/Self-Forcing/prompts/vbench"


def sha256(path):
    h = hashlib.sha256()
    with open(path, "rb") as f:
        for chunk in iter(lambda: f.read(1 << 20), b""):
            h.update(chunk)
    return h.hexdigest()


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--full-info", default=DEFAULT_FULL_INFO)
    ap.add_argument("--prompt-dir", default=DEFAULT_PROMPT_DIR)
    ap.add_argument("--out", default=os.path.join(ROOT, "assets/vbench8_extended_subset_mapping.json"))
    args = ap.parse_args()

    short_path = os.path.join(args.prompt_dir, "all_dimension.txt")
    ext_path = os.path.join(args.prompt_dir, "all_dimension_extended.txt")

    with open(args.full_info) as f:
        info = json.load(f)
    with open(short_path, encoding="utf-8") as f:
        short = [l.rstrip("\n") for l in f]
    with open(ext_path, encoding="utf-8") as f:
        ext = [l.rstrip("\n") for l in f]

    errors = []
    if len(short) != 946:
        errors.append(f"short prompts == {len(short)}, expected 946")
    if len(ext) != 946:
        errors.append(f"extended prompts == {len(ext)}, expected 946")
    if len(info) != 946:
        errors.append(f"VBench_full_info rows == {len(info)}, expected 946")
    for i, (row, s) in enumerate(zip(info, short)):
        if row["prompt_en"].strip() != s.strip():
            errors.append(f"short prompt order differs from VBench_full_info at line {i}")
            break
    for i, e in enumerate(ext):
        if not e.strip():
            errors.append(f"empty extended prompt at line {i}")
            break

    rows = []
    for suite in SUITES:
        suite_index = 0
        for gi, row in enumerate(info):
            if suite not in row["dimension"]:
                continue
            entry = {
                "global_index": gi,
                "prompt_suite": suite,
                "suite_index": suite_index,
                "original_prompt": short[gi],
                "extended_prompt": ext[gi],
                "official_dimensions": list(row["dimension"]),
            }
            # Standard VBench 0.1.5 needs the official scene keywords to score `scene`.
            for key in ("auxiliary_info",):
                if key in row:
                    entry[key] = row[key]
            rows.append(entry)
            suite_index += 1

    counts = {s: sum(1 for r in rows if r["prompt_suite"] == s) for s in SUITES}
    for s, n in EXPECTED.items():
        if counts.get(s) != n:
            errors.append(f"{s} == {counts.get(s)}, expected {n}")
    if len(rows) != 251:
        errors.append(f"mapping rows == {len(rows)}, expected 251")
    if len({r["global_index"] for r in rows}) != len(rows):
        errors.append("duplicate global_index")
    for s in SUITES:
        idx = [r["suite_index"] for r in rows if r["prompt_suite"] == s]
        if idx != list(range(len(idx))):
            errors.append(f"{s} suite_index not contiguous")

    if errors:
        print("MAPPING VALIDATION FAILED:")
        for e in errors:
            print("  -", e)
        return 1

    payload = {
        "protocol": "Self-Forcing Extended-251 Full Evaluation",
        "num_rows": len(rows),
        "counts": counts,
        "sources": {
            "vbench_full_info": {"path": args.full_info, "sha256": sha256(args.full_info)},
            "short_prompts": {"path": short_path, "sha256": sha256(short_path)},
            "extended_prompts": {"path": ext_path, "sha256": sha256(ext_path)},
        },
        "rows": rows,
    }
    os.makedirs(os.path.dirname(args.out), exist_ok=True)
    with open(args.out, "w") as f:
        json.dump(payload, f, indent=2)

    print("mapping OK")
    print(f"  rows            {len(rows)}")
    for s in SUITES:
        print(f"  {s:22s} {counts[s]}")
    for k, v in payload["sources"].items():
        print(f"  sha256 {k:18s} {v['sha256'][:16]}...")
    print(f"  scene rows with auxiliary_info: "
          f"{sum(1 for r in rows if r['prompt_suite'] == 'scene' and 'auxiliary_info' in r)}")
    print(f"wrote {args.out}")
    return 0


if __name__ == "__main__":
    sys.exit(main())