Download scripts/prepare_three_benchmarks.py from Orangerl/umm: direct link, hf CLI and curl.
- Browser
- Download file 11.2 kB
-
https://huggingface.co/Orangerl/umm/resolve/main/scripts/prepare_three_benchmarks.py
- Command line
-
hf download hf://Orangerl/umm/scripts/prepare_three_benchmarks.py
-
curl -L -o prepare_three_benchmarks.py https://huggingface.co/Orangerl/umm/resolve/main/scripts/prepare_three_benchmarks.py
11.2 kB
| #!/usr/bin/env python3 | |
| """Prepare pinned CV-Bench, BLINK-val, and VStar data as auditable JSONL manifests.""" | |
| from __future__ import annotations | |
| import argparse | |
| import hashlib | |
| import io | |
| import json | |
| import re | |
| from collections import Counter | |
| from pathlib import Path | |
| from typing import Any, Iterable | |
| import pyarrow.parquet as pq | |
| from PIL import Image | |
| REVISIONS = { | |
| "cvbench": "bc284db50d036958861cb60cdd7b77612052ce0d", | |
| "blink": "a3666eb249237ba3d5eca8db21176cc47967e040", | |
| "vstar": "d9ae62c903da0c98336e85c5ee89cd863b04b4da", | |
| } | |
| IMAGE_EXTENSIONS = { | |
| "BMP": ".bmp", | |
| "GIF": ".gif", | |
| "JPEG": ".jpg", | |
| "PNG": ".png", | |
| "TIFF": ".tiff", | |
| "WEBP": ".webp", | |
| } | |
| OPTION_RE = re.compile(r"^\s*\(?([A-Z])\)?\s*$", re.IGNORECASE) | |
| def _sha256(path: Path) -> str: | |
| digest = hashlib.sha256() | |
| with path.open("rb") as handle: | |
| for chunk in iter(lambda: handle.read(8 * 1024 * 1024), b""): | |
| digest.update(chunk) | |
| return digest.hexdigest() | |
| def _write_json(path: Path, value: Any) -> None: | |
| path.write_text( | |
| json.dumps(value, ensure_ascii=False, indent=2, sort_keys=True) + "\n", | |
| encoding="utf-8", | |
| ) | |
| def _write_manifest(output_dir: Path, records: list[dict], metadata: dict) -> None: | |
| output_dir.mkdir(parents=True, exist_ok=True) | |
| manifest = output_dir / "manifest.jsonl" | |
| with manifest.open("w", encoding="utf-8") as handle: | |
| for record in records: | |
| handle.write(json.dumps(record, ensure_ascii=False) + "\n") | |
| metadata = { | |
| **metadata, | |
| "samples": len(records), | |
| "manifest_sha256": _sha256(manifest), | |
| "config_counts": dict(sorted(Counter(row["config"] for row in records).items())), | |
| "image_count": sum(len(row["images"]) for row in records), | |
| } | |
| _write_json(output_dir / "metadata.json", metadata) | |
| def _expected(answer: Any, choices: list[Any]) -> str: | |
| match = OPTION_RE.fullmatch(str(answer)) | |
| if not match: | |
| raise ValueError(f"Answer is not an option letter: {answer!r}") | |
| letter = match.group(1).upper() | |
| if ord(letter) - ord("A") >= len(choices): | |
| raise ValueError(f"Answer {letter} is outside {len(choices)} choices") | |
| return letter | |
| def _image_extension(data: bytes, suggested: str | None = None) -> str: | |
| if suggested: | |
| suffix = Path(suggested).suffix.lower() | |
| if suffix in {".jpg", ".jpeg", ".png", ".webp", ".bmp", ".gif", ".tif", ".tiff"}: | |
| return ".jpg" if suffix == ".jpeg" else suffix | |
| with Image.open(io.BytesIO(data)) as image: | |
| return IMAGE_EXTENSIONS.get(str(image.format).upper(), ".img") | |
| def _save_embedded_image(value: dict, output_base: Path, stem: str) -> tuple[str, int, int]: | |
| data = value.get("bytes") | |
| if not data: | |
| raise ValueError(f"Embedded image has no bytes: {value!r}") | |
| extension = _image_extension(data, value.get("path")) | |
| path = output_base / f"{stem}{extension}" | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| if not path.exists(): | |
| path.write_bytes(data) | |
| elif path.read_bytes() != data: | |
| raise RuntimeError(f"Refusing to overwrite non-identical image: {path}") | |
| with Image.open(io.BytesIO(data)) as image: | |
| width, height = image.size | |
| image.verify() | |
| return path.name, width, height | |
| def _iter_parquet(path: Path) -> Iterable[tuple[int, dict]]: | |
| row_index = 0 | |
| parquet = pq.ParquetFile(path) | |
| for batch in parquet.iter_batches(batch_size=32): | |
| for row in batch.to_pylist(): | |
| yield row_index, row | |
| row_index += 1 | |
| def prepare_cvbench(raw: Path, output: Path) -> None: | |
| records: list[dict] = [] | |
| image_dir = output / "images" | |
| sources = [("2D", raw / "test_2d.parquet"), ("3D", raw / "test_3d.parquet")] | |
| for config, parquet in sources: | |
| if not parquet.is_file(): | |
| raise FileNotFoundError(parquet) | |
| for row_index, row in _iter_parquet(parquet): | |
| sample_id = f"{config}-{int(row['idx']):04d}" | |
| filename, width, height = _save_embedded_image( | |
| row["image"], image_dir, sample_id | |
| ) | |
| choices = [str(choice) for choice in row["choices"]] | |
| _expected(row["answer"], choices) | |
| records.append( | |
| { | |
| "sample_id": sample_id, | |
| "benchmark": "cvbench", | |
| "split": "test", | |
| "config": config, | |
| "task": str(row["task"]), | |
| "row_index": len(records), | |
| "source_row_index": row_index, | |
| "images": [f"images/{filename}"], | |
| "image_sizes": [[width, height]], | |
| "question": str(row["question"]), | |
| "choices": choices, | |
| "answer": str(row["answer"]), | |
| "prompt": str(row["prompt"]), | |
| "source_idx": int(row["idx"]), | |
| "source_filename": row.get("filename"), | |
| } | |
| ) | |
| if len(records) != 2638 or Counter(row["config"] for row in records) != Counter({"2D": 1438, "3D": 1200}): | |
| raise RuntimeError("CV-Bench count contract failed") | |
| _write_manifest( | |
| output, | |
| records, | |
| { | |
| "dataset": "nyu-visionx/CV-Bench", | |
| "revision": REVISIONS["cvbench"], | |
| "split": "test", | |
| "scope": "full official test", | |
| }, | |
| ) | |
| def prepare_blink(raw: Path, output: Path) -> None: | |
| records: list[dict] = [] | |
| image_dir = output / "images" | |
| parquets = sorted(raw.glob("*/val-*.parquet")) | |
| if len(parquets) != 14: | |
| raise RuntimeError(f"Expected 14 BLINK val parquet files, got {len(parquets)}") | |
| for parquet in parquets: | |
| config = parquet.parent.name | |
| for source_row_index, row in _iter_parquet(parquet): | |
| sample_id = str(row["idx"]) | |
| image_paths: list[str] = [] | |
| image_sizes: list[list[int]] = [] | |
| for image_number in range(1, 5): | |
| embedded = row.get(f"image_{image_number}") | |
| if embedded is None: | |
| continue | |
| filename, width, height = _save_embedded_image( | |
| embedded, | |
| image_dir, | |
| f"{config}__{sample_id}__{image_number}", | |
| ) | |
| image_paths.append(f"images/{filename}") | |
| image_sizes.append([width, height]) | |
| if not image_paths: | |
| raise RuntimeError(f"BLINK sample has no images: {sample_id}") | |
| choices = [str(choice) for choice in row["choices"]] | |
| _expected(row["answer"], choices) | |
| records.append( | |
| { | |
| "sample_id": sample_id, | |
| "benchmark": "blink", | |
| "split": "val", | |
| "config": config, | |
| "task": str(row["sub_task"]), | |
| "row_index": len(records), | |
| "source_row_index": source_row_index, | |
| "images": image_paths, | |
| "image_sizes": image_sizes, | |
| "question": str(row["question"]), | |
| "choices": choices, | |
| "answer": str(row["answer"]), | |
| "prompt": str(row["prompt"]), | |
| } | |
| ) | |
| expected_counts = { | |
| "Art_Style": 117, | |
| "Counting": 120, | |
| "Forensic_Detection": 132, | |
| "Functional_Correspondence": 130, | |
| "IQ_Test": 150, | |
| "Jigsaw": 150, | |
| "Multi-view_Reasoning": 133, | |
| "Object_Localization": 122, | |
| "Relative_Depth": 124, | |
| "Relative_Reflectance": 134, | |
| "Semantic_Correspondence": 139, | |
| "Spatial_Relation": 143, | |
| "Visual_Correspondence": 172, | |
| "Visual_Similarity": 135, | |
| } | |
| if len(records) != 1901 or Counter(row["config"] for row in records) != Counter(expected_counts): | |
| raise RuntimeError("BLINK val count contract failed") | |
| _write_manifest( | |
| output, | |
| records, | |
| { | |
| "dataset": "BLINK-Benchmark/BLINK", | |
| "revision": REVISIONS["blink"], | |
| "split": "val", | |
| "scope": "full official val; public test labels are hidden", | |
| }, | |
| ) | |
| def _vstar_choices(text: str) -> list[str]: | |
| matches = re.findall(r"(?m)^\s*\(([A-Z])\)\s*(.+?)\s*$", text) | |
| if not matches: | |
| matches = re.findall(r"(?m)^\s*([A-Z])[\.:]\s*(.+?)\s*$", text) | |
| letters = [letter for letter, _ in matches] | |
| expected_letters = [chr(ord("A") + index) for index in range(len(letters))] | |
| if letters != expected_letters: | |
| raise ValueError(f"Cannot parse contiguous VStar choices from: {text!r}") | |
| return [choice for _, choice in matches] | |
| def prepare_vstar(raw: Path, output: Path) -> None: | |
| questions = raw / "test_questions.jsonl" | |
| if not questions.is_file(): | |
| raise FileNotFoundError(questions) | |
| records: list[dict] = [] | |
| for source_row_index, line in enumerate(questions.read_text(encoding="utf-8").splitlines()): | |
| if not line.strip(): | |
| continue | |
| row = json.loads(line) | |
| image = (raw / str(row["image"])).resolve() | |
| if not image.is_file() or raw.resolve() not in image.parents: | |
| raise FileNotFoundError(image) | |
| with Image.open(image) as opened: | |
| width, height = opened.size | |
| opened.verify() | |
| choices = _vstar_choices(str(row["text"])) | |
| _expected(row["label"], choices) | |
| records.append( | |
| { | |
| "sample_id": str(row["question_id"]), | |
| "benchmark": "vstar", | |
| "split": "test", | |
| "config": str(row["category"]), | |
| "task": str(row["category"]), | |
| "row_index": len(records), | |
| "source_row_index": source_row_index, | |
| "images": [str(image)], | |
| "image_sizes": [[width, height]], | |
| "question": str(row["text"]).split("\n", 1)[0], | |
| "choices": choices, | |
| "answer": str(row["label"]), | |
| "prompt": str(row["text"]), | |
| } | |
| ) | |
| if len(records) != 191: | |
| raise RuntimeError(f"Expected 191 VStar rows, got {len(records)}") | |
| _write_manifest( | |
| output, | |
| records, | |
| { | |
| "dataset": "craigwu/vstar_bench", | |
| "revision": REVISIONS["vstar"], | |
| "split": "test", | |
| "scope": "full official test", | |
| }, | |
| ) | |
| def main() -> None: | |
| parser = argparse.ArgumentParser() | |
| parser.add_argument("--raw-root", type=Path, default=Path("data/benchmarks/raw")) | |
| parser.add_argument("--output-root", type=Path, default=Path("data/benchmarks/prepared")) | |
| args = parser.parse_args() | |
| raw = args.raw_root.resolve() | |
| output = args.output_root.resolve() | |
| prepare_cvbench(raw / "CV-Bench", output / "cvbench_full") | |
| prepare_blink(raw / "BLINK", output / "blink_val") | |
| prepare_vstar(raw / "vstar_bench", output / "vstar_test") | |
| print(f"Prepared all benchmarks under {output}") | |
| if __name__ == "__main__": | |
| main() | |