File size: 6,465 Bytes
932bc69 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 | #!/usr/bin/env python3
"""Split SPEED-Bench flat files into per-category/subcategory JSONL files.
**Step 1** — run NVIDIA's prepare.py to materialise prompts from their
original sources (fetched at runtime due to redistribution restrictions)::
curl -LsSf https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/ \\
refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py \\
| python3 - --output_dir ./speedbench_data
.. note::
The URL above points to the ``main`` branch of an external repository
maintained by NVIDIA. Save a local copy of the script if you anticipate
running data preparation again::
curl -LsSf <url> -o prepare_nvidia_speedbench.py
The output files produced by prepare.py contain data fetched from
third-party sources that carry their own licences. Do not redistribute
the materialised JSONL files.
**Step 2** — run this script to split the flat files into per-category files
that evaluate.py can consume directly::
python prepare_speedbench.py --data-dir ./speedbench_data
Output files follow the naming convention ``{config}_{key}.jsonl`` where
``key`` is the category (qualitative) or ``{entropy}__{subcategory}``
(throughput configs). Each file has a single ``turns`` column.
Usage::
python prepare_speedbench.py --data-dir ./speedbench_data \
[--configs qualitative,throughput_1k]
"""
from __future__ import annotations
import argparse
import json
import logging
import subprocess
import sys
import urllib.request
from pathlib import Path
logger = logging.getLogger("prepare_speedbench")
_PLACEHOLDER_MARKERS = (
"FULL BENCHMARK DATA SHOULD BE FETCHED FROM THE SOURCE",
"{article}",
)
_ALL_CONFIGS = [
"qualitative",
"throughput_1k",
"throughput_2k",
"throughput_8k",
"throughput_32k",
]
def _extract_text(row: dict) -> str:
"""Return prompt text — supports both ``turns`` and ``messages`` columns.
.. note::
For multi-turn rows only the first turn is extracted. Full multi-turn
conversation support is out of scope for this initial implementation.
"""
text = row.get("turns") or ""
if not text:
msgs = row.get("messages") or []
# TODO: support multi-turn by concatenating all turns
text = msgs[0].get("content", "") if msgs else ""
return text
def split_config(flat_file: Path, out_dir: Path, config: str) -> int:
"""Split *flat_file* into per-category/subcategory files in *out_dir*.
Returns the number of output files written.
"""
logger.info("Loading %s ...", flat_file)
with flat_file.open() as f:
rows = [json.loads(line) for line in f if line.strip()]
use_category_only = config == "qualitative"
buckets: dict[str, list[str]] = {}
for row in rows:
text = _extract_text(row)
if not text:
continue
for marker in _PLACEHOLDER_MARKERS:
if marker in text:
cat = row.get("category", "?")
logger.error(
"Placeholder text in %s/%s — re-run NVIDIA prepare.py first.",
config,
cat,
)
sys.exit(1)
cat = (row.get("category") or "unknown").replace(" ", "_").replace("/", "_")
sub = (row.get("sub_category") or "unknown").replace(" ", "_").replace("/", "_")
key = cat if use_category_only else f"{cat}__{sub}"
buckets.setdefault(key, []).append(text)
written = 0
for key, texts in sorted(buckets.items()):
out_path = out_dir / f"{config}_{key}.jsonl"
with out_path.open("w") as f:
for t in texts:
f.write(json.dumps({"turns": t}) + "\n")
logger.info(" wrote %s (%d rows)", out_path.name, len(texts))
written += 1
return written
_NVIDIA_PREPARE_URL = (
"https://raw.githubusercontent.com/NVIDIA-NeMo/Skills/"
"refs/heads/main/nemo_skills/dataset/speed-bench/prepare.py"
)
def _run_nvidia_prepare(data_dir: Path, configs: list[str]) -> None:
"""Download and run NVIDIA's prepare.py to materialise prompts."""
logger.info("Downloading NVIDIA prepare.py from %s ...", _NVIDIA_PREPARE_URL)
with urllib.request.urlopen(_NVIDIA_PREPARE_URL) as resp: # noqa: S310
script_bytes = resp.read()
for config in configs:
logger.info("Running prepare.py for config=%s ...", config)
subprocess.run( # noqa: S603
[sys.executable, "-", "--config", config, "--output_dir", str(data_dir)],
input=script_bytes,
check=True,
)
logger.info("prepare.py done for config=%s", config)
def main() -> None:
logging.basicConfig(level=logging.INFO, format="[%(levelname)s] %(message)s")
parser = argparse.ArgumentParser(
description="Split SPEED-Bench flat files into per-category JSONL files.",
)
parser.add_argument(
"--data-dir",
required=True,
type=Path,
help=(
"Directory with flat JSONL files from NVIDIA prepare.py"
" (e.g. qualitative.jsonl)."
),
)
parser.add_argument(
"--configs",
default=",".join(_ALL_CONFIGS),
help=(f"Comma-separated configs to split (default: {','.join(_ALL_CONFIGS)})"),
)
parser.add_argument(
"--download",
action="store_true",
default=False,
help=(
"Download and run NVIDIA's prepare.py to materialise prompts before "
"splitting. Fetches from the NVIDIA-NeMo/Skills repository."
),
)
args = parser.parse_args()
data_dir: Path = args.data_dir
data_dir.mkdir(parents=True, exist_ok=True)
configs = [c.strip() for c in args.configs.split(",") if c.strip()]
if args.download:
_run_nvidia_prepare(data_dir, configs)
if not data_dir.exists():
logger.error("--data-dir '%s' does not exist.", data_dir)
sys.exit(1)
total = 0
for config in configs:
flat_file = data_dir / f"{config}.jsonl"
if not flat_file.exists():
logger.warning("Skipping %s — %s not found.", config, flat_file)
continue
n = split_config(flat_file, data_dir, config)
total += n
logger.info("%s: %d files written.", config, n)
logger.info("Done. %d total files written to %s", total, data_dir)
if __name__ == "__main__":
main()
|