Spaces:
Running
Running
Download scripts/materialize_dataset.py from StentorLabs/SLM_Arena: direct link, hf CLI and curl.
- Browser
- Download file 2.47 kB
-
https://huggingface.co/spaces/StentorLabs/SLM_Arena/resolve/main/scripts/materialize_dataset.py
- Command line
-
hf download hf://spaces/StentorLabs/SLM_Arena/scripts/materialize_dataset.py
-
curl -L -o materialize_dataset.py https://huggingface.co/spaces/StentorLabs/SLM_Arena/resolve/main/scripts/materialize_dataset.py
2.47 kB
| #!/usr/bin/env python3 | |
| """Build a deterministic JSONL round snapshot from a local dataset checkout. | |
| This is an offline derived artifact. It never commits or uploads files. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| from pathlib import Path | |
| from typing import Any | |
| def read_events(root: Path) -> tuple[list[dict[str, Any]], int]: | |
| events: list[dict[str, Any]] = [] | |
| malformed = 0 | |
| for path in sorted(root.rglob("*.json")): | |
| if path.name.endswith("schema.json"): | |
| continue | |
| try: | |
| record = json.loads(path.read_text(encoding="utf-8")) | |
| except (OSError, json.JSONDecodeError): | |
| malformed += 1 | |
| continue | |
| if isinstance(record, dict) and record.get("round_id") and record.get("event_type"): | |
| events.append(record) | |
| return events, malformed | |
| def materialize(events: list[dict[str, Any]]) -> list[dict[str, Any]]: | |
| rounds: dict[str, dict[str, Any]] = {} | |
| for event in sorted(events, key=lambda item: (item.get("created_at", ""), item.get("event_id", ""))): | |
| round_id = str(event["round_id"]) | |
| record = rounds.setdefault(round_id, {"round_id": round_id, "events": {}}) | |
| record["events"].setdefault(event["event_type"], []).append(event) | |
| if event["event_type"] == "round_completed": | |
| for key, value in event.items(): | |
| if key not in {"event_id", "event_type", "created_at", "round_id"}: | |
| record[key] = value | |
| return [rounds[key] for key in sorted(rounds)] | |
| def main() -> int: | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument("dataset_dir", type=Path, help="Local clone or download of the dataset repo") | |
| parser.add_argument("output", type=Path, help="Destination JSONL path") | |
| args = parser.parse_args() | |
| if not args.dataset_dir.is_dir(): | |
| parser.error(f"Not a directory: {args.dataset_dir}") | |
| events, malformed = read_events(args.dataset_dir) | |
| rounds = materialize(events) | |
| args.output.parent.mkdir(parents=True, exist_ok=True) | |
| with args.output.open("w", encoding="utf-8", newline="\n") as target: | |
| for round_record in rounds: | |
| target.write(json.dumps(round_record, ensure_ascii=False, sort_keys=True) + "\n") | |
| print(f"Wrote {len(rounds)} derived rounds from {len(events)} events; skipped {malformed} malformed files.") | |
| return 0 | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |