File size: 2,472 Bytes
cd13720
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
#!/usr/bin/env python3
"""Build a deterministic JSONL round snapshot from a local dataset checkout.

This is an offline derived artifact. It never commits or uploads files.
"""

from __future__ import annotations

import argparse
import json
from pathlib import Path
from typing import Any


def read_events(root: Path) -> tuple[list[dict[str, Any]], int]:
    events: list[dict[str, Any]] = []
    malformed = 0
    for path in sorted(root.rglob("*.json")):
        if path.name.endswith("schema.json"):
            continue
        try:
            record = json.loads(path.read_text(encoding="utf-8"))
        except (OSError, json.JSONDecodeError):
            malformed += 1
            continue
        if isinstance(record, dict) and record.get("round_id") and record.get("event_type"):
            events.append(record)
    return events, malformed


def materialize(events: list[dict[str, Any]]) -> list[dict[str, Any]]:
    rounds: dict[str, dict[str, Any]] = {}
    for event in sorted(events, key=lambda item: (item.get("created_at", ""), item.get("event_id", ""))):
        round_id = str(event["round_id"])
        record = rounds.setdefault(round_id, {"round_id": round_id, "events": {}})
        record["events"].setdefault(event["event_type"], []).append(event)
        if event["event_type"] == "round_completed":
            for key, value in event.items():
                if key not in {"event_id", "event_type", "created_at", "round_id"}:
                    record[key] = value
    return [rounds[key] for key in sorted(rounds)]


def main() -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("dataset_dir", type=Path, help="Local clone or download of the dataset repo")
    parser.add_argument("output", type=Path, help="Destination JSONL path")
    args = parser.parse_args()
    if not args.dataset_dir.is_dir():
        parser.error(f"Not a directory: {args.dataset_dir}")
    events, malformed = read_events(args.dataset_dir)
    rounds = materialize(events)
    args.output.parent.mkdir(parents=True, exist_ok=True)
    with args.output.open("w", encoding="utf-8", newline="\n") as target:
        for round_record in rounds:
            target.write(json.dumps(round_record, ensure_ascii=False, sort_keys=True) + "\n")
    print(f"Wrote {len(rounds)} derived rounds from {len(events)} events; skipped {malformed} malformed files.")
    return 0


if __name__ == "__main__":
    raise SystemExit(main())