Download scripts/bench_threads.py from cazyundee/training: direct link, hf CLI and curl.
- Browser
- Download file 3.18 kB
-
https://huggingface.co/spaces/cazyundee/training/resolve/main/scripts/bench_threads.py
- Command line
-
hf download hf://spaces/cazyundee/training/scripts/bench_threads.py
-
curl -L -o bench_threads.py https://huggingface.co/spaces/cazyundee/training/resolve/main/scripts/bench_threads.py
3.18 kB
| #!/usr/bin/env python3 | |
| """ | |
| Measure self-play throughput for (parallel experiments x threads each). | |
| The research machine has 16 cores. The question is whether one 16-thread job or | |
| several narrow jobs give more total positions/second. Tiny models usually | |
| parallelise badly across threads, so N x 1-thread often wins -- but measure it. | |
| python scripts/bench_threads.py --total-cores 16 --games 8 --plies 40 | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import itertools | |
| import json | |
| import multiprocessing as mp | |
| import os | |
| import sys | |
| import time | |
| sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) | |
| def worker(threads, games, plies, steps, seed, q): | |
| os.environ["OMP_NUM_THREADS"] = str(threads) | |
| os.environ["MKL_NUM_THREADS"] = str(threads) | |
| import torch | |
| torch.set_num_threads(threads) | |
| from tinychess.config import baseline_config | |
| from tinychess.model import TinyChess | |
| from tinychess.selfplay import SelfPlayConfig, play_games | |
| torch.manual_seed(seed) | |
| m = TinyChess(baseline_config()) | |
| t = time.time() | |
| _, st = play_games(m, cfg=SelfPlayConfig(n_games=games, max_plies=plies, | |
| steps=steps, seed=seed, | |
| adjudicate="material")) | |
| q.put({"moves": st["total_moves"], "seconds": time.time() - t}) | |
| def run(nproc, threads, games, plies, steps): | |
| q = mp.Queue() | |
| ps = [mp.Process(target=worker, args=(threads, games, plies, steps, 100 + i, q)) | |
| for i in range(nproc)] | |
| t0 = time.time() | |
| for p in ps: | |
| p.start() | |
| res = [q.get() for _ in ps] | |
| for p in ps: | |
| p.join() | |
| wall = time.time() - t0 | |
| moves = sum(r["moves"] for r in res) | |
| return {"procs": nproc, "threads": threads, "wall": round(wall, 2), | |
| "moves": moves, "moves_per_s": round(moves / wall, 1)} | |
| def main(): | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--total-cores", type=int, default=os.cpu_count()) | |
| ap.add_argument("--games", type=int, default=8) | |
| ap.add_argument("--plies", type=int, default=40) | |
| ap.add_argument("--steps", type=int, default=4) | |
| ap.add_argument("--out", default="runs/bench_threads.json") | |
| a = ap.parse_args() | |
| C = a.total_cores | |
| combos = [] | |
| n = 1 | |
| while n <= C: | |
| combos.append((n, max(1, C // n))) | |
| n *= 2 | |
| print(f"cores={C} benchmarking {len(combos)} configurations") | |
| print(f"{'procs':>6} {'thr':>4} {'wall(s)':>9} {'moves':>8} {'moves/s':>9}") | |
| rows = [] | |
| for nproc, thr in combos: | |
| r = run(nproc, thr, a.games, a.plies, a.steps) | |
| rows.append(r) | |
| print(f"{r['procs']:>6} {r['threads']:>4} {r['wall']:>9.2f} " | |
| f"{r['moves']:>8} {r['moves_per_s']:>9}") | |
| best = max(rows, key=lambda r: r["moves_per_s"]) | |
| print(f"\nBEST: {best['procs']} process(es) x {best['threads']} thread(s) " | |
| f"= {best['moves_per_s']} moves/s") | |
| os.makedirs(os.path.dirname(os.path.abspath(a.out)), exist_ok=True) | |
| json.dump({"cores": C, "rows": rows, "best": best}, open(a.out, "w"), indent=2) | |
| if __name__ == "__main__": | |
| mp.set_start_method("spawn", force=True) | |
| main() | |