serialize the host netlist unit by unit and instantiate each generation from the emitted bytes
0654a8d | """Throughput of the three evaluators, measured on one idle machine. | |
| Each evaluator runs the same prefix of the self-describing instance and is | |
| timed after a warm-up, so the rates reported in the paper are not those of runs | |
| that shared the processor with each other. | |
| """ | |
| import json | |
| import os | |
| import sys | |
| import time | |
| sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) | |
| REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) | |
| def runs_path(name: str) -> str: | |
| d = os.path.join(REPO, "paper", "runs") | |
| os.makedirs(d, exist_ok=True) | |
| return os.path.join(d, name) | |
| def main() -> int: | |
| import torch | |
| from selfrep import (M_P, LevEvaluator, NetEvaluator, describe, read_host, | |
| run_reference) | |
| sigma = read_host() | |
| r_H = describe(sigma) | |
| out = {} | |
| t0 = time.perf_counter() | |
| o, n = run_reference(M_P, r_H, expect=sigma) | |
| dt = time.perf_counter() - t0 | |
| assert o == sigma | |
| out["reference"] = {"steps": n, "seconds": dt, "steps_per_second": n / dt} | |
| print(f" integer reference: {n:,} steps in {dt:.2f} s " | |
| f"({n / dt:,.0f} steps/s)", flush=True) | |
| N = 20000 | |
| for key, what, ev in ( | |
| ("net_cpu", "the netlist of sigma unit by unit, one core", | |
| lambda: NetEvaluator(sigma, device="cpu")), | |
| ("net_gpu", "the netlist of sigma unit by unit, GPU", | |
| lambda: NetEvaluator(sigma, device="cuda", graph=True))): | |
| G = ev() | |
| G.run(M_P, r_H, max_steps=200) # warm up | |
| t0 = time.perf_counter() | |
| G.run(M_P, r_H, max_steps=N, expect=sigma) | |
| dt = time.perf_counter() - t0 | |
| out[key] = {"steps": N, "seconds": dt, "steps_per_second": N / dt, | |
| "units": G.info["units"], "entries": G.info["entries"]} | |
| print(f" {what}: {N:,} steps in {dt:.1f} s ({N / dt:,.0f} steps/s)", | |
| flush=True) | |
| D = LevEvaluator(sigma, device="cuda", dense=True) | |
| entries = sum(int(W.numel()) for W in D.W) | |
| nonzero = sum(int((W != 0).sum()) for W in D.W) | |
| D.run(M_P, r_H, max_steps=200) | |
| t0 = time.perf_counter() | |
| D.run(M_P, r_H, max_steps=N, expect=sigma) | |
| dt = time.perf_counter() - t0 | |
| rate = N / dt | |
| out["dense_gpu"] = {"steps": N, "seconds": dt, "steps_per_second": rate, | |
| "entries": entries, "nonzero": nonzero, | |
| "bytes_per_step": entries * 4, | |
| "bandwidth_gbs": rate * entries * 4 / 1e9} | |
| print(f" Lev(N_host) dense on the GPU: {N:,} steps in {dt:.1f} s " | |
| f"({rate:,.0f} steps/s); {entries:,} entries of which {nonzero:,} " | |
| f"are nonzero, {entries * 4 / 1e6:.0f} MB read per step, " | |
| f"{rate * entries * 4 / 1e9:.0f} GB/s", flush=True) | |
| G2 = LevEvaluator(sigma, device="cuda", dense=False, graph=True) | |
| G2.run(M_P, r_H, max_steps=200) | |
| t0 = time.perf_counter() | |
| G2.run(M_P, r_H, max_steps=N, expect=sigma) | |
| dt = time.perf_counter() - t0 | |
| out["graph_gpu"] = {"steps": N, "seconds": dt, "steps_per_second": N / dt} | |
| print(f" the same map from its nonzero entries, captured as a CUDA graph: " | |
| f"{N:,} steps in {dt:.1f} s ({N / dt:,.0f} steps/s)", flush=True) | |
| S = LevEvaluator(sigma, device="cpu", dense=False) | |
| S.run(M_P, r_H, max_steps=200) | |
| t0 = time.perf_counter() | |
| S.run(M_P, r_H, max_steps=N, expect=sigma) | |
| dt = time.perf_counter() - t0 | |
| out["sparse_cpu"] = {"steps": N, "seconds": dt, "steps_per_second": N / dt} | |
| print(f" the same map from its nonzero entries on one core: {N:,} steps in " | |
| f"{dt:.1f} s ({N / dt:,.0f} steps/s)", flush=True) | |
| json.dump(out, open(runs_path("paper_throughput.json"), "w"), indent=1) | |
| return 0 | |
| if __name__ == "__main__": | |
| sys.exit(main()) | |