"""Throughput of the three evaluators, measured on one idle machine. Each evaluator runs the same prefix of the self-describing instance and is timed after a warm-up, so the rates reported in the paper are not those of runs that shared the processor with each other. """ import json import os import sys import time sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) def runs_path(name: str) -> str: d = os.path.join(REPO, "paper", "runs") os.makedirs(d, exist_ok=True) return os.path.join(d, name) def main() -> int: import torch from selfrep import (M_P, LevEvaluator, NetEvaluator, describe, read_host, run_reference) sigma = read_host() r_H = describe(sigma) out = {} t0 = time.perf_counter() o, n = run_reference(M_P, r_H, expect=sigma) dt = time.perf_counter() - t0 assert o == sigma out["reference"] = {"steps": n, "seconds": dt, "steps_per_second": n / dt} print(f" integer reference: {n:,} steps in {dt:.2f} s " f"({n / dt:,.0f} steps/s)", flush=True) N = 20000 for key, what, ev in ( ("net_cpu", "the netlist of sigma unit by unit, one core", lambda: NetEvaluator(sigma, device="cpu")), ("net_gpu", "the netlist of sigma unit by unit, GPU", lambda: NetEvaluator(sigma, device="cuda", graph=True))): G = ev() G.run(M_P, r_H, max_steps=200) # warm up t0 = time.perf_counter() G.run(M_P, r_H, max_steps=N, expect=sigma) dt = time.perf_counter() - t0 out[key] = {"steps": N, "seconds": dt, "steps_per_second": N / dt, "units": G.info["units"], "entries": G.info["entries"]} print(f" {what}: {N:,} steps in {dt:.1f} s ({N / dt:,.0f} steps/s)", flush=True) D = LevEvaluator(sigma, device="cuda", dense=True) entries = sum(int(W.numel()) for W in D.W) nonzero = sum(int((W != 0).sum()) for W in D.W) D.run(M_P, r_H, max_steps=200) t0 = time.perf_counter() D.run(M_P, r_H, max_steps=N, expect=sigma) dt = time.perf_counter() - t0 rate = N / dt out["dense_gpu"] = {"steps": N, "seconds": dt, "steps_per_second": rate, "entries": entries, "nonzero": nonzero, "bytes_per_step": entries * 4, "bandwidth_gbs": rate * entries * 4 / 1e9} print(f" Lev(N_host) dense on the GPU: {N:,} steps in {dt:.1f} s " f"({rate:,.0f} steps/s); {entries:,} entries of which {nonzero:,} " f"are nonzero, {entries * 4 / 1e6:.0f} MB read per step, " f"{rate * entries * 4 / 1e9:.0f} GB/s", flush=True) G2 = LevEvaluator(sigma, device="cuda", dense=False, graph=True) G2.run(M_P, r_H, max_steps=200) t0 = time.perf_counter() G2.run(M_P, r_H, max_steps=N, expect=sigma) dt = time.perf_counter() - t0 out["graph_gpu"] = {"steps": N, "seconds": dt, "steps_per_second": N / dt} print(f" the same map from its nonzero entries, captured as a CUDA graph: " f"{N:,} steps in {dt:.1f} s ({N / dt:,.0f} steps/s)", flush=True) S = LevEvaluator(sigma, device="cpu", dense=False) S.run(M_P, r_H, max_steps=200) t0 = time.perf_counter() S.run(M_P, r_H, max_steps=N, expect=sigma) dt = time.perf_counter() - t0 out["sparse_cpu"] = {"steps": N, "seconds": dt, "steps_per_second": N / dt} print(f" the same map from its nonzero entries on one core: {N:,} steps in " f"{dt:.1f} s ({N / dt:,.0f} steps/s)", flush=True) json.dump(out, open(runs_path("paper_throughput.json"), "w"), indent=1) return 0 if __name__ == "__main__": sys.exit(main())