threshold-computers / src /throughput.py
phanerozoic's picture
serialize the host netlist unit by unit and instantiate each generation from the emitted bytes
0654a8d
Raw
History Blame Contribute Delete
3.79 kB
"""Throughput of the three evaluators, measured on one idle machine.
Each evaluator runs the same prefix of the self-describing instance and is
timed after a warm-up, so the rates reported in the paper are not those of runs
that shared the processor with each other.
"""
import json
import os
import sys
import time
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
def runs_path(name: str) -> str:
d = os.path.join(REPO, "paper", "runs")
os.makedirs(d, exist_ok=True)
return os.path.join(d, name)
def main() -> int:
import torch
from selfrep import (M_P, LevEvaluator, NetEvaluator, describe, read_host,
run_reference)
sigma = read_host()
r_H = describe(sigma)
out = {}
t0 = time.perf_counter()
o, n = run_reference(M_P, r_H, expect=sigma)
dt = time.perf_counter() - t0
assert o == sigma
out["reference"] = {"steps": n, "seconds": dt, "steps_per_second": n / dt}
print(f" integer reference: {n:,} steps in {dt:.2f} s "
f"({n / dt:,.0f} steps/s)", flush=True)
N = 20000
for key, what, ev in (
("net_cpu", "the netlist of sigma unit by unit, one core",
lambda: NetEvaluator(sigma, device="cpu")),
("net_gpu", "the netlist of sigma unit by unit, GPU",
lambda: NetEvaluator(sigma, device="cuda", graph=True))):
G = ev()
G.run(M_P, r_H, max_steps=200) # warm up
t0 = time.perf_counter()
G.run(M_P, r_H, max_steps=N, expect=sigma)
dt = time.perf_counter() - t0
out[key] = {"steps": N, "seconds": dt, "steps_per_second": N / dt,
"units": G.info["units"], "entries": G.info["entries"]}
print(f" {what}: {N:,} steps in {dt:.1f} s ({N / dt:,.0f} steps/s)",
flush=True)
D = LevEvaluator(sigma, device="cuda", dense=True)
entries = sum(int(W.numel()) for W in D.W)
nonzero = sum(int((W != 0).sum()) for W in D.W)
D.run(M_P, r_H, max_steps=200)
t0 = time.perf_counter()
D.run(M_P, r_H, max_steps=N, expect=sigma)
dt = time.perf_counter() - t0
rate = N / dt
out["dense_gpu"] = {"steps": N, "seconds": dt, "steps_per_second": rate,
"entries": entries, "nonzero": nonzero,
"bytes_per_step": entries * 4,
"bandwidth_gbs": rate * entries * 4 / 1e9}
print(f" Lev(N_host) dense on the GPU: {N:,} steps in {dt:.1f} s "
f"({rate:,.0f} steps/s); {entries:,} entries of which {nonzero:,} "
f"are nonzero, {entries * 4 / 1e6:.0f} MB read per step, "
f"{rate * entries * 4 / 1e9:.0f} GB/s", flush=True)
G2 = LevEvaluator(sigma, device="cuda", dense=False, graph=True)
G2.run(M_P, r_H, max_steps=200)
t0 = time.perf_counter()
G2.run(M_P, r_H, max_steps=N, expect=sigma)
dt = time.perf_counter() - t0
out["graph_gpu"] = {"steps": N, "seconds": dt, "steps_per_second": N / dt}
print(f" the same map from its nonzero entries, captured as a CUDA graph: "
f"{N:,} steps in {dt:.1f} s ({N / dt:,.0f} steps/s)", flush=True)
S = LevEvaluator(sigma, device="cpu", dense=False)
S.run(M_P, r_H, max_steps=200)
t0 = time.perf_counter()
S.run(M_P, r_H, max_steps=N, expect=sigma)
dt = time.perf_counter() - t0
out["sparse_cpu"] = {"steps": N, "seconds": dt, "steps_per_second": N / dt}
print(f" the same map from its nonzero entries on one core: {N:,} steps in "
f"{dt:.1f} s ({N / dt:,.0f} steps/s)", flush=True)
json.dump(out, open(runs_path("paper_throughput.json"), "w"), indent=1)
return 0
if __name__ == "__main__":
sys.exit(main())