File size: 5,121 Bytes
79ed4eb 4a3e194 79ed4eb 4a3e194 79ed4eb 4a3e194 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 | """Throughput of the four evaluators, measured on one idle machine.
Each evaluator runs the same prefix of the self-describing instance and is
timed after a warm-up, so the rates reported in the paper are not those of runs
that shared the processor with each other.
"""
import json
import os
import sys
import time
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
def runs_path(name: str) -> str:
d = os.path.join(REPO, "paper", "runs")
os.makedirs(d, exist_ok=True)
return os.path.join(d, name)
def main() -> int:
import torch
from selfrep import (M_P, LevEvaluator, NetEvaluator, describe, read_host,
run_reference)
torch.set_num_threads(1) # the processor rates are for one core
sigma = read_host()
r_H = describe(sigma)
out = {}
t0 = time.perf_counter()
o, n = run_reference(M_P, r_H, expect=sigma)
dt = time.perf_counter() - t0
assert o == sigma
out["reference"] = {"steps": n, "seconds": dt, "steps_per_second": n / dt}
print(f" integer reference: {n:,} steps in {dt:.2f} s "
f"({n / dt:,.0f} steps/s)", flush=True)
N = 20000
for key, what, ev in (
("net_cpu", "the netlist of sigma unit by unit, one core",
lambda: NetEvaluator(sigma, device="cpu")),
("net_gpu", "the netlist of sigma unit by unit, GPU",
lambda: NetEvaluator(sigma, device="cuda", graph=True))):
G = ev()
G.run(M_P, r_H, max_steps=200) # warm up
t0 = time.perf_counter()
G.run(M_P, r_H, max_steps=N, expect=sigma)
dt = time.perf_counter() - t0
out[key] = {"steps": N, "seconds": dt, "steps_per_second": N / dt,
"units": G.info["units"], "entries": G.info["entries"]}
print(f" {what}: {N:,} steps in {dt:.1f} s ({N / dt:,.0f} steps/s)",
flush=True)
D = LevEvaluator(sigma, device="cuda", dense=True)
entries = sum(int(W.numel()) for W in D.W)
nonzero = sum(int((W != 0).sum()) for W in D.W)
D.run(M_P, r_H, max_steps=200)
t0 = time.perf_counter()
D.run(M_P, r_H, max_steps=N, expect=sigma)
dt = time.perf_counter() - t0
rate = N / dt
out["dense_gpu"] = {"steps": N, "seconds": dt, "steps_per_second": rate,
"entries": entries, "nonzero": nonzero,
"bytes_per_step": entries * 4,
"bandwidth_gbs": rate * entries * 4 / 1e9}
print(f" Lev(N_host) dense on the GPU: {N:,} steps in {dt:.1f} s "
f"({rate:,.0f} steps/s); {entries:,} entries of which {nonzero:,} "
f"are nonzero, {entries * 4 / 1e6:.0f} MB read per step, "
f"{rate * entries * 4 / 1e9:.0f} GB/s", flush=True)
G2 = LevEvaluator(sigma, device="cuda", dense=False, graph=True)
G2.run(M_P, r_H, max_steps=200)
t0 = time.perf_counter()
G2.run(M_P, r_H, max_steps=N, expect=sigma)
dt = time.perf_counter() - t0
out["graph_gpu"] = {"steps": N, "seconds": dt, "steps_per_second": N / dt}
print(f" the same map from its nonzero entries, captured as a CUDA graph: "
f"{N:,} steps in {dt:.1f} s ({N / dt:,.0f} steps/s)", flush=True)
S = LevEvaluator(sigma, device="cpu", dense=False)
S.run(M_P, r_H, max_steps=200)
t0 = time.perf_counter()
S.run(M_P, r_H, max_steps=N, expect=sigma)
dt = time.perf_counter() - t0
out["sparse_cpu"] = {"steps": N, "seconds": dt, "steps_per_second": N / dt}
print(f" the same map from its nonzero entries on one core: {N:,} steps in "
f"{dt:.1f} s ({N / dt:,.0f} steps/s)", flush=True)
# evaluator (d): the netlist in C, one core, on the self-describing instance
import subprocess
import tempfile
from check_c import parse_sigma, write_flat
work = tempfile.mkdtemp(prefix="throughput_")
exe = os.path.join(work, "netlist_eval")
subprocess.run(["gcc", "-O3", "-march=native", "-fopenmp", "-o", exe,
os.path.join(REPO, "src", "netlist_eval.c")], check=True)
nb, units, nxt = parse_sigma(sigma)
write_flat(os.path.join(work, "host.bin"), nb, units, nxt)
open(os.path.join(work, "mP"), "wb").write(bytes(M_P))
open(os.path.join(work, "rH"), "wb").write(r_H)
env = dict(os.environ, OMP_NUM_THREADS="1")
t0 = time.perf_counter()
subprocess.run([exe, "run", os.path.join(work, "host.bin"), os.path.join(work, "mP"),
os.path.join(work, "rH"), os.path.join(work, "out"), "net"],
check=True, capture_output=True, env=env)
dt = time.perf_counter() - t0
assert open(os.path.join(work, "out"), "rb").read() == sigma
out["c_evaluator"] = {"steps": n, "seconds": dt, "steps_per_second": n / dt}
print(f" the netlist of sigma in C, one core: {n:,} steps in {dt:.1f} s "
f"({n / dt:,.0f} steps/s)", flush=True)
json.dump(out, open(runs_path("paper_throughput.json"), "w"), indent=1)
return 0
if __name__ == "__main__":
sys.exit(main())
|