File size: 5,121 Bytes
79ed4eb
4a3e194
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
79ed4eb
4a3e194
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
79ed4eb
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4a3e194
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
"""Throughput of the four evaluators, measured on one idle machine.

Each evaluator runs the same prefix of the self-describing instance and is
timed after a warm-up, so the rates reported in the paper are not those of runs
that shared the processor with each other.
"""
import json
import os
import sys
import time

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))

REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))


def runs_path(name: str) -> str:
    d = os.path.join(REPO, "paper", "runs")
    os.makedirs(d, exist_ok=True)
    return os.path.join(d, name)


def main() -> int:
    import torch
    from selfrep import (M_P, LevEvaluator, NetEvaluator, describe, read_host,
                         run_reference)
    torch.set_num_threads(1)                 # the processor rates are for one core
    sigma = read_host()
    r_H = describe(sigma)
    out = {}

    t0 = time.perf_counter()
    o, n = run_reference(M_P, r_H, expect=sigma)
    dt = time.perf_counter() - t0
    assert o == sigma
    out["reference"] = {"steps": n, "seconds": dt, "steps_per_second": n / dt}
    print(f"  integer reference: {n:,} steps in {dt:.2f} s "
          f"({n / dt:,.0f} steps/s)", flush=True)

    N = 20000
    for key, what, ev in (
            ("net_cpu", "the netlist of sigma unit by unit, one core",
             lambda: NetEvaluator(sigma, device="cpu")),
            ("net_gpu", "the netlist of sigma unit by unit, GPU",
             lambda: NetEvaluator(sigma, device="cuda", graph=True))):
        G = ev()
        G.run(M_P, r_H, max_steps=200)                   # warm up
        t0 = time.perf_counter()
        G.run(M_P, r_H, max_steps=N, expect=sigma)
        dt = time.perf_counter() - t0
        out[key] = {"steps": N, "seconds": dt, "steps_per_second": N / dt,
                    "units": G.info["units"], "entries": G.info["entries"]}
        print(f"  {what}: {N:,} steps in {dt:.1f} s ({N / dt:,.0f} steps/s)",
              flush=True)

    D = LevEvaluator(sigma, device="cuda", dense=True)
    entries = sum(int(W.numel()) for W in D.W)
    nonzero = sum(int((W != 0).sum()) for W in D.W)
    D.run(M_P, r_H, max_steps=200)
    t0 = time.perf_counter()
    D.run(M_P, r_H, max_steps=N, expect=sigma)
    dt = time.perf_counter() - t0
    rate = N / dt
    out["dense_gpu"] = {"steps": N, "seconds": dt, "steps_per_second": rate,
                        "entries": entries, "nonzero": nonzero,
                        "bytes_per_step": entries * 4,
                        "bandwidth_gbs": rate * entries * 4 / 1e9}
    print(f"  Lev(N_host) dense on the GPU: {N:,} steps in {dt:.1f} s "
          f"({rate:,.0f} steps/s); {entries:,} entries of which {nonzero:,} "
          f"are nonzero, {entries * 4 / 1e6:.0f} MB read per step, "
          f"{rate * entries * 4 / 1e9:.0f} GB/s", flush=True)

    G2 = LevEvaluator(sigma, device="cuda", dense=False, graph=True)
    G2.run(M_P, r_H, max_steps=200)
    t0 = time.perf_counter()
    G2.run(M_P, r_H, max_steps=N, expect=sigma)
    dt = time.perf_counter() - t0
    out["graph_gpu"] = {"steps": N, "seconds": dt, "steps_per_second": N / dt}
    print(f"  the same map from its nonzero entries, captured as a CUDA graph: "
          f"{N:,} steps in {dt:.1f} s ({N / dt:,.0f} steps/s)", flush=True)

    S = LevEvaluator(sigma, device="cpu", dense=False)
    S.run(M_P, r_H, max_steps=200)
    t0 = time.perf_counter()
    S.run(M_P, r_H, max_steps=N, expect=sigma)
    dt = time.perf_counter() - t0
    out["sparse_cpu"] = {"steps": N, "seconds": dt, "steps_per_second": N / dt}
    print(f"  the same map from its nonzero entries on one core: {N:,} steps in "
          f"{dt:.1f} s ({N / dt:,.0f} steps/s)", flush=True)

    # evaluator (d): the netlist in C, one core, on the self-describing instance
    import subprocess
    import tempfile
    from check_c import parse_sigma, write_flat
    work = tempfile.mkdtemp(prefix="throughput_")
    exe = os.path.join(work, "netlist_eval")
    subprocess.run(["gcc", "-O3", "-march=native", "-fopenmp", "-o", exe,
                    os.path.join(REPO, "src", "netlist_eval.c")], check=True)
    nb, units, nxt = parse_sigma(sigma)
    write_flat(os.path.join(work, "host.bin"), nb, units, nxt)
    open(os.path.join(work, "mP"), "wb").write(bytes(M_P))
    open(os.path.join(work, "rH"), "wb").write(r_H)
    env = dict(os.environ, OMP_NUM_THREADS="1")
    t0 = time.perf_counter()
    subprocess.run([exe, "run", os.path.join(work, "host.bin"), os.path.join(work, "mP"),
                    os.path.join(work, "rH"), os.path.join(work, "out"), "net"],
                   check=True, capture_output=True, env=env)
    dt = time.perf_counter() - t0
    assert open(os.path.join(work, "out"), "rb").read() == sigma
    out["c_evaluator"] = {"steps": n, "seconds": dt, "steps_per_second": n / dt}
    print(f"  the netlist of sigma in C, one core: {n:,} steps in {dt:.1f} s "
          f"({n / dt:,.0f} steps/s)", flush=True)

    json.dump(out, open(runs_path("paper_throughput.json"), "w"), indent=1)
    return 0


if __name__ == "__main__":
    sys.exit(main())