-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathplot_results.py
More file actions
85 lines (66 loc) · 2.75 KB
/
Copy pathplot_results.py
File metadata and controls
85 lines (66 loc) · 2.75 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
"""Turn bench.json into the two figures the README leads with.
Only the DRAM regime is plotted. The cache regime sits on the ~0.04 ms timing
floor, so its numbers say more about the launch overhead than about the
kernels. L = 512 and L = 1024 are dropped as well: in the final run the clocks
were still settling there, and level 2 -- which acts as the calibration control
because it is the most predictable kernel in the set -- reads 6.17 ms instead
of its stable 4.31.
Run: build_env.bat plot_results.py
"""
import json
import pathlib
import matplotlib
matplotlib.use("Agg")
import matplotlib.pyplot as plt
HERE = pathlib.Path(__file__).parent
RESULTS = HERE / "results" / "NVIDIA_GeForce_RTX_4050_Laptop_GPU"
# Below this length the memory clock had not settled; see the module docstring.
MIN_TRUSTED_L = 2048
SERIES = [
("level2_C1024", "Level 2: three passes", "#888888", "o"),
("level3_C1024", "Level 3: serial look-back", "#d62728", "s"),
("level3w_C1024", "Level 3b: warp look-back", "#ff7f0e", "^"),
("level3i_C4096i8", "Level 3c: + 8 items/thread", "#2ca02c", "D"),
]
def load():
with open(RESULTS / "bench.json") as f:
data = json.load(f)
rows = [r for r in data["records"]
if r["regime"] == "dram" and r["L"] >= MIN_TRUSTED_L]
return data["meta"], rows
def series(rows, impl, field):
got = sorted((r["L"], r[field]) for r in rows if r["impl"] == impl)
return [x for x, _ in got], [y for _, y in got]
def main():
meta, rows = load()
fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 4.6))
for impl, label, colour, marker in SERIES:
xs, ys = series(rows, impl, "median_ms")
ax1.plot(xs, ys, marker=marker, color=colour, label=label, lw=1.8, ms=5)
xs, ys = series(rows, impl, "pct_of_theoretical")
ax2.plot(xs, ys, marker=marker, color=colour, label=label, lw=1.8, ms=5)
ax1.set_xscale("log", base=2)
ax1.set_xlabel("sequence length L")
ax1.set_ylabel("ms per call")
ax1.set_title("Time (lower is better)")
ax1.set_ylim(bottom=0)
ax1.grid(alpha=0.3)
ax1.legend(fontsize=8)
ax2.set_xscale("log", base=2)
ax2.set_xlabel("sequence length L")
ax2.set_ylabel("% of 192 GB/s theoretical peak")
ax2.set_title("Bandwidth efficiency (higher is better)")
ax2.axhline(100, color="black", ls="--", lw=1)
ax2.text(rows[0]["L"], 101, "theoretical peak", fontsize=8, va="bottom")
ax2.set_ylim(0, 110)
ax2.grid(alpha=0.3)
ax2.legend(fontsize=8, loc="lower left")
fig.suptitle(
f"Selective-SSM scan on {meta['gpu']} "
f"(memory pinned at 8001 MHz, 42-45 W)", fontsize=11)
fig.tight_layout()
out = RESULTS / "scan_results.png"
fig.savefig(out, dpi=150)
print("wrote", out)
if __name__ == "__main__":
main()