"""Render the published charts from evaluation/2026-09-24/performance.json. Usage: python evaluation/render_charts.py """ import json from pathlib import Path import matplotlib matplotlib.use("Agg") import matplotlib.pyplot as plt import numpy as np RESULTS = Path(__file__).resolve().parent / "2026-09-24" data = json.loads((RESULTS / "performance.json").read_text(encoding="utf-8")) bench, speed = data["benchmark"], data["speed"] cohorts = bench["cohorts"] names = list(cohorts) accuracies = [100 * cohorts[name]["accuracy"] for name in names] ece = [cohorts[name]["ece_10"] for name in names] palette = ["#2a58d7", "#129184", "#dc8240"] plt.rcParams.update({ "font.family": "DejaVu Sans", "font.size": 11, "axes.spines.top": False, "axes.spines.right": False, "axes.edgecolor": "#c8d3e4", "axes.facecolor": "#f5f7fb", "text.color": "#182640", }) def public_chart(ax): bars = ax.barh(names[::-1], accuracies[::-1], color=palette[::-1], height=.58) ax.set_xlim(0, 105) ax.set_xlabel("Accuracy (%)") ax.set_title("JevBench v1.4.1 public tasks", loc="left", fontweight="bold") for bar, name, value in zip(bars, names[::-1], accuracies[::-1]): ax.text(value + 2, bar.get_y() + bar.get_height() / 2, f"{value:.2f}% (n={cohorts[name]['n']})", va="center", fontsize=10) ax.text(.02, -.18, f"Total {bench['public_correct']} / {bench['public_n']} = {100*bench['public_accuracy']:.2f}%", transform=ax.transAxes, fontweight="bold") def latency_chart(ax): x = np.arange(2) p50 = [speed[key]["single_decision"]["p50_ms"] for key in ("h200", "cpu")] p95 = [speed[key]["single_decision"]["p95_ms"] for key in ("h200", "cpu")] ax.bar(x - .18, p50, .35, color=palette[0], label="p50") ax.bar(x + .18, p95, .35, color=palette[2], label="p95") ax.set_yscale("log") ax.set_ylim(5, 2100) ax.set_xticks(x, ["H200 GPU", "Xeon CPU"]) ax.set_ylabel("Milliseconds (log scale)") ax.set_title("Warm single-decision latency", loc="left", fontweight="bold") ax.legend(frameon=False) for pos, values, offset in ((0, p50, -.18), (1, p95, .18)): for index, value in enumerate(values): ax.text(index + offset, value * 1.15, f"{value:.1f} ms", ha="center", fontsize=9) def throughput_chart(ax): values = [speed[key]["throughput"]["decisions_per_second"] for key in ("h200", "cpu")] batches = [speed[key]["throughput"]["batch_size"] for key in ("h200", "cpu")] ax.bar([f"H200\nbatch {batches[0]}", f"Xeon CPU\nbatch {batches[1]}"], values, color=palette[:2], width=.58) ax.set_yscale("log") ax.set_ylim(1, 2000) ax.set_ylabel("Decisions/s (log scale)") ax.set_title("Warm batched throughput", loc="left", fontweight="bold") for index, value in enumerate(values): ax.text(index, value * 1.15, f"{value:.2f}/s", ha="center", fontsize=10) fig, ax = plt.subplots(figsize=(8.2, 4.8), layout="constrained") public_chart(ax) fig.savefig(RESULTS / "accuracy.png", dpi=180) plt.close(fig) fig, axes = plt.subplots(1, 2, figsize=(11, 4.7), layout="constrained") latency_chart(axes[0]) throughput_chart(axes[1]) fig.suptitle("Pivot | local FP32 inference", fontsize=16, fontweight="bold") fig.savefig(RESULTS / "latency_throughput.png", dpi=180) plt.close(fig) fig, axes = plt.subplots(2, 2, figsize=(12.6, 8.8), layout="constrained") public_chart(axes[0, 0]) axes[0, 1].barh(names[::-1], ece[::-1], color=palette[::-1], height=.58) axes[0, 1].set_xlim(0, .68) axes[0, 1].set_xlabel("ECE, 10 bins (lower is better)") axes[0, 1].set_title("Public calibration", loc="left", fontweight="bold") for index, value in enumerate(ece[::-1]): axes[0, 1].text(value + .01, index, f"{value:.3f}", va="center") latency_chart(axes[1, 0]) throughput_chart(axes[1, 1]) fig.suptitle("Pivot | measured public performance", fontsize=19, fontweight="bold") fig.text(.02, -.015, "Pinned checkpoint | warm local tokenization + inference + scoring | throughput batch sizes differ", fontsize=9, color="#53637b") fig.savefig(RESULTS / "performance_overview.png", dpi=180, bbox_inches="tight") plt.close(fig) print(f"Saved charts in {RESULTS}")