Buckets:
| # /// script | |
| # requires-python = ">=3.9" | |
| # dependencies = ["matplotlib>=3.7", "numpy>=1.24"] | |
| # /// | |
| """Distribution plot for codex_route_ab.py results: one density curve per route. | |
| Usage: | |
| uv run codex_route_ab_plot.py RESULTS_DIR [--title TEXT] [--out FILE.png] | |
| Two separate figures: reasoning output tokens and wall seconds. Each route is a Gaussian kernel | |
| density (Scott's bandwidth) with a dashed median line and rug ticks for every run, so | |
| smoothing never hides individual results (e.g. a zero-reasoning run). Only successful | |
| runs (exit 0 with usage) are plotted. Writes RESULTS_DIR/reasoning_tokens.png and | |
| RESULTS_DIR/wall_seconds.png (or --out-prefix PREFIX -> PREFIX_<metric>.png). | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| from pathlib import Path | |
| import matplotlib | |
| matplotlib.use("Agg") | |
| import matplotlib.pyplot as plt # noqa: E402 | |
| import numpy as np # noqa: E402 | |
| COLOURS = {"api": "#1f77b4", "oauth": "#d62728"} | |
| LABELS = {"api": "API key", "oauth": "ChatGPT OAuth"} | |
| PANELS = [("reasoning_output_tokens", "reasoning output tokens", "reasoning_tokens"), | |
| ("seconds", "wall seconds", "wall_seconds")] | |
| def kde(values: np.ndarray, grid: np.ndarray) -> np.ndarray: | |
| if len(values) < 2 or values.std() == 0: | |
| return np.zeros_like(grid) | |
| bandwidth = values.std(ddof=1) * len(values) ** (-1 / 5) # Scott's rule | |
| z = (grid[:, None] - values[None, :]) / bandwidth | |
| return np.exp(-0.5 * z ** 2).sum(axis=1) / (len(values) * bandwidth * np.sqrt(2 * np.pi)) | |
| def top_of(values, grid): | |
| """Median line height: up to the curve's peak, below the legend.""" | |
| return kde(values, grid).max() * 1.05 | |
| def main(): | |
| p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) | |
| p.add_argument("results_dir") | |
| p.add_argument("--title") | |
| p.add_argument("--out-prefix") | |
| args = p.parse_args() | |
| base = Path(args.results_dir) | |
| rows = [json.loads(l) for l in (base / "results.jsonl").read_text().splitlines() if l.strip()] | |
| meta_path = base / "summary.json" | |
| meta = json.loads(meta_path.read_text()) if meta_path.exists() else {} | |
| ok = [r for r in rows if r["exit"] == 0 and r["usage"]] | |
| routes = [r for r in ("api", "oauth") if any(x["route"] == r for x in ok)] | |
| base_title = args.title or (f"Codex CLI {meta.get('codex_version', '').replace('codex-cli ', '')} · " | |
| f"{meta.get('model', '?')} {meta.get('effort', '')} · API key vs ChatGPT OAuth") | |
| for key, label, stem in PANELS: | |
| fig, ax = plt.subplots(figsize=(8, 4.8)) | |
| data = {r: np.array([x["seconds"] if key == "seconds" else x["usage"].get(key) | |
| for x in ok if x["route"] == r | |
| and (key == "seconds" or x["usage"].get(key) is not None)], float) | |
| for r in routes} | |
| lo = min(v.min() for v in data.values()) | |
| hi = max(v.max() for v in data.values()) | |
| span = hi - lo or 1.0 | |
| grid = np.linspace(max(0.0, lo - 0.15 * span), hi + 0.15 * span, 400) | |
| top = 0.0 | |
| for r, v in data.items(): | |
| density = kde(v, grid) | |
| top = max(top, density.max()) | |
| ax.plot(grid, density, color=COLOURS[r], lw=2, | |
| label=f"{LABELS[r]} (n={len(v)}, median {np.median(v):,.0f})") | |
| ax.fill_between(grid, density, color=COLOURS[r], alpha=0.15) | |
| ax.vlines(np.median(v), 0, top_of(v, grid), color=COLOURS[r], ls="--", lw=1.2) | |
| for r, v in data.items(): # rug: every run | |
| offset = -0.04 * top if r == "api" else -0.08 * top | |
| ax.plot(v, np.full_like(v, offset), "|", color=COLOURS[r], ms=9, alpha=0.7) | |
| ax.set_ylim(-0.11 * top, top * 1.35) # headroom so the legend clears the curves | |
| ax.set_yticks([]) | |
| ax.set_xlabel(label) | |
| ax.set_ylabel("density") | |
| ax.spines[["top", "right", "left"]].set_visible(False) | |
| ax.legend(frameon=False, fontsize=9, loc="upper center", ncol=len(routes)) | |
| ax.set_title(f"{base_title}\n{label}", fontsize=11) | |
| fig.text(0.5, 0.01, "Kernel density; dashed line = median; one tick per run.", | |
| ha="center", fontsize=8, color="0.4") | |
| fig.tight_layout(rect=(0, 0.03, 1, 1)) | |
| out = Path(f"{args.out_prefix}_{stem}.png") if args.out_prefix else base / f"{stem}.png" | |
| fig.savefig(out, dpi=150) | |
| plt.close(fig) | |
| print(out) | |
| if __name__ == "__main__": | |
| main() | |
Xet Storage Details
- Size:
- 4.51 kB
- Xet hash:
- b0e8ca88be69f8aae59c24450d013b6ce2921410852f799e582d761123cca10c
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.