Buckets:
| #!/usr/bin/env python3 | |
| """Report for codex_route_ab.py output: time / output-token statistics and a basic chart. | |
| Usage: | |
| python3 codex_route_ab_report.py RESULTS_DIR [--title TEXT] | |
| Reads RESULTS_DIR/results.jsonl (+ summary.json metadata) and writes, into RESULTS_DIR: | |
| report.md - per-route n, correctness, median/mean/p10/p90 of wall seconds, output, | |
| reasoning and input tokens; OAuth/API median ratios with bootstrap 95% CIs; | |
| Mann-Whitney U (normal approximation) p-values. | |
| chart.svg - left: wall seconds vs output tokens per run; right: reasoning tokens | |
| per route (strip plot with median bar). Stdlib only, no plotting library. | |
| Only successful runs (exit 0 with usage) enter statistics; failures are counted. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| import math | |
| import random | |
| import statistics | |
| from pathlib import Path | |
| COLOURS = {"api": "#1f77b4", "oauth": "#d62728"} | |
| METRICS = [("seconds", "wall seconds"), ("output_tokens", "output tokens"), | |
| ("reasoning_output_tokens", "reasoning tokens"), ("input_tokens", "input tokens")] | |
| def value(row, key): | |
| return row["seconds"] if key == "seconds" else row["usage"].get(key) | |
| def quantile(values, q): | |
| values = sorted(values) | |
| if not values: | |
| return None | |
| pos = (len(values) - 1) * q | |
| lo, hi = math.floor(pos), math.ceil(pos) | |
| return values[lo] + (values[hi] - values[lo]) * (pos - lo) | |
| def mann_whitney_p(a, b): | |
| """Two-sided p, normal approximation with tie correction (fine for n >= ~8).""" | |
| pooled = sorted((v, i) for i, v in enumerate(a + b)) | |
| ranks, i = [0.0] * len(pooled), 0 | |
| ties = 0.0 | |
| while i < len(pooled): | |
| j = i | |
| while j + 1 < len(pooled) and pooled[j + 1][0] == pooled[i][0]: | |
| j += 1 | |
| for k in range(i, j + 1): | |
| ranks[pooled[k][1]] = (i + j) / 2 + 1 | |
| t = j - i + 1 | |
| ties += t ** 3 - t | |
| i = j + 1 | |
| n1, n2 = len(a), len(b) | |
| u = sum(ranks[:n1]) - n1 * (n1 + 1) / 2 | |
| n = n1 + n2 | |
| var = n1 * n2 / 12 * ((n + 1) - ties / (n * (n - 1))) | |
| if var <= 0: | |
| return 1.0 | |
| z = (abs(u - n1 * n2 / 2) - 0.5) / math.sqrt(var) | |
| return math.erfc(max(z, 0) / math.sqrt(2)) | |
| def bootstrap_ratio(a, b, iterations=5000, seed=1): | |
| """95% CI of median(b)/median(a) by resampling each arm.""" | |
| rng = random.Random(seed) | |
| ratios = sorted(statistics.median(rng.choices(b, k=len(b))) / | |
| statistics.median(rng.choices(a, k=len(a))) for _ in range(iterations)) | |
| return ratios[int(0.025 * iterations)], ratios[int(0.975 * iterations) - 1] | |
| def fmt(v, digits=0): | |
| return "—" if v is None else f"{v:,.{digits}f}" | |
| def report(rows, meta, title): | |
| routes = [r for r in ("api", "oauth") if any(x["route"] == r for x in rows)] | |
| ok = {r: [x for x in rows if x["route"] == r and x["exit"] == 0 and x["usage"]] for r in routes} | |
| lines = [f"# {title}", "", | |
| f"Codex CLI `{meta.get('codex_version')}` · model `{meta.get('model')}` · effort " | |
| f"`{meta.get('effort')}` · prompt sha256 `{str(meta.get('prompt_sha256'))[:12]}` · " | |
| f"parallel {meta.get('parallel', 1)} · started {meta.get('started_at')}", "", | |
| "| route | runs | ok | correct | metric | median | mean | p10 | p90 |", | |
| "|---|---|---|---|---|---|---|---|---|"] | |
| for route in routes: | |
| runs = sum(x["route"] == route for x in rows) | |
| correct = sum(bool(x.get("correct")) for x in ok[route]) | |
| for i, (key, name) in enumerate(METRICS): | |
| vals = [value(x, key) for x in ok[route] if value(x, key) is not None] | |
| digits = 1 if key == "seconds" else 0 | |
| head = f"| {route} | {runs} | {len(ok[route])} | {correct} |" if i == 0 else "| | | | |" | |
| lines.append(f"{head} {name} | {fmt(quantile(vals, .5), digits)} | " | |
| f"{fmt(statistics.mean(vals) if vals else None, digits)} | " | |
| f"{fmt(quantile(vals, .1), digits)} | {fmt(quantile(vals, .9), digits)} |") | |
| if set(routes) == {"api", "oauth"}: | |
| lines += ["", "| metric | OAuth/API median ratio | bootstrap 95% CI | Mann-Whitney p |", | |
| "|---|---|---|---|"] | |
| for key, name in METRICS: | |
| a = [value(x, key) for x in ok["api"] if value(x, key) is not None] | |
| b = [value(x, key) for x in ok["oauth"] if value(x, key) is not None] | |
| if len(a) < 2 or len(b) < 2: | |
| continue | |
| lo, hi = bootstrap_ratio(a, b) | |
| ratio = statistics.median(b) / statistics.median(a) | |
| lines.append(f"| {name} | {ratio:.2f} | {lo:.2f}–{hi:.2f} | {mann_whitney_p(a, b):.2g} |") | |
| tps = {r: [value(x, "output_tokens") / x["seconds"] for x in ok[r] if x["seconds"]] | |
| for r in routes} | |
| lines += ["", "Output tokens per wall second (median): " + " · ".join( | |
| f"{r} {statistics.median(v):.1f}" for r, v in tps.items() if v)] | |
| failed = [x for x in rows if not (x["exit"] == 0 and x["usage"])] | |
| lines += ["", f"Failed/unusable runs: {len(failed)}", "", | |
| "Wall seconds include Codex CLI start-up and queueing; with `parallel > 1` " | |
| "both routes share the same concurrency.", ""] | |
| return "\n".join(lines), ok | |
| def svg_chart(ok, title): | |
| width, height, pad = 1000, 420, 75 | |
| panel_w = (width - 3 * pad) / 2 | |
| parts = [f'<svg xmlns="http://www.w3.org/2000/svg" width="{width}" height="{height}" ' | |
| f'font-family="sans-serif" font-size="12">', | |
| f'<rect width="{width}" height="{height}" fill="white"/>', | |
| f'<text x="{width / 2}" y="22" text-anchor="middle" font-size="15">{title}</text>'] | |
| points = [(value(x, "seconds"), value(x, "output_tokens"), r) | |
| for r, xs in ok.items() for x in xs if value(x, "output_tokens") is not None] | |
| if not points: | |
| return "\n".join(parts + ["</svg>"]) | |
| max_s = max(p[0] for p in points) * 1.05 | |
| max_o = max(p[1] for p in points) * 1.05 | |
| top, bottom = 50, height - pad | |
| def axes(x0, xlabel, ylabel, ymax, xticks): | |
| out = [f'<line x1="{x0}" y1="{bottom}" x2="{x0 + panel_w}" y2="{bottom}" stroke="black"/>', | |
| f'<line x1="{x0}" y1="{top}" x2="{x0}" y2="{bottom}" stroke="black"/>', | |
| f'<text x="{x0 + panel_w / 2}" y="{height - 18}" text-anchor="middle">{xlabel}</text>', | |
| f'<text x="{x0 - 58}" y="{(top + bottom) / 2}" text-anchor="middle" ' | |
| f'transform="rotate(-90 {x0 - 58} {(top + bottom) / 2})">{ylabel}</text>'] | |
| for k in range(5): | |
| v = ymax * k / 4 | |
| y = bottom - (bottom - top) * k / 4 | |
| out.append(f'<text x="{x0 - 6}" y="{y + 4}" text-anchor="end">{v:,.0f}</text>') | |
| out.append(f'<line x1="{x0}" y1="{y}" x2="{x0 + panel_w}" y2="{y}" stroke="#eee"/>') | |
| for x, text in xticks: | |
| out.append(f'<text x="{x}" y="{bottom + 16}" text-anchor="middle">{text}</text>') | |
| return out | |
| x0 = pad | |
| ticks = [(x0 + panel_w * k / 4, f"{max_s * k / 4:.0f}") for k in range(5)] | |
| parts += axes(x0, "wall seconds", "output tokens", max_o, ticks) | |
| for s, o, r in points: | |
| cx = x0 + panel_w * s / max_s | |
| cy = bottom - (bottom - top) * o / max_o | |
| parts.append(f'<circle cx="{cx:.1f}" cy="{cy:.1f}" r="3.5" fill="{COLOURS[r]}" fill-opacity="0.6"/>') | |
| x1 = 2 * pad + panel_w | |
| reasoning = {r: [value(x, "reasoning_output_tokens") for x in xs | |
| if value(x, "reasoning_output_tokens") is not None] for r, xs in ok.items()} | |
| max_r = max((max(v) for v in reasoning.values() if v), default=1) * 1.05 | |
| routes = list(reasoning) | |
| centres = {r: x1 + panel_w * (i + 0.5) / len(routes) for i, r in enumerate(routes)} | |
| parts += axes(x1, "route", "reasoning tokens", max_r, | |
| [(centres[r], f"{r} (n={len(reasoning[r])})") for r in routes]) | |
| rng = __import__("random").Random(7) | |
| for r, vals in reasoning.items(): | |
| for v in vals: | |
| cx = centres[r] + rng.uniform(-panel_w / 8, panel_w / 8) | |
| cy = bottom - (bottom - top) * v / max_r | |
| parts.append(f'<circle cx="{cx:.1f}" cy="{cy:.1f}" r="3.5" fill="{COLOURS[r]}" fill-opacity="0.5"/>') | |
| if vals: | |
| my = bottom - (bottom - top) * statistics.median(vals) / max_r | |
| parts.append(f'<line x1="{centres[r] - panel_w / 6}" y1="{my:.1f}" x2="{centres[r] + panel_w / 6}" ' | |
| f'y2="{my:.1f}" stroke="black" stroke-width="2.5"/>') | |
| parts.append(f'<text x="{centres[r] + panel_w / 6 + 4}" y="{my + 4:.1f}">median {statistics.median(vals):,.0f}</text>') | |
| for i, r in enumerate(ok): | |
| parts.append(f'<circle cx="{pad + 10}" cy="{top + 8 + 16 * i}" r="5" fill="{COLOURS[r]}"/>') | |
| parts.append(f'<text x="{pad + 20}" y="{top + 12 + 16 * i}">{r}</text>') | |
| parts.append("</svg>") | |
| return "\n".join(parts) | |
| def main(): | |
| p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) | |
| p.add_argument("results_dir") | |
| p.add_argument("--title") | |
| args = p.parse_args() | |
| base = Path(args.results_dir) | |
| rows = [json.loads(l) for l in (base / "results.jsonl").read_text().splitlines() if l.strip()] | |
| summary = base / "summary.json" | |
| meta = json.loads(summary.read_text()) if summary.exists() else {} | |
| title = args.title or f"Codex CLI API vs OAuth · {meta.get('model', '?')} {meta.get('effort', '')}" | |
| text, ok = report(rows, meta, title) | |
| (base / "report.md").write_text(text + "\n\n") | |
| (base / "chart.svg").write_text(svg_chart(ok, title)) | |
| print(text) | |
| if __name__ == "__main__": | |
| main() | |
Xet Storage Details
- Size:
- 9.76 kB
- Xet hash:
- d53b6d0881aab6311a13c208278cd344053f872edfaf35a779b140970e73a9cc
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.