evalstate/codex-159-test / scripts /codex_route_ab_report.py
evalstate's picture
download
raw
9.76 kB
#!/usr/bin/env python3
"""Report for codex_route_ab.py output: time / output-token statistics and a basic chart.
Usage:
python3 codex_route_ab_report.py RESULTS_DIR [--title TEXT]
Reads RESULTS_DIR/results.jsonl (+ summary.json metadata) and writes, into RESULTS_DIR:
report.md - per-route n, correctness, median/mean/p10/p90 of wall seconds, output,
reasoning and input tokens; OAuth/API median ratios with bootstrap 95% CIs;
Mann-Whitney U (normal approximation) p-values.
chart.svg - left: wall seconds vs output tokens per run; right: reasoning tokens
per route (strip plot with median bar). Stdlib only, no plotting library.
Only successful runs (exit 0 with usage) enter statistics; failures are counted.
"""
from __future__ import annotations
import argparse
import json
import math
import random
import statistics
from pathlib import Path
COLOURS = {"api": "#1f77b4", "oauth": "#d62728"}
METRICS = [("seconds", "wall seconds"), ("output_tokens", "output tokens"),
("reasoning_output_tokens", "reasoning tokens"), ("input_tokens", "input tokens")]
def value(row, key):
return row["seconds"] if key == "seconds" else row["usage"].get(key)
def quantile(values, q):
values = sorted(values)
if not values:
return None
pos = (len(values) - 1) * q
lo, hi = math.floor(pos), math.ceil(pos)
return values[lo] + (values[hi] - values[lo]) * (pos - lo)
def mann_whitney_p(a, b):
"""Two-sided p, normal approximation with tie correction (fine for n >= ~8)."""
pooled = sorted((v, i) for i, v in enumerate(a + b))
ranks, i = [0.0] * len(pooled), 0
ties = 0.0
while i < len(pooled):
j = i
while j + 1 < len(pooled) and pooled[j + 1][0] == pooled[i][0]:
j += 1
for k in range(i, j + 1):
ranks[pooled[k][1]] = (i + j) / 2 + 1
t = j - i + 1
ties += t ** 3 - t
i = j + 1
n1, n2 = len(a), len(b)
u = sum(ranks[:n1]) - n1 * (n1 + 1) / 2
n = n1 + n2
var = n1 * n2 / 12 * ((n + 1) - ties / (n * (n - 1)))
if var <= 0:
return 1.0
z = (abs(u - n1 * n2 / 2) - 0.5) / math.sqrt(var)
return math.erfc(max(z, 0) / math.sqrt(2))
def bootstrap_ratio(a, b, iterations=5000, seed=1):
"""95% CI of median(b)/median(a) by resampling each arm."""
rng = random.Random(seed)
ratios = sorted(statistics.median(rng.choices(b, k=len(b))) /
statistics.median(rng.choices(a, k=len(a))) for _ in range(iterations))
return ratios[int(0.025 * iterations)], ratios[int(0.975 * iterations) - 1]
def fmt(v, digits=0):
return "—" if v is None else f"{v:,.{digits}f}"
def report(rows, meta, title):
routes = [r for r in ("api", "oauth") if any(x["route"] == r for x in rows)]
ok = {r: [x for x in rows if x["route"] == r and x["exit"] == 0 and x["usage"]] for r in routes}
lines = [f"# {title}", "",
f"Codex CLI `{meta.get('codex_version')}` · model `{meta.get('model')}` · effort "
f"`{meta.get('effort')}` · prompt sha256 `{str(meta.get('prompt_sha256'))[:12]}` · "
f"parallel {meta.get('parallel', 1)} · started {meta.get('started_at')}", "",
"| route | runs | ok | correct | metric | median | mean | p10 | p90 |",
"|---|---|---|---|---|---|---|---|---|"]
for route in routes:
runs = sum(x["route"] == route for x in rows)
correct = sum(bool(x.get("correct")) for x in ok[route])
for i, (key, name) in enumerate(METRICS):
vals = [value(x, key) for x in ok[route] if value(x, key) is not None]
digits = 1 if key == "seconds" else 0
head = f"| {route} | {runs} | {len(ok[route])} | {correct} |" if i == 0 else "| | | | |"
lines.append(f"{head} {name} | {fmt(quantile(vals, .5), digits)} | "
f"{fmt(statistics.mean(vals) if vals else None, digits)} | "
f"{fmt(quantile(vals, .1), digits)} | {fmt(quantile(vals, .9), digits)} |")
if set(routes) == {"api", "oauth"}:
lines += ["", "| metric | OAuth/API median ratio | bootstrap 95% CI | Mann-Whitney p |",
"|---|---|---|---|"]
for key, name in METRICS:
a = [value(x, key) for x in ok["api"] if value(x, key) is not None]
b = [value(x, key) for x in ok["oauth"] if value(x, key) is not None]
if len(a) < 2 or len(b) < 2:
continue
lo, hi = bootstrap_ratio(a, b)
ratio = statistics.median(b) / statistics.median(a)
lines.append(f"| {name} | {ratio:.2f} | {lo:.2f}–{hi:.2f} | {mann_whitney_p(a, b):.2g} |")
tps = {r: [value(x, "output_tokens") / x["seconds"] for x in ok[r] if x["seconds"]]
for r in routes}
lines += ["", "Output tokens per wall second (median): " + " · ".join(
f"{r} {statistics.median(v):.1f}" for r, v in tps.items() if v)]
failed = [x for x in rows if not (x["exit"] == 0 and x["usage"])]
lines += ["", f"Failed/unusable runs: {len(failed)}", "",
"Wall seconds include Codex CLI start-up and queueing; with `parallel > 1` "
"both routes share the same concurrency.", ""]
return "\n".join(lines), ok
def svg_chart(ok, title):
width, height, pad = 1000, 420, 75
panel_w = (width - 3 * pad) / 2
parts = [f'<svg xmlns="http://www.w3.org/2000/svg" width="{width}" height="{height}" '
f'font-family="sans-serif" font-size="12">',
f'<rect width="{width}" height="{height}" fill="white"/>',
f'<text x="{width / 2}" y="22" text-anchor="middle" font-size="15">{title}</text>']
points = [(value(x, "seconds"), value(x, "output_tokens"), r)
for r, xs in ok.items() for x in xs if value(x, "output_tokens") is not None]
if not points:
return "\n".join(parts + ["</svg>"])
max_s = max(p[0] for p in points) * 1.05
max_o = max(p[1] for p in points) * 1.05
top, bottom = 50, height - pad
def axes(x0, xlabel, ylabel, ymax, xticks):
out = [f'<line x1="{x0}" y1="{bottom}" x2="{x0 + panel_w}" y2="{bottom}" stroke="black"/>',
f'<line x1="{x0}" y1="{top}" x2="{x0}" y2="{bottom}" stroke="black"/>',
f'<text x="{x0 + panel_w / 2}" y="{height - 18}" text-anchor="middle">{xlabel}</text>',
f'<text x="{x0 - 58}" y="{(top + bottom) / 2}" text-anchor="middle" '
f'transform="rotate(-90 {x0 - 58} {(top + bottom) / 2})">{ylabel}</text>']
for k in range(5):
v = ymax * k / 4
y = bottom - (bottom - top) * k / 4
out.append(f'<text x="{x0 - 6}" y="{y + 4}" text-anchor="end">{v:,.0f}</text>')
out.append(f'<line x1="{x0}" y1="{y}" x2="{x0 + panel_w}" y2="{y}" stroke="#eee"/>')
for x, text in xticks:
out.append(f'<text x="{x}" y="{bottom + 16}" text-anchor="middle">{text}</text>')
return out
x0 = pad
ticks = [(x0 + panel_w * k / 4, f"{max_s * k / 4:.0f}") for k in range(5)]
parts += axes(x0, "wall seconds", "output tokens", max_o, ticks)
for s, o, r in points:
cx = x0 + panel_w * s / max_s
cy = bottom - (bottom - top) * o / max_o
parts.append(f'<circle cx="{cx:.1f}" cy="{cy:.1f}" r="3.5" fill="{COLOURS[r]}" fill-opacity="0.6"/>')
x1 = 2 * pad + panel_w
reasoning = {r: [value(x, "reasoning_output_tokens") for x in xs
if value(x, "reasoning_output_tokens") is not None] for r, xs in ok.items()}
max_r = max((max(v) for v in reasoning.values() if v), default=1) * 1.05
routes = list(reasoning)
centres = {r: x1 + panel_w * (i + 0.5) / len(routes) for i, r in enumerate(routes)}
parts += axes(x1, "route", "reasoning tokens", max_r,
[(centres[r], f"{r} (n={len(reasoning[r])})") for r in routes])
rng = __import__("random").Random(7)
for r, vals in reasoning.items():
for v in vals:
cx = centres[r] + rng.uniform(-panel_w / 8, panel_w / 8)
cy = bottom - (bottom - top) * v / max_r
parts.append(f'<circle cx="{cx:.1f}" cy="{cy:.1f}" r="3.5" fill="{COLOURS[r]}" fill-opacity="0.5"/>')
if vals:
my = bottom - (bottom - top) * statistics.median(vals) / max_r
parts.append(f'<line x1="{centres[r] - panel_w / 6}" y1="{my:.1f}" x2="{centres[r] + panel_w / 6}" '
f'y2="{my:.1f}" stroke="black" stroke-width="2.5"/>')
parts.append(f'<text x="{centres[r] + panel_w / 6 + 4}" y="{my + 4:.1f}">median {statistics.median(vals):,.0f}</text>')
for i, r in enumerate(ok):
parts.append(f'<circle cx="{pad + 10}" cy="{top + 8 + 16 * i}" r="5" fill="{COLOURS[r]}"/>')
parts.append(f'<text x="{pad + 20}" y="{top + 12 + 16 * i}">{r}</text>')
parts.append("</svg>")
return "\n".join(parts)
def main():
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
p.add_argument("results_dir")
p.add_argument("--title")
args = p.parse_args()
base = Path(args.results_dir)
rows = [json.loads(l) for l in (base / "results.jsonl").read_text().splitlines() if l.strip()]
summary = base / "summary.json"
meta = json.loads(summary.read_text()) if summary.exists() else {}
title = args.title or f"Codex CLI API vs OAuth · {meta.get('model', '?')} {meta.get('effort', '')}"
text, ok = report(rows, meta, title)
(base / "report.md").write_text(text + "\n![chart](chart.svg)\n")
(base / "chart.svg").write_text(svg_chart(ok, title))
print(text)
if __name__ == "__main__":
main()

Xet Storage Details

Size:
9.76 kB
·
Xet hash:
d53b6d0881aab6311a13c208278cd344053f872edfaf35a779b140970e73a9cc

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.