Spaces:
Running
Running
Commit ·
1efaba3
1
Parent(s): a593bad
fix cost
Browse files- app.py +1 -0
- results/deepseek_deepseek-v4-flash__browser-use.json +2 -1
- results/deepseek_deepseek-v4-flash__openhands.json +2 -1
- results/deepseek_deepseek-v4-flash__openmanus.json +2 -1
- results/deepseek_deepseek-v4-flash__ouroboros-cut.json +2 -1
- results/deepseek_deepseek-v4-flash__ouroboros-full-evolving.json +2 -1
- results/deepseek_deepseek-v4-flash__ouroboros-full-isolated.json +2 -1
- results/google_gemini-2.5-flash__browser-use.json +2 -1
- results/google_gemini-2.5-flash__openhands.json +2 -1
- results/google_gemini-2.5-flash__openmanus.json +2 -1
- results/google_gemini-2.5-flash__ouroboros-cut.json +2 -1
- results/google_gemini-2.5-flash__ouroboros-full-evolving.json +2 -1
- results/google_gemini-2.5-flash__ouroboros-full-isolated.json +2 -1
- src/export_results.py +43 -0
- src/formatting.py +6 -0
- src/load_entries.py +6 -0
- src/models.py +5 -0
- src/ui.py +4 -4
app.py
CHANGED
|
@@ -347,6 +347,7 @@ def build_demo(entries=None):
|
|
| 347 |
"Tasks": f"{entry.passed_tasks}/{entry.total_tasks}",
|
| 348 |
"Avg duration": round2(entry.avg_duration_seconds),
|
| 349 |
"Total duration": round2(entry.total_duration_seconds),
|
|
|
|
| 350 |
"Avg steps": round2(entry.avg_agent_steps),
|
| 351 |
"Tokens/task": round2(entry.avg_tokens_per_task),
|
| 352 |
"Total tokens": entry.total_tokens,
|
|
|
|
| 347 |
"Tasks": f"{entry.passed_tasks}/{entry.total_tasks}",
|
| 348 |
"Avg duration": round2(entry.avg_duration_seconds),
|
| 349 |
"Total duration": round2(entry.total_duration_seconds),
|
| 350 |
+
"Total steps": entry.total_agent_steps,
|
| 351 |
"Avg steps": round2(entry.avg_agent_steps),
|
| 352 |
"Tokens/task": round2(entry.avg_tokens_per_task),
|
| 353 |
"Total tokens": entry.total_tokens,
|
results/deepseek_deepseek-v4-flash__browser-use.json
CHANGED
|
@@ -21,7 +21,8 @@
|
|
| 21 |
"agent_completion_rate": 0.8235294117647058,
|
| 22 |
"agent_dab_agreement_rate": 0.6274509803921569,
|
| 23 |
"section_avg_rate": 0.5166143380429095,
|
| 24 |
-
"total_duration_seconds": 21972.302697375417
|
|
|
|
| 25 |
},
|
| 26 |
"sections": {
|
| 27 |
"shop": {
|
|
|
|
| 21 |
"agent_completion_rate": 0.8235294117647058,
|
| 22 |
"agent_dab_agreement_rate": 0.6274509803921569,
|
| 23 |
"section_avg_rate": 0.5166143380429095,
|
| 24 |
+
"total_duration_seconds": 21972.302697375417,
|
| 25 |
+
"total_agent_steps": 616
|
| 26 |
},
|
| 27 |
"sections": {
|
| 28 |
"shop": {
|
results/deepseek_deepseek-v4-flash__openhands.json
CHANGED
|
@@ -21,7 +21,8 @@
|
|
| 21 |
"agent_completion_rate": 0.49019607843137253,
|
| 22 |
"agent_dab_agreement_rate": 0.7647058823529411,
|
| 23 |
"section_avg_rate": 0.3830455259026687,
|
| 24 |
-
"total_duration_seconds": 10914.631350658834
|
|
|
|
| 25 |
},
|
| 26 |
"sections": {
|
| 27 |
"shop": {
|
|
|
|
| 21 |
"agent_completion_rate": 0.49019607843137253,
|
| 22 |
"agent_dab_agreement_rate": 0.7647058823529411,
|
| 23 |
"section_avg_rate": 0.3830455259026687,
|
| 24 |
+
"total_duration_seconds": 10914.631350658834,
|
| 25 |
+
"total_agent_steps": 967
|
| 26 |
},
|
| 27 |
"sections": {
|
| 28 |
"shop": {
|
results/deepseek_deepseek-v4-flash__openmanus.json
CHANGED
|
@@ -21,7 +21,8 @@
|
|
| 21 |
"agent_completion_rate": 0.82,
|
| 22 |
"agent_dab_agreement_rate": 0.74,
|
| 23 |
"section_avg_rate": 0.5972004186289901,
|
| 24 |
-
"total_duration_seconds": 18126.616548523307
|
|
|
|
| 25 |
},
|
| 26 |
"sections": {
|
| 27 |
"shop": {
|
|
|
|
| 21 |
"agent_completion_rate": 0.82,
|
| 22 |
"agent_dab_agreement_rate": 0.74,
|
| 23 |
"section_avg_rate": 0.5972004186289901,
|
| 24 |
+
"total_duration_seconds": 18126.616548523307,
|
| 25 |
+
"total_agent_steps": 602
|
| 26 |
},
|
| 27 |
"sections": {
|
| 28 |
"shop": {
|
results/deepseek_deepseek-v4-flash__ouroboros-cut.json
CHANGED
|
@@ -21,7 +21,8 @@
|
|
| 21 |
"agent_completion_rate": 0.8235294117647058,
|
| 22 |
"agent_dab_agreement_rate": 0.6666666666666666,
|
| 23 |
"section_avg_rate": 0.5385923600209315,
|
| 24 |
-
"total_duration_seconds": 24993.06771468371
|
|
|
|
| 25 |
},
|
| 26 |
"sections": {
|
| 27 |
"shop": {
|
|
|
|
| 21 |
"agent_completion_rate": 0.8235294117647058,
|
| 22 |
"agent_dab_agreement_rate": 0.6666666666666666,
|
| 23 |
"section_avg_rate": 0.5385923600209315,
|
| 24 |
+
"total_duration_seconds": 24993.06771468371,
|
| 25 |
+
"total_agent_steps": 592
|
| 26 |
},
|
| 27 |
"sections": {
|
| 28 |
"shop": {
|
results/deepseek_deepseek-v4-flash__ouroboros-full-evolving.json
CHANGED
|
@@ -21,7 +21,8 @@
|
|
| 21 |
"agent_completion_rate": 0.7058823529411765,
|
| 22 |
"agent_dab_agreement_rate": 0.7843137254901961,
|
| 23 |
"section_avg_rate": 0.5300889586603873,
|
| 24 |
-
"total_duration_seconds": 50013.8336566519
|
|
|
|
| 25 |
},
|
| 26 |
"sections": {
|
| 27 |
"shop": {
|
|
|
|
| 21 |
"agent_completion_rate": 0.7058823529411765,
|
| 22 |
"agent_dab_agreement_rate": 0.7843137254901961,
|
| 23 |
"section_avg_rate": 0.5300889586603873,
|
| 24 |
+
"total_duration_seconds": 50013.8336566519,
|
| 25 |
+
"total_agent_steps": 1159
|
| 26 |
},
|
| 27 |
"sections": {
|
| 28 |
"shop": {
|
results/deepseek_deepseek-v4-flash__ouroboros-full-isolated.json
CHANGED
|
@@ -21,7 +21,8 @@
|
|
| 21 |
"agent_completion_rate": 0.6862745098039216,
|
| 22 |
"agent_dab_agreement_rate": 0.7450980392156863,
|
| 23 |
"section_avg_rate": 0.64141810570382,
|
| 24 |
-
"total_duration_seconds": 27027.0369386971
|
|
|
|
| 25 |
},
|
| 26 |
"sections": {
|
| 27 |
"shop": {
|
|
|
|
| 21 |
"agent_completion_rate": 0.6862745098039216,
|
| 22 |
"agent_dab_agreement_rate": 0.7450980392156863,
|
| 23 |
"section_avg_rate": 0.64141810570382,
|
| 24 |
+
"total_duration_seconds": 27027.0369386971,
|
| 25 |
+
"total_agent_steps": 866
|
| 26 |
},
|
| 27 |
"sections": {
|
| 28 |
"shop": {
|
results/google_gemini-2.5-flash__browser-use.json
CHANGED
|
@@ -21,7 +21,8 @@
|
|
| 21 |
"agent_completion_rate": 0.8235294117647058,
|
| 22 |
"agent_dab_agreement_rate": 0.5882352941176471,
|
| 23 |
"section_avg_rate": 0.45866038723181585,
|
| 24 |
-
"total_duration_seconds": 11037.299745082855
|
|
|
|
| 25 |
},
|
| 26 |
"sections": {
|
| 27 |
"shop": {
|
|
|
|
| 21 |
"agent_completion_rate": 0.8235294117647058,
|
| 22 |
"agent_dab_agreement_rate": 0.5882352941176471,
|
| 23 |
"section_avg_rate": 0.45866038723181585,
|
| 24 |
+
"total_duration_seconds": 11037.299745082855,
|
| 25 |
+
"total_agent_steps": 723
|
| 26 |
},
|
| 27 |
"sections": {
|
| 28 |
"shop": {
|
results/google_gemini-2.5-flash__openhands.json
CHANGED
|
@@ -21,7 +21,8 @@
|
|
| 21 |
"agent_completion_rate": 0.9607843137254902,
|
| 22 |
"agent_dab_agreement_rate": 0.13725490196078433,
|
| 23 |
"section_avg_rate": 0.2099686028257457,
|
| 24 |
-
"total_duration_seconds": 11133.858564271126
|
|
|
|
| 25 |
},
|
| 26 |
"sections": {
|
| 27 |
"shop": {
|
|
|
|
| 21 |
"agent_completion_rate": 0.9607843137254902,
|
| 22 |
"agent_dab_agreement_rate": 0.13725490196078433,
|
| 23 |
"section_avg_rate": 0.2099686028257457,
|
| 24 |
+
"total_duration_seconds": 11133.858564271126,
|
| 25 |
+
"total_agent_steps": 523
|
| 26 |
},
|
| 27 |
"sections": {
|
| 28 |
"shop": {
|
results/google_gemini-2.5-flash__openmanus.json
CHANGED
|
@@ -21,7 +21,8 @@
|
|
| 21 |
"agent_completion_rate": 0.8235294117647058,
|
| 22 |
"agent_dab_agreement_rate": 0.6470588235294118,
|
| 23 |
"section_avg_rate": 0.5173338566195709,
|
| 24 |
-
"total_duration_seconds": 10862.327687103301
|
|
|
|
| 25 |
},
|
| 26 |
"sections": {
|
| 27 |
"shop": {
|
|
|
|
| 21 |
"agent_completion_rate": 0.8235294117647058,
|
| 22 |
"agent_dab_agreement_rate": 0.6470588235294118,
|
| 23 |
"section_avg_rate": 0.5173338566195709,
|
| 24 |
+
"total_duration_seconds": 10862.327687103301,
|
| 25 |
+
"total_agent_steps": 659
|
| 26 |
},
|
| 27 |
"sections": {
|
| 28 |
"shop": {
|
results/google_gemini-2.5-flash__ouroboros-cut.json
CHANGED
|
@@ -21,7 +21,8 @@
|
|
| 21 |
"agent_completion_rate": 0.8235294117647058,
|
| 22 |
"agent_dab_agreement_rate": 0.6078431372549019,
|
| 23 |
"section_avg_rate": 0.4696493982208268,
|
| 24 |
-
"total_duration_seconds": 10506.739309936762
|
|
|
|
| 25 |
},
|
| 26 |
"sections": {
|
| 27 |
"shop": {
|
|
|
|
| 21 |
"agent_completion_rate": 0.8235294117647058,
|
| 22 |
"agent_dab_agreement_rate": 0.6078431372549019,
|
| 23 |
"section_avg_rate": 0.4696493982208268,
|
| 24 |
+
"total_duration_seconds": 10506.739309936762,
|
| 25 |
+
"total_agent_steps": 726
|
| 26 |
},
|
| 27 |
"sections": {
|
| 28 |
"shop": {
|
results/google_gemini-2.5-flash__ouroboros-full-evolving.json
CHANGED
|
@@ -21,7 +21,8 @@
|
|
| 21 |
"agent_completion_rate": 0.6862745098039216,
|
| 22 |
"agent_dab_agreement_rate": 0.45098039215686275,
|
| 23 |
"section_avg_rate": 0.18524332810047098,
|
| 24 |
-
"total_duration_seconds": 8925.054827867076
|
|
|
|
| 25 |
},
|
| 26 |
"sections": {
|
| 27 |
"shop": {
|
|
|
|
| 21 |
"agent_completion_rate": 0.6862745098039216,
|
| 22 |
"agent_dab_agreement_rate": 0.45098039215686275,
|
| 23 |
"section_avg_rate": 0.18524332810047098,
|
| 24 |
+
"total_duration_seconds": 8925.054827867076,
|
| 25 |
+
"total_agent_steps": 579
|
| 26 |
},
|
| 27 |
"sections": {
|
| 28 |
"shop": {
|
results/google_gemini-2.5-flash__ouroboros-full-isolated.json
CHANGED
|
@@ -21,7 +21,8 @@
|
|
| 21 |
"agent_completion_rate": 0.5882352941176471,
|
| 22 |
"agent_dab_agreement_rate": 0.47058823529411764,
|
| 23 |
"section_avg_rate": 0.17857142857142858,
|
| 24 |
-
"total_duration_seconds": 30995.941995203495
|
|
|
|
| 25 |
},
|
| 26 |
"sections": {
|
| 27 |
"shop": {
|
|
|
|
| 21 |
"agent_completion_rate": 0.5882352941176471,
|
| 22 |
"agent_dab_agreement_rate": 0.47058823529411764,
|
| 23 |
"section_avg_rate": 0.17857142857142858,
|
| 24 |
+
"total_duration_seconds": 30995.941995203495,
|
| 25 |
+
"total_agent_steps": 679
|
| 26 |
},
|
| 27 |
"sections": {
|
| 28 |
"shop": {
|
src/export_results.py
CHANGED
|
@@ -189,6 +189,18 @@ def backfill_summary_costs(results_dir: Path | None = None) -> int:
|
|
| 189 |
return updated
|
| 190 |
|
| 191 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 192 |
def export_entry(payload: dict[str, Any], *, source_path: str | None = None) -> LeaderboardEntry:
|
| 193 |
config = load_bench_config()
|
| 194 |
run = payload.get("run") or {}
|
|
@@ -241,6 +253,7 @@ def export_entry(payload: dict[str, Any], *, source_path: str | None = None) ->
|
|
| 241 |
total_tasks=int(run.get("total_tasks") or 0),
|
| 242 |
avg_duration_seconds=run.get("avg_duration_seconds"),
|
| 243 |
total_duration_seconds=run.get("total_duration_seconds"),
|
|
|
|
| 244 |
avg_agent_steps=run.get("avg_agent_steps"),
|
| 245 |
avg_tokens_per_task=run.get("avg_tokens_per_task"),
|
| 246 |
total_tokens=run.get("total_tokens"),
|
|
@@ -323,6 +336,36 @@ def export_latest_runs(
|
|
| 323 |
return written
|
| 324 |
|
| 325 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 326 |
def backfill_summary_total_duration(results_dir: Path | None = None) -> int:
|
| 327 |
"""Add total_duration_seconds to summary JSON from copied raw results when missing."""
|
| 328 |
root = results_dir or (_space_root() / "results")
|
|
|
|
| 189 |
return updated
|
| 190 |
|
| 191 |
|
| 192 |
+
def _total_agent_steps_from_payload(payload: dict[str, Any]) -> int | None:
|
| 193 |
+
run = payload.get("run") or {}
|
| 194 |
+
total = run.get("total_agent_steps")
|
| 195 |
+
if isinstance(total, int):
|
| 196 |
+
return total
|
| 197 |
+
if isinstance(total, float):
|
| 198 |
+
return int(total)
|
| 199 |
+
tests = payload.get("tests") or []
|
| 200 |
+
steps = [row["agent_steps"] for row in tests if isinstance(row.get("agent_steps"), int)]
|
| 201 |
+
return sum(steps) if steps else None
|
| 202 |
+
|
| 203 |
+
|
| 204 |
def export_entry(payload: dict[str, Any], *, source_path: str | None = None) -> LeaderboardEntry:
|
| 205 |
config = load_bench_config()
|
| 206 |
run = payload.get("run") or {}
|
|
|
|
| 253 |
total_tasks=int(run.get("total_tasks") or 0),
|
| 254 |
avg_duration_seconds=run.get("avg_duration_seconds"),
|
| 255 |
total_duration_seconds=run.get("total_duration_seconds"),
|
| 256 |
+
total_agent_steps=_total_agent_steps_from_payload(payload),
|
| 257 |
avg_agent_steps=run.get("avg_agent_steps"),
|
| 258 |
avg_tokens_per_task=run.get("avg_tokens_per_task"),
|
| 259 |
total_tokens=run.get("total_tokens"),
|
|
|
|
| 336 |
return written
|
| 337 |
|
| 338 |
|
| 339 |
+
def backfill_summary_total_agent_steps(results_dir: Path | None = None) -> int:
|
| 340 |
+
"""Add total_agent_steps to summary JSON from copied raw results when missing."""
|
| 341 |
+
root = results_dir or (_space_root() / "results")
|
| 342 |
+
if not root.is_dir():
|
| 343 |
+
return 0
|
| 344 |
+
updated = 0
|
| 345 |
+
for path in sorted(root.glob("*.json")):
|
| 346 |
+
payload = json.loads(path.read_text(encoding="utf-8"))
|
| 347 |
+
if "metrics" not in payload:
|
| 348 |
+
continue
|
| 349 |
+
metrics = dict(payload.get("metrics") or {})
|
| 350 |
+
if metrics.get("total_agent_steps") is not None:
|
| 351 |
+
continue
|
| 352 |
+
entry = LeaderboardEntry.from_json(payload)
|
| 353 |
+
raw_path = raw_results_path_for(entry, root.parent)
|
| 354 |
+
if not raw_path.is_file() and entry.raw_results_path:
|
| 355 |
+
raw_path = root.parent / entry.raw_results_path
|
| 356 |
+
if not raw_path.is_file():
|
| 357 |
+
continue
|
| 358 |
+
raw = json.loads(raw_path.read_text(encoding="utf-8"))
|
| 359 |
+
total_steps = _total_agent_steps_from_payload(raw)
|
| 360 |
+
if total_steps is None:
|
| 361 |
+
continue
|
| 362 |
+
metrics["total_agent_steps"] = total_steps
|
| 363 |
+
payload["metrics"] = metrics
|
| 364 |
+
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
| 365 |
+
updated += 1
|
| 366 |
+
return updated
|
| 367 |
+
|
| 368 |
+
|
| 369 |
def backfill_summary_total_duration(results_dir: Path | None = None) -> int:
|
| 370 |
"""Add total_duration_seconds to summary JSON from copied raw results when missing."""
|
| 371 |
root = results_dir or (_space_root() / "results")
|
src/formatting.py
CHANGED
|
@@ -41,6 +41,12 @@ def fmt_pct_value(value: float | None) -> str:
|
|
| 41 |
return fmt_number(value, suffix="%")
|
| 42 |
|
| 43 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 44 |
def fmt_duration(seconds: float | None) -> str:
|
| 45 |
if seconds is None:
|
| 46 |
return "—"
|
|
|
|
| 41 |
return fmt_number(value, suffix="%")
|
| 42 |
|
| 43 |
|
| 44 |
+
def fmt_steps(value: float | int | None) -> str:
|
| 45 |
+
if value is None or isinstance(value, bool):
|
| 46 |
+
return "—"
|
| 47 |
+
return str(int(round(float(value))))
|
| 48 |
+
|
| 49 |
+
|
| 50 |
def fmt_duration(seconds: float | None) -> str:
|
| 51 |
if seconds is None:
|
| 52 |
return "—"
|
src/load_entries.py
CHANGED
|
@@ -10,6 +10,7 @@ from src.bench_config import load_bench_config
|
|
| 10 |
from src.export_results import (
|
| 11 |
_cost_from_payload,
|
| 12 |
_recorded_cost_from_payload,
|
|
|
|
| 13 |
raw_results_path_for,
|
| 14 |
ui_classes_from_payload,
|
| 15 |
)
|
|
@@ -77,6 +78,10 @@ def _enrich_entry_from_raw(entry: LeaderboardEntry, raw: dict[str, Any]) -> None
|
|
| 77 |
entry.avg_tokens_per_task = float(run["avg_tokens_per_task"])
|
| 78 |
if entry.total_duration_seconds is None and isinstance(run.get("total_duration_seconds"), (int, float)):
|
| 79 |
entry.total_duration_seconds = float(run["total_duration_seconds"])
|
|
|
|
|
|
|
|
|
|
|
|
|
| 80 |
slots = run.get("ouroboros_model_slots")
|
| 81 |
if isinstance(slots, dict) and slots and not entry.ouroboros_model_slots:
|
| 82 |
entry.ouroboros_model_slots = {
|
|
@@ -104,6 +109,7 @@ def load_entry_file(path: Path) -> LeaderboardEntry | None:
|
|
| 104 |
or entry.total_tokens is None
|
| 105 |
or entry.avg_tokens_per_task is None
|
| 106 |
or entry.total_duration_seconds is None
|
|
|
|
| 107 |
or (entry.harness.startswith("ouroboros-full") and not entry.ouroboros_model_slots)
|
| 108 |
)
|
| 109 |
if needs_raw:
|
|
|
|
| 10 |
from src.export_results import (
|
| 11 |
_cost_from_payload,
|
| 12 |
_recorded_cost_from_payload,
|
| 13 |
+
_total_agent_steps_from_payload,
|
| 14 |
raw_results_path_for,
|
| 15 |
ui_classes_from_payload,
|
| 16 |
)
|
|
|
|
| 78 |
entry.avg_tokens_per_task = float(run["avg_tokens_per_task"])
|
| 79 |
if entry.total_duration_seconds is None and isinstance(run.get("total_duration_seconds"), (int, float)):
|
| 80 |
entry.total_duration_seconds = float(run["total_duration_seconds"])
|
| 81 |
+
if entry.total_agent_steps is None:
|
| 82 |
+
total_steps = _total_agent_steps_from_payload(raw)
|
| 83 |
+
if total_steps is not None:
|
| 84 |
+
entry.total_agent_steps = total_steps
|
| 85 |
slots = run.get("ouroboros_model_slots")
|
| 86 |
if isinstance(slots, dict) and slots and not entry.ouroboros_model_slots:
|
| 87 |
entry.ouroboros_model_slots = {
|
|
|
|
| 109 |
or entry.total_tokens is None
|
| 110 |
or entry.avg_tokens_per_task is None
|
| 111 |
or entry.total_duration_seconds is None
|
| 112 |
+
or entry.total_agent_steps is None
|
| 113 |
or (entry.harness.startswith("ouroboros-full") and not entry.ouroboros_model_slots)
|
| 114 |
)
|
| 115 |
if needs_raw:
|
src/models.py
CHANGED
|
@@ -52,6 +52,7 @@ class LeaderboardEntry:
|
|
| 52 |
|
| 53 |
avg_duration_seconds: float | None = None
|
| 54 |
total_duration_seconds: float | None = None
|
|
|
|
| 55 |
avg_agent_steps: float | None = None
|
| 56 |
avg_tokens_per_task: float | None = None
|
| 57 |
total_tokens: int | None = None
|
|
@@ -112,6 +113,8 @@ class LeaderboardEntry:
|
|
| 112 |
return self.avg_duration_seconds
|
| 113 |
if name == "total_duration_seconds":
|
| 114 |
return self.total_duration_seconds
|
|
|
|
|
|
|
| 115 |
if name == "total_cost_usd":
|
| 116 |
return self.total_cost_usd
|
| 117 |
if name == "value_score":
|
|
@@ -134,6 +137,7 @@ class LeaderboardEntry:
|
|
| 134 |
"total_tasks": self.total_tasks,
|
| 135 |
"avg_duration_seconds": self.avg_duration_seconds,
|
| 136 |
"total_duration_seconds": self.total_duration_seconds,
|
|
|
|
| 137 |
"avg_agent_steps": self.avg_agent_steps,
|
| 138 |
"avg_tokens_per_task": self.avg_tokens_per_task,
|
| 139 |
"total_tokens": self.total_tokens,
|
|
@@ -193,6 +197,7 @@ class LeaderboardEntry:
|
|
| 193 |
total_tasks=int(metrics.get("total_tasks") or 0),
|
| 194 |
avg_duration_seconds=metrics.get("avg_duration_seconds"),
|
| 195 |
total_duration_seconds=metrics.get("total_duration_seconds"),
|
|
|
|
| 196 |
avg_agent_steps=metrics.get("avg_agent_steps"),
|
| 197 |
avg_tokens_per_task=metrics.get("avg_tokens_per_task"),
|
| 198 |
total_tokens=metrics.get("total_tokens"),
|
|
|
|
| 52 |
|
| 53 |
avg_duration_seconds: float | None = None
|
| 54 |
total_duration_seconds: float | None = None
|
| 55 |
+
total_agent_steps: int | None = None
|
| 56 |
avg_agent_steps: float | None = None
|
| 57 |
avg_tokens_per_task: float | None = None
|
| 58 |
total_tokens: int | None = None
|
|
|
|
| 113 |
return self.avg_duration_seconds
|
| 114 |
if name == "total_duration_seconds":
|
| 115 |
return self.total_duration_seconds
|
| 116 |
+
if name == "total_agent_steps":
|
| 117 |
+
return float(self.total_agent_steps) if self.total_agent_steps is not None else None
|
| 118 |
if name == "total_cost_usd":
|
| 119 |
return self.total_cost_usd
|
| 120 |
if name == "value_score":
|
|
|
|
| 137 |
"total_tasks": self.total_tasks,
|
| 138 |
"avg_duration_seconds": self.avg_duration_seconds,
|
| 139 |
"total_duration_seconds": self.total_duration_seconds,
|
| 140 |
+
"total_agent_steps": self.total_agent_steps,
|
| 141 |
"avg_agent_steps": self.avg_agent_steps,
|
| 142 |
"avg_tokens_per_task": self.avg_tokens_per_task,
|
| 143 |
"total_tokens": self.total_tokens,
|
|
|
|
| 197 |
total_tasks=int(metrics.get("total_tasks") or 0),
|
| 198 |
avg_duration_seconds=metrics.get("avg_duration_seconds"),
|
| 199 |
total_duration_seconds=metrics.get("total_duration_seconds"),
|
| 200 |
+
total_agent_steps=metrics.get("total_agent_steps"),
|
| 201 |
avg_agent_steps=metrics.get("avg_agent_steps"),
|
| 202 |
avg_tokens_per_task=metrics.get("avg_tokens_per_task"),
|
| 203 |
total_tokens=metrics.get("total_tokens"),
|
src/ui.py
CHANGED
|
@@ -7,7 +7,7 @@ from pathlib import Path
|
|
| 7 |
from typing import Any
|
| 8 |
|
| 9 |
from src.bench_config import load_bench_config, task_count
|
| 10 |
-
from src.formatting import fmt_cost, fmt_duration, fmt_number, fmt_pct_value, round2
|
| 11 |
from src.load_entries import sort_entries
|
| 12 |
from src.models import LeaderboardEntry
|
| 13 |
from src.token_display import fmt_tokens, format_model_usage_breakdown_html
|
|
@@ -282,7 +282,7 @@ def _speed_row(idx: int, entry: LeaderboardEntry) -> str:
|
|
| 282 |
<div class="wab-badges"><span class="wab-badge">{html.escape(entry.harness)}</span></div></td>
|
| 283 |
<td>{html.escape(fmt_duration(entry.total_duration_seconds))}</td>
|
| 284 |
<td>{html.escape(fmt_duration(entry.avg_duration_seconds))}</td>
|
| 285 |
-
<td>{html.escape(
|
| 286 |
<td class="wab-score-cell">
|
| 287 |
<div class="wab-score-value wab-score-{tier}">{html.escape(fmt_pct_display(entry))}</div>
|
| 288 |
<div class="wab-bar"><span class="wab-bar-{tier}" style="width:{max(pct, 0):.1f}%"></span></div>
|
|
@@ -320,11 +320,11 @@ def build_table(entries: list[LeaderboardEntry], view_id: str) -> str:
|
|
| 320 |
sorted_entries = sort_entries(entries, view_id)
|
| 321 |
|
| 322 |
if view_id == "speed":
|
| 323 |
-
head = "<tr><th>#</th><th>Model</th><th>Total time</th><th>Avg time</th><th>
|
| 324 |
rows = [_speed_row(i, entry) for i, entry in enumerate(sorted_entries)]
|
| 325 |
subtitle = (
|
| 326 |
"Sorted by total run duration (run.total_duration_seconds). "
|
| 327 |
-
"Avg time is mean per-task duration
|
| 328 |
)
|
| 329 |
elif view_id == "cost":
|
| 330 |
head = (
|
|
|
|
| 7 |
from typing import Any
|
| 8 |
|
| 9 |
from src.bench_config import load_bench_config, task_count
|
| 10 |
+
from src.formatting import fmt_cost, fmt_duration, fmt_number, fmt_pct_value, fmt_steps, round2
|
| 11 |
from src.load_entries import sort_entries
|
| 12 |
from src.models import LeaderboardEntry
|
| 13 |
from src.token_display import fmt_tokens, format_model_usage_breakdown_html
|
|
|
|
| 282 |
<div class="wab-badges"><span class="wab-badge">{html.escape(entry.harness)}</span></div></td>
|
| 283 |
<td>{html.escape(fmt_duration(entry.total_duration_seconds))}</td>
|
| 284 |
<td>{html.escape(fmt_duration(entry.avg_duration_seconds))}</td>
|
| 285 |
+
<td>{html.escape(fmt_steps(entry.total_agent_steps))}</td>
|
| 286 |
<td class="wab-score-cell">
|
| 287 |
<div class="wab-score-value wab-score-{tier}">{html.escape(fmt_pct_display(entry))}</div>
|
| 288 |
<div class="wab-bar"><span class="wab-bar-{tier}" style="width:{max(pct, 0):.1f}%"></span></div>
|
|
|
|
| 320 |
sorted_entries = sort_entries(entries, view_id)
|
| 321 |
|
| 322 |
if view_id == "speed":
|
| 323 |
+
head = "<tr><th>#</th><th>Model</th><th>Total time</th><th>Avg time</th><th>Total steps</th><th>Success %</th></tr>"
|
| 324 |
rows = [_speed_row(i, entry) for i, entry in enumerate(sorted_entries)]
|
| 325 |
subtitle = (
|
| 326 |
"Sorted by total run duration (run.total_duration_seconds). "
|
| 327 |
+
"Avg time is mean per-task duration. Total steps is the sum of agent_steps across tasks."
|
| 328 |
)
|
| 329 |
elif view_id == "cost":
|
| 330 |
head = (
|