ai-forever commited on
Commit
1efaba3
·
1 Parent(s): a593bad
app.py CHANGED
@@ -347,6 +347,7 @@ def build_demo(entries=None):
347
  "Tasks": f"{entry.passed_tasks}/{entry.total_tasks}",
348
  "Avg duration": round2(entry.avg_duration_seconds),
349
  "Total duration": round2(entry.total_duration_seconds),
 
350
  "Avg steps": round2(entry.avg_agent_steps),
351
  "Tokens/task": round2(entry.avg_tokens_per_task),
352
  "Total tokens": entry.total_tokens,
 
347
  "Tasks": f"{entry.passed_tasks}/{entry.total_tasks}",
348
  "Avg duration": round2(entry.avg_duration_seconds),
349
  "Total duration": round2(entry.total_duration_seconds),
350
+ "Total steps": entry.total_agent_steps,
351
  "Avg steps": round2(entry.avg_agent_steps),
352
  "Tokens/task": round2(entry.avg_tokens_per_task),
353
  "Total tokens": entry.total_tokens,
results/deepseek_deepseek-v4-flash__browser-use.json CHANGED
@@ -21,7 +21,8 @@
21
  "agent_completion_rate": 0.8235294117647058,
22
  "agent_dab_agreement_rate": 0.6274509803921569,
23
  "section_avg_rate": 0.5166143380429095,
24
- "total_duration_seconds": 21972.302697375417
 
25
  },
26
  "sections": {
27
  "shop": {
 
21
  "agent_completion_rate": 0.8235294117647058,
22
  "agent_dab_agreement_rate": 0.6274509803921569,
23
  "section_avg_rate": 0.5166143380429095,
24
+ "total_duration_seconds": 21972.302697375417,
25
+ "total_agent_steps": 616
26
  },
27
  "sections": {
28
  "shop": {
results/deepseek_deepseek-v4-flash__openhands.json CHANGED
@@ -21,7 +21,8 @@
21
  "agent_completion_rate": 0.49019607843137253,
22
  "agent_dab_agreement_rate": 0.7647058823529411,
23
  "section_avg_rate": 0.3830455259026687,
24
- "total_duration_seconds": 10914.631350658834
 
25
  },
26
  "sections": {
27
  "shop": {
 
21
  "agent_completion_rate": 0.49019607843137253,
22
  "agent_dab_agreement_rate": 0.7647058823529411,
23
  "section_avg_rate": 0.3830455259026687,
24
+ "total_duration_seconds": 10914.631350658834,
25
+ "total_agent_steps": 967
26
  },
27
  "sections": {
28
  "shop": {
results/deepseek_deepseek-v4-flash__openmanus.json CHANGED
@@ -21,7 +21,8 @@
21
  "agent_completion_rate": 0.82,
22
  "agent_dab_agreement_rate": 0.74,
23
  "section_avg_rate": 0.5972004186289901,
24
- "total_duration_seconds": 18126.616548523307
 
25
  },
26
  "sections": {
27
  "shop": {
 
21
  "agent_completion_rate": 0.82,
22
  "agent_dab_agreement_rate": 0.74,
23
  "section_avg_rate": 0.5972004186289901,
24
+ "total_duration_seconds": 18126.616548523307,
25
+ "total_agent_steps": 602
26
  },
27
  "sections": {
28
  "shop": {
results/deepseek_deepseek-v4-flash__ouroboros-cut.json CHANGED
@@ -21,7 +21,8 @@
21
  "agent_completion_rate": 0.8235294117647058,
22
  "agent_dab_agreement_rate": 0.6666666666666666,
23
  "section_avg_rate": 0.5385923600209315,
24
- "total_duration_seconds": 24993.06771468371
 
25
  },
26
  "sections": {
27
  "shop": {
 
21
  "agent_completion_rate": 0.8235294117647058,
22
  "agent_dab_agreement_rate": 0.6666666666666666,
23
  "section_avg_rate": 0.5385923600209315,
24
+ "total_duration_seconds": 24993.06771468371,
25
+ "total_agent_steps": 592
26
  },
27
  "sections": {
28
  "shop": {
results/deepseek_deepseek-v4-flash__ouroboros-full-evolving.json CHANGED
@@ -21,7 +21,8 @@
21
  "agent_completion_rate": 0.7058823529411765,
22
  "agent_dab_agreement_rate": 0.7843137254901961,
23
  "section_avg_rate": 0.5300889586603873,
24
- "total_duration_seconds": 50013.8336566519
 
25
  },
26
  "sections": {
27
  "shop": {
 
21
  "agent_completion_rate": 0.7058823529411765,
22
  "agent_dab_agreement_rate": 0.7843137254901961,
23
  "section_avg_rate": 0.5300889586603873,
24
+ "total_duration_seconds": 50013.8336566519,
25
+ "total_agent_steps": 1159
26
  },
27
  "sections": {
28
  "shop": {
results/deepseek_deepseek-v4-flash__ouroboros-full-isolated.json CHANGED
@@ -21,7 +21,8 @@
21
  "agent_completion_rate": 0.6862745098039216,
22
  "agent_dab_agreement_rate": 0.7450980392156863,
23
  "section_avg_rate": 0.64141810570382,
24
- "total_duration_seconds": 27027.0369386971
 
25
  },
26
  "sections": {
27
  "shop": {
 
21
  "agent_completion_rate": 0.6862745098039216,
22
  "agent_dab_agreement_rate": 0.7450980392156863,
23
  "section_avg_rate": 0.64141810570382,
24
+ "total_duration_seconds": 27027.0369386971,
25
+ "total_agent_steps": 866
26
  },
27
  "sections": {
28
  "shop": {
results/google_gemini-2.5-flash__browser-use.json CHANGED
@@ -21,7 +21,8 @@
21
  "agent_completion_rate": 0.8235294117647058,
22
  "agent_dab_agreement_rate": 0.5882352941176471,
23
  "section_avg_rate": 0.45866038723181585,
24
- "total_duration_seconds": 11037.299745082855
 
25
  },
26
  "sections": {
27
  "shop": {
 
21
  "agent_completion_rate": 0.8235294117647058,
22
  "agent_dab_agreement_rate": 0.5882352941176471,
23
  "section_avg_rate": 0.45866038723181585,
24
+ "total_duration_seconds": 11037.299745082855,
25
+ "total_agent_steps": 723
26
  },
27
  "sections": {
28
  "shop": {
results/google_gemini-2.5-flash__openhands.json CHANGED
@@ -21,7 +21,8 @@
21
  "agent_completion_rate": 0.9607843137254902,
22
  "agent_dab_agreement_rate": 0.13725490196078433,
23
  "section_avg_rate": 0.2099686028257457,
24
- "total_duration_seconds": 11133.858564271126
 
25
  },
26
  "sections": {
27
  "shop": {
 
21
  "agent_completion_rate": 0.9607843137254902,
22
  "agent_dab_agreement_rate": 0.13725490196078433,
23
  "section_avg_rate": 0.2099686028257457,
24
+ "total_duration_seconds": 11133.858564271126,
25
+ "total_agent_steps": 523
26
  },
27
  "sections": {
28
  "shop": {
results/google_gemini-2.5-flash__openmanus.json CHANGED
@@ -21,7 +21,8 @@
21
  "agent_completion_rate": 0.8235294117647058,
22
  "agent_dab_agreement_rate": 0.6470588235294118,
23
  "section_avg_rate": 0.5173338566195709,
24
- "total_duration_seconds": 10862.327687103301
 
25
  },
26
  "sections": {
27
  "shop": {
 
21
  "agent_completion_rate": 0.8235294117647058,
22
  "agent_dab_agreement_rate": 0.6470588235294118,
23
  "section_avg_rate": 0.5173338566195709,
24
+ "total_duration_seconds": 10862.327687103301,
25
+ "total_agent_steps": 659
26
  },
27
  "sections": {
28
  "shop": {
results/google_gemini-2.5-flash__ouroboros-cut.json CHANGED
@@ -21,7 +21,8 @@
21
  "agent_completion_rate": 0.8235294117647058,
22
  "agent_dab_agreement_rate": 0.6078431372549019,
23
  "section_avg_rate": 0.4696493982208268,
24
- "total_duration_seconds": 10506.739309936762
 
25
  },
26
  "sections": {
27
  "shop": {
 
21
  "agent_completion_rate": 0.8235294117647058,
22
  "agent_dab_agreement_rate": 0.6078431372549019,
23
  "section_avg_rate": 0.4696493982208268,
24
+ "total_duration_seconds": 10506.739309936762,
25
+ "total_agent_steps": 726
26
  },
27
  "sections": {
28
  "shop": {
results/google_gemini-2.5-flash__ouroboros-full-evolving.json CHANGED
@@ -21,7 +21,8 @@
21
  "agent_completion_rate": 0.6862745098039216,
22
  "agent_dab_agreement_rate": 0.45098039215686275,
23
  "section_avg_rate": 0.18524332810047098,
24
- "total_duration_seconds": 8925.054827867076
 
25
  },
26
  "sections": {
27
  "shop": {
 
21
  "agent_completion_rate": 0.6862745098039216,
22
  "agent_dab_agreement_rate": 0.45098039215686275,
23
  "section_avg_rate": 0.18524332810047098,
24
+ "total_duration_seconds": 8925.054827867076,
25
+ "total_agent_steps": 579
26
  },
27
  "sections": {
28
  "shop": {
results/google_gemini-2.5-flash__ouroboros-full-isolated.json CHANGED
@@ -21,7 +21,8 @@
21
  "agent_completion_rate": 0.5882352941176471,
22
  "agent_dab_agreement_rate": 0.47058823529411764,
23
  "section_avg_rate": 0.17857142857142858,
24
- "total_duration_seconds": 30995.941995203495
 
25
  },
26
  "sections": {
27
  "shop": {
 
21
  "agent_completion_rate": 0.5882352941176471,
22
  "agent_dab_agreement_rate": 0.47058823529411764,
23
  "section_avg_rate": 0.17857142857142858,
24
+ "total_duration_seconds": 30995.941995203495,
25
+ "total_agent_steps": 679
26
  },
27
  "sections": {
28
  "shop": {
src/export_results.py CHANGED
@@ -189,6 +189,18 @@ def backfill_summary_costs(results_dir: Path | None = None) -> int:
189
  return updated
190
 
191
 
 
 
 
 
 
 
 
 
 
 
 
 
192
  def export_entry(payload: dict[str, Any], *, source_path: str | None = None) -> LeaderboardEntry:
193
  config = load_bench_config()
194
  run = payload.get("run") or {}
@@ -241,6 +253,7 @@ def export_entry(payload: dict[str, Any], *, source_path: str | None = None) ->
241
  total_tasks=int(run.get("total_tasks") or 0),
242
  avg_duration_seconds=run.get("avg_duration_seconds"),
243
  total_duration_seconds=run.get("total_duration_seconds"),
 
244
  avg_agent_steps=run.get("avg_agent_steps"),
245
  avg_tokens_per_task=run.get("avg_tokens_per_task"),
246
  total_tokens=run.get("total_tokens"),
@@ -323,6 +336,36 @@ def export_latest_runs(
323
  return written
324
 
325
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
326
  def backfill_summary_total_duration(results_dir: Path | None = None) -> int:
327
  """Add total_duration_seconds to summary JSON from copied raw results when missing."""
328
  root = results_dir or (_space_root() / "results")
 
189
  return updated
190
 
191
 
192
+ def _total_agent_steps_from_payload(payload: dict[str, Any]) -> int | None:
193
+ run = payload.get("run") or {}
194
+ total = run.get("total_agent_steps")
195
+ if isinstance(total, int):
196
+ return total
197
+ if isinstance(total, float):
198
+ return int(total)
199
+ tests = payload.get("tests") or []
200
+ steps = [row["agent_steps"] for row in tests if isinstance(row.get("agent_steps"), int)]
201
+ return sum(steps) if steps else None
202
+
203
+
204
  def export_entry(payload: dict[str, Any], *, source_path: str | None = None) -> LeaderboardEntry:
205
  config = load_bench_config()
206
  run = payload.get("run") or {}
 
253
  total_tasks=int(run.get("total_tasks") or 0),
254
  avg_duration_seconds=run.get("avg_duration_seconds"),
255
  total_duration_seconds=run.get("total_duration_seconds"),
256
+ total_agent_steps=_total_agent_steps_from_payload(payload),
257
  avg_agent_steps=run.get("avg_agent_steps"),
258
  avg_tokens_per_task=run.get("avg_tokens_per_task"),
259
  total_tokens=run.get("total_tokens"),
 
336
  return written
337
 
338
 
339
+ def backfill_summary_total_agent_steps(results_dir: Path | None = None) -> int:
340
+ """Add total_agent_steps to summary JSON from copied raw results when missing."""
341
+ root = results_dir or (_space_root() / "results")
342
+ if not root.is_dir():
343
+ return 0
344
+ updated = 0
345
+ for path in sorted(root.glob("*.json")):
346
+ payload = json.loads(path.read_text(encoding="utf-8"))
347
+ if "metrics" not in payload:
348
+ continue
349
+ metrics = dict(payload.get("metrics") or {})
350
+ if metrics.get("total_agent_steps") is not None:
351
+ continue
352
+ entry = LeaderboardEntry.from_json(payload)
353
+ raw_path = raw_results_path_for(entry, root.parent)
354
+ if not raw_path.is_file() and entry.raw_results_path:
355
+ raw_path = root.parent / entry.raw_results_path
356
+ if not raw_path.is_file():
357
+ continue
358
+ raw = json.loads(raw_path.read_text(encoding="utf-8"))
359
+ total_steps = _total_agent_steps_from_payload(raw)
360
+ if total_steps is None:
361
+ continue
362
+ metrics["total_agent_steps"] = total_steps
363
+ payload["metrics"] = metrics
364
+ path.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
365
+ updated += 1
366
+ return updated
367
+
368
+
369
  def backfill_summary_total_duration(results_dir: Path | None = None) -> int:
370
  """Add total_duration_seconds to summary JSON from copied raw results when missing."""
371
  root = results_dir or (_space_root() / "results")
src/formatting.py CHANGED
@@ -41,6 +41,12 @@ def fmt_pct_value(value: float | None) -> str:
41
  return fmt_number(value, suffix="%")
42
 
43
 
 
 
 
 
 
 
44
  def fmt_duration(seconds: float | None) -> str:
45
  if seconds is None:
46
  return "—"
 
41
  return fmt_number(value, suffix="%")
42
 
43
 
44
+ def fmt_steps(value: float | int | None) -> str:
45
+ if value is None or isinstance(value, bool):
46
+ return "—"
47
+ return str(int(round(float(value))))
48
+
49
+
50
  def fmt_duration(seconds: float | None) -> str:
51
  if seconds is None:
52
  return "—"
src/load_entries.py CHANGED
@@ -10,6 +10,7 @@ from src.bench_config import load_bench_config
10
  from src.export_results import (
11
  _cost_from_payload,
12
  _recorded_cost_from_payload,
 
13
  raw_results_path_for,
14
  ui_classes_from_payload,
15
  )
@@ -77,6 +78,10 @@ def _enrich_entry_from_raw(entry: LeaderboardEntry, raw: dict[str, Any]) -> None
77
  entry.avg_tokens_per_task = float(run["avg_tokens_per_task"])
78
  if entry.total_duration_seconds is None and isinstance(run.get("total_duration_seconds"), (int, float)):
79
  entry.total_duration_seconds = float(run["total_duration_seconds"])
 
 
 
 
80
  slots = run.get("ouroboros_model_slots")
81
  if isinstance(slots, dict) and slots and not entry.ouroboros_model_slots:
82
  entry.ouroboros_model_slots = {
@@ -104,6 +109,7 @@ def load_entry_file(path: Path) -> LeaderboardEntry | None:
104
  or entry.total_tokens is None
105
  or entry.avg_tokens_per_task is None
106
  or entry.total_duration_seconds is None
 
107
  or (entry.harness.startswith("ouroboros-full") and not entry.ouroboros_model_slots)
108
  )
109
  if needs_raw:
 
10
  from src.export_results import (
11
  _cost_from_payload,
12
  _recorded_cost_from_payload,
13
+ _total_agent_steps_from_payload,
14
  raw_results_path_for,
15
  ui_classes_from_payload,
16
  )
 
78
  entry.avg_tokens_per_task = float(run["avg_tokens_per_task"])
79
  if entry.total_duration_seconds is None and isinstance(run.get("total_duration_seconds"), (int, float)):
80
  entry.total_duration_seconds = float(run["total_duration_seconds"])
81
+ if entry.total_agent_steps is None:
82
+ total_steps = _total_agent_steps_from_payload(raw)
83
+ if total_steps is not None:
84
+ entry.total_agent_steps = total_steps
85
  slots = run.get("ouroboros_model_slots")
86
  if isinstance(slots, dict) and slots and not entry.ouroboros_model_slots:
87
  entry.ouroboros_model_slots = {
 
109
  or entry.total_tokens is None
110
  or entry.avg_tokens_per_task is None
111
  or entry.total_duration_seconds is None
112
+ or entry.total_agent_steps is None
113
  or (entry.harness.startswith("ouroboros-full") and not entry.ouroboros_model_slots)
114
  )
115
  if needs_raw:
src/models.py CHANGED
@@ -52,6 +52,7 @@ class LeaderboardEntry:
52
 
53
  avg_duration_seconds: float | None = None
54
  total_duration_seconds: float | None = None
 
55
  avg_agent_steps: float | None = None
56
  avg_tokens_per_task: float | None = None
57
  total_tokens: int | None = None
@@ -112,6 +113,8 @@ class LeaderboardEntry:
112
  return self.avg_duration_seconds
113
  if name == "total_duration_seconds":
114
  return self.total_duration_seconds
 
 
115
  if name == "total_cost_usd":
116
  return self.total_cost_usd
117
  if name == "value_score":
@@ -134,6 +137,7 @@ class LeaderboardEntry:
134
  "total_tasks": self.total_tasks,
135
  "avg_duration_seconds": self.avg_duration_seconds,
136
  "total_duration_seconds": self.total_duration_seconds,
 
137
  "avg_agent_steps": self.avg_agent_steps,
138
  "avg_tokens_per_task": self.avg_tokens_per_task,
139
  "total_tokens": self.total_tokens,
@@ -193,6 +197,7 @@ class LeaderboardEntry:
193
  total_tasks=int(metrics.get("total_tasks") or 0),
194
  avg_duration_seconds=metrics.get("avg_duration_seconds"),
195
  total_duration_seconds=metrics.get("total_duration_seconds"),
 
196
  avg_agent_steps=metrics.get("avg_agent_steps"),
197
  avg_tokens_per_task=metrics.get("avg_tokens_per_task"),
198
  total_tokens=metrics.get("total_tokens"),
 
52
 
53
  avg_duration_seconds: float | None = None
54
  total_duration_seconds: float | None = None
55
+ total_agent_steps: int | None = None
56
  avg_agent_steps: float | None = None
57
  avg_tokens_per_task: float | None = None
58
  total_tokens: int | None = None
 
113
  return self.avg_duration_seconds
114
  if name == "total_duration_seconds":
115
  return self.total_duration_seconds
116
+ if name == "total_agent_steps":
117
+ return float(self.total_agent_steps) if self.total_agent_steps is not None else None
118
  if name == "total_cost_usd":
119
  return self.total_cost_usd
120
  if name == "value_score":
 
137
  "total_tasks": self.total_tasks,
138
  "avg_duration_seconds": self.avg_duration_seconds,
139
  "total_duration_seconds": self.total_duration_seconds,
140
+ "total_agent_steps": self.total_agent_steps,
141
  "avg_agent_steps": self.avg_agent_steps,
142
  "avg_tokens_per_task": self.avg_tokens_per_task,
143
  "total_tokens": self.total_tokens,
 
197
  total_tasks=int(metrics.get("total_tasks") or 0),
198
  avg_duration_seconds=metrics.get("avg_duration_seconds"),
199
  total_duration_seconds=metrics.get("total_duration_seconds"),
200
+ total_agent_steps=metrics.get("total_agent_steps"),
201
  avg_agent_steps=metrics.get("avg_agent_steps"),
202
  avg_tokens_per_task=metrics.get("avg_tokens_per_task"),
203
  total_tokens=metrics.get("total_tokens"),
src/ui.py CHANGED
@@ -7,7 +7,7 @@ from pathlib import Path
7
  from typing import Any
8
 
9
  from src.bench_config import load_bench_config, task_count
10
- from src.formatting import fmt_cost, fmt_duration, fmt_number, fmt_pct_value, round2
11
  from src.load_entries import sort_entries
12
  from src.models import LeaderboardEntry
13
  from src.token_display import fmt_tokens, format_model_usage_breakdown_html
@@ -282,7 +282,7 @@ def _speed_row(idx: int, entry: LeaderboardEntry) -> str:
282
  <div class="wab-badges"><span class="wab-badge">{html.escape(entry.harness)}</span></div></td>
283
  <td>{html.escape(fmt_duration(entry.total_duration_seconds))}</td>
284
  <td>{html.escape(fmt_duration(entry.avg_duration_seconds))}</td>
285
- <td>{html.escape(fmt_number(entry.avg_agent_steps))}</td>
286
  <td class="wab-score-cell">
287
  <div class="wab-score-value wab-score-{tier}">{html.escape(fmt_pct_display(entry))}</div>
288
  <div class="wab-bar"><span class="wab-bar-{tier}" style="width:{max(pct, 0):.1f}%"></span></div>
@@ -320,11 +320,11 @@ def build_table(entries: list[LeaderboardEntry], view_id: str) -> str:
320
  sorted_entries = sort_entries(entries, view_id)
321
 
322
  if view_id == "speed":
323
- head = "<tr><th>#</th><th>Model</th><th>Total time</th><th>Avg time</th><th>Steps</th><th>Success %</th></tr>"
324
  rows = [_speed_row(i, entry) for i, entry in enumerate(sorted_entries)]
325
  subtitle = (
326
  "Sorted by total run duration (run.total_duration_seconds). "
327
- "Avg time is mean per-task duration (run.avg_duration_seconds)."
328
  )
329
  elif view_id == "cost":
330
  head = (
 
7
  from typing import Any
8
 
9
  from src.bench_config import load_bench_config, task_count
10
+ from src.formatting import fmt_cost, fmt_duration, fmt_number, fmt_pct_value, fmt_steps, round2
11
  from src.load_entries import sort_entries
12
  from src.models import LeaderboardEntry
13
  from src.token_display import fmt_tokens, format_model_usage_breakdown_html
 
282
  <div class="wab-badges"><span class="wab-badge">{html.escape(entry.harness)}</span></div></td>
283
  <td>{html.escape(fmt_duration(entry.total_duration_seconds))}</td>
284
  <td>{html.escape(fmt_duration(entry.avg_duration_seconds))}</td>
285
+ <td>{html.escape(fmt_steps(entry.total_agent_steps))}</td>
286
  <td class="wab-score-cell">
287
  <div class="wab-score-value wab-score-{tier}">{html.escape(fmt_pct_display(entry))}</div>
288
  <div class="wab-bar"><span class="wab-bar-{tier}" style="width:{max(pct, 0):.1f}%"></span></div>
 
320
  sorted_entries = sort_entries(entries, view_id)
321
 
322
  if view_id == "speed":
323
+ head = "<tr><th>#</th><th>Model</th><th>Total time</th><th>Avg time</th><th>Total steps</th><th>Success %</th></tr>"
324
  rows = [_speed_row(i, entry) for i, entry in enumerate(sorted_entries)]
325
  subtitle = (
326
  "Sorted by total run duration (run.total_duration_seconds). "
327
+ "Avg time is mean per-task duration. Total steps is the sum of agent_steps across tasks."
328
  )
329
  elif view_id == "cost":
330
  head = (