posttrain-arena / dev /mock_server.py
Xiangyi Li
SkillsBench challenge; Terminal-Bench 2 out of the arena; gates check SkillsBench; practice off the board
77d8d06
Raw History Blame Contribute Delete
9.97 kB
"""Dev only, not served by the Space: the real app on synthetic, offline data, to see populated states
(scored and verified runs, a live training run, multi-step trainer logs, gate pass rates over two trials).
python dev/mock_server.py [port]
Every data source the page reads is patched in-process and HF Hub is forced offline; nothing touches HF.
All run IDs, scores and notes below are made up.
"""
import os, sys, time
from datetime import datetime, timedelta, timezone
from pathlib import Path
from types import SimpleNamespace
from unittest.mock import patch
os.environ.update(HF_HUB_OFFLINE='1', HF_TOKEN='mock')
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
import uvicorn
import app, auth, challenges, collab, environments as env, arena_jobs as jobs
NOW = datetime.now(timezone.utc)
def at(minutes): return (NOW + timedelta(minutes=minutes)).isoformat().replace('+00:00', 'Z')
ENV = {'id': 'env-mock0000001', 'challenge_id': 'skillsbench', 'repo_type': 'dataset', 'repo_id': 'mock/pack', 'revision': 'a' * 40, 'entry_path': '', 'title': 'Mock pack · shell repair', 'author': 'mock-team',
'agent_id': 'mock-agent', 'status': 'Validated', 'task_count': 40, 'source_url': 'https://huggingface.co/datasets/mock/pack', 'created_at': at(-3000)}
ENV2 = {**ENV, 'id': 'env-mock0000002', 'revision': 'b' * 40, 'title': 'Mock pack · log triage', 'author': 'other-team', 'agent_id': None, 'task_count': 12}
def eval_stage(start, marker, passed, failed, errors, tasks=None, timeouts=0, minutes=50):
lines = [(at(start), f'[posttrainarena] {marker}: bench eval run'), (at(start), f'Job: {passed + failed + errors} tasks, 0 done')]
names = tasks or [f'{marker}-t{i}' for i in range(passed + failed + errors)]
kinds = ['PASS'] * passed + ['FAIL'] * failed + ['ERR'] * errors
for i, (name, kind) in enumerate(zip(names, kinds)):
note = ' (Agent prompt exceeded wall-clock budget 900s)' if kind == 'FAIL' and i < passed + timeouts else (' (ACP initialize timed out after 180.0s before the…)' if kind == 'ERR' else '')
lines.append((at(start + 1 + i * minutes // max(1, len(kinds))), f'[{kind}] {name} (tools=9){note}'))
return lines + [(at(start + minutes), f'Job complete: {passed}/{len(kinds)} ({100 * passed / len(kinds):.1f}%), errors={errors}, idle_timeouts=0, time={minutes}.0min')]
def boot(start):
return [(at(start), 'pipeline ref 20ab5c45ff01aec89b650e80c701a412c04d8233 at 20ab5c4'), (at(start + 5), 'vllm up'), (at(start + 5), 'bridge up'), (at(start + 6), 'relay reachable via https://mock/relay/v1'),
(at(start + 6), '[posttrainarena] snapshot_train_tasks: snapshot'), (at(start + 7), '[posttrainarena] snapshot_eval_tasks: snapshot')]
def training(start, steps, rewards):
lines = []
for step in range(steps):
t = start + step * 12
lines += [(at(t), f'[posttrainarena] grpo_rollout_{step:06d}_000000: bench eval run'), (at(t + 5), '[PASS] g0 (tools=4)'), (at(t + 6), '[FAIL] g1 (tools=4)')]
if step % 2: lines.append((at(t + 7), '[ERR] g2 (tools=0) (Failed to execute command: )'))
r = rewards[step]
lines.append((at(t + 10), str({'loss': 0.02 - 0.003 * step, 'grad_norm': 0.8 - 0.1 * step, 'learning_rate': 1e-6, 'reward': r, 'reward_std': 0.4, 'frac_reward_zero_std': 0.5 - 0.08 * step,
'kl': 0.001 * step, 'entropy': 0.9 - 0.05 * step, 'completions/mean_length': 5200 + 300 * step, 'clip_ratio/region_mean': 0.02 + 0.01 * step, 'epoch': step + 1.0})))
return lines
GATE_TASKS = [f'task-{i:02d}' for i in range(12)]
LOGS = {
'job-a1': boot(-900) + eval_stage(-893, 'baseline_eval', 3, 28, 1, timeouts=12) + eval_stage(-840, 'grpo_gate_eval', 5, 6, 1, tasks=GATE_TASKS) + training(-785, 4, [0.25, 0.375, 0.375, 0.5])
+ [(at(-735), '[posttrainarena] sync_grpo_endpoint: sync')] + eval_stage(-734, 'posttrain_eval', 5, 27, 0, timeouts=10) + [(at(-680), 'TRAINER_EXIT=0')],
'job-a2': boot(-600) + eval_stage(-593, 'baseline_eval', 2, 29, 1, timeouts=15) + eval_stage(-540, 'grpo_gate_eval', 3, 8, 1, tasks=GATE_TASKS[3:] + GATE_TASKS[:3])
+ training(-485, 2, [0.25, 0.25]) + [(at(-460), '[posttrainarena] sync_grpo_endpoint: sync')] + eval_stage(-459, 'posttrain_eval', 1, 31, 0, timeouts=17) + [(at(-400), 'TRAINER_EXIT=0')],
'job-a3': boot(-380) + eval_stage(-373, 'baseline_eval', 2, 28, 2, timeouts=15) + eval_stage(-320, 'grpo_gate_eval', 1, 2, 9, tasks=GATE_TASKS)
+ [(at(-270), 'RuntimeError: OpenCode evaluation contains agent or verifier errors (9 > 4 tolerated)'), (at(-270), 'TRAINER_EXIT=1')],
'job-a4': boot(-150) + eval_stage(-143, 'baseline_eval', 4, 27, 1, timeouts=14) + eval_stage(-90, 'grpo_gate_eval', 6, 5, 1, tasks=GATE_TASKS[:12]) + training(-38, 3, [0.25, 0.375, 0.5]),
'job-o1': boot(-2000) + eval_stage(-1993, 'baseline_eval', 4, 27, 1, timeouts=13) + [(at(-1940), 'subprocess.CalledProcessError: bench eval run exited 1'), (at(-1940), 'TRAINER_EXIT=1')],
}
STAGES = {'job-a1': 'COMPLETED', 'job-a2': 'COMPLETED', 'job-a3': 'ERROR', 'job-a4': 'RUNNING', 'job-o1': 'COMPLETED'}
def record(run_id, job, created, env_row, **extra):
return {'run_id': run_id, 'request_key': run_id, 'kind': 'challenge-run', 'author': env_row['author'], 'status': 'SCHEDULING', 'created_at': at(created), 'job_id': job, 'job_url': 'https://huggingface.co/jobs/benchflow/' + job,
'config': {'challenge_id': 'skillsbench-9b', 'environment_id': env_row['id'], 'environment_revision': env_row['revision'], 'train_task_count': env_row['task_count']}, 'max_compute_usd': 160.0, **extra}
LEDGER = {'prior_allowance_usd': 50, 'runs': [record('challenge-d9e0f1a20004', 'job-a4', -152, ENV), record('challenge-c3d4e5f60003', 'job-a3', -382, ENV2, settled_usd=40.0),
record('challenge-b7e8f9a00002', 'job-a2', -602, ENV2, settled_usd=72.5), record('challenge-a1b2c3d40001', 'job-a1', -902, ENV, settled_usd=81.0)]}
SUMMARIES = {'challenge-a1b2c3d40001': {'baseline_score': 3 / 32, 'score_after_posttrain': 5 / 32, 'grpo_ran': True}, 'challenge-b7e8f9a00002': {'baseline_score': 2 / 32, 'score_after_posttrain': 1 / 32, 'grpo_ran': True}}
def result(run_id, env_row, b, a, verification):
return {'run_id': run_id, 'challenge_id': 'skillsbench-9b', 'environment_id': env_row['id'], 'environment_revision': env_row['revision'], 'author': env_row['author'], 'agent_id': env_row['agent_id'],
'baseline_pass_rate': b / 32, 'after_pass_rate': a / 32, 'delta_pp': round(100 * (a - b) / 32, 4), 'stderr_pp': round(100 * ((b / 32 * (1 - b / 32) + a / 32 * (1 - a / 32)) / 32) ** .5, 4), 'n_tasks': 32,
'trials': 1, 'grpo_ran': True, 'report_url': 'https://huggingface.co/datasets/mock/runs/blob/' + 'c' * 40 + '/score.json', 'verification': verification, 'collected_at': at(-600)}
REGISTRY = {
challenges.RESULTS: [result('challenge-a1b2c3d40001', ENV, 3, 5, 'valid'), result('challenge-b7e8f9a00002', ENV2, 2, 1, 'pending')],
collab.MESSAGES: [{'agent_id': 'arena-system', 'owner': 'benchflow', 'type': 'agent', 'refs': [], 'filename': '20260924-000000_arena-system_mock01.md', 'created_at': at(-152),
'body': 'Run challenge-d9e0f1a20004 started on challenge skillsbench-9b for submission env-mock0000001 (Mock pack · shell repair, 40 tasks) by mock-team.'}],
challenges.NOTICES: [{'t': at(-100), 'text': 'Mock notice: the gate now drops tasks the base model always fails.'}],
collab.AGENTS: [], collab.EXPERIMENTS: [],
}
PHASE2 = [{'run_name': 'phase2-grpo-mock-r30', 'job_id': 'job-o1', 'job_url': 'https://huggingface.co/jobs/benchflow/job-o1', 'status': 'COMPLETED', 'created_at': at(-2002), 'note': 'mock organizer run: baseline accepted, gate aborted'}]
PRICING = [{'name': 'a100x8', 'unitLabel': 'minute', 'unitCostUSD': 0.333333}, {'name': 'a100-large', 'unitLabel': 'minute', 'unitCostUSD': 0.041667}, {'name': 'cpu-upgrade', 'unitLabel': 'minute', 'unitCostUSD': 0.0005}]
def read_registry(path=env.PATH, head=None, default=None):
if path == env.PATH: return [ENV, ENV2]
return REGISTRY.get(path, [] if default is None else default)
api = SimpleNamespace(token='mock', repo_info=lambda *a, **k: SimpleNamespace(sha='d' * 40), inspect_job=lambda job_id, namespace: SimpleNamespace(status=SimpleNamespace(stage=STAGES[job_id])),
fetch_job_logs=lambda job_id, namespace: [text for _, text in LOGS[job_id]])
def refuse(*a, **k): raise RuntimeError('mock server is read-only')
PATCHES = [patch.object(jobs, 'api', return_value=api), patch.object(jobs, 'read', side_effect=lambda head=None: LEDGER), patch.object(jobs, 'write', side_effect=refuse),
patch.object(jobs.httpx, 'get', return_value=SimpleNamespace(raise_for_status=lambda: None, json=lambda: PRICING)),
patch.object(env, 'read', side_effect=read_registry), patch.object(env, 'replace_file', side_effect=refuse),
patch.object(challenges, 'job_log', side_effect=lambda job_id: LOGS.get(job_id)), patch.object(challenges, 'uploaded_log', return_value=None),
patch.object(challenges, 'report', side_effect=lambda run_id: SUMMARIES.get(run_id)), patch.object(challenges, 'isolation', side_effect=lambda run_id: 0),
patch.object(challenges, 'phase2_rows', side_effect=lambda: PHASE2), patch.object(collab, 'system_post', side_effect=refuse),
patch.object(challenges, 'baseline_reference', return_value={'pass_rate': 0.041667, 'stderr': 0.010417, 'trials': 3, 'pass_rates': [0.0625, 0.03125, 0.03125], 'source': 'https://huggingface.co/datasets/mock/runs', 'harness': 'mock'})]
if __name__ == '__main__':
for p in PATCHES: p.start()
challenges.relays.RELAYS['challenge-d9e0f1a20004'] = SimpleNamespace(requests=2900, connected_at=time.time() - 145 * 60, reconnects=0)
uvicorn.run(app.app, host='127.0.0.1', port=int(sys.argv[1]) if len(sys.argv) > 1 else 7862, access_log=False)