Spaces:
Running
Running
Xiangyi Li
SkillsBench challenge; Terminal-Bench 2 out of the arena; gates check SkillsBench; practice off the board
77d8d06 Download dev/mock_server.py from benchflow/posttrain-arena: direct link, hf CLI and curl.
- Browser
- Download file 9.97 kB
-
https://huggingface.co/spaces/benchflow/posttrain-arena/resolve/main/dev/mock_server.py
- Command line
-
hf download hf://spaces/benchflow/posttrain-arena/dev/mock_server.py
-
curl -L -o mock_server.py https://huggingface.co/spaces/benchflow/posttrain-arena/resolve/main/dev/mock_server.py
9.97 kB
| """Dev only, not served by the Space: the real app on synthetic, offline data, to see populated states | |
| (scored and verified runs, a live training run, multi-step trainer logs, gate pass rates over two trials). | |
| python dev/mock_server.py [port] | |
| Every data source the page reads is patched in-process and HF Hub is forced offline; nothing touches HF. | |
| All run IDs, scores and notes below are made up. | |
| """ | |
| import os, sys, time | |
| from datetime import datetime, timedelta, timezone | |
| from pathlib import Path | |
| from types import SimpleNamespace | |
| from unittest.mock import patch | |
| os.environ.update(HF_HUB_OFFLINE='1', HF_TOKEN='mock') | |
| sys.path.insert(0, str(Path(__file__).resolve().parents[1])) | |
| import uvicorn | |
| import app, auth, challenges, collab, environments as env, arena_jobs as jobs | |
| NOW = datetime.now(timezone.utc) | |
| def at(minutes): return (NOW + timedelta(minutes=minutes)).isoformat().replace('+00:00', 'Z') | |
| ENV = {'id': 'env-mock0000001', 'challenge_id': 'skillsbench', 'repo_type': 'dataset', 'repo_id': 'mock/pack', 'revision': 'a' * 40, 'entry_path': '', 'title': 'Mock pack · shell repair', 'author': 'mock-team', | |
| 'agent_id': 'mock-agent', 'status': 'Validated', 'task_count': 40, 'source_url': 'https://huggingface.co/datasets/mock/pack', 'created_at': at(-3000)} | |
| ENV2 = {**ENV, 'id': 'env-mock0000002', 'revision': 'b' * 40, 'title': 'Mock pack · log triage', 'author': 'other-team', 'agent_id': None, 'task_count': 12} | |
| def eval_stage(start, marker, passed, failed, errors, tasks=None, timeouts=0, minutes=50): | |
| lines = [(at(start), f'[posttrainarena] {marker}: bench eval run'), (at(start), f'Job: {passed + failed + errors} tasks, 0 done')] | |
| names = tasks or [f'{marker}-t{i}' for i in range(passed + failed + errors)] | |
| kinds = ['PASS'] * passed + ['FAIL'] * failed + ['ERR'] * errors | |
| for i, (name, kind) in enumerate(zip(names, kinds)): | |
| note = ' (Agent prompt exceeded wall-clock budget 900s)' if kind == 'FAIL' and i < passed + timeouts else (' (ACP initialize timed out after 180.0s before the…)' if kind == 'ERR' else '') | |
| lines.append((at(start + 1 + i * minutes // max(1, len(kinds))), f'[{kind}] {name} (tools=9){note}')) | |
| return lines + [(at(start + minutes), f'Job complete: {passed}/{len(kinds)} ({100 * passed / len(kinds):.1f}%), errors={errors}, idle_timeouts=0, time={minutes}.0min')] | |
| def boot(start): | |
| return [(at(start), 'pipeline ref 20ab5c45ff01aec89b650e80c701a412c04d8233 at 20ab5c4'), (at(start + 5), 'vllm up'), (at(start + 5), 'bridge up'), (at(start + 6), 'relay reachable via https://mock/relay/v1'), | |
| (at(start + 6), '[posttrainarena] snapshot_train_tasks: snapshot'), (at(start + 7), '[posttrainarena] snapshot_eval_tasks: snapshot')] | |
| def training(start, steps, rewards): | |
| lines = [] | |
| for step in range(steps): | |
| t = start + step * 12 | |
| lines += [(at(t), f'[posttrainarena] grpo_rollout_{step:06d}_000000: bench eval run'), (at(t + 5), '[PASS] g0 (tools=4)'), (at(t + 6), '[FAIL] g1 (tools=4)')] | |
| if step % 2: lines.append((at(t + 7), '[ERR] g2 (tools=0) (Failed to execute command: )')) | |
| r = rewards[step] | |
| lines.append((at(t + 10), str({'loss': 0.02 - 0.003 * step, 'grad_norm': 0.8 - 0.1 * step, 'learning_rate': 1e-6, 'reward': r, 'reward_std': 0.4, 'frac_reward_zero_std': 0.5 - 0.08 * step, | |
| 'kl': 0.001 * step, 'entropy': 0.9 - 0.05 * step, 'completions/mean_length': 5200 + 300 * step, 'clip_ratio/region_mean': 0.02 + 0.01 * step, 'epoch': step + 1.0}))) | |
| return lines | |
| GATE_TASKS = [f'task-{i:02d}' for i in range(12)] | |
| LOGS = { | |
| 'job-a1': boot(-900) + eval_stage(-893, 'baseline_eval', 3, 28, 1, timeouts=12) + eval_stage(-840, 'grpo_gate_eval', 5, 6, 1, tasks=GATE_TASKS) + training(-785, 4, [0.25, 0.375, 0.375, 0.5]) | |
| + [(at(-735), '[posttrainarena] sync_grpo_endpoint: sync')] + eval_stage(-734, 'posttrain_eval', 5, 27, 0, timeouts=10) + [(at(-680), 'TRAINER_EXIT=0')], | |
| 'job-a2': boot(-600) + eval_stage(-593, 'baseline_eval', 2, 29, 1, timeouts=15) + eval_stage(-540, 'grpo_gate_eval', 3, 8, 1, tasks=GATE_TASKS[3:] + GATE_TASKS[:3]) | |
| + training(-485, 2, [0.25, 0.25]) + [(at(-460), '[posttrainarena] sync_grpo_endpoint: sync')] + eval_stage(-459, 'posttrain_eval', 1, 31, 0, timeouts=17) + [(at(-400), 'TRAINER_EXIT=0')], | |
| 'job-a3': boot(-380) + eval_stage(-373, 'baseline_eval', 2, 28, 2, timeouts=15) + eval_stage(-320, 'grpo_gate_eval', 1, 2, 9, tasks=GATE_TASKS) | |
| + [(at(-270), 'RuntimeError: OpenCode evaluation contains agent or verifier errors (9 > 4 tolerated)'), (at(-270), 'TRAINER_EXIT=1')], | |
| 'job-a4': boot(-150) + eval_stage(-143, 'baseline_eval', 4, 27, 1, timeouts=14) + eval_stage(-90, 'grpo_gate_eval', 6, 5, 1, tasks=GATE_TASKS[:12]) + training(-38, 3, [0.25, 0.375, 0.5]), | |
| 'job-o1': boot(-2000) + eval_stage(-1993, 'baseline_eval', 4, 27, 1, timeouts=13) + [(at(-1940), 'subprocess.CalledProcessError: bench eval run exited 1'), (at(-1940), 'TRAINER_EXIT=1')], | |
| } | |
| STAGES = {'job-a1': 'COMPLETED', 'job-a2': 'COMPLETED', 'job-a3': 'ERROR', 'job-a4': 'RUNNING', 'job-o1': 'COMPLETED'} | |
| def record(run_id, job, created, env_row, **extra): | |
| return {'run_id': run_id, 'request_key': run_id, 'kind': 'challenge-run', 'author': env_row['author'], 'status': 'SCHEDULING', 'created_at': at(created), 'job_id': job, 'job_url': 'https://huggingface.co/jobs/benchflow/' + job, | |
| 'config': {'challenge_id': 'skillsbench-9b', 'environment_id': env_row['id'], 'environment_revision': env_row['revision'], 'train_task_count': env_row['task_count']}, 'max_compute_usd': 160.0, **extra} | |
| LEDGER = {'prior_allowance_usd': 50, 'runs': [record('challenge-d9e0f1a20004', 'job-a4', -152, ENV), record('challenge-c3d4e5f60003', 'job-a3', -382, ENV2, settled_usd=40.0), | |
| record('challenge-b7e8f9a00002', 'job-a2', -602, ENV2, settled_usd=72.5), record('challenge-a1b2c3d40001', 'job-a1', -902, ENV, settled_usd=81.0)]} | |
| SUMMARIES = {'challenge-a1b2c3d40001': {'baseline_score': 3 / 32, 'score_after_posttrain': 5 / 32, 'grpo_ran': True}, 'challenge-b7e8f9a00002': {'baseline_score': 2 / 32, 'score_after_posttrain': 1 / 32, 'grpo_ran': True}} | |
| def result(run_id, env_row, b, a, verification): | |
| return {'run_id': run_id, 'challenge_id': 'skillsbench-9b', 'environment_id': env_row['id'], 'environment_revision': env_row['revision'], 'author': env_row['author'], 'agent_id': env_row['agent_id'], | |
| 'baseline_pass_rate': b / 32, 'after_pass_rate': a / 32, 'delta_pp': round(100 * (a - b) / 32, 4), 'stderr_pp': round(100 * ((b / 32 * (1 - b / 32) + a / 32 * (1 - a / 32)) / 32) ** .5, 4), 'n_tasks': 32, | |
| 'trials': 1, 'grpo_ran': True, 'report_url': 'https://huggingface.co/datasets/mock/runs/blob/' + 'c' * 40 + '/score.json', 'verification': verification, 'collected_at': at(-600)} | |
| REGISTRY = { | |
| challenges.RESULTS: [result('challenge-a1b2c3d40001', ENV, 3, 5, 'valid'), result('challenge-b7e8f9a00002', ENV2, 2, 1, 'pending')], | |
| collab.MESSAGES: [{'agent_id': 'arena-system', 'owner': 'benchflow', 'type': 'agent', 'refs': [], 'filename': '20260924-000000_arena-system_mock01.md', 'created_at': at(-152), | |
| 'body': 'Run challenge-d9e0f1a20004 started on challenge skillsbench-9b for submission env-mock0000001 (Mock pack · shell repair, 40 tasks) by mock-team.'}], | |
| challenges.NOTICES: [{'t': at(-100), 'text': 'Mock notice: the gate now drops tasks the base model always fails.'}], | |
| collab.AGENTS: [], collab.EXPERIMENTS: [], | |
| } | |
| PHASE2 = [{'run_name': 'phase2-grpo-mock-r30', 'job_id': 'job-o1', 'job_url': 'https://huggingface.co/jobs/benchflow/job-o1', 'status': 'COMPLETED', 'created_at': at(-2002), 'note': 'mock organizer run: baseline accepted, gate aborted'}] | |
| PRICING = [{'name': 'a100x8', 'unitLabel': 'minute', 'unitCostUSD': 0.333333}, {'name': 'a100-large', 'unitLabel': 'minute', 'unitCostUSD': 0.041667}, {'name': 'cpu-upgrade', 'unitLabel': 'minute', 'unitCostUSD': 0.0005}] | |
| def read_registry(path=env.PATH, head=None, default=None): | |
| if path == env.PATH: return [ENV, ENV2] | |
| return REGISTRY.get(path, [] if default is None else default) | |
| api = SimpleNamespace(token='mock', repo_info=lambda *a, **k: SimpleNamespace(sha='d' * 40), inspect_job=lambda job_id, namespace: SimpleNamespace(status=SimpleNamespace(stage=STAGES[job_id])), | |
| fetch_job_logs=lambda job_id, namespace: [text for _, text in LOGS[job_id]]) | |
| def refuse(*a, **k): raise RuntimeError('mock server is read-only') | |
| PATCHES = [patch.object(jobs, 'api', return_value=api), patch.object(jobs, 'read', side_effect=lambda head=None: LEDGER), patch.object(jobs, 'write', side_effect=refuse), | |
| patch.object(jobs.httpx, 'get', return_value=SimpleNamespace(raise_for_status=lambda: None, json=lambda: PRICING)), | |
| patch.object(env, 'read', side_effect=read_registry), patch.object(env, 'replace_file', side_effect=refuse), | |
| patch.object(challenges, 'job_log', side_effect=lambda job_id: LOGS.get(job_id)), patch.object(challenges, 'uploaded_log', return_value=None), | |
| patch.object(challenges, 'report', side_effect=lambda run_id: SUMMARIES.get(run_id)), patch.object(challenges, 'isolation', side_effect=lambda run_id: 0), | |
| patch.object(challenges, 'phase2_rows', side_effect=lambda: PHASE2), patch.object(collab, 'system_post', side_effect=refuse), | |
| patch.object(challenges, 'baseline_reference', return_value={'pass_rate': 0.041667, 'stderr': 0.010417, 'trials': 3, 'pass_rates': [0.0625, 0.03125, 0.03125], 'source': 'https://huggingface.co/datasets/mock/runs', 'harness': 'mock'})] | |
| if __name__ == '__main__': | |
| for p in PATCHES: p.start() | |
| challenges.relays.RELAYS['challenge-d9e0f1a20004'] = SimpleNamespace(requests=2900, connected_at=time.time() - 145 * 60, reconnects=0) | |
| uvicorn.run(app.app, host='127.0.0.1', port=int(sys.argv[1]) if len(sys.argv) > 1 else 7862, access_log=False) | |