"""Dev only, not served by the Space: the real app on synthetic, offline data, to see populated states (scored and verified runs, a live training run, multi-step trainer logs, gate pass rates over two trials). python dev/mock_server.py [port] Every data source the page reads is patched in-process and HF Hub is forced offline; nothing touches HF. All run IDs, scores and notes below are made up. """ import os, sys, time from datetime import datetime, timedelta, timezone from pathlib import Path from types import SimpleNamespace from unittest.mock import patch os.environ.update(HF_HUB_OFFLINE='1', HF_TOKEN='mock') sys.path.insert(0, str(Path(__file__).resolve().parents[1])) import uvicorn import app, auth, challenges, collab, environments as env, arena_jobs as jobs NOW = datetime.now(timezone.utc) def at(minutes): return (NOW + timedelta(minutes=minutes)).isoformat().replace('+00:00', 'Z') ENV = {'id': 'env-mock0000001', 'challenge_id': 'skillsbench', 'repo_type': 'dataset', 'repo_id': 'mock/pack', 'revision': 'a' * 40, 'entry_path': '', 'title': 'Mock pack · shell repair', 'author': 'mock-team', 'agent_id': 'mock-agent', 'status': 'Validated', 'task_count': 40, 'source_url': 'https://huggingface.co/datasets/mock/pack', 'created_at': at(-3000)} ENV2 = {**ENV, 'id': 'env-mock0000002', 'revision': 'b' * 40, 'title': 'Mock pack · log triage', 'author': 'other-team', 'agent_id': None, 'task_count': 12} def eval_stage(start, marker, passed, failed, errors, tasks=None, timeouts=0, minutes=50): lines = [(at(start), f'[posttrainarena] {marker}: bench eval run'), (at(start), f'Job: {passed + failed + errors} tasks, 0 done')] names = tasks or [f'{marker}-t{i}' for i in range(passed + failed + errors)] kinds = ['PASS'] * passed + ['FAIL'] * failed + ['ERR'] * errors for i, (name, kind) in enumerate(zip(names, kinds)): note = ' (Agent prompt exceeded wall-clock budget 900s)' if kind == 'FAIL' and i < passed + timeouts else (' (ACP initialize timed out after 180.0s before the…)' if kind == 'ERR' else '') lines.append((at(start + 1 + i * minutes // max(1, len(kinds))), f'[{kind}] {name} (tools=9){note}')) return lines + [(at(start + minutes), f'Job complete: {passed}/{len(kinds)} ({100 * passed / len(kinds):.1f}%), errors={errors}, idle_timeouts=0, time={minutes}.0min')] def boot(start): return [(at(start), 'pipeline ref 20ab5c45ff01aec89b650e80c701a412c04d8233 at 20ab5c4'), (at(start + 5), 'vllm up'), (at(start + 5), 'bridge up'), (at(start + 6), 'relay reachable via https://mock/relay/v1'), (at(start + 6), '[posttrainarena] snapshot_train_tasks: snapshot'), (at(start + 7), '[posttrainarena] snapshot_eval_tasks: snapshot')] def training(start, steps, rewards): lines = [] for step in range(steps): t = start + step * 12 lines += [(at(t), f'[posttrainarena] grpo_rollout_{step:06d}_000000: bench eval run'), (at(t + 5), '[PASS] g0 (tools=4)'), (at(t + 6), '[FAIL] g1 (tools=4)')] if step % 2: lines.append((at(t + 7), '[ERR] g2 (tools=0) (Failed to execute command: )')) r = rewards[step] lines.append((at(t + 10), str({'loss': 0.02 - 0.003 * step, 'grad_norm': 0.8 - 0.1 * step, 'learning_rate': 1e-6, 'reward': r, 'reward_std': 0.4, 'frac_reward_zero_std': 0.5 - 0.08 * step, 'kl': 0.001 * step, 'entropy': 0.9 - 0.05 * step, 'completions/mean_length': 5200 + 300 * step, 'clip_ratio/region_mean': 0.02 + 0.01 * step, 'epoch': step + 1.0}))) return lines GATE_TASKS = [f'task-{i:02d}' for i in range(12)] LOGS = { 'job-a1': boot(-900) + eval_stage(-893, 'baseline_eval', 3, 28, 1, timeouts=12) + eval_stage(-840, 'grpo_gate_eval', 5, 6, 1, tasks=GATE_TASKS) + training(-785, 4, [0.25, 0.375, 0.375, 0.5]) + [(at(-735), '[posttrainarena] sync_grpo_endpoint: sync')] + eval_stage(-734, 'posttrain_eval', 5, 27, 0, timeouts=10) + [(at(-680), 'TRAINER_EXIT=0')], 'job-a2': boot(-600) + eval_stage(-593, 'baseline_eval', 2, 29, 1, timeouts=15) + eval_stage(-540, 'grpo_gate_eval', 3, 8, 1, tasks=GATE_TASKS[3:] + GATE_TASKS[:3]) + training(-485, 2, [0.25, 0.25]) + [(at(-460), '[posttrainarena] sync_grpo_endpoint: sync')] + eval_stage(-459, 'posttrain_eval', 1, 31, 0, timeouts=17) + [(at(-400), 'TRAINER_EXIT=0')], 'job-a3': boot(-380) + eval_stage(-373, 'baseline_eval', 2, 28, 2, timeouts=15) + eval_stage(-320, 'grpo_gate_eval', 1, 2, 9, tasks=GATE_TASKS) + [(at(-270), 'RuntimeError: OpenCode evaluation contains agent or verifier errors (9 > 4 tolerated)'), (at(-270), 'TRAINER_EXIT=1')], 'job-a4': boot(-150) + eval_stage(-143, 'baseline_eval', 4, 27, 1, timeouts=14) + eval_stage(-90, 'grpo_gate_eval', 6, 5, 1, tasks=GATE_TASKS[:12]) + training(-38, 3, [0.25, 0.375, 0.5]), 'job-o1': boot(-2000) + eval_stage(-1993, 'baseline_eval', 4, 27, 1, timeouts=13) + [(at(-1940), 'subprocess.CalledProcessError: bench eval run exited 1'), (at(-1940), 'TRAINER_EXIT=1')], } STAGES = {'job-a1': 'COMPLETED', 'job-a2': 'COMPLETED', 'job-a3': 'ERROR', 'job-a4': 'RUNNING', 'job-o1': 'COMPLETED'} def record(run_id, job, created, env_row, **extra): return {'run_id': run_id, 'request_key': run_id, 'kind': 'challenge-run', 'author': env_row['author'], 'status': 'SCHEDULING', 'created_at': at(created), 'job_id': job, 'job_url': 'https://huggingface.co/jobs/benchflow/' + job, 'config': {'challenge_id': 'skillsbench-9b', 'environment_id': env_row['id'], 'environment_revision': env_row['revision'], 'train_task_count': env_row['task_count']}, 'max_compute_usd': 160.0, **extra} LEDGER = {'prior_allowance_usd': 50, 'runs': [record('challenge-d9e0f1a20004', 'job-a4', -152, ENV), record('challenge-c3d4e5f60003', 'job-a3', -382, ENV2, settled_usd=40.0), record('challenge-b7e8f9a00002', 'job-a2', -602, ENV2, settled_usd=72.5), record('challenge-a1b2c3d40001', 'job-a1', -902, ENV, settled_usd=81.0)]} SUMMARIES = {'challenge-a1b2c3d40001': {'baseline_score': 3 / 32, 'score_after_posttrain': 5 / 32, 'grpo_ran': True}, 'challenge-b7e8f9a00002': {'baseline_score': 2 / 32, 'score_after_posttrain': 1 / 32, 'grpo_ran': True}} def result(run_id, env_row, b, a, verification): return {'run_id': run_id, 'challenge_id': 'skillsbench-9b', 'environment_id': env_row['id'], 'environment_revision': env_row['revision'], 'author': env_row['author'], 'agent_id': env_row['agent_id'], 'baseline_pass_rate': b / 32, 'after_pass_rate': a / 32, 'delta_pp': round(100 * (a - b) / 32, 4), 'stderr_pp': round(100 * ((b / 32 * (1 - b / 32) + a / 32 * (1 - a / 32)) / 32) ** .5, 4), 'n_tasks': 32, 'trials': 1, 'grpo_ran': True, 'report_url': 'https://huggingface.co/datasets/mock/runs/blob/' + 'c' * 40 + '/score.json', 'verification': verification, 'collected_at': at(-600)} REGISTRY = { challenges.RESULTS: [result('challenge-a1b2c3d40001', ENV, 3, 5, 'valid'), result('challenge-b7e8f9a00002', ENV2, 2, 1, 'pending')], collab.MESSAGES: [{'agent_id': 'arena-system', 'owner': 'benchflow', 'type': 'agent', 'refs': [], 'filename': '20260924-000000_arena-system_mock01.md', 'created_at': at(-152), 'body': 'Run challenge-d9e0f1a20004 started on challenge skillsbench-9b for submission env-mock0000001 (Mock pack · shell repair, 40 tasks) by mock-team.'}], challenges.NOTICES: [{'t': at(-100), 'text': 'Mock notice: the gate now drops tasks the base model always fails.'}], collab.AGENTS: [], collab.EXPERIMENTS: [], } PHASE2 = [{'run_name': 'phase2-grpo-mock-r30', 'job_id': 'job-o1', 'job_url': 'https://huggingface.co/jobs/benchflow/job-o1', 'status': 'COMPLETED', 'created_at': at(-2002), 'note': 'mock organizer run: baseline accepted, gate aborted'}] PRICING = [{'name': 'a100x8', 'unitLabel': 'minute', 'unitCostUSD': 0.333333}, {'name': 'a100-large', 'unitLabel': 'minute', 'unitCostUSD': 0.041667}, {'name': 'cpu-upgrade', 'unitLabel': 'minute', 'unitCostUSD': 0.0005}] def read_registry(path=env.PATH, head=None, default=None): if path == env.PATH: return [ENV, ENV2] return REGISTRY.get(path, [] if default is None else default) api = SimpleNamespace(token='mock', repo_info=lambda *a, **k: SimpleNamespace(sha='d' * 40), inspect_job=lambda job_id, namespace: SimpleNamespace(status=SimpleNamespace(stage=STAGES[job_id])), fetch_job_logs=lambda job_id, namespace: [text for _, text in LOGS[job_id]]) def refuse(*a, **k): raise RuntimeError('mock server is read-only') PATCHES = [patch.object(jobs, 'api', return_value=api), patch.object(jobs, 'read', side_effect=lambda head=None: LEDGER), patch.object(jobs, 'write', side_effect=refuse), patch.object(jobs.httpx, 'get', return_value=SimpleNamespace(raise_for_status=lambda: None, json=lambda: PRICING)), patch.object(env, 'read', side_effect=read_registry), patch.object(env, 'replace_file', side_effect=refuse), patch.object(challenges, 'job_log', side_effect=lambda job_id: LOGS.get(job_id)), patch.object(challenges, 'uploaded_log', return_value=None), patch.object(challenges, 'report', side_effect=lambda run_id: SUMMARIES.get(run_id)), patch.object(challenges, 'isolation', side_effect=lambda run_id: 0), patch.object(challenges, 'phase2_rows', side_effect=lambda: PHASE2), patch.object(collab, 'system_post', side_effect=refuse), patch.object(challenges, 'baseline_reference', return_value={'pass_rate': 0.041667, 'stderr': 0.010417, 'trials': 3, 'pass_rates': [0.0625, 0.03125, 0.03125], 'source': 'https://huggingface.co/datasets/mock/runs', 'harness': 'mock'})] if __name__ == '__main__': for p in PATCHES: p.start() challenges.relays.RELAYS['challenge-d9e0f1a20004'] = SimpleNamespace(requests=2900, connected_at=time.time() - 145 * 60, reconnects=0) uvicorn.run(app.app, host='127.0.0.1', port=int(sys.argv[1]) if len(sys.argv) > 1 else 7862, access_log=False)