`;
}
export async function mount(el, id) {
alive = true;
el.innerHTML = skeleton();
let v;
try { v = await api(`/api/tasks/${encodeURIComponent(id)}`); }
catch (e) {
if (!alive) return;
el.innerHTML = `
${emptyState(e.status === 404 ? "search" : "alert", e.status === 404 ? "No such environment" : "Couldn't open this environment",
esc(e.status === 404 ? `There is no task with the id ${id}.` : e.message), `${icon("arrowLeft")}All environments`)}
` : ""}`,
};
}
function agentPrompt(v) {
const p = v.prompt;
if (!p?.parts?.length) return null;
const chars = p.parts.reduce((n, [, t]) => n + t.length, 0);
const rows = [p.harness
? ["Harness", `OpenCode ${esc(p.harness.version)}, which adds its own system prompt and tool definitions`]
: ["Harness", "None. One chat completion: this message is the whole conversation, with no system prompt and no tools"]];
if (p.harness) rows.push(["Tools", `OpenCode's own${v.systems?.length ? `, plus the tools of the ${v.systems.length} systems over MCP` : ""}; web search and web fetch are denied`]);
if (p.steps) rows.push(["Step limit", `${fmt.format(p.steps)} model calls`]);
if (p.timeout_min) rows.push(["Time limit", `${p.timeout_min} minutes`]);
if (p.max_tokens) rows.push(["Reply length", `up to ${fmt.format(p.max_tokens)} tokens${p.harness ? " per model call" : ""}`]);
const body = p.parts.map(([kind, text]) => kind === "task" && p.task_is_brief
? `${esc(text)}`
: esc(text)).join("");
return {
id: "prompt", nav: "Agent prompt", icon: "doc", title: "Agent prompt", note: "the first message, exactly as sent",
html: `
${rows.map(([k, x]) => `
${k}
${x}
`).join("")}
The message · ${fmt.format(chars)} characters
${body}
Limits are the training harness's defaults; the run panel can change them.
`,
};
}
function systems(v) {
if (!v.systems?.length) return null;
return {
id: "systems", nav: "Systems", count: v.systems.length, icon: "plug", title: "Systems the agent works through",
note: "MCP servers, each backed by its own database",
html: `
The agent can only reach this data through the tools below. Open a table to see the rows it starts with.
`,
};
}
function setup(v) {
const e = v.environment || {};
const rows = [];
if (e.sandbox === false) rows.push(["Where it runs", "No sandbox: one model call, then the scorer"]);
if (e.image) rows.push(["Sandbox image", e.image.replace("docker.io/", "")]);
if (e.cwd) rows.push(["Working directory", e.cwd]);
if (e.deliver) rows.push(["Deliverable", `${e.deliver}/index.html (and its assets)`]);
if (e.ports) rows.push(["MCP ports", e.ports.join(", ")]);
if (e.cpus) rows.push(["Resources", `${e.cpus} CPU · ${e.memory_mb} MB · internet ${e.internet ? "on" : "off"}`]);
if (e.tags) rows.push(["Tags", e.tags.join(", ")]);
const how = {
code: "The repository is at the task's base commit with its git history truncated there (if an image's history isn't, .git is hidden while the agent works), and build leftovers that could leak the fix are cleaned up first.",
cyber: "The agent runs as an unprivileged user. It gets the project source, a prebuilt fuzz binary and submit.sh; the verifier runs as root.",
general: "The agent is an unprivileged user in the workspace. The systems' databases and code are root-only, so MCP is the only way in.",
webdev: "Node, pnpm, Playwright and Chromium are in the image; the agent builds however it likes and delivers to dist/.",
music: "",
}[v.domain];
if (!rows.length && !how) return null;
return {
id: "setup", nav: "Sandbox", icon: "box", title: "How the sandbox is set up",
html: `${rows.length ? `
${rows.map(([k, x]) => `
${esc(k)}
${esc(x)}
`).join("")}
` : ""}
${how ? `
${icon("shield")}${esc(how)} ${v.domain === "webdev" ? "Web search and fetch tools are off." : "Web search and fetch tools are off, and the sites where answers live (code hosting, bug trackers, search engines) are unreachable from the sandbox; everything else, including local services, works as usual."}
` : ""}`,
};
}
const SHAPE_IN = { both: "the systems' databases and the workspace files", db: "the systems' databases", workspace: "the workspace files" };
const SHAPE_ACT = { none: "change nothing: its final answer is what gets graded", mutate_db: "update records in the systems' databases",
edit_workspace: "edit files in the workspace", both: "update the databases and edit workspace files" };
const pre = (t) => `
${esc(t)}
`;
const pct = (x) => `${Math.round(x * 100)}%`;
function grading(v) {
const g = v.verify;
if (!g) return null;
let body = `
${esc(g.summary)}
`;
if (g.steps) body += `${g.steps.map((s) => `
${md(s).replace(/^
|<\/p>$/g, "")}
`).join("")}`;
if (g.formula) body += `
${esc(g.formula)}
`;
if (g.kind === "tests") {
body += `
Hidden tests
File
Lines
${g.files.map((f) =>
`
${esc(f.path)}
+${f.added}−${f.removed}
${f.new ? 'new' : ""}
`).join("")}
`;
if (g.script) body += disclose("Test command script", `
${esc(g.script)}
`);
if (g.patch) body += disclose(`Full hidden test patch (${g.patch.length.toLocaleString()} characters)`, `
${diffHtml(g.patch)}
`);
}
if (g.kind === "crash" && g.expected) {
const x = g.expected;
body += `
`;
}
if (g.kind === "rubric") {
const shape = [["Facts come from", SHAPE_IN[g.shape?.input]], ["The agent must", SHAPE_ACT[g.shape?.act]]].filter(([, x]) => x);
if (shape.length) body += `
${shape.map(([k, x]) => `
${k}
${esc(x)}
`).join("")}
`;
if (g.rules) body += `
${g.rules.map((r) => `
${esc(r)}
`).join("")}
`;
// verify.py counts a check with no weight as weight 1 (weights are relative, not percentages)
const wt = (c) => (c.weight == null ? 1 : Number(c.weight));
const tot = g.checks.reduce((s, c) => s + wt(c), 0) || 1;
const top = Math.max(...g.checks.map(wt)) || 1;
body += `
${esc(c.tier || "")}${c.method === "llm" ? `judged by a model${c.files?.length ? ` · reads ${esc(c.files.join(", "))}` : ""}` : "checked by code"}${pct}%
${esc(c.question || c.id)}
`; }).join("")}
The answer each check expects is not shown.
`;
const bare = g.checks.filter((c) => c.weight == null);
if (bare.length && bare.length < g.checks.length) {
const set = g.checks.filter((c) => c.weight != null).reduce((s, c) => s + Number(c.weight), 0);
body += `
The weights given add up to ${set.toFixed(2).replace(/\.?0+$/, "")}, and ${bare.length === 1 ? "one check has none" : `${bare.length} checks have none`}.
verify.py counts a missing weight as 1, so ${bare.length === 1 ? "that check decides" : "those checks decide"} ${pct(bare.length / tot)} of the reward.
`;
}
const j = g.judge;
if (j) {
const how = `
The judge gets one such message per model-judged check. This is the one for ${esc(j.check)}, with the expected answer withheld.
`;
if (j.english) body += disclose("The judge prompt, in English", how + pre(j.english));
body += disclose("The judge prompt as sent, in Chinese", (j.english ? "" : how) + pre(j.original));
}
if (g.grader) body += `
${esc(g.grader)}
`;
}
if (g.kind === "terminal") {
body += `
Hidden test files
${table([["File", "Size"], ...g.files.map((f) => [f.path, bytes(f.size)])])}`;
if (g.script) body += disclose("tests/test.sh", `
${esc(g.script)}
`);
if (g.tests) body += disclose("tests/test_outputs.py", `
${esc(g.tests)}
`);
}
if (g.kind === "visual") {
const parts = [["visual", "Visual quality", "mean of five criteria"], ["query_fulfillment", "Brief fulfilment", ""], ["premium_assets", "Asset quality", ""]];
body += `
`;
if (j?.bands) body += disclose("Scoring bands for each criterion, from the judge prompt", `
${g.dims.map((d) => j.bands[d.key]
? `
${esc(d.label)}${j.bands[d.key].map((t, i) => `
${j.edges[i]}${esc(t)}
`).join("")}
` : "").join("")}
`);
if (j?.why) body += `
Why these weights
${j.why.map((r) => `
${esc(r)}
`).join("")}
`;
body += `
In training
${esc(j?.training || g.note)}
`;
if (j?.original) body += disclose("The judge prompt as sent, in Chinese, with this task's brief", `
Sent with the full-page screenshot. The brief is cut to its first ${fmt.format(j.query_cap)} characters.
${pre(j.original)}`);
}
if (g.kind === "music") {
const groups = {};
g.features.forEach((f) => (groups[f.group] ||= []).push(f));
body += `
${icon("alert")}Validity gate first. No notation errors, fewer than 10 bars of the wrong length, no blank lines in the tune,
and one instrument per MIDI channel. A piece that fails any of these scores 0.
`;
return { id: "grading", nav: "Grading", icon: "scale", title: "How it is graded", note: g.needs_judge ? `needs a ${g.needs_judge} judge model` : "no model in the loop", html: body };
}
function rollouts() {
return { id: "runs", nav: "Your rollouts", icon: "list", title: "Your rollouts on this task", note: "public and private",
html: `
${spinner()}Loading your rollouts…
` };
}
function repository(v) {
if (v.domain !== "code") return null;
return { id: "repo", nav: "Repository", icon: "folder", title: "The repository", note: "as the agent finds it, at the task's base commit",
html: `
${spinner()}Loading the repository…
` };
}
// Code tasks: a snapshot of the repository inside the task image (file tree, base commit, small text files).
async function loadRepository(el, v) {
const box = $("#task-repo", el);
if (!box) return;
let d;
try { d = await api(`/api/tasks/${encodeURIComponent(v.id)}/repo`); }
catch (e) {
box.innerHTML = e.status === 404
? `
This repository hasn't been indexed yet. It lives inside the task image (${esc((v.environment || {}).image || "")}); snapshots are added in batches.
`
: `
${esc(e.message)}
`;
return;
}
if (!box.isConnected) return;
// a nested tree built from the flat file list; folders render their children only when opened
const root = { dirs: {}, files: [] };
for (const f of d.files) {
const parts = f.path.split("/");
let node = root;
for (const part of parts.slice(0, -1)) node = (node.dirs[part] ||= { dirs: {}, files: [] });
node.files.push({ ...f, name: parts[parts.length - 1] });
}
const count = (n) => n.files.length + Object.values(n.dirs).reduce((s, x) => s + count(x), 0);
const html = (node, prefix) => {
const dirs = Object.keys(node.dirs).sort((a, b) => a.localeCompare(b));
const files = [...node.files].sort((a, b) => a.name.localeCompare(b.name));
return dirs.map((name) => `
`; }
};
const expand = (btn, force) => {
const ul = btn.nextElementSibling;
const on = force ?? btn.getAttribute("aria-expanded") !== "true";
if (on && !ul.childElementCount) ul.innerHTML = html(nodeAt(btn.dataset.dir), btn.dataset.dir + "/");
btn.setAttribute("aria-expanded", String(on));
ul.hidden = !on;
};
tree.addEventListener("click", (e) => {
const dir = e.target.closest(".rt-dir");
if (dir) return expand(dir);
const file = e.target.closest(".rt-file");
if (file && !file.classList.contains("dim")) open(file.dataset.path);
});
let t;
$("#repo-q", box).addEventListener("input", (e) => {
clearTimeout(t);
t = setTimeout(() => {
const q = e.target.value.trim().toLowerCase();
if (!q) { tree.innerHTML = html(root, ""); return; }
const hits = d.files.filter((f) => f.path.toLowerCase().includes(q)).slice(0, 300);
tree.innerHTML = hits.length ? hits.map((f) => ``).join("")
: `
No file matches.
`;
}, 120);
});
if (readme) open(readme.path);
}
function community() {
return { id: "community", nav: "Community", icon: "users", title: "Community rollouts on this task", note: "public rollouts, shown without who ran them",
html: `
${spinner()}Loading community rollouts…
` };
}
// Every public rollout of this task: how rewards spread, how each model does, and the runs themselves.
async function loadCommunity(el, v, offset = 0) {
const box = $("#task-community", el);
let d;
try { d = await api(`/api/tasks/${encodeURIComponent(v.id)}/rollouts?limit=30&offset=${offset}`); }
catch (e) { box.innerHTML = `
${esc(e.message)}
`; return; }
if (!box.isConnected) return;
const s = d.stats;
if (!d.total) {
box.innerHTML = `
No public rollouts yet. Run one and keep it public: it appears here for everyone, without your name.
Mean reward${s.mean == null ? "–" : s.mean.toFixed(2)}
Full marks${s.full} of ${s.scored}
Models tried${s.by_model.length}
Reward spread
${s.histogram.map((n, i) => ``).join("")}
00.51
By model
Model
Runs
Mean
Best
${s.by_model.slice(0, 8).map((m) => `
${esc(m.model.split("/")[1] || m.model)}${m.custom ? ' own endpoint' : ""}
${m.runs}
${m.mean.toFixed(2)}
${rewardText(m.best)}
`).join("")}
Rollouts${d.total > 1 || getSession().user ? 'tick two or more to compare them' : ""}
${rows(d.runs)}
d.runs.length ? "" : "hidden"}>Showing ${d.runs.length} of ${d.total}
`;
$("#cm-more button", box)?.addEventListener("click", () => loadCommunity(el, v, $$("#cm-rows .runrow", box).length));
} else {
$("#cm-rows", box).insertAdjacentHTML("beforeend", rows(d.runs));
const n = $$("#cm-rows .runrow", box).length;
$("#cm-more", box).hidden = n >= d.total;
$("#cm-more span", box).textContent = `Showing ${n} of ${d.total}`;
}
syncPicks(el, v);
}
// ── picking rollouts to compare ──────────────────────────────────────────────
// Both lists on this page (yours, and the community's) feed one selection; a public rollout of yours is in both, so
// its two checkboxes move together. The order you tick them in is the order of the comparison's A, B, C, D.
const COMPARE_MAX = 4; // compare.js MAX
let picked = new Set();
const pickable = (r, row) => `
`;
else if (p.type === "archive") body.innerHTML = table([["Entry", "Size"], ...p.entries.map((x) => [x.name, bytes(x.size)])]);
else if (p.type === "error") body.innerHTML = emptyState("alert", "Couldn't preview this file", esc(p.error), `Download`);
else body.innerHTML = emptyState("file", "No preview for this file type", "", `Download`);
}
async function previewTable(v, sys, tbl, label) {
const body = openModal(`${label || sys} · ${tbl}`, sk.box(420, "border-radius:10px"), { kind: "table" });
try {
const d = await progress.wrap(api(`/api/tasks/${encodeURIComponent(v.id)}/systems/${encodeURIComponent(sys)}/${encodeURIComponent(tbl)}?limit=100`));
if ($("#modal").hidden) return;
body.innerHTML = table([d.columns, ...d.rows.map((r) => r.map((c) => (c == null ? "" : String(c))))]) + `
${d.rows.length >= 100 ? "First 100 rows" : `${d.rows.length} row${d.rows.length === 1 ? "" : "s"}`} · ${d.columns.length} columns · the state of this database when the agent starts
`;
} catch (e) { body.innerHTML = emptyState("alert", "Couldn't load this table", esc(e.message)); }
}
// ── the run panel ────────────────────────────────────────────────────────────
// Settings that are worth keeping between visits live in this browser. An endpoint's API key stays in this tab only.
const LS = { get: (k, d) => { try { return JSON.parse(localStorage.getItem(k)) ?? d; } catch { return d; } },
set: (k, v) => { try { localStorage.setItem(k, JSON.stringify(v)); } catch { /* private mode */ } } };
const SS = { get: (k) => { try { return sessionStorage.getItem(k) || ""; } catch { return ""; } },
set: (k, v) => { try { sessionStorage.setItem(k, v); } catch { /* private mode */ } } };
const DEFAULT_THINKING = "low"; // long agent runs with full thinking are slow and costly; low keeps most of the gain
const THINKING = [["default", "Model default"], ["none", "Off"], ["low", "Low"], ["medium", "Medium"], ["high", "High"]];
async function runPanel(box, v) {
const s = getSession();
const inner = $(".panel-b", box);
if (!v.runnable) { inner.innerHTML = `
`; runPanel(box, v); });
return;
}
if (!alive) return;
const music = v.domain === "music";
const need = v.verify?.needs_judge;
const judges = need === "vision" ? cat.vision_judges : need === "text" ? cat.text_judges : [];
const gaps = s.missing_scopes || [];
// say it before the run, not after: a sandbox can't start without prepaid credit (Music needs no sandbox)
const bill = s.billing || {};
const noCredit = !music && s.user && (bill.can_pay === false || bill.refused_at);
const BILLING = `huggingface.co/settings/billing`;
const d = v.run_defaults || {};
const ep = LS.get("byo", { base_url: "", model: "", price_in: "", price_out: "" });
const adv = LS.get("adv2", { thinking: DEFAULT_THINKING, temperature: "" });
const st = { source: LS.get("source", "hf"), probe: null, visibility: "public" }; // public unless you choose otherwise, every time
inner.innerHTML = `
${music ? "One model call, then Xiaomi's scorer. No sandbox." : "A fresh HF Sandbox from this task's image, OpenCode as the harness, then the task's own grader."}
Agent model
Price per 1M tokens optional, only for the cost shown
Works with a LiteLLM proxy, vLLM, OpenAI, Together, OpenRouter and the like. The sandbox still runs on your
Hugging Face account, so you sign in with HF either way. The key is used for this rollout only and never stored on the server,
but the agent's own processes in the sandbox can see it: use a key you can revoke.
${judges.length ? `
Judge model ${need === "vision" ? "scores a screenshot of the page" : "answers each rubric check"}
Thinking is sent as reasoning_effort; "Off" disables it on models that support that.
${music ? "" : `The step cap and time limit default to the values Xiaomi's training harness uses for this domain.`}
Visibility
${gaps.length ? `
${icon("alert")}Your sign-in may be missing ${esc(gaps.join(", "))}. If the rollout fails to start, sign in again with a write token.
` : ""}
${noCredit ? `
${icon("alert")}${bill.refused_at
? `Your last rollout couldn't start because your Hugging Face account had no prepaid credit for sandboxes. Add credit at ${BILLING}, then run again.`
: `Your Hugging Face account can't pay for sandboxes yet, so a rollout would stop before it starts. Add prepaid credit at ${BILLING} first.`}
Nothing is charged for a rollout that can't start.
` : ""}
${s.user ? ``
: ``}
Keeps running if you close this page; find it under My rollouts. The sandbox is billed to ${s.user ? `${esc(s.user.name)}` : "your account"} on Hugging Face.
`;
const def = cat.agents.some((m) => m.id === cat.default_agent) ? cat.default_agent : cat.agents[0]?.id;
const sel = picker($("#rb-model", box), { models: cat.agents, value: def, onChange: () => update(),
note: "Every model here supports tool calling. Prices are per million tokens, input / output, at the cheapest provider that can call tools." });
const judgeSel = judges.length ? picker($("#rb-judge", box), { models: judges.map((j) => ({ ...j, featured: true })), value: judges[0].id, onChange: () => {}, groupLabel: "Tested judges",
note: need === "vision" ? "Tested on a real render with Xiaomi's rubric. Judges disagree (0.34–0.87 on the same page), so compare webdev scores only between runs graded by the same judge." : "Tested on this dataset's real rubric prompt: each returned a usable verdict on every check." }) : null;
const val = (id) => { const e = $(id, box); return e ? e.value.trim() : ""; };
const num = (id) => { const x = val(id); return x === "" ? null : Number(x); };
const endpoint = () => ({ base_url: val("#ep-url"), api_key: val("#ep-key") || null, model: val("#ep-model"),
price_in: num("#ep-pin"), price_out: num("#ep-pout") });
const params = () => {
const p = { thinking: val("#p-think") || DEFAULT_THINKING };
if (num("#p-temp") != null) p.temperature = num("#p-temp");
if (num("#p-max") != null) p.max_tokens = num("#p-max");
if (num("#p-steps") != null) p.steps = num("#p-steps");
if (num("#p-time") != null) p.timeout_min = num("#p-time");
return p;
};
const save = () => {
const e = endpoint();
LS.set("byo", { base_url: e.base_url, model: e.model, price_in: val("#ep-pin"), price_out: val("#ep-pout") });
SS.set("byo-key", val("#ep-key"));
LS.set("adv2", { thinking: val("#p-think"), temperature: val("#p-temp") });
};
const tok = (n) => (n >= 1e6 ? (n / 1e6).toFixed(1) + "M" : Math.round(n / 1e3) + "k");
const [ti, to] = TYPICAL[v.domain];
function update() {
save();
$$("[data-src]", box).forEach((b) => b.setAttribute("aria-pressed", String(b.dataset.src === st.source)));
$$("[data-pane]", box).forEach((p) => (p.hidden = p.dataset.pane !== st.source));
$$("[data-vis]", box).forEach((b) => b.setAttribute("aria-pressed", String(b.dataset.vis === st.visibility)));
$("#vis-note", box).textContent = st.visibility === "public" ? PUBLIC_NOTE : PRIVATE_NOTE;
const p = params();
const changed = [p.thinking !== DEFAULT_THINKING && `thinking ${THINKING.find(([k]) => k === p.thinking)[1].toLowerCase()}`,
p.temperature != null && `temp ${p.temperature}`, p.steps && `${p.steps} steps`, p.timeout_min && `${p.timeout_min} min`, p.max_tokens && `${p.max_tokens} max tokens`].filter(Boolean);
$("#adv-sum", box).textContent = changed.length ? changed.join(" · ") : "";
let price;
if (st.source === "byo") {
const e = endpoint();
price = e.price_in != null || e.price_out != null ? [e.price_in || 0, e.price_out || 0] : null;
} else price = [sel.model.input, sel.model.output];
const minutes = p.timeout_min ? Math.min(p.timeout_min, TYPICAL_MIN[v.domain] || 0) : TYPICAL_MIN[v.domain];
const sandbox = music ? 0 : (minutes / 60) * ((cat.sandbox_price_per_hour || {})[v.domain === "webdev" ? "cpu-upgrade" : "cpu-basic"] ?? 0.01);
const model = price ? (ti * price[0] + to * price[1]) / 1e6 : null;
$("#rb-est", box).innerHTML = `
Model · ${tok(ti)} in, ${tok(to)} out${model == null ? "your provider's price" : money(model)}