// Task page: one scrolling page built from sections, an outline to jump between them, and a run panel. // Each section builder returns null when the task has nothing for it, so every domain gets only what applies. import { $, $$, api, esc, fmt, md, money, ago, bytes, openModal, table, sheetGrid, statusPill, rewardBadge, toast, DOMAIN_NAME, vals, sk, spinner, emptyState, progress, confetti, PUBLIC_NOTE, PRIVATE_NOTE, reportUrl, visibilityBadge, rewardClass, rewardText } from "./util.js"; import { icon, DOMAIN_ICON, FILE_ICON } from "./icons.js"; import { getSession, refreshActive } from "./session.js"; import { picker } from "./picker.js"; let modelsP = null; const getModels = () => (modelsP ||= api("/api/models").catch((e) => { modelsP = null; throw e; })); // Tokens (input, output) of the test rollouts per domain. Agents re-send their whole context every step, so // input dominates: a General task read ~1.2M input tokens against ~20k written. const TYPICAL = { code: [100e3, 3e3], cyber: [1.2e6, 30e3], general: [1.2e6, 20e3], webdev: [60e3, 20e3], music: [500, 6e3] }; const TYPICAL_MIN = { code: 8, cyber: 12, general: 15, webdev: 10, music: 0 }; const KIND_LABEL = { spreadsheet: "Excel", document: "Word", pdf: "PDF", slides: "PowerPoint", image: "image", web: "web page", text: "text", archive: "archive" }; const KIND_NOUN = { spreadsheet: "spreadsheet", document: "document", pdf: "PDF", slides: "slide deck", image: "image", web: "web page", text: "text file", archive: "archive", other: "file" }; const LEFTOVER = /^(None named|Unspecified|Unknown|Unrated|Other)$/; let spy = null, alive = false; export function unmount() { alive = false; if (spy) spy.disconnect(); spy = null; } function skeleton() { return `
${sk.line(22)}
${sk.line(24, 22)}${sk.line(62, 28)}${sk.line(40)}
${[5, 3, 4].map((n) => `
${sk.line(24, 14)}
${sk.lines(...[100, 96, 88, 92, 70].slice(0, n))}
`).join("")}
`; } export async function mount(el, id) { alive = true; el.innerHTML = skeleton(); let v; try { v = await api(`/api/tasks/${encodeURIComponent(id)}`); } catch (e) { if (!alive) return; el.innerHTML = `
${emptyState(e.status === 404 ? "search" : "alert", e.status === 404 ? "No such environment" : "Couldn't open this environment", esc(e.status === 404 ? `There is no task with the id ${id}.` : e.message), `${icon("arrowLeft")}All environments`)}
`; return; } if (!alive) return; getModels().catch(() => {}); // start fetching models while the page renders const sections = [brief, agentPrompt, systems, workspace, repository, setup, grading, community, rollouts].map((f) => f(v)).filter(Boolean); el.innerHTML = `
${header(v)}
${sections.map((s) => `

${icon(s.icon, 15)}${esc(s.title)}

${s.note ? `${esc(s.note)}` : ""}
${s.html}
`).join("")}
`; picked = new Set(); wire(el, v); runPanel($("#runbox", el), v); loadRollouts(el, v); loadCommunity(el, v); loadRepository(el, v); } function header(v) { const vf = v.verify || {}; const chips = Object.values(v.facets || {}).flatMap(vals).filter((x) => !LEFTOVER.test(x)).slice(0, 6); const facts = []; if (v.systems) facts.push(["plug", `${v.systems.length} systems`], ["terminal", `${v.systems.reduce((n, s) => n + s.tools.length, 0)} tools`], ["folder", `${v.files.length} files`]); if (vf.checks && vf.kind === "rubric") facts.push(["check", `${vf.checks.length} rubric checks`]); if (vf.kind === "tests" || vf.kind === "terminal") facts.push(["flask", `${vf.files.length} hidden test files`]); if (vf.kind === "crash" && vf.expected?.function) facts.push(["target", `crash in ${esc(vf.expected.function)}`]); if (vf.kind === "visual") facts.push(["image", "graded from a screenshot"]); if (vf.kind === "music") facts.push(["gauge", "scored by code, no model"]); if (vf.needs_judge) facts.push(["scale", `needs a ${vf.needs_judge} judge`]); return `
${icon(DOMAIN_ICON[v.domain], 13)}${esc(DOMAIN_NAME[v.domain])}${chips.map((c) => `${esc(c)}`).join("")}

${esc(v.title)}

${facts.map(([i, f]) => `${icon(i, 14)}${f}`).join("")} ${icon("flag", 14)}Report an issue with this task
`; } // ── sections ───────────────────────────────────────────────────────────────── function brief(v) { const vf = v.verify || {}; let spec = ""; if (vf.kind === "music" && vf.spec) { const s = vf.spec; spec = `
${[["Style", s.style], ["Tempo", s.bpm && `${s.bpm} BPM`], ["Meter", s.meter], ["Length", s.bars && `${s.bars} bars`], ["Voices", s.voices]] .filter(([, x]) => x).map(([k, x]) => `
${k}${esc(x)}
`).join("")}
`; } const meta = (v.meta || []).filter(([, x]) => x && String(x).length < 200 && !/^(none named|unspecified|unknown|n\/a|none)$/i.test(String(x).trim())); return { id: "brief", nav: "The task", icon: "message", title: "The task", note: v.prompt?.task_is_brief ? "word for word, inside the agent prompt below" : "", html: `${spec}
${md(v.brief)}
${meta.length ? `
${meta.map(([k, x]) => `
${esc(k)}
${esc(x)}
`).join("")}
` : ""} ${v.links?.length ? `` : ""}`, }; } function agentPrompt(v) { const p = v.prompt; if (!p?.parts?.length) return null; const chars = p.parts.reduce((n, [, t]) => n + t.length, 0); const rows = [p.harness ? ["Harness", `OpenCode ${esc(p.harness.version)}, which adds its own system prompt and tool definitions`] : ["Harness", "None. One chat completion: this message is the whole conversation, with no system prompt and no tools"]]; if (p.harness) rows.push(["Tools", `OpenCode's own${v.systems?.length ? `, plus the tools of the ${v.systems.length} systems over MCP` : ""}; web search and web fetch are denied`]); if (p.steps) rows.push(["Step limit", `${fmt.format(p.steps)} model calls`]); if (p.timeout_min) rows.push(["Time limit", `${p.timeout_min} minutes`]); if (p.max_tokens) rows.push(["Reply length", `up to ${fmt.format(p.max_tokens)} tokens${p.harness ? " per model call" : ""}`]); const body = p.parts.map(([kind, text]) => kind === "task" && p.task_is_brief ? `` : esc(text)).join(""); return { id: "prompt", nav: "Agent prompt", icon: "doc", title: "Agent prompt", note: "the first message, exactly as sent", html: `
${rows.map(([k, x]) => `
${k}
${x}
`).join("")}
The message · ${fmt.format(chars)} characters
${body}

Limits are the training harness's defaults; the run panel can change them.

`, }; } function systems(v) { if (!v.systems?.length) return null; return { id: "systems", nav: "Systems", count: v.systems.length, icon: "plug", title: "Systems the agent works through", note: "MCP servers, each backed by its own database", html: `

The agent can only reach this data through the tools below. Open a table to see the rows it starts with.

${v.systems.map((s, i) => `
${icon("chevronRight", 14, "chev")}${esc(s.label)}${s.tools.length} tools · ${s.tables.length} tables · ${s.tables.reduce((n, t) => n + t.rows, 0).toLocaleString()} rows
Tools
    ${s.tools.map((t) => `
  • ${esc(t.name)}(${t.params.map((p) => esc(p.name) + (p.default != null ? "?" : "")).join(", ")}) ${t.doc ? `

    ${esc(t.doc)}

    ` : ""}
  • `).join("")}
Data
${s.tables.map((t) => ``).join("")}
`).join("")}
`, }; } function workspace(v) { if (!v.files?.length) return null; const kinds = {}; v.files.forEach((f) => (kinds[f.kind] = (kinds[f.kind] || 0) + 1)); return { id: "files", nav: "Workspace", count: v.files.length, icon: "folder", title: "Workspace files", note: Object.entries(kinds).map(([k, n]) => `${n} ${KIND_NOUN[k] || k}${n > 1 ? "s" : ""}`).join(" · "), html: `
${v.files.map((f) => ``).join("")}
`, }; } function setup(v) { const e = v.environment || {}; const rows = []; if (e.sandbox === false) rows.push(["Where it runs", "No sandbox: one model call, then the scorer"]); if (e.image) rows.push(["Sandbox image", e.image.replace("docker.io/", "")]); if (e.cwd) rows.push(["Working directory", e.cwd]); if (e.deliver) rows.push(["Deliverable", `${e.deliver}/index.html (and its assets)`]); if (e.ports) rows.push(["MCP ports", e.ports.join(", ")]); if (e.cpus) rows.push(["Resources", `${e.cpus} CPU · ${e.memory_mb} MB · internet ${e.internet ? "on" : "off"}`]); if (e.tags) rows.push(["Tags", e.tags.join(", ")]); const how = { code: "The repository is at the task's base commit with its git history truncated there (if an image's history isn't, .git is hidden while the agent works), and build leftovers that could leak the fix are cleaned up first.", cyber: "The agent runs as an unprivileged user. It gets the project source, a prebuilt fuzz binary and submit.sh; the verifier runs as root.", general: "The agent is an unprivileged user in the workspace. The systems' databases and code are root-only, so MCP is the only way in.", webdev: "Node, pnpm, Playwright and Chromium are in the image; the agent builds however it likes and delivers to dist/.", music: "", }[v.domain]; if (!rows.length && !how) return null; return { id: "setup", nav: "Sandbox", icon: "box", title: "How the sandbox is set up", html: `${rows.length ? `
${rows.map(([k, x]) => `
${esc(k)}
${esc(x)}
`).join("")}
` : ""} ${how ? `
${icon("shield")}${esc(how)} ${v.domain === "webdev" ? "Web search and fetch tools are off." : "Web search and fetch tools are off, and the sites where answers live (code hosting, bug trackers, search engines) are unreachable from the sandbox; everything else, including local services, works as usual."}
` : ""}`, }; } const SHAPE_IN = { both: "the systems' databases and the workspace files", db: "the systems' databases", workspace: "the workspace files" }; const SHAPE_ACT = { none: "change nothing: its final answer is what gets graded", mutate_db: "update records in the systems' databases", edit_workspace: "edit files in the workspace", both: "update the databases and edit workspace files" }; const pre = (t) => `
${esc(t)}
`; const pct = (x) => `${Math.round(x * 100)}%`; function grading(v) { const g = v.verify; if (!g) return null; let body = `

${esc(g.summary)}

`; if (g.steps) body += `
    ${g.steps.map((s) => `
  1. ${md(s).replace(/^

    |<\/p>$/g, "")}

  2. `).join("")}
`; if (g.formula) body += `

${esc(g.formula)}

`; if (g.kind === "tests") { body += `
Hidden tests
${g.files.map((f) => ``).join("")}
FileLines
${esc(f.path)}+${f.added} −${f.removed}${f.new ? 'new' : ""}
`; if (g.script) body += disclose("Test command script", `
${esc(g.script)}
`); if (g.patch) body += disclose(`Full hidden test patch (${g.patch.length.toLocaleString()} characters)`, `
${diffHtml(g.patch)}
`); } if (g.kind === "crash" && g.expected) { const x = g.expected; body += `
${[["Sanitizer", esc(x.sanitizer)], ["Bug class", esc(x.error_type)], ["Function", `${esc(x.function)}`], ["File", `${esc(x.file)}`]] .map(([k, val]) => `
${k}${val}
`).join("")}
`; } if (g.kind === "rubric") { const shape = [["Facts come from", SHAPE_IN[g.shape?.input]], ["The agent must", SHAPE_ACT[g.shape?.act]]].filter(([, x]) => x); if (shape.length) body += `
${shape.map(([k, x]) => `
${k}
${esc(x)}
`).join("")}
`; if (g.rules) body += ``; // verify.py counts a check with no weight as weight 1 (weights are relative, not percentages) const wt = (c) => (c.weight == null ? 1 : Number(c.weight)); const tot = g.checks.reduce((s, c) => s + wt(c), 0) || 1; const top = Math.max(...g.checks.map(wt)) || 1; body += `
${g.checks.length} checks, weighted
    ${g.checks.map((c) => { const pct = Math.round((wt(c) / tot) * 100); return `
  1. ${esc(c.tier || "")} ${c.method === "llm" ? `judged by a model${c.files?.length ? ` · reads ${esc(c.files.join(", "))}` : ""}` : "checked by code"} ${pct}%

    ${esc(c.question || c.id)}

  2. `; }).join("")}

The answer each check expects is not shown.

`; const bare = g.checks.filter((c) => c.weight == null); if (bare.length && bare.length < g.checks.length) { const set = g.checks.filter((c) => c.weight != null).reduce((s, c) => s + Number(c.weight), 0); body += `

The weights given add up to ${set.toFixed(2).replace(/\.?0+$/, "")}, and ${bare.length === 1 ? "one check has none" : `${bare.length} checks have none`}. verify.py counts a missing weight as 1, so ${bare.length === 1 ? "that check decides" : "those checks decide"} ${pct(bare.length / tot)} of the reward.

`; } const j = g.judge; if (j) { const how = `

The judge gets one such message per model-judged check. This is the one for ${esc(j.check)}, with the expected answer withheld.

`; if (j.english) body += disclose("The judge prompt, in English", how + pre(j.english)); body += disclose("The judge prompt as sent, in Chinese", (j.english ? "" : how) + pre(j.original)); } if (g.grader) body += `

${esc(g.grader)}

`; } if (g.kind === "terminal") { body += `
Hidden test files
${table([["File", "Size"], ...g.files.map((f) => [f.path, bytes(f.size)])])}`; if (g.script) body += disclose("tests/test.sh", `
${esc(g.script)}
`); if (g.tests) body += disclose("tests/test_outputs.py", `
${esc(g.tests)}
`); } if (g.kind === "visual") { const parts = [["visual", "Visual quality", "mean of five criteria"], ["query_fulfillment", "Brief fulfilment", ""], ["premium_assets", "Asset quality", ""]]; body += `
${parts.map(([key, name, how]) => { const ds = g.dims.filter((d) => d.group === key); return ds.length ? `
${name}⅓ of the score${how ? ` · ${how}` : ""}
${ds.map((d) => `
${esc(d.label)}
${esc(d.desc)}
`).join("")}
` : ""; }).join("")}
`; const j = g.judge; if (j?.rules) body += `
The judge's instructions
`; if (j?.bands) body += disclose("Scoring bands for each criterion, from the judge prompt", `
${g.dims.map((d) => j.bands[d.key] ? `
${esc(d.label)}
    ${j.bands[d.key].map((t, i) => `
  1. ${j.edges[i]}${esc(t)}
  2. `).join("")}
` : "").join("")}
`); if (j?.why) body += `
Why these weights
`; body += `
In training

${esc(j?.training || g.note)}

`; if (j?.original) body += disclose("The judge prompt as sent, in Chinese, with this task's brief", `

Sent with the full-page screenshot. The brief is cut to its first ${fmt.format(j.query_cap)} characters.

${pre(j.original)}`); } if (g.kind === "music") { const groups = {}; g.features.forEach((f) => (groups[f.group] ||= []).push(f)); body += `
${icon("alert")}Validity gate first. No notation errors, fewer than 10 bars of the wrong length, no blank lines in the tune, and one instrument per MIDI channel. A piece that fails any of these scores 0.
${Object.entries(groups).map(([gname, fs]) => `
${esc(gname)}${g.weights?.[gname] != null ? ` · ${pct(g.weights[gname])}` : ""}
${fs.map((f) => `
${esc(f.label || f.name)}${f.rule === "band" ? "within the human range" : f.rule === "high" ? "the higher the better" : "the lower the better"} · ${esc(f.name)} ${f.band ? `${fmtNum(f.band[0])} – ${fmtNum(f.band[1])}` : ""}
`).join("")}
`).join("")}
`; if (g.curves) body += `
How a feature scores
${[["Within the human range", g.curves.band], ["The higher the better", g.curves.high], ["The lower the better", g.curves.low], ["Histogram similarity", g.histograms]].filter(([, x]) => x).map(([k, x]) => `
${k}
${esc(x)}
`).join("")}

Group weights are the share of the 85% that the six groups make up together.

`; } if (g.not_scored) body += `
Not scored
${esc(g.not_scored)}${/^Never/.test(g.not_scored) ? "" : " A rollout that isn't scored has no reward, rather than 0."}
`; body += `

Every kind of task side by side, and how the rubrics are built across the dataset

`; return { id: "grading", nav: "Grading", icon: "scale", title: "How it is graded", note: g.needs_judge ? `needs a ${g.needs_judge} judge model` : "no model in the loop", html: body }; } function rollouts() { return { id: "runs", nav: "Your rollouts", icon: "list", title: "Your rollouts on this task", note: "public and private", html: `
${spinner()}Loading your rollouts…
` }; } function repository(v) { if (v.domain !== "code") return null; return { id: "repo", nav: "Repository", icon: "folder", title: "The repository", note: "as the agent finds it, at the task's base commit", html: `
${spinner()}Loading the repository…
` }; } // Code tasks: a snapshot of the repository inside the task image (file tree, base commit, small text files). async function loadRepository(el, v) { const box = $("#task-repo", el); if (!box) return; let d; try { d = await api(`/api/tasks/${encodeURIComponent(v.id)}/repo`); } catch (e) { box.innerHTML = e.status === 404 ? `

This repository hasn't been indexed yet. It lives inside the task image (${esc((v.environment || {}).image || "")}); snapshots are added in batches.

` : `

${esc(e.message)}

`; return; } if (!box.isConnected) return; // a nested tree built from the flat file list; folders render their children only when opened const root = { dirs: {}, files: [] }; for (const f of d.files) { const parts = f.path.split("/"); let node = root; for (const part of parts.slice(0, -1)) node = (node.dirs[part] ||= { dirs: {}, files: [] }); node.files.push({ ...f, name: parts[parts.length - 1] }); } const count = (n) => n.files.length + Object.values(n.dirs).reduce((s, x) => s + count(x), 0); const html = (node, prefix) => { const dirs = Object.keys(node.dirs).sort((a, b) => a.localeCompare(b)); const files = [...node.files].sort((a, b) => a.name.localeCompare(b.name)); return dirs.map((name) => `
  • `).join("") + files.map((f) => `
  • `).join(""); }; const nodeAt = (path) => path.split("/").reduce((n, part) => n && n.dirs[part], root); const readme = d.files.find((f) => /^readme(\.[a-z]+)?$/i.test(f.path) && f.preview); box.innerHTML = `
    Files${d.files.length.toLocaleString()}
    Base commit${esc((d.base || "").slice(0, 10))}
    ${d.remote ? `
    Upstream${esc(d.remote.replace(/^https?:\/\//, "").replace(/\.git$/, ""))}
    ` : ""}
    History${d.history_truncated ? "truncated at base" : "not truncated, hidden from the agent"}
    ${icon("file", 26)}

    Pick a file to read it.

    ${esc(d.cwd || "")} in the task image, as the agent finds it. Text files up to 200 KB open here; the hidden tests only arrive at grading.

    `; const tree = $("#rt-tree", box), view = $("#rt-view", box); const open = async (path) => { $$(".rt-file.on", box).forEach((b) => b.classList.remove("on")); box.querySelector(`.rt-file[data-path="${CSS.escape(path)}"]`)?.classList.add("on"); const meta = d.files.find((f) => f.path === path); view.innerHTML = `
    ${esc(path)}${bytes(meta?.size)}
    ${sk.lines(90, 70, 85, 60, 95, 75)}
    `; if (!meta?.preview) { $(".rt-body", view).innerHTML = `

    No preview: this file is binary or larger than 200 KB.

    `; return; } try { const f = await api(`/api/tasks/${encodeURIComponent(v.id)}/repo/file?path=${encodeURIComponent(path)}`); const lines = f.text.split("\n"); $(".rt-body", view).innerHTML = `
    ${lines.map((l, i) => `${i + 1}${esc(l)}`).join("\n")}
    `; } catch (err) { $(".rt-body", view).innerHTML = `

    ${esc(err.message)}

    `; } }; const expand = (btn, force) => { const ul = btn.nextElementSibling; const on = force ?? btn.getAttribute("aria-expanded") !== "true"; if (on && !ul.childElementCount) ul.innerHTML = html(nodeAt(btn.dataset.dir), btn.dataset.dir + "/"); btn.setAttribute("aria-expanded", String(on)); ul.hidden = !on; }; tree.addEventListener("click", (e) => { const dir = e.target.closest(".rt-dir"); if (dir) return expand(dir); const file = e.target.closest(".rt-file"); if (file && !file.classList.contains("dim")) open(file.dataset.path); }); let t; $("#repo-q", box).addEventListener("input", (e) => { clearTimeout(t); t = setTimeout(() => { const q = e.target.value.trim().toLowerCase(); if (!q) { tree.innerHTML = html(root, ""); return; } const hits = d.files.filter((f) => f.path.toLowerCase().includes(q)).slice(0, 300); tree.innerHTML = hits.length ? hits.map((f) => `
  • `).join("") : `
  • No file matches.
  • `; }, 120); }); if (readme) open(readme.path); } function community() { return { id: "community", nav: "Community", icon: "users", title: "Community rollouts on this task", note: "public rollouts, shown without who ran them", html: `
    ${spinner()}Loading community rollouts…
    ` }; } // Every public rollout of this task: how rewards spread, how each model does, and the runs themselves. async function loadCommunity(el, v, offset = 0) { const box = $("#task-community", el); let d; try { d = await api(`/api/tasks/${encodeURIComponent(v.id)}/rollouts?limit=30&offset=${offset}`); } catch (e) { box.innerHTML = `

    ${esc(e.message)}

    `; return; } if (!box.isConnected) return; const s = d.stats; if (!d.total) { box.innerHTML = `

    No public rollouts yet. Run one and keep it public: it appears here for everyone, without your name.

    `; return; } const peak = Math.max(1, ...s.histogram); const rows = (runs) => runs.map((r) => pickable(r, `${rewardBadge(r.reward, r.status)} ${esc(r.endpoint ? r.model : r.model.split("/")[1] || r.model)}${r.endpoint ? ' own endpoint' : ""} ${esc(paramsShort(r.params))}${money((r.cost || {}).total)}${ago(r.created_at)} ${icon("chevronRight", 14)}`)).join(""); if (offset === 0) { box.innerHTML = `
    Public rollouts${s.runs}
    Mean reward${s.mean == null ? "–" : s.mean.toFixed(2)}
    Full marks${s.full} of ${s.scored}
    Models tried${s.by_model.length}
    Reward spread
    ${s.histogram.map((n, i) => ``).join("")}
    00.51
    By model
    ${s.by_model.slice(0, 8).map((m) => ``).join("")}
    ModelRunsMeanBest
    ${esc(m.model.split("/")[1] || m.model)}${m.custom ? ' own endpoint' : ""}${m.runs} ${m.mean.toFixed(2)}${rewardText(m.best)}
    Rollouts${d.total > 1 || getSession().user ? 'tick two or more to compare them' : ""}
    ${rows(d.runs)}
    d.runs.length ? "" : "hidden"}>Showing ${d.runs.length} of ${d.total}
    `; $("#cm-more button", box)?.addEventListener("click", () => loadCommunity(el, v, $$("#cm-rows .runrow", box).length)); } else { $("#cm-rows", box).insertAdjacentHTML("beforeend", rows(d.runs)); const n = $$("#cm-rows .runrow", box).length; $("#cm-more", box).hidden = n >= d.total; $("#cm-more span", box).textContent = `Showing ${n} of ${d.total}`; } syncPicks(el, v); } // ── picking rollouts to compare ────────────────────────────────────────────── // Both lists on this page (yours, and the community's) feed one selection; a public rollout of yours is in both, so // its two checkboxes move together. The order you tick them in is the order of the comparison's A, B, C, D. const COMPARE_MAX = 4; // compare.js MAX let picked = new Set(); const pickable = (r, row) => `
    ${row}
    `; function syncPicks(el, v) { const full = picked.size >= COMPARE_MAX; $$("[data-cmp]", el).forEach((c) => { c.checked = picked.has(c.dataset.cmp); c.disabled = full && !c.checked; c.closest(".pick").title = c.disabled ? `Up to ${COMPARE_MAX} at once` : "Select to compare"; }); const bar = $("#cmp-bar", el); if (!bar) return; bar.hidden = !picked.size; const n = picked.size; bar.innerHTML = `${n} selected${n < 2 ? " · pick one more to compare" : ""} ${n >= 2 ? `${icon("columns", 13)}Compare ${n}` : ``}`; } const paramsShort = (p) => !p ? "" : [p.thinking && p.thinking !== "default" && `thinking ${p.thinking === "none" ? "off" : p.thinking}`, p.temperature != null && `t=${p.temperature}`, p.steps && `${p.steps} steps`].filter(Boolean).join(" · "); const disclose = (label, inner) => `
    ${icon("chevronRight", 14, "chev")}${esc(label)}${inner}
    `; const fmtNum = (x) => (x == null ? "" : Math.abs(x) >= 100 ? x.toFixed(0) : Math.abs(x) >= 1 ? x.toFixed(2) : x.toFixed(3)); function diffHtml(patch) { return patch.split("\n").slice(0, 4000).map((l) => { const cls = l.startsWith("+++") || l.startsWith("---") || l.startsWith("diff ") ? "dh" : l.startsWith("+") ? "da" : l.startsWith("-") ? "dd" : l.startsWith("@@") ? "dc" : ""; return `${esc(l)}`; }).join("\n"); } // ── interactions: file previews, tables, outline ───────────────────────────── function wire(el, v) { el.addEventListener("click", async (ev) => { const j = ev.target.closest("[data-jump]"); if (j) { ev.preventDefault(); $(`#sec-${j.dataset.jump}`, el).scrollIntoView({ behavior: "smooth", block: "start" }); return; } if (ev.target.closest("[data-copy]")) { try { await navigator.clipboard.writeText(location.href); toast("Link copied"); } catch { toast("Copy the address bar"); } return; } const fold = ev.target.closest("[data-pr-expand]"); if (fold) { fold.nextElementSibling.hidden = false; fold.remove(); return; } if (ev.target.closest("[data-pr-copy]")) { try { await navigator.clipboard.writeText(v.prompt.parts.map(([, t]) => t).join("")); toast("Message copied"); } catch { toast("Couldn't copy"); } return; } const f = ev.target.closest(".file"); if (f) return previewFile(v, f.dataset.path, f.dataset.kind); const t = ev.target.closest(".tbl-btn"); if (t) return previewTable(v, t.dataset.sys, t.dataset.table, t.dataset.label); if (ev.target.closest("[data-cmp-clear]")) { picked.clear(); syncPicks(el, v); } }); el.addEventListener("change", (ev) => { const c = ev.target.closest("[data-cmp]"); if (!c) return; if (c.checked) picked.add(c.dataset.cmp); else picked.delete(c.dataset.cmp); syncPicks(el, v); }); const links = new Map($$("[data-jump]", el).map((a) => [a.dataset.jump, a])); spy = new IntersectionObserver((es) => es.forEach((e) => { if (e.isIntersecting) { links.forEach((a) => a.classList.remove("on")); links.get(e.target.id.slice(4))?.classList.add("on"); } }), { rootMargin: "-20% 0px -70% 0px" }); $$(".tp-sec", el).forEach((s) => spy.observe(s)); } const previewSkeleton = (kind) => kind === "spreadsheet" ? `
    ${sk.box(28, "width:90px;border-radius:7px")}${sk.box(28, "width:70px;border-radius:7px")}
    ${sk.box(360, "border-radius:10px")}` : `
    ${sk.line(40, 18)}${sk.lines(100, 96, 90, 98, 70)}
    ${sk.lines(94, 100, 88, 60)}
    `; async function previewFile(v, path, kind) { const raw = `/api/tasks/${encodeURIComponent(v.id)}/file?path=${encodeURIComponent(path)}`; const body = openModal(path, previewSkeleton(kind), { raw, kind }); if (kind === "image") { body.innerHTML = `
    ${esc(path)}
    `; return; } if (kind === "web") { body.innerHTML = ``; return; } let p; try { p = await progress.wrap(api(`/api/tasks/${encodeURIComponent(v.id)}/preview?path=${encodeURIComponent(path)}`)); } catch (e) { body.innerHTML = emptyState("alert", "Couldn't open this file", esc(e.message)); return; } if (!body.isConnected || $("#modal").hidden) return; if (p.type === "sheets") { body.innerHTML = `
    ${p.sheets.map((s, i) => ``).join("")}
    `; const show = (i) => { $("#pv-sheet", body).innerHTML = sheetGrid(p.sheets[i].rows) + (p.sheets[i].rows.length >= 60 ? `

    Showing the first 60 rows and 24 columns. Open the original for the whole workbook.

    ` : ""); $$(".pv-tabs button", body).forEach((b) => b.setAttribute("aria-selected", b.dataset.i == i)); }; body.querySelector(".pv-tabs").addEventListener("click", (e) => { const b = e.target.closest("button"); if (b) show(+b.dataset.i); }); show(0); } else if (p.type === "document") { body.innerHTML = `
    ${p.blocks.map((b) => b.table ? table(b.table, { header: true }) : b.h ? `${esc(b.text)}` : `

    ${esc(b.text)}

    `).join("")}
    `; } else if (p.type === "slides") { body.innerHTML = `${p.repeated?.length ? `
    ${icon("info")}On every slide, not repeated below: ${p.repeated.map((t) => `${esc(t)}`).join(" · ")}
    ` : ""}
    ${p.slides.map((s) => `
    ${s.n}
    ${esc(s.title || "Untitled slide")}${s.text.map((t) => `

    ${esc(t).replace(/\n/g, "
    ")}

    `).join("")} ${(s.tables || []).map((t) => table(t)).join("")}
    `).join("")}
    `; } else if (p.type === "pdf") { body.innerHTML = `
    `; body.querySelector(".pv-tabs").addEventListener("click", (e) => { const b = e.target.closest("button"); if (!b) return; $$(".pv-tabs button", body).forEach((x) => x.setAttribute("aria-selected", x === b)); $("#pv-pdf", body).innerHTML = b.dataset.m === "view" ? `` : `
    ${p.pages.map((pg) => `
    Page ${pg.n}
    ${esc(pg.text)}
    `).join("")}
    `; }); } else if (p.type === "text") body.innerHTML = `
    ${esc(p.text)}
    `; else if (p.type === "archive") body.innerHTML = table([["Entry", "Size"], ...p.entries.map((x) => [x.name, bytes(x.size)])]); else if (p.type === "error") body.innerHTML = emptyState("alert", "Couldn't preview this file", esc(p.error), `Download`); else body.innerHTML = emptyState("file", "No preview for this file type", "", `Download`); } async function previewTable(v, sys, tbl, label) { const body = openModal(`${label || sys} · ${tbl}`, sk.box(420, "border-radius:10px"), { kind: "table" }); try { const d = await progress.wrap(api(`/api/tasks/${encodeURIComponent(v.id)}/systems/${encodeURIComponent(sys)}/${encodeURIComponent(tbl)}?limit=100`)); if ($("#modal").hidden) return; body.innerHTML = table([d.columns, ...d.rows.map((r) => r.map((c) => (c == null ? "" : String(c))))]) + `

    ${d.rows.length >= 100 ? "First 100 rows" : `${d.rows.length} row${d.rows.length === 1 ? "" : "s"}`} · ${d.columns.length} columns · the state of this database when the agent starts

    `; } catch (e) { body.innerHTML = emptyState("alert", "Couldn't load this table", esc(e.message)); } } // ── the run panel ──────────────────────────────────────────────────────────── // Settings that are worth keeping between visits live in this browser. An endpoint's API key stays in this tab only. const LS = { get: (k, d) => { try { return JSON.parse(localStorage.getItem(k)) ?? d; } catch { return d; } }, set: (k, v) => { try { localStorage.setItem(k, JSON.stringify(v)); } catch { /* private mode */ } } }; const SS = { get: (k) => { try { return sessionStorage.getItem(k) || ""; } catch { return ""; } }, set: (k, v) => { try { sessionStorage.setItem(k, v); } catch { /* private mode */ } } }; const DEFAULT_THINKING = "low"; // long agent runs with full thinking are slow and costly; low keeps most of the gain const THINKING = [["default", "Model default"], ["none", "Off"], ["low", "Low"], ["medium", "Medium"], ["high", "High"]]; async function runPanel(box, v) { const s = getSession(); const inner = $(".panel-b", box); if (!v.runnable) { inner.innerHTML = `

    This task can't be run here yet.

    `; return; } let cat; try { cat = await getModels(); } catch (e) { if (!alive) return; inner.innerHTML = `
    ${icon("alert")}Couldn't reach HF Inference Providers: ${esc(e.message)}
    `; $("#rb-retry", inner).addEventListener("click", () => { inner.innerHTML = `
    ${spinner()}Loading models…
    `; runPanel(box, v); }); return; } if (!alive) return; const music = v.domain === "music"; const need = v.verify?.needs_judge; const judges = need === "vision" ? cat.vision_judges : need === "text" ? cat.text_judges : []; const gaps = s.missing_scopes || []; // say it before the run, not after: a sandbox can't start without prepaid credit (Music needs no sandbox) const bill = s.billing || {}; const noCredit = !music && s.user && (bill.can_pay === false || bill.refused_at); const BILLING = `huggingface.co/settings/billing`; const d = v.run_defaults || {}; const ep = LS.get("byo", { base_url: "", model: "", price_in: "", price_out: "" }); const adv = LS.get("adv2", { thinking: DEFAULT_THINKING, temperature: "" }); const st = { source: LS.get("source", "hf"), probe: null, visibility: "public" }; // public unless you choose otherwise, every time inner.innerHTML = `

    ${music ? "One model call, then Xiaomi's scorer. No sandbox." : "A fresh HF Sandbox from this task's image, OpenCode as the harness, then the task's own grader."}

    Agent model
    Price per 1M tokens optional, only for the cost shown

    Works with a LiteLLM proxy, vLLM, OpenAI, Together, OpenRouter and the like. The sandbox still runs on your Hugging Face account, so you sign in with HF either way. The key is used for this rollout only and never stored on the server, but the agent's own processes in the sandbox can see it: use a key you can revoke.

    ${judges.length ? `
    Judge model ${need === "vision" ? "scores a screenshot of the page" : "answers each rubric check"}
    ` : ""}
    ${icon("chevronRight", 14, "chev")}Advanced settings
    ${music ? "" : ` `}

    Thinking is sent as reasoning_effort; "Off" disables it on models that support that. ${music ? "" : `The step cap and time limit default to the values Xiaomi's training harness uses for this domain.`}

    Visibility

    ${gaps.length ? `
    ${icon("alert")}Your sign-in may be missing ${esc(gaps.join(", "))}. If the rollout fails to start, sign in again with a write token.
    ` : ""} ${noCredit ? `
    ${icon("alert")}${bill.refused_at ? `Your last rollout couldn't start because your Hugging Face account had no prepaid credit for sandboxes. Add credit at ${BILLING}, then run again.` : `Your Hugging Face account can't pay for sandboxes yet, so a rollout would stop before it starts. Add prepaid credit at ${BILLING} first.`} Nothing is charged for a rollout that can't start.
    ` : ""} ${s.user ? `` : ``}

    Keeps running if you close this page; find it under My rollouts. The sandbox is billed to ${s.user ? `${esc(s.user.name)}` : "your account"} on Hugging Face.

    `; const def = cat.agents.some((m) => m.id === cat.default_agent) ? cat.default_agent : cat.agents[0]?.id; const sel = picker($("#rb-model", box), { models: cat.agents, value: def, onChange: () => update(), note: "Every model here supports tool calling. Prices are per million tokens, input / output, at the cheapest provider that can call tools." }); const judgeSel = judges.length ? picker($("#rb-judge", box), { models: judges.map((j) => ({ ...j, featured: true })), value: judges[0].id, onChange: () => {}, groupLabel: "Tested judges", note: need === "vision" ? "Tested on a real render with Xiaomi's rubric. Judges disagree (0.34–0.87 on the same page), so compare webdev scores only between runs graded by the same judge." : "Tested on this dataset's real rubric prompt: each returned a usable verdict on every check." }) : null; const val = (id) => { const e = $(id, box); return e ? e.value.trim() : ""; }; const num = (id) => { const x = val(id); return x === "" ? null : Number(x); }; const endpoint = () => ({ base_url: val("#ep-url"), api_key: val("#ep-key") || null, model: val("#ep-model"), price_in: num("#ep-pin"), price_out: num("#ep-pout") }); const params = () => { const p = { thinking: val("#p-think") || DEFAULT_THINKING }; if (num("#p-temp") != null) p.temperature = num("#p-temp"); if (num("#p-max") != null) p.max_tokens = num("#p-max"); if (num("#p-steps") != null) p.steps = num("#p-steps"); if (num("#p-time") != null) p.timeout_min = num("#p-time"); return p; }; const save = () => { const e = endpoint(); LS.set("byo", { base_url: e.base_url, model: e.model, price_in: val("#ep-pin"), price_out: val("#ep-pout") }); SS.set("byo-key", val("#ep-key")); LS.set("adv2", { thinking: val("#p-think"), temperature: val("#p-temp") }); }; const tok = (n) => (n >= 1e6 ? (n / 1e6).toFixed(1) + "M" : Math.round(n / 1e3) + "k"); const [ti, to] = TYPICAL[v.domain]; function update() { save(); $$("[data-src]", box).forEach((b) => b.setAttribute("aria-pressed", String(b.dataset.src === st.source))); $$("[data-pane]", box).forEach((p) => (p.hidden = p.dataset.pane !== st.source)); $$("[data-vis]", box).forEach((b) => b.setAttribute("aria-pressed", String(b.dataset.vis === st.visibility))); $("#vis-note", box).textContent = st.visibility === "public" ? PUBLIC_NOTE : PRIVATE_NOTE; const p = params(); const changed = [p.thinking !== DEFAULT_THINKING && `thinking ${THINKING.find(([k]) => k === p.thinking)[1].toLowerCase()}`, p.temperature != null && `temp ${p.temperature}`, p.steps && `${p.steps} steps`, p.timeout_min && `${p.timeout_min} min`, p.max_tokens && `${p.max_tokens} max tokens`].filter(Boolean); $("#adv-sum", box).textContent = changed.length ? changed.join(" · ") : ""; let price; if (st.source === "byo") { const e = endpoint(); price = e.price_in != null || e.price_out != null ? [e.price_in || 0, e.price_out || 0] : null; } else price = [sel.model.input, sel.model.output]; const minutes = p.timeout_min ? Math.min(p.timeout_min, TYPICAL_MIN[v.domain] || 0) : TYPICAL_MIN[v.domain]; const sandbox = music ? 0 : (minutes / 60) * ((cat.sandbox_price_per_hour || {})[v.domain === "webdev" ? "cpu-upgrade" : "cpu-basic"] ?? 0.01); const model = price ? (ti * price[0] + to * price[1]) / 1e6 : null; $("#rb-est", box).innerHTML = `
    Model · ${tok(ti)} in, ${tok(to)} out${model == null ? "your provider's price" : money(model)}
    ${music ? "" : `
    Sandbox · ~${minutes} min${money(sandbox)}
    `}
    Typical ${esc(DOMAIN_NAME[v.domain])} rollout${model == null ? `${money(sandbox)} + tokens` : `~${money(model + sandbox)}`}
    ${(v.domain === "general" || v.domain === "cyber") && st.source === "hf" ? `
    Agents re-read their context every step, so input tokens dominate. A cheaper model makes a big difference here.
    ` : ""}`; const go = $("#rb-go", box); if (go && !go.dataset.busy) { const ready = st.source === "hf" || (st.probe?.ok && (music || st.probe.tools)); go.disabled = !ready; go.title = ready ? "" : "Test the connection first"; } } const status = (kind, html) => { $("#ep-status", box).innerHTML = html ? `
    ${icon(kind === "err" ? "alert" : kind === "warn" ? "alert" : "check")}${html}
    ` : ""; }; box.addEventListener("click", async (e) => { const vb = e.target.closest("[data-vis]"); if (vb) { const was = st.visibility; st.visibility = vb.dataset.vis; update(); if (was === "private" && st.visibility === "public") { const r = vb.getBoundingClientRect(); confetti(r.left + r.width / 2, r.top); toast("Thank you for sharing with the community"); } return; } const b = e.target.closest("[data-src]"); if (b) { st.source = b.dataset.src; LS.set("source", st.source); update(); return; } if (e.target.closest("#ep-load")) { const btn = $("#ep-load", box); btn.disabled = true; try { const r = await api("/api/endpoints/models", { method: "POST", body: { base_url: val("#ep-url"), api_key: val("#ep-key") || null } }); $("#ep-models", box).innerHTML = r.models.map((m) => `