"use strict"; // The server owns the model list and its order (/api/status "order"). This default only covers the first paint. let MODELS = []; // filled from /api/status; nothing is assumed before the server answers const $ = (id) => document.getElementById(id); const esc = (s) => String(s ?? "").replace(/[&<>"']/g, (c) => ({ "&": "&", "<": "<", ">": ">", '"': """, "'": "'" }[c])); const pct = (p) => `${(100 * p).toFixed(1)}%`; const nModels = () => MODELS.length; // ------------------------------------------------------------------ API access from the page // The API is open (no token). If the server cannot be reached, say so plainly. async function api(url, opts = {}) { try { return await fetch(url, { ...opts, credentials: "same-origin" }); } catch { throw new Error("The page lost its connection to the lab server. Reload the page (Ctrl+F5)."); } } // CSP (DL-SA-006) forbids inline style attributes; bars carry data-scale and get their transform here. function applyScales(root) { root.querySelectorAll("[data-scale]").forEach((el) => { el.style.transform = `scaleX(${+el.dataset.scale || 0})`; }); } let DEMOS = []; let current = null; // selected demo let last = null; // {questions, reference, response, roundTrip} let status = null; // ------------------------------------------------------------------ status async function pollStatus() { try { status = await (await api("/api/status")).json(); syncModels(); renderStatus(); const busy = Object.values(status.models).some((m) => m.status === "loading" || m.status === "idle"); setTimeout(pollStatus, busy ? 2000 : 15000); } catch { $("readiness").textContent = "Lab server not reachable"; setTimeout(pollStatus, 4000); } } function syncModels() { const order = status.order || Object.keys(status.models); const next = order.filter((k) => status.models[k]).map((k) => { const d = status.models[k]; return { key: k, name: d.name || k, side: d.side || k }; }); const changed = JSON.stringify(next) !== JSON.stringify(MODELS); MODELS = next; if (changed) { applyModelCount(); renderHero(); renderScoreHead(); } } function applyModelCount() { document.documentElement.style.setProperty("--n-models", String(nModels())); $("run").textContent = runLabel(); const n = { 1: "One", 2: "Both", 3: "All three" }[nModels()] || `All ${nModels()}`; document.querySelectorAll("[data-n-models]").forEach((el) => { el.textContent = el.dataset.nModels.replace("{N}", n).replace("{n}", n.toLowerCase()); }); } const runLabel = () => (nModels() <= 1 ? "Run the model" : nModels() === 2 ? "Run both models" : `Run all ${nModels()} models`); function renderHero() { // "vs" between boxes reads well on one row (up to 3 models); with more, the boxes wrap into rows without it. const vs = MODELS.length <= 3; $("versus").classList.toggle("many", !vs); $("versus").innerHTML = MODELS.map((m, i) => `${i && vs ? `
vs
` : ""}
${esc(m.name)} – parameters
`).join(""); } function sourceLine(m, d) { if (d.path) return `Local folder ${esc(d.path)}`; const repo = d.repo || ""; return repo ? `${esc(repo)}` : ""; } function renderStatus() { const ms = status.models; const dir = status.env?.models_dir || "/models"; const found = MODELS.filter((m) => m.key !== "laya").length; $("models-found").textContent = found ? `Found ${found} model${found === 1 ? "" : "s"} in ${dir} (your DecisionLab\\models folder). Add or remove a folder there and restart the container to change the line-up.` : ""; const none = found ? "" : `

No models found in ${esc(dir)}. Unzip a model into your DecisionLab\\models folder (each in its own folder: a LightDec folder has falcondec_config.json, an Arthur folder has config.json and model.safetensors) and restart the container. Laya still runs.

`; $("models").innerHTML = none + MODELS.map((m) => { const d = ms[m.key] || {}; const label = { ready: "Ready", loading: "Loading", error: "Failed to load", idle: "Waiting" }[d.status] || d.status; const facts = [ ["Parameters", d.params_m ? `${d.params_m}M` : "–"], ["Weights", d.weights_mb ? `${d.weights_mb} MB` : m.key === "laya" ? "≈808 MB" : "–"], ["Device", d.device ? d.device.toUpperCase() : "–"], ["Load time", d.load_seconds ? `${d.load_seconds} s` : "–"], ["Version", d.version || "–"], ["Native confidence", d.confidence_native || "–"], ]; return `

${esc(m.name)}${esc(label)}

${sourceLine(m, d)}${d.note ? ` · ${esc(d.note)}` : ""}

${facts.map(([k, v]) => `
${k}
${esc(v)}
`).join("")}
${d.status === "error" ? `

${esc(d.error)}

` : ""}
`; }).join(""); for (const m of MODELS) { const d = ms[m.key] || {}, el = $(`hero-params-${m.key}`); if (el) el.textContent = d.params_m ? `${Math.round(d.params_m)}M` : "–"; } document.querySelectorAll("[data-reload]").forEach((b) => b.addEventListener("click", async () => { b.disabled = true; b.textContent = "Loading…"; await api(`/api/reload/${encodeURIComponent(b.dataset.reload)}`, { method: "POST" }); setTimeout(pollStatus, 500); })); const ready = MODELS.filter((m) => ms[m.key]?.status === "ready").map((m) => m.name); const loading = MODELS.filter((m) => ms[m.key]?.status === "loading").map((m) => m.name); const r = $("readiness"); r.classList.toggle("ready", ready.length === nModels()); r.textContent = ready.length === nModels() ? (nModels() === 2 ? "Both models ready" : `All ${nModels()} models ready`) : loading.length ? `Loading ${loading.join(", ")}…` : ready.length ? `${ready.join(", ")} ready` : "Models not loaded"; const e = status.env; $("env").textContent = `Server: ${e.gpu ? `GPU ${e.gpu}` : `CPU, ${e.threads} threads`}, PyTorch ${e.torch}, Python ${e.python}.`; } // ------------------------------------------------------------------ demos: tabs, tiles, filter, prev/next let GROUPS = []; let activeGroup = null; const groupOf = (id) => GROUPS.find((g) => g.id === id); async function loadDemos() { [DEMOS, GROUPS] = await Promise.all([ api("/api/demos").then((r) => r.json()), api("/api/groups").then((r) => r.json()).catch(() => []), ]); if (!GROUPS.length) { // older server: derive groups from the demos GROUPS = [...new Set(DEMOS.map((d) => d.group))].map((id) => ({ id, label: id, family: "", blurb: "" })); } DEMOS.forEach((d, i) => { d.num = i + 1; }); for (const g of GROUPS) g.count = DEMOS.filter((d) => d.group === g.id).length; activeGroup = GROUPS[0]?.id; renderTabs(); renderScope(); $("demo-filter").placeholder = `Filter all ${DEMOS.length} demos`; $("demo-filter").addEventListener("input", renderTiles); $("prev").addEventListener("click", () => step(-1)); $("next").addEventListener("click", () => step(1)); selectDemo(DEMOS[0].id); } function renderTabs() { const families = [...new Set(GROUPS.map((g) => g.family))]; $("tabs").innerHTML = families.map((f) => `
${f ? `${esc(f)}` : ""}
${GROUPS.filter((g) => g.family === f).map((g) => ` `).join("")}
`).join(""); document.querySelectorAll(".tab").forEach((t) => t.addEventListener("click", () => { $("demo-filter").value = ""; activeGroup = t.dataset.group; const first = DEMOS.find((d) => d.group === activeGroup); if (first && current?.group !== activeGroup) selectDemo(first.id); else renderTiles(); })); } function renderTiles() { const q = $("demo-filter").value.trim().toLowerCase(); const list = q ? DEMOS.filter((d) => `${d.num} ${d.title} ${d.blurb} ${groupOf(d.group)?.label ?? d.group}`.toLowerCase().includes(q)) : DEMOS.filter((d) => d.group === activeGroup); document.querySelectorAll(".tab").forEach((t) => t.setAttribute("aria-selected", String(!q && t.dataset.group === activeGroup))); $("group-blurb").textContent = q ? `${list.length} of ${DEMOS.length} demos match.` : (groupOf(activeGroup)?.blurb || ""); $("tiles").innerHTML = list.length ? list.map((d) => ` `).join("") : `

No demo matches “${esc(q)}”.

`; $("tiles").querySelectorAll(".tile").forEach((b) => b.addEventListener("click", () => selectDemo(b.dataset.demo))); } function renderScope() { const nq = (ds) => ds.reduce((t, d) => t + Object.keys(d.questions).length, 0); const all = [...new Set(GROUPS.map((g) => g.family))].filter(Boolean); const families = all.length > 1 ? all : []; // a single family would only repeat "All demos" $("scope").innerHTML = `` + families.map((f) => { const ds = DEMOS.filter((d) => groupOf(d.group)?.family === f); return ``; }).join("") + GROUPS.map((g) => ``).join(""); $("scope").addEventListener("change", updateRunLabel); updateRunLabel(); } function scopedDemos() { const v = $("scope").value; if (v.startsWith("family:")) return DEMOS.filter((d) => groupOf(d.group)?.family === v.slice(7)); if (v.startsWith("group:")) return DEMOS.filter((d) => d.group === v.slice(6)); return DEMOS; } function updateRunLabel() { const n = scopedDemos().length; $("run-all").textContent = n === DEMOS.length ? `Run all ${n} demos` : `Run ${n} demos`; } function selectDemo(id) { current = DEMOS.find((d) => d.id === id); if (!$("demo-filter").value.trim()) activeGroup = current.group; renderTiles(); $("demo-title").innerHTML = `${current.num}${esc(current.title)}`; $("demo-blurb").textContent = current.blurb; $("state").value = typeof current.state === "string" ? current.state : JSON.stringify(current.state, null, 2); $("questions").value = JSON.stringify(current.questions, null, 2); $("form-error").textContent = ""; $("prev").disabled = current.num === 1; $("next").disabled = current.num === DEMOS.length; } function step(delta) { const i = DEMOS.indexOf(current) + delta; if (i >= 0 && i < DEMOS.length) { $("demo-filter").value = ""; selectDemo(DEMOS[i].id); } } function parseState(text) { const t = text.trim(); if (t.startsWith("{") || t.startsWith("[")) { try { return JSON.parse(t); } catch { /* plain text that happens to start with a brace */ } } return t; } // ------------------------------------------------------------------ running async function decide(state, questions) { const t0 = performance.now(); const res = await api("/api/decide", { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify(MODELS.length ? { state, questions, models: MODELS.map((m) => m.key) } : { state, questions }), }); const body = await res.json(); if (!res.ok) throw new Error(body.detail || `Request failed (${res.status})`); return { response: body, roundTrip: performance.now() - t0 }; } async function runCurrent() { const err = $("form-error"); err.textContent = ""; let questions; try { questions = JSON.parse($("questions").value); } catch (e) { err.textContent = `Questions aren't valid JSON: ${e.message}`; return; } const state = parseState($("state").value); if (!state || (typeof state === "string" && !state.length)) { err.textContent = "Add a state for the models to read."; return; } const btn = $("run"); btn.disabled = true; btn.textContent = "Running…"; try { const { response, roundTrip } = await decide(state, questions); const sameDemo = current && JSON.stringify(current.questions) === JSON.stringify(questions); last = { questions, reference: sameDemo ? current.reference : {}, response, roundTrip }; renderVerdicts(); $("verdicts").scrollIntoView(); } catch (e) { err.textContent = e.message; } finally { btn.disabled = false; btn.textContent = runLabel(); } } // ------------------------------------------------------------------ rendering helpers const measure = () => $("measure").value; const threshold = () => parseFloat($("threshold").value); const conf = (rec) => rec[measure()]; function optionLabels(q) { const t = q.type || "choice"; const crit = q.criteria ?? q.options; if (t === "noul") return [["true", "Yes"], ["false", "No"]]; if (t === "score") return crit.map((c, i) => [String(i), `${i} · ${c}`]); if (Array.isArray(crit)) return crit.map((c) => [String(c), String(c)]); return Object.entries(crit).map(([k, v]) => [k, v ? `${k}: ${v}` : k]); } function answerText(q, rec) { if (!rec) return "–"; const t = q.type || "choice"; if (t === "noul") return `${rec.choice === "true" ? "Yes" : "No"} (P(yes) ${pct(rec.p_true)})`; if (t === "score") { const crit = q.criteria ?? q.options; return `${crit[+rec.choice]} (level ${rec.expected_level.toFixed(2)})`; } return rec.choice; } function answerBlock(m, q, rec, err, ref) { const head = `
${esc(m.name)}
`; if (err) return `
${head}
${esc(err)}
`; if (!rec) return `
${head}
No answer returned.
`; const c = conf(rec); const act = c >= threshold(); const refTxt = ref === undefined ? "" : rec.choice === ref ? `matches reference` : `differs from reference`; return `
${head}
${esc(answerText(q, rec))}
confidence ${c.toFixed(3)}${act ? "act" : "defer"}${refTxt}
`; } // how many of the models that answered give the same top answer: {agree, of} function agreement(recs) { const got = recs.filter(Boolean); if (got.length < 2) return null; const counts = {}; for (const r of got) counts[r.choice] = (counts[r.choice] || 0) + 1; return { agree: Math.max(...Object.values(counts)), of: got.length }; } function agreeTag(ag) { if (!ag) return ""; const all = ag.agree === ag.of; const text = all ? (ag.of === 2 ? "Models agree" : `All ${ag.of} agree`) : ag.agree === 1 ? "All differ" : `${ag.agree} of ${ag.of} agree`; return `${text}`; } function renderVerdicts() { if (!last) return; const { questions, reference, response, roundTrip } = last; const R = response.results; const res = MODELS.map((m) => ({ m, r: R[m.key] || {} })); const tm = $("timing"); tm.hidden = false; const timed = res.filter((x) => x.r.ms != null); const fast = timed.length ? timed.reduce((a, b) => (b.r.ms < a.r.ms ? b : a)) : null; const slow = timed.length ? timed.reduce((a, b) => (b.r.ms > a.r.ms ? b : a)) : null; tm.innerHTML = res.map((x) => `
${x.r.ms != null ? `${x.r.ms} ms` : "–"}${esc(x.m.name)} model time
`).join("") + `
${fast && slow && fast !== slow ? `${(slow.r.ms / fast.r.ms).toFixed(1)}×` : "–"}${fast && slow && fast !== slow ? `${esc(fast.m.name)} fastest, vs ${esc(slow.m.name)}` : "speed ratio"}
` + `
${Math.round(roundTrip)} msround trip from your browser
`; $("verdict-list").innerHTML = Object.entries(questions).map(([name, q]) => { const recs = res.map((x) => x.r.answers?.[name]); const ref = reference?.[name]; const rows = optionLabels(q).map(([k, label]) => `
${esc(label)}
${res.map((x, i) => { const rec = recs[i], p = rec?.probs?.[k] ?? 0; return `
${rec ? pct(p) : ""}
`; }).join("")}
`).join(""); return `

${esc(q.instructions || q.question || name)}

${esc(name)} · ${esc(q.type || "choice")}

${agreeTag(agreement(recs))}
${res.map((x, i) => answerBlock(x.m, q, recs[i], x.r.error, ref)).join("")}
${rows}
`; }).join(""); applyScales($("verdict-list")); $("raw").hidden = false; $("raw-json").textContent = JSON.stringify(response, null, 2); } // ------------------------------------------------------------------ scoreboard let board = null; // {rows: [{demo, name, q, ref, high, ans: {modelKey: rec}}], times: {modelKey: [ms]}} async function runAll() { const btn = $("run-all"); btn.disabled = true; const rows = [], times = Object.fromEntries(MODELS.map((m) => [m.key, []])); try { const list = scopedDemos(); for (let i = 0; i < list.length; i++) { const d = list[i]; $("progress").textContent = `Running demo ${i + 1} of ${list.length}: ${d.title}`; const { response } = await decide(d.state, d.questions); for (const k of Object.keys(times)) if (response.results[k]?.ms != null) times[k].push(response.results[k].ms); for (const [name, q] of Object.entries(d.questions)) { rows.push({ demo: d, name, q, ref: d.reference?.[name], high: d.stakes?.[name] === "high", ans: Object.fromEntries(MODELS.map((m) => [m.key, response.results[m.key]?.answers?.[name]])) }); } } board = { rows, times }; $("progress").textContent = `Done: ${rows.length} questions across ${list.length} demos.`; renderScoreHead(); renderBoard(); } catch (e) { $("progress").textContent = `Stopped: ${e.message}`; } finally { btn.disabled = false; } } function median(xs) { if (!xs.length) return null; const s = [...xs].sort((x, y) => x - y), m = Math.floor(s.length / 2); return s.length % 2 ? s[m] : (s[m - 1] + s[m]) / 2; } // Agentic Use Score (AUS): a 0-100 score skewed toward what matters when an agent acts on the answer. // Every question counts once; questions marked high-stakes (irreversible, security, money) count twice. const AUS_WEIGHTS = { precision: 0.35, caught: 0.25, accuracy: 0.15, autonomy: 0.15, speed: 0.10 }; const AUS_LABELS = { precision: "Right when it acts", caught: "Flags its own mistakes", accuracy: "Overall accuracy", autonomy: "Handles on its own", speed: "Speed", }; function speedScore(ms) { // 1.0 at 50 ms or faster, 0 at 1 s or slower, log scale in between if (ms == null) return 0; return Math.max(0, Math.min(1, 1 - Math.log10(Math.max(ms, 1) / 50) / Math.log10(20))); } function modelStats(rows, key, times, thr) { let n = 0, hits = 0, acted = 0, actedHits = 0, wrong = 0, wrongDeferred = 0; // raw counts let W = 0, wHits = 0, wActed = 0, wActedHits = 0, wWrong = 0, wWrongDeferred = 0; // stakes-weighted for (const r of rows) { const rec = r.ans[key]; if (!rec || r.ref === undefined) continue; const w = r.high ? 2 : 1, ok = rec.choice === r.ref, act = conf(rec) >= thr; n++; W += w; if (ok) { hits++; wHits += w; } if (act) { acted++; wActed += w; if (ok) { actedHits++; wActedHits += w; } } if (!ok) { wrong++; wWrong += w; if (!act) { wrongDeferred++; wWrongDeferred += w; } } } const med = median(times || []); const parts = { precision: (wActedHits + 1) / (wActed + 2), // small prior: a model that acts twice isn't "perfect" caught: wWrong ? wWrongDeferred / wWrong : 1, accuracy: W ? wHits / W : 0, autonomy: W ? wActed / W : 0, speed: speedScore(med), }; const aus = 100 * Object.entries(AUS_WEIGHTS).reduce((t, [k, w]) => t + w * parts[k], 0); return { n, hits, acted, actedHits, wrong, wrongDeferred, confidentWrong: acted - actedHits, med, parts, aus }; } // Which models are best on a metric. values: {modelKey: number|null}. Returns the Set of keys within eps of the // best value, or an empty Set when every model is tied (nothing to highlight) or fewer than two have a value. function best(values, higherIsBetter = true, eps = 1e-9) { const got = Object.entries(values).filter(([, v]) => v != null); if (got.length < 2) return new Set(); const top = higherIsBetter ? Math.max(...got.map(([, v]) => v)) : Math.min(...got.map(([, v]) => v)); const keys = got.filter(([, v]) => Math.abs(v - top) < eps).map(([k]) => k); return keys.length === got.length ? new Set() : new Set(keys); } const modelHeads = () => MODELS.map((m) => `${esc(m.name)}`).join(""); // per-group matches and confident mistakes, so a model's weak spots show up by use case function groupBreakdown(rows, thr) { const ids = [...new Set(rows.map((r) => r.demo.group))]; if (ids.length < 2) return ""; const line = (rs, key) => { const scored = rs.filter((r) => r.ans[key] && r.ref !== undefined); const hits = scored.filter((r) => r.ans[key].choice === r.ref).length; const cw = scored.filter((r) => r.ans[key].choice !== r.ref && conf(r.ans[key]) >= thr).length; return { hits, n: scored.length, cw, rate: scored.length ? hits / scored.length : null }; }; const body = ids.map((id) => { const rs = rows.filter((r) => r.demo.group === id); const L = Object.fromEntries(MODELS.map((m) => [m.key, line(rs, m.key)])); const w = best(Object.fromEntries(MODELS.map((m) => [m.key, L[m.key].rate]))); return `${esc(groupOf(id)?.label ?? id)}${esc(groupOf(id)?.family ?? "")}${MODELS.map((m) => { const x = L[m.key]; if (!x.n) return `–no reference answers`; // e.g. Agent security: states and questions only return `${x.hits} / ${x.n}${x.cw} confident mistake${x.cw === 1 ? "" : "s"}`; }).join("")}`; }).join(""); return `

By group

${modelHeads()}${body}
Group
`; } function renderScoreHead() { $("score-head").innerHTML = `DemoQuestionReference${modelHeads()}Agree`; } // A run with no reference answers (e.g. only the Agent security demos) cannot be scored: the Agentic Use Score // would fall back to the same defaults for every model. Show what CAN be compared instead. function renderUnscoredSummary() { const { rows, times } = board; const thr = threshold(); const card = (m) => { const recs = rows.map((r) => r.ans[m.key]).filter(Boolean); const acts = recs.filter((rec) => conf(rec) >= thr).length; const med = median(times[m.key] || []); const facts = [ ["Questions answered", `${recs.length} / ${rows.length}`], ["Median model time", med != null ? `${Math.round(med)} ms` : "–"], [`Acts on (conf ≥ ${thr.toFixed(2)})`, `${acts} / ${recs.length}`], ["Defers", `${recs.length - acts} / ${recs.length}`], ]; return `

${esc(m.name)}

${facts.map(([k, v]) => `
${k}
${v}
`).join("")}
`; }; const answered = rows.filter((r) => MODELS.every((m) => r.ans[m.key])); const allAgree = answered.filter((r) => new Set(MODELS.map((m) => r.ans[m.key].choice)).size === 1).length; const sum = $("summary"); sum.hidden = false; sum.innerHTML = `

Not scored

The Agentic Use Score compares each answer with a reference answer, and the questions in this run have none. So there is no score or ranking here; compare the models below by what they would act on at your threshold, their speed, and how often they agree.

` + MODELS.map(card).join("") + `

All ${MODELS.length} models give the same top answer on ${allAgree} of ${answered.length} questions. Confidence uses the measure selected above; changing it or the threshold updates every number here.

`; } function renderSummary() { const { rows, times } = board; if (!rows.some((r) => r.ref !== undefined)) return renderUnscoredSummary(); const thr = threshold(); const S = Object.fromEntries(MODELS.map((m) => [m.key, modelStats(rows, m.key, times[m.key], thr)])); const ratio = (x, y) => (y ? x / y : null); const per = (f) => Object.fromEntries(MODELS.map((m) => [m.key, f(S[m.key])])); const metrics = [ { label: "Matches reference", show: (x) => `${x.hits} / ${x.n}`, w: best(per((x) => x.hits)) }, { label: "Median model time", show: (x) => (x.med != null ? `${Math.round(x.med)} ms` : "–"), w: best(per((x) => x.med), false, 0.5) }, { label: `Acts on (conf ≥ ${thr.toFixed(2)})`, show: (x) => `${x.acted} / ${x.n}`, w: best(per((x) => x.acted)) }, { label: "Correct when acting", show: (x) => (x.acted ? `${x.actedHits} / ${x.acted}` : "–"), w: best(per((x) => ratio(x.actedHits, x.acted))) }, { label: "Confident mistakes", show: (x) => `${x.confidentWrong}`, w: best(per((x) => x.confidentWrong), false) }, { label: "Mistakes it deferred", show: (x) => (x.wrong ? `${x.wrongDeferred} / ${x.wrong}` : "–"), w: best(per((x) => ratio(x.wrongDeferred, x.wrong) ?? 1)) }, ]; const overall = best(per((x) => x.aus), true, 0.05); const answered = rows.filter((r) => MODELS.every((m) => r.ans[m.key])); const allAgree = answered.filter((r) => new Set(MODELS.map((m) => r.ans[m.key].choice)).size === 1).length; const card = (m) => { const st = S[m.key], win = overall.has(m.key); return `

${esc(m.name)}${win ? `Best for agents` : ""}

${metrics.map((x) => `
${x.label}
${x.show(st)}
`).join("")}
`; }; const ranked = [...MODELS].sort((x, y) => S[y.key].aus - S[x.key].aus); let pos = 0, prev = null; const rankItems = ranked.map((m, i) => { const v = S[m.key].aus; if (prev === null || Math.abs(prev - v) >= 0.05) pos = i + 1; const tied = ranked.some((o) => o !== m && Math.abs(S[o.key].aus - v) < 0.05); prev = v; return `
  • ${tied ? `=${pos}` : pos} ${esc(m.name)} ${v.toFixed(1)}
  • `; }).join(""); const aus = `

    Agentic Use Score

    How safe and useful each model is when an agent acts on its answers. Weighted toward being right when it acts and flagging its own mistakes; high-stakes questions count double.

      ${rankItems}
    ${modelHeads()}${Object.keys(AUS_WEIGHTS).map((k) => { const max = AUS_WEIGHTS[k] * 100; const pts = Object.fromEntries(MODELS.map((m) => [m.key, max * S[m.key].parts[k]])); const w = best(pts, true, 0.05); return `${MODELS.map((m) => ``).join("")}`; }).join("")}${MODELS.map((m) => ``).join("")}
    ComponentMax points
    ${AUS_LABELS[k]}${max.toFixed(0)}${pts[m.key].toFixed(1)}${Math.round(100 * S[m.key].parts[k])}% of max
    Agentic Use Score100${S[m.key].aus.toFixed(1)}
    `; const sum = $("summary"); sum.hidden = false; sum.innerHTML = aus + MODELS.map(card).join("") + groupBreakdown(rows, thr) + `

    Highlighted values are the best model on that metric (none are highlighted when all tie). All ${MODELS.length} models give the same top answer on ${allAgree} of ${answered.length} questions. Confidence uses the measure selected above; changing it or the threshold updates every number here.

    `; applyScales(sum); } function renderScoreRows() { const { rows } = board; const thr = threshold(); const cell = (q, rec, ref) => { if (!rec) return `–`; const cls = ref === undefined ? "" : rec.choice === ref ? "hit" : "miss"; return `${esc(answerText(q, rec))}conf ${conf(rec).toFixed(3)} · ${conf(rec) >= thr ? "act" : "defer"}`; }; const refText = (q, ref) => { if (ref === undefined) return "–"; if ((q.type || "choice") === "noul") return ref === "true" ? "Yes" : "No"; if (q.type === "score") return (q.criteria ?? q.options)[+ref]; return ref; }; const table = $("score-table"); table.hidden = false; table.querySelector("tbody").innerHTML = rows.map((r) => { const ag = agreement(MODELS.map((m) => r.ans[m.key])); return ` ${esc(r.demo.title)}${esc(groupOf(r.demo.group)?.label ?? r.demo.group)} ${esc(r.q.instructions)}${esc(r.name)} · ${esc(r.q.type || "choice")}${r.high ? ` · high stakes` : ""} ${esc(refText(r.q, r.ref))} ${MODELS.map((m) => cell(r.q, r.ans[m.key], r.ref)).join("")} ${ag ? `${ag.agree} / ${ag.of}` : "–"} `; }).join(""); } // DL1: the summary paints in the current frame; the long row table follows in its own task, so no single // task has to rebuild everything. A newer request cancels a pending row render. let rowsTimer = 0; function renderBoard() { if (!board) return; renderSummary(); clearTimeout(rowsTimer); rowsTimer = setTimeout(renderScoreRows, 0); } // ------------------------------------------------------------------ wiring // DL1: the slider can fire faster than a frame; update the number at once and redraw at most once per frame. let rerenderFrame = 0; function onThresholdChange() { $("thr-out").textContent = threshold().toFixed(2); if (rerenderFrame) return; rerenderFrame = requestAnimationFrame(() => { rerenderFrame = 0; renderVerdicts(); renderBoard(); }); } $("run").addEventListener("click", runCurrent); $("run-all").addEventListener("click", runAll); $("threshold").addEventListener("input", onThresholdChange); $("measure").addEventListener("change", onThresholdChange); document.addEventListener("keydown", (e) => { if ((e.ctrlKey || e.metaKey) && e.key === "Enter") runCurrent(); }); applyModelCount(); renderHero(); renderScoreHead(); pollStatus(); loadDemos();