"use strict";
// The server owns the model list and its order (/api/status "order"). This default only covers the first paint.
let MODELS = []; // filled from /api/status; nothing is assumed before the server answers
const $ = (id) => document.getElementById(id);
const esc = (s) => String(s ?? "").replace(/[&<>"']/g, (c) => ({ "&": "&", "<": "<", ">": ">", '"': """, "'": "'" }[c]));
const pct = (p) => `${(100 * p).toFixed(1)}%`;
const nModels = () => MODELS.length;
// ------------------------------------------------------------------ API access from the page
// The API is open (no token). If the server cannot be reached, say so plainly.
async function api(url, opts = {}) {
try {
return await fetch(url, { ...opts, credentials: "same-origin" });
} catch {
throw new Error("The page lost its connection to the lab server. Reload the page (Ctrl+F5).");
}
}
// CSP (DL-SA-006) forbids inline style attributes; bars carry data-scale and get their transform here.
function applyScales(root) {
root.querySelectorAll("[data-scale]").forEach((el) => { el.style.transform = `scaleX(${+el.dataset.scale || 0})`; });
}
let DEMOS = [];
let current = null; // selected demo
let last = null; // {questions, reference, response, roundTrip}
let status = null;
// ------------------------------------------------------------------ status
async function pollStatus() {
try {
status = await (await api("/api/status")).json();
syncModels();
renderStatus();
const busy = Object.values(status.models).some((m) => m.status === "loading" || m.status === "idle");
setTimeout(pollStatus, busy ? 2000 : 15000);
} catch {
$("readiness").textContent = "Lab server not reachable";
setTimeout(pollStatus, 4000);
}
}
function syncModels() {
const order = status.order || Object.keys(status.models);
const next = order.filter((k) => status.models[k]).map((k) => {
const d = status.models[k];
return { key: k, name: d.name || k, side: d.side || k };
});
const changed = JSON.stringify(next) !== JSON.stringify(MODELS);
MODELS = next;
if (changed) { applyModelCount(); renderHero(); renderScoreHead(); }
}
function applyModelCount() {
document.documentElement.style.setProperty("--n-models", String(nModels()));
$("run").textContent = runLabel();
const n = { 1: "One", 2: "Both", 3: "All three" }[nModels()] || `All ${nModels()}`;
document.querySelectorAll("[data-n-models]").forEach((el) => { el.textContent = el.dataset.nModels.replace("{N}", n).replace("{n}", n.toLowerCase()); });
}
const runLabel = () => (nModels() <= 1 ? "Run the model" : nModels() === 2 ? "Run both models" : `Run all ${nModels()} models`);
function renderHero() {
// "vs" between boxes reads well on one row (up to 3 models); with more, the boxes wrap into rows without it.
const vs = MODELS.length <= 3;
$("versus").classList.toggle("many", !vs);
$("versus").innerHTML = MODELS.map((m, i) => `${i && vs ? `
vs
` : ""}
${esc(m.name)}–parameters
`).join("");
}
function sourceLine(m, d) {
if (d.path) return `Local folder ${esc(d.path)}`;
const repo = d.repo || "";
return repo ? `${esc(repo)}` : "";
}
function renderStatus() {
const ms = status.models;
const dir = status.env?.models_dir || "/models";
const found = MODELS.filter((m) => m.key !== "laya").length;
$("models-found").textContent = found
? `Found ${found} model${found === 1 ? "" : "s"} in ${dir} (your DecisionLab\\models folder). Add or remove a folder there and restart the container to change the line-up.`
: "";
const none = found ? "" : `
No models found in ${esc(dir)}. Unzip a model into your DecisionLab\\models folder (each in its own folder: a LightDec folder has falcondec_config.json, an Arthur folder has config.json and model.safetensors) and restart the container. Laya still runs.
`;
}
function renderScoreHead() {
$("score-head").innerHTML = `
Demo
Question
Reference
${modelHeads()}
Agree
`;
}
// A run with no reference answers (e.g. only the Agent security demos) cannot be scored: the Agentic Use Score
// would fall back to the same defaults for every model. Show what CAN be compared instead.
function renderUnscoredSummary() {
const { rows, times } = board;
const thr = threshold();
const card = (m) => {
const recs = rows.map((r) => r.ans[m.key]).filter(Boolean);
const acts = recs.filter((rec) => conf(rec) >= thr).length;
const med = median(times[m.key] || []);
const facts = [
["Questions answered", `${recs.length} / ${rows.length}`],
["Median model time", med != null ? `${Math.round(med)} ms` : "–"],
[`Acts on (conf ≥ ${thr.toFixed(2)})`, `${acts} / ${recs.length}`],
["Defers", `${recs.length - acts} / ${recs.length}`],
];
return `
The Agentic Use Score compares each answer with a reference answer, and the questions in this run have none. So there is no score or ranking here; compare the models below by what they would act on at your threshold, their speed, and how often they agree.
`
+ MODELS.map(card).join("")
+ `
All ${MODELS.length} models give the same top answer on ${allAgree} of ${answered.length} questions. Confidence uses the measure selected above; changing it or the threshold updates every number here.
How safe and useful each model is when an agent acts on its answers. Weighted toward being right when it acts and flagging its own mistakes; high-stakes questions count double.
${rankItems}
Component
Max points
${modelHeads()}
${Object.keys(AUS_WEIGHTS).map((k) => {
const max = AUS_WEIGHTS[k] * 100;
const pts = Object.fromEntries(MODELS.map((m) => [m.key, max * S[m.key].parts[k]]));
const w = best(pts, true, 0.05);
return `
${AUS_LABELS[k]}
${max.toFixed(0)}
${MODELS.map((m) =>
`
${pts[m.key].toFixed(1)}${Math.round(100 * S[m.key].parts[k])}% of max
`).join("")}
`;
}).join("")}
Agentic Use Score
100
${MODELS.map((m) =>
`
${S[m.key].aus.toFixed(1)}
`).join("")}
`;
const sum = $("summary");
sum.hidden = false;
sum.innerHTML = aus + MODELS.map(card).join("")
+ groupBreakdown(rows, thr)
+ `
Highlighted values are the best model on that metric (none are highlighted when all tie). All ${MODELS.length} models give the same top answer on ${allAgree} of ${answered.length} questions. Confidence uses the measure selected above; changing it or the threshold updates every number here.
`;
}).join("");
}
// DL1: the summary paints in the current frame; the long row table follows in its own task, so no single
// task has to rebuild everything. A newer request cancels a pending row render.
let rowsTimer = 0;
function renderBoard() {
if (!board) return;
renderSummary();
clearTimeout(rowsTimer);
rowsTimer = setTimeout(renderScoreRows, 0);
}
// ------------------------------------------------------------------ wiring
// DL1: the slider can fire faster than a frame; update the number at once and redraw at most once per frame.
let rerenderFrame = 0;
function onThresholdChange() {
$("thr-out").textContent = threshold().toFixed(2);
if (rerenderFrame) return;
rerenderFrame = requestAnimationFrame(() => {
rerenderFrame = 0;
renderVerdicts();
renderBoard();
});
}
$("run").addEventListener("click", runCurrent);
$("run-all").addEventListener("click", runAll);
$("threshold").addEventListener("input", onThresholdChange);
$("measure").addEventListener("change", onThresholdChange);
document.addEventListener("keydown", (e) => { if ((e.ctrlKey || e.metaKey) && e.key === "Enter") runCurrent(); });
applyModelCount();
renderHero();
renderScoreHead();
pollStatus();
loadDemos();