"""Gradio dashboard: the live Stateless state-machine graph is the centrepiece.
Everything shown is folded from the harness's EVENT lines (see server.Live) and rendered server-side on a timer, so there is
no custom JavaScript to keep in sync. The graph itself is drawn from the machine's own GetInfo() export, not hard-coded.
"""
import io, json
from html import escape
from pathlib import Path
import gradio as gr
import numpy as np
from fastapi import HTTPException
from PIL import Image
COLORS = {"Queued": "#8b93a1", "Inferring": "#4f8cff", "Auditing": "#38bdf8", "Enforcing": "#a78bfa", "Correct": "#34d399", "WrongButSafe": "#2dd4bf",
"Overblocked": "#fbbf24", "Unsafe": "#f87171", "Misclassified": "#fb923c", "Errored": "#f472b6"}
STATES = list(COLORS) # order must match server.TEST_STATES (the compact per-test state codes)
GROUPS = [("activated", ["Inferring", "Auditing", "Enforcing"]), ("passed", ["Correct", "WrongButSafe"]),
("failed", ["Overblocked", "Unsafe", "Misclassified", "Errored"]), ("queued", ["Queued"])]
CRUMBS = ["Pending", "ModelReady", "SessionLoaded", "WarmedUp", "Measured", "Scored"]
# Hand-placed layout for the known test machine. Unknown states fall into a spare row so a machine change never hides a node.
POS = {"Queued": (60, 190), "Inferring": (200, 190), "Auditing": (340, 190), "Enforcing": (480, 190), "Correct": (670, 50), "WrongButSafe": (670, 120),
"Overblocked": (670, 215), "Unsafe": (670, 285), "Misclassified": (670, 350), "Errored": (830, 285)}
W, H = 118, 46
TXT = "var(--body-text-color)"
MUT = "var(--body-text-color-subdued)"
def graph_svg(graph, counts, edges):
if not graph:
return f'
Waiting for the harness to report its state machine — start a run from the Run tab.
'
states = graph["states"]
pos, spare = dict(POS), 0
for s in states:
if s["id"] not in pos and not any(x["parent"] == s["id"] for x in states):
pos[s["id"]] = (70 + 120 * spare, 400); spare += 1
boxes = {}
for cid in {s["parent"] for s in states if s["parent"]}:
kids = [k for k in states if k["parent"] == cid and k["id"] in pos]
if kids:
xs, ys = [pos[k["id"]][0] for k in kids], [pos[k["id"]][1] for k in kids]
boxes[cid] = (min(xs) - W / 2 - 14, min(ys) - H / 2 - 24, max(xs) + W / 2 + 14, max(ys) + H / 2 + 14)
def centre(i):
if i in pos: return pos[i]
b = boxes[i]; return ((b[0] + b[2]) / 2, (b[1] + b[3]) / 2)
def anchor(i, toward):
cx, cy = centre(i)
w, h = (W, H) if i in pos else (boxes[i][2] - boxes[i][0], boxes[i][3] - boxes[i][1])
dx, dy = toward[0] - cx, toward[1] - cy
k = min((w / 2) / max(abs(dx), 1e-9), (h / 2) / max(abs(dy), 1e-9))
return cx + dx * k, cy + dy * k
out = ['")
return "".join(out)
def dataset_html(v, registry):
"""One card per registered dataset with live progress; the dataset being processed right now is highlighted.
A test's suite comes from its domain's policy pack (packDomains) or is 'classification'; the registry entry whose `provides`
lists that suite is its dataset. Tests run in file order (suite by suite), so the active dataset is the one holding the furthest
started test."""
plan, states = v["plan"], v["states"]
if not plan:
return f'
No run yet.
'
packs = plan.get("packDomains", {})
by_suite = {sn: e for e in registry for sn in e["provides"]}
ds_of = []
for k in plan["keys"]:
dom = k.split("|")[1]
suite = packs.get(dom) if isinstance(packs.get(dom), str) else ("automation" if dom in packs else "classification")
ds_of.append(by_suite.get(suite, {}).get("id", "?"))
info = {e["id"]: e for e in registry}
tot, done = {}, {}
for i, d in enumerate(ds_of):
tot[d] = tot.get(d, 0) + 1
if states[i] not in (0, 1, 2, 3): # not Queued / Inferring / Auditing / Enforcing
done[d] = done.get(d, 0) + 1
started = [i for i in range(len(states)) if states[i] != 0]
cur = ds_of[max(started)] if started else ds_of[0]
measuring = v["runState"] == "Measured" # the run machine's state while tests are being processed
cards = []
for e in registry:
name = e["id"]
if name not in tot:
continue
n, dn = tot[name], done.get(name, 0)
is_active = measuring and name == cur
state = "processing now" if is_active else ("done" if dn == n else "waiting")
border = "#4f8cff" if is_active else "var(--border-color-primary)"
glow = "box-shadow:0 0 0 3px #4f8cff55;" if is_active else ""
badge = {"processing now": "background:#4f8cff;color:#fff", "done": "background:#34d399;color:#fff", "waiting": f"border:1px solid {MUT};color:{MUT}"}[state]
cards.append(
f'
'
f'
{escape(e["title"])}'
f'{state}
'
f'
{escape(e.get("what", ""))}
'
f'
{" · ".join(f"{escape(r.strip())}" for r in e.get("repo", "").split(",") if r.strip())} · '
f'suites: {", ".join(e["provides"])} · {escape(e.get("license", ""))}
'
f'
'
f'
{dn}/{n} tests
')
head = (f'
Now processing: {escape(info[cur]["title"])} '
f'({", ".join(info[cur]["provides"])})
') if measuring and cur in info else ""
return head + f'
{"".join(cards)}
'
def checks_html(listing, results):
"""Cedar checks: status, summary, findings and the check-specific tables."""
if not listing:
return f'
Check list unavailable.
'
out = []
for c in listing:
r = results.get(c["id"])
if r is None:
badge, col = "not run", MUT
elif r["passed"]:
badge, col = ("PASS" if c["hard"] else "done"), "#34d399"
else:
badge, col = "FAIL", "#f87171"
body = ""
if r:
d = r.get("details")
if c["id"] == "conformance" and d:
body += "
")
if c["id"] == "promotion-gate" and d:
body += "
candidate
champion
comparable
decision
by
" + "".join(
f'
{escape(x["candidate"])}
{escape(x["champion"])}
{x["comparable"]}
{"allow" if x["allow"] else "deny"}
{escape(", ".join(x["by"]))}
' for x in d) + "
"
if r["findings"]:
body += "
findings
" + "".join(f"
{escape(f)}
" for f in r["findings"][:15]) + "
"
out.append(
f'
'
f'
{escape(c["title"])}'
f'{badge}{" · gate" if c["hard"] else ""}
'
f'
{escape(c["description"])}
'
+ (f'
{escape(r["summary"])} ({r["seconds"]}s)
' if r else "") + body + "
")
return "".join(out)
def status_html(v, job):
rs = v["runState"] or "Pending"; i = CRUMBS.index(rs) if rs in CRUMBS else -1
chips = []
for k, c in enumerate(CRUMBS):
col = "#f87171" if rs == "Failed" else ("#34d399" if k < i else "#4f8cff" if k == i else MUT)
fill = "background:" + col + ";color:#fff;" if (k == i and rs != "Failed") else ""
chips.append(f'{c}')
if rs == "Failed":
chips.append('Failed')
pol = "".join(f''
f'{"✓" if p["allow"] else "✗"} Cedar {p["action"]} '
f'{escape(", ".join(p["by"]) or "no permit matched")}' for p in v["runPolicy"])
line = "no job yet"
if job:
p, e = job.get("progress") or {}, job.get("eta") or {}
unk = f" (+ unknown: {', '.join(e['queued_unknown'])})" if e.get("queued_unknown") else ""
line = (f'{job["status"]} · {job.get("stage") or "idle"} · {p.get("phase", "")} {p.get("done", "")}/{p.get("total", "")} '
f'({job.get("percent", 0)}%) · mean {p.get("meanMs", "?")} ms · elapsed {round(job["elapsed_s"])}s · '
f'ETA this model {e.get("current_s", "?")}s · queued {e.get("queued_s", 0)}s{unk} · done: {", ".join(job["done_models"]) or "none"}')
if job["errors"]:
line += " · errors: " + " | ".join(f"{k}: {escape(str(x))}" for k, x in job["errors"].items())
pct = (job or {}).get("percent") or 0
return (f'
{escape(v["model"] or "")}
{"".join(chips)}
'
f'
'
f'
{line}
{pol}
')
def legend_html(counts):
parts = []
for g, ss in GROUPS:
inner = " ".join(f' {s} {counts.get(s, 0)}' for s in ss)
parts.append(f'{g} {sum(counts.get(s, 0) for s in ss)} {inner}')
return "".join(parts)
CELL, GAP, COLS = 10, 2, 60
def grid_image(states, n, flagged=b""):
"""One cell per test, coloured by state. Returns (PIL image, cols) — cell (i) sits at column i % COLS, row i // COLS."""
if not n:
return None
rows = -(-n // COLS)
step = CELL + GAP
img = np.zeros((rows * step, COLS * step, 3), dtype=np.uint8)
img[:] = (24, 26, 33)
pal = [tuple(int(COLORS[s][k:k + 2], 16) for k in (1, 3, 5)) for s in STATES]
for i in range(n):
y, x = (i // COLS) * step, (i % COLS) * step
img[y:y + CELL, x:x + CELL] = pal[states[i]]
if i < len(flagged) and flagged[i]: # oracle-flagged: a white centre dot
img[y + 3:y + CELL - 3, x + 3:x + CELL - 3] = (255, 255, 255)
return Image.fromarray(img)
def oracle_rows(rules):
"""Live view of the label-free oracle rules: how often each flags, how often a flag was a real error, and whether it also fires on gold."""
rows = []
for rule, o in sorted(rules.items(), key=lambda kv: -kv[1]["flagged"]):
prec = f'{100 * o["tp"] / o["flagged"]:.0f}%' if o["flagged"] else "–"
rows.append([rule, o["flagged"], prec, o["gold"]])
return rows
def policy_rows(hits):
return [[k, h["allow"], h["deny"]] for k, h in sorted(hits.items(), key=lambda kv: -(kv[1]["allow"] + kv[1]["deny"]))]
def feed_rows(recent):
rows = []
for r in reversed(recent):
flips = "; ".join(f'{a["a"]}: model {"allow" if a["p"] else "deny"} / gold {"allow" if a["g"] else "deny"}' for a in (r.get("acts") or []) if a["p"] != a["g"])
heads = ", ".join(f'{h["t"]}: {h["p"]} ({h["c"]}%) vs {h["g"]}' for h in (r.get("heads") or []))
rows.append([r["i"], r["to"], r["k"], heads, flips or (r.get("err") or "")])
return rows
def domain_rows(v):
plan, states = v["plan"], v["states"]
if not plan:
return []
doms, packs = {}, plan.get("packDomains", {})
for i, k in enumerate(plan["keys"]):
d = k.split("|")[1]
row = doms.setdefault(d, [0] * len(STATES))
row[states[i]] += 1
return [[d, packs.get(d) if isinstance(packs.get(d), str) else ("automation" if d in packs else "classification"), *doms[d]] for d in sorted(doms)]
def build_ui(srv):
"""Build the Blocks app against the server module `srv` (LIVE, jobs, start_job, ranking...)."""
registry = srv.suite_registry()
defaults = srv.policy_params()
try:
listing = srv.checks_endpoint()["checks"]
except HTTPException:
listing = []
def latest_job():
return list(srv._jobs.values())[-1] if srv._jobs else None
def tick():
v = srv.LIVE.view()
job = latest_job()
snap = srv.snapshot(job) if job else None
counts = v["counts"]
img = grid_image(v["states"], v["plan"]["tests"], v["flagged"]) if v["plan"] else None
g = v["graphs"].get("test")
return (status_html(v, snap), dataset_html(v, registry), graph_svg(g, counts, v["edges"]), legend_html(counts), img,
policy_rows(v["policyHits"]), feed_rows(v["recent"]), domain_rows(v), checks_html(listing, srv.CHECKS["results"]), oracle_rows(v["oracle"]))
def inspect(i):
try:
i = int(i)
except (TypeError, ValueError):
return "enter a test index"
with srv.LIVE.lock:
ver = srv.LIVE.verdicts.get(i)
return json.dumps(ver, indent=1) if ver else f"no verdict for test {i} (yet)"
def on_select(evt: gr.SelectData):
x, y = evt.index
i = int(y // (CELL + GAP)) * COLS + int(x // (CELL + GAP))
return i, inspect(i)
def start(model, all_models, skip_done, threads, limit, key, suites, *pvals):
need = srv.os.environ.get("API_KEY")
if need and key != need:
return "API key required (Space secret API_KEY)."
overrides = {k: int(v) for k, v in zip(defaults, pvals) if v is not None and int(v) != defaults[k]}
try:
models = list(srv.REGISTRY) if all_models else [model]
r = srv.start_runs(models, int(threads), int(limit), 20, bool(skip_done), list(suites or []), overrides)
return f"started: {json.dumps(r)}"
except HTTPException as e:
return f"not started: {e.detail}"
def start_checks(selected, key):
need = srv.os.environ.get("API_KEY")
if need and key != need:
return "API key required (Space secret API_KEY)."
try:
return f"started: {json.dumps(srv.run_checks_endpoint(srv.CheckRequest(checks=list(selected or []) or None)))}"
except HTTPException as e:
return f"not started: {e.detail}"
def ranking():
try:
return srv.ranking()["markdown"]
except HTTPException as e:
return f"ranking unavailable: {e.detail}"
with gr.Blocks(title="FindAJev live", fill_width=True) as demo:
gr.Markdown("# FindAJev — live test state machines with Cedar policy enforcement\n"
"Each test is a [Stateless](https://github.com/dotnet-state-machine/stateless) machine; [Cedar](https://www.cedarpolicy.com/) "
"decides every action with the model's labels and again with gold labels — a difference is a guardrail failure.")
with gr.Tabs():
with gr.Tab("Live"):
status = gr.HTML()
datasets = gr.HTML()
graph = gr.HTML()
legend = gr.HTML()
with gr.Row():
grid = gr.Image(label="tests (click a cell to inspect)", interactive=False, buttons=[], scale=3)
pol = gr.Dataframe(headers=["Cedar policy (model labels)", "allow", "deny"], interactive=False, scale=2, max_height=420)
gr.Markdown("**Oracle rules** — label-free Cedar audits that flag suspicious predictions (white dot in the grid). *flagged* = how often the rule fired; "
"*precision* = share of flags that were real errors (compare with the base error rate); *on gold* = times the rule also fires on the gold labels "
"(a sound rule ≈ 0).")
oracle = gr.Dataframe(headers=["rule", "flagged", "precision", "on gold"], interactive=False, max_height=260)
with gr.Tab("Failures & inspect"):
feed = gr.Dataframe(headers=["test", "state", "key", "model vs gold", "decision flips"], interactive=False, max_height=420)
with gr.Row():
idx = gr.Number(label="test index", precision=0, scale=1)
btn = gr.Button("Inspect", scale=1)
detail = gr.Code(label="verdict (labels, confidence, every Cedar decision with policy ids)", language="json")
with gr.Tab("By domain"):
dom = gr.Dataframe(headers=["domain", "suite", *STATES], interactive=False, max_height=600)
with gr.Tab("Ranking"):
rank = gr.Markdown()
gr.Button("Refresh").click(ranking, outputs=rank)
with gr.Tab("Cedar checks"):
gr.Markdown("Tests of the Cedar policies and of Cedar itself — independent of any model. **Gate** checks run automatically before every "
"benchmark job and must pass. They use CPU, so they share the single job slot with benchmark runs.")
with gr.Row():
check_pick = gr.CheckboxGroup([(c["title"], c["id"]) for c in listing], value=[c["id"] for c in listing], label="checks (discovered from the harness)")
check_key = gr.Textbox(label="API key", type="password")
run_checks_btn = gr.Button("Run selected checks", variant="primary")
check_out = gr.Textbox(label="result", interactive=False)
checks_view = gr.HTML()
run_checks_btn.click(start_checks, [check_pick, check_key], check_out)
with gr.Tab("Run"):
gr.Markdown(f"Runs are sequential (one at a time) on **{srv.CPUS} CPU(s)**. Models: {', '.join(srv.REGISTRY)}.")
with gr.Row():
model = gr.Dropdown(list(srv.REGISTRY), value=list(srv.REGISTRY)[0], label="model")
threads = gr.Number(value=srv.CPUS, precision=0, label="threads")
limit = gr.Number(value=0, precision=0, label="limit (0 = all)")
suite_pick = gr.CheckboxGroup([(f'{e["title"]} — {", ".join(e["provides"])}', e["id"]) for e in registry],
value=[e["id"] for e in registry], label="suites (from suites.json)")
with gr.Accordion("Cedar policy parameters (defaults from policies/params.json; a changed value gets its own ranking)", open=False):
pnums = [gr.Number(value=v, precision=0, label=k) for k, v in defaults.items()]
with gr.Row():
allm = gr.Checkbox(label="run every model")
skip = gr.Checkbox(label="skip models that already have a result", value=True)
key = gr.Textbox(label="API key", type="password")
go = gr.Button("Start", variant="primary")
out = gr.Textbox(label="result", interactive=False)
go.click(start, [model, allm, skip, threads, limit, key, suite_pick, *pnums], out)
timer = gr.Timer(0.5)
timer.tick(tick, outputs=[status, datasets, graph, legend, grid, pol, feed, dom, checks_view, oracle])
grid.select(on_select, outputs=[idx, detail])
btn.click(inspect, idx, detail)
demo.load(ranking, outputs=rank)
return demo