agent-observability / index.html
ai-systems's picture
Upload 2 files
1875878 verified
Raw History Blame Contribute Delete
13.5 kB
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width,initial-scale=1">
<meta name="description" content="Explore agent observability across planning, tools, memory, verification, recovery and long-horizon execution.">
<title>Agent Observability</title>
<style>
:root{
--bg:#f6fbff;--panel:#fff;--line:#d2e3ee;--text:#102033;--muted:#667b90;
--cyan:#00b7e4;--blue:#3477ff;--violet:#735cff;
}
*{box-sizing:border-box}
body{
margin:0;
background:
radial-gradient(circle at 14% 0,rgba(0,183,228,.14),transparent 28%),
radial-gradient(circle at 88% 0,rgba(115,92,255,.10),transparent 24%),
var(--bg);
color:var(--text);
font:15px/1.55 Inter,system-ui,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif;
}
.wrap{max-width:1180px;margin:auto;padding:0 22px}
.hero{text-align:center;padding:62px 0 28px}
.kicker{display:inline-block;border:1px solid #bddce8;border-radius:999px;padding:7px 12px;background:#ffffffdc;color:#1687a8;font-size:11px;font-weight:900;text-transform:uppercase;letter-spacing:.1em}
h1{font-size:clamp(44px,8vw,76px);line-height:1;margin:18px 0 12px;letter-spacing:-.055em}
.grad{background:linear-gradient(90deg,var(--cyan),var(--blue),var(--violet));background-clip:text;-webkit-background-clip:text;color:transparent}
.hero p{max-width:820px;margin:auto;color:var(--muted);font-size:18px}
.controls{display:flex;gap:8px;flex-wrap:wrap;padding:18px 0 22px}
.controls button{
border:1px solid var(--line);border-radius:999px;background:#fff;color:var(--text);
padding:9px 12px;cursor:pointer;font-size:11px;font-weight:800
}
.controls button.on{border-color:var(--cyan);box-shadow:0 0 0 3px rgba(0,183,228,.1);background:#f2fbff}
.layout{display:grid;grid-template-columns:1.1fr .9fr;gap:15px;padding-bottom:50px}
.panel{border:1px solid var(--line);border-radius:18px;background:#ffffffef;padding:18px;box-shadow:0 18px 44px rgba(38,76,112,.08)}
.ey{font-size:10px;color:#1687a8;font-weight:900;text-transform:uppercase;letter-spacing:.08em}
h2{margin:3px 0 7px;font-size:25px;letter-spacing:-.03em}
.desc{color:var(--muted);font-size:13px}
.flow{margin:16px 0;padding:15px;border:1px solid #d6e6ef;border-radius:14px;background:#f9fcff;display:flex;align-items:center;gap:7px;flex-wrap:wrap}
.node{border:1px solid #bfd9e6;border-radius:9px;background:#fff;padding:8px 10px;font-size:10px;font-weight:900}
.node.focus{border-color:var(--violet);background:#f4f1ff}
.arrow{color:var(--cyan);font-weight:900}
.cards{display:grid;grid-template-columns:repeat(2,1fr);gap:9px}
.card{border:1px solid #d5e4ed;border-radius:12px;background:#fbfdff;padding:11px}
.card b{display:block;font-size:11px}.card span{font-size:10px;color:var(--muted)}
.tags{display:flex;flex-wrap:wrap;gap:6px;margin-top:8px}
.tag{border:1px solid #cfe0eb;border-radius:999px;background:#f6fbff;padding:5px 7px;font-size:10px;color:#50697d}
.metric{margin:11px 0}
.mt{display:flex;justify-content:space-between;font-size:11px;font-weight:800}
.bar{height:7px;background:#e6eef4;border-radius:99px;overflow:hidden;margin-top:5px}
.fill{height:100%;background:linear-gradient(90deg,var(--cyan),var(--blue),var(--violet))}
.failure{margin-top:14px;padding:12px;border:1px solid #e3d9f6;border-radius:12px;background:#fbf8ff}
.failure b{display:block;font-size:11px}.failure span{font-size:10px;color:var(--muted)}
.note{margin-top:14px;padding-top:12px;border-top:1px solid var(--line);color:var(--muted);font-size:11px}
footer{border-top:1px solid var(--line);padding:28px 0 44px;color:var(--muted);font-size:13px}
footer strong{color:var(--text)}
@media(max-width:900px){.layout{grid-template-columns:1fr}}
@media(max-width:560px){.cards{grid-template-columns:1fr}}
</style>
</head>
<body>
<header class="hero">
<div class="wrap">
<span class="kicker">Agent Tracing · Memory · Tools · Recovery</span>
<h1>Agent <span class="grad">Observability</span></h1>
<p>Trace how advanced AI agents plan, act, use tools, update memory, recover and reach final outcomes.</p>
</div>
</header>
<main class="wrap">
<section class="controls" id="controls"></section>
<section class="layout">
<div class="panel">
<div class="ey" id="category"></div>
<h2 id="name"></h2>
<div class="desc" id="desc"></div>
<div class="flow" id="flow"></div>
<div class="cards" id="cards"></div>
<div class="ey" style="margin-top:16px">Recommended fields</div>
<div class="tags" id="fields"></div>
</div>
<aside class="panel">
<div class="ey">Operational profile</div>
<h2 style="font-size:21px">Observability value</h2>
<div class="desc">Conceptual importance for production agent systems.</div>
<div id="metrics"></div>
<div class="failure">
<b>Typical blind spot</b>
<span id="failure"></span>
</div>
<div class="note">Conceptual values only; not benchmark scores.</div>
</aside>
</section>
</main>
<footer>
<div class="wrap">
<strong>Agent Observability</strong> — an independent Hugging Face Space.<br>
Collaboration and partnerships: <strong>agenten@magenta.de</strong>
</div>
</footer>
<script>
const topics=[
{
n:"Goal Tracing",cat:"Intent",
d:"Preserves the original objective, constraints and success criteria across the full agent execution.",
flow:["Goal","Constraints","Plan","Action","Outcome"],
cards:[
["Why it matters","Makes goal drift detectable."],
["Best for","Long-horizon and autonomous agents."],
["Key question","Is the current behavior still aligned with the original objective?"],
["Related","Planning, verification, human oversight."]
],
fields:["goal_id","objective","constraints","success_criteria","budget"],
m:{DebugValue:92,LongHorizonValue:100,AuditValue:95,ImplementationComplexity:50},
fail:"The agent remains active, but no one can determine whether the current actions still serve the original goal."
},
{
n:"Plan Tracing",cat:"Planning",
d:"Tracks plan creation, task decomposition, revisions and dependency changes.",
flow:["Create Plan","Execute","Observe","Revise","Continue"],
cards:[
["Why it matters","Shows how execution structure changes over time."],
["Best for","Planner-executor and hierarchical agents."],
["Key question","When and why did the plan change?"],
["Related","Task graphs, replanning, goal stability."]
],
fields:["plan_id","plan_version","tasks","dependencies","replan_reason"],
m:{DebugValue:98,LongHorizonValue:100,AuditValue:92,ImplementationComplexity:70},
fail:"The final result is wrong, but the plan revisions that led there were never captured."
},
{
n:"Step Tracing",cat:"Execution",
d:"Captures each discrete agent step, status, duration and output.",
flow:["Step 1","Step 2","Step 3","Step 4","Complete"],
cards:[
["Why it matters","Provides a chronological execution history."],
["Best for","Agent loops and multi-step workflows."],
["Key question","Which exact step caused the failure?"],
["Related","Traces, checkpoints, retries."]
],
fields:["step_id","step_number","action","status","duration","output"],
m:{DebugValue:100,LongHorizonValue:98,AuditValue:100,ImplementationComplexity:58},
fail:"Only the final response is visible, so intermediate errors cannot be isolated."
},
{
n:"Tool Tracing",cat:"Action",
d:"Tracks tool selection, arguments, execution results, permissions and retries.",
flow:["Select Tool","Authorize","Call","Result","Validate"],
cards:[
["Why it matters","External actions must be reconstructable."],
["Best for","Browser, API, code and enterprise tools."],
["Key question","What did the agent actually do outside the model?"],
["Related","Permissions, tool registry, retries."]
],
fields:["tool_name","arguments","permission","result","error","retry_count"],
m:{DebugValue:100,LongHorizonValue:96,AuditValue:100,ImplementationComplexity:72},
fail:"A tool changed an external system, but the action and authorization context are missing."
},
{
n:"Memory Tracing",cat:"State",
d:"Captures what the agent reads, writes, updates and retrieves from persistent memory.",
flow:["Read","Use","Update","Validate","Persist"],
cards:[
["Why it matters","Explains which stored information influenced behavior."],
["Best for","Persistent and long-horizon agents."],
["Key question","Did stale or wrong memory cause the decision?"],
["Related","State management, provenance, freshness."]
],
fields:["memory_id","operation","source","freshness","confidence","conflict"],
m:{DebugValue:99,LongHorizonValue:100,AuditValue:94,ImplementationComplexity:80},
fail:"The agent acts on stale memory, but there is no record of what was retrieved."
},
{
n:"Verification Tracing",cat:"Quality",
d:"Records which checks were run, what evidence was used and whether the result passed.",
flow:["Output","Verifier","Evidence","Decision","Continue"],
cards:[
["Why it matters","Makes correctness gates visible."],
["Best for","High-reliability and high-impact workflows."],
["Key question","Was this action independently checked?"],
["Related","Validation, evaluation, human approval."]
],
fields:["verifier","evidence","result","failure_reason","next_action"],
m:{DebugValue:95,LongHorizonValue:98,AuditValue:100,ImplementationComplexity:74},
fail:"A wrong intermediate result is accepted, and there is no evidence showing whether verification ran."
},
{
n:"Recovery Tracing",cat:"Resilience",
d:"Tracks retries, fallback paths, checkpoint restores, replanning and escalation.",
flow:["Failure","Diagnose","Recover","Verify","Resume"],
cards:[
["Why it matters","Shows whether the agent can recover intelligently."],
["Best for","Long-running production agents."],
["Key question","How did the system respond after failure?"],
["Related","Checkpoints, fallbacks, retries, replanning."]
],
fields:["failure_type","recovery_action","retry_count","fallback","checkpoint","outcome"],
m:{DebugValue:100,LongHorizonValue:100,AuditValue:92,ImplementationComplexity:78},
fail:"The system retries repeatedly, but recovery behavior is not distinguishable from normal execution."
},
{
n:"Human Approval Tracing",cat:"Oversight",
d:"Records approval requests, decisions, approvers and the high-impact action being controlled.",
flow:["Propose","Risk Check","Human Review","Approve / Reject"],
cards:[
["Why it matters","Creates an auditable boundary around autonomy."],
["Best for","Financial, publishing, deployment and privileged actions."],
["Key question","Who approved this action and why?"],
["Related","Permissions, policy, governance."]
],
fields:["approval_id","action","risk_level","approver","decision","timestamp"],
m:{DebugValue:84,LongHorizonValue:90,AuditValue:100,ImplementationComplexity:62},
fail:"A consequential action executes, but the approval event cannot be reconstructed."
},
{
n:"Cost Tracing",cat:"Operations",
d:"Attributes cost to specific agent steps, tools, models, retries and verification calls.",
flow:["Step","Model / Tool","Meter","Attribute","Aggregate"],
cards:[
["Why it matters","Long-horizon agents can accumulate hidden cost."],
["Best for","Multi-model agent systems."],
["Key question","Which step or retry drives task cost?"],
["Related","Budgets, routing, optimization."]
],
fields:["step_cost","model_cost","tool_cost","retry_cost","total_cost","budget"],
m:{DebugValue:76,LongHorizonValue:96,AuditValue:88,ImplementationComplexity:60},
fail:"Total spend rises, but costs cannot be linked to individual steps or retries."
},
{
n:"Drift Detection",cat:"Long-Horizon",
d:"Uses execution telemetry to detect repeated actions, plan drift, goal drift and runaway loops.",
flow:["Trace","Compare","Detect Drift","Alert","Correct"],
cards:[
["Why it matters","Long tasks amplify small deviations."],
["Best for","Persistent autonomous agents."],
["Key question","Is the agent still making meaningful progress?"],
["Related","Goal tracing, plan tracing, memory tracing."]
],
fields:["repeat_rate","goal_alignment","plan_changes","step_count","progress_score"],
m:{DebugValue:98,LongHorizonValue:100,AuditValue:90,ImplementationComplexity:88},
fail:"The agent keeps working but makes little progress because drift is not detected."
}
];
let selected=0;
const controls=document.getElementById("controls");
function buttons(){
controls.innerHTML=topics.map((x,i)=>`<button class="${i===selected?'on':''}" data-i="${i}">${x.n}</button>`).join("");
controls.querySelectorAll("button").forEach(b=>b.onclick=()=>{selected=Number(b.dataset.i);buttons();render();});
}
function render(){
const x=topics[selected];
category.textContent=x.cat;
name.textContent=x.n;
desc.textContent=x.d;
flow.innerHTML=x.flow.map((v,i)=>`<span class="node ${i===x.flow.length-1?'focus':''}">${v}</span>${i<x.flow.length-1?'<span class="arrow">→</span>':''}`).join("");
cards.innerHTML=x.cards.map(v=>`<div class="card"><b>${v[0]}</b><span>${v[1]}</span></div>`).join("");
fields.innerHTML=x.fields.map(v=>`<span class="tag">${v}</span>`).join("");
const labels={DebugValue:"Debug value",LongHorizonValue:"Long-horizon value",AuditValue:"Audit value",ImplementationComplexity:"Implementation complexity"};
metrics.innerHTML=Object.entries(x.m).map(([k,v])=>`
<div class="metric"><div class="mt"><span>${labels[k]}</span><span>${v}</span></div><div class="bar"><div class="fill" style="width:${v}%"></div></div></div>`).join("");
failure.textContent=x.fail;
}
buttons();render();
</script>
</body>
</html>