ai-systems's picture
Upload 2 files
48f7db7 verified
Raw History Blame Contribute Delete
14.1 kB
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width,initial-scale=1">
<meta name="description" content="Explore AI Observability across LLMs, agents, tools, memory, retrieval, cost and evaluation.">
<title>Observability Explorer</title>
<style>
:root{
--bg:#f6fbff;--panel:#fff;--line:#d2e3ee;--text:#102033;--muted:#667b90;
--cyan:#00b7e4;--blue:#3477ff;--violet:#735cff;
}
*{box-sizing:border-box}
body{
margin:0;
background:
radial-gradient(circle at 14% 0,rgba(0,183,228,.14),transparent 28%),
radial-gradient(circle at 88% 0,rgba(115,92,255,.10),transparent 24%),
var(--bg);
color:var(--text);
font:15px/1.55 Inter,system-ui,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif;
}
.wrap{max-width:1180px;margin:auto;padding:0 22px}
.hero{text-align:center;padding:62px 0 28px}
.kicker{display:inline-block;border:1px solid #bddce8;border-radius:999px;padding:7px 12px;background:#ffffffdc;color:#1687a8;font-size:11px;font-weight:900;text-transform:uppercase;letter-spacing:.1em}
h1{font-size:clamp(44px,8vw,76px);line-height:1;margin:18px 0 12px;letter-spacing:-.055em}
.grad{background:linear-gradient(90deg,var(--cyan),var(--blue),var(--violet));background-clip:text;-webkit-background-clip:text;color:transparent}
.hero p{max-width:820px;margin:auto;color:var(--muted);font-size:18px}
.controls{display:flex;gap:8px;flex-wrap:wrap;padding:18px 0 22px}
.controls button{
border:1px solid var(--line);border-radius:999px;background:#fff;color:var(--text);
padding:9px 12px;cursor:pointer;font-size:11px;font-weight:800
}
.controls button.on{border-color:var(--cyan);box-shadow:0 0 0 3px rgba(0,183,228,.1);background:#f2fbff}
.layout{display:grid;grid-template-columns:1.1fr .9fr;gap:15px;padding-bottom:50px}
.panel{border:1px solid var(--line);border-radius:18px;background:#ffffffef;padding:18px;box-shadow:0 18px 44px rgba(38,76,112,.08)}
.ey{font-size:10px;color:#1687a8;font-weight:900;text-transform:uppercase;letter-spacing:.08em}
h2{margin:3px 0 7px;font-size:25px;letter-spacing:-.03em}
.desc{color:var(--muted);font-size:13px}
.trace{
margin:16px 0;padding:15px;border:1px solid #d6e6ef;border-radius:14px;background:#f9fcff;
display:flex;align-items:center;gap:7px;flex-wrap:wrap
}
.node{border:1px solid #bfd9e6;border-radius:9px;background:#fff;padding:8px 10px;font-size:10px;font-weight:900}
.node.focus{border-color:var(--violet);background:#f4f1ff}
.arrow{color:var(--cyan);font-weight:900}
.cards{display:grid;grid-template-columns:repeat(2,1fr);gap:9px}
.card{border:1px solid #d5e4ed;border-radius:12px;background:#fbfdff;padding:11px}
.card b{display:block;font-size:11px}.card span{font-size:10px;color:var(--muted)}
.tags{display:flex;flex-wrap:wrap;gap:6px;margin-top:8px}
.tag{border:1px solid #cfe0eb;border-radius:999px;background:#f6fbff;padding:5px 7px;font-size:10px;color:#50697d}
.metric{margin:11px 0}
.mt{display:flex;justify-content:space-between;font-size:11px;font-weight:800}
.bar{height:7px;background:#e6eef4;border-radius:99px;overflow:hidden;margin-top:5px}
.fill{height:100%;background:linear-gradient(90deg,var(--cyan),var(--blue),var(--violet))}
.failure{margin-top:14px;padding:12px;border:1px solid #e3d9f6;border-radius:12px;background:#fbf8ff}
.failure b{display:block;font-size:11px}.failure span{font-size:10px;color:var(--muted)}
.note{margin-top:14px;padding-top:12px;border-top:1px solid var(--line);color:var(--muted);font-size:11px}
footer{border-top:1px solid var(--line);padding:28px 0 44px;color:var(--muted);font-size:13px}
footer strong{color:var(--text)}
@media(max-width:900px){.layout{grid-template-columns:1fr}}
@media(max-width:560px){.cards{grid-template-columns:1fr}}
</style>
</head>
<body>
<header class="hero">
<div class="wrap">
<span class="kicker">AI Observability · Tracing · Telemetry</span>
<h1>Observability <span class="grad">Explorer</span></h1>
<p>Explore the signals that make modern AI systems traceable, measurable and debuggable.</p>
</div>
</header>
<main class="wrap">
<section class="controls" id="controls"></section>
<section class="layout">
<div class="panel">
<div class="ey" id="category"></div>
<h2 id="name"></h2>
<div class="desc" id="desc"></div>
<div class="trace" id="trace"></div>
<div class="cards" id="cards"></div>
<div class="ey" style="margin-top:16px">Capture these signals</div>
<div class="tags" id="signals"></div>
</div>
<aside class="panel">
<div class="ey">Operational profile</div>
<h2 style="font-size:21px">Why it matters</h2>
<div class="desc">Conceptual importance for production AI systems.</div>
<div id="metrics"></div>
<div class="failure">
<b>Typical failure mode</b>
<span id="failure"></span>
</div>
<div class="note">
Scores are educational and conceptual, not benchmark results.
</div>
</aside>
</section>
</main>
<footer>
<div class="wrap">
<strong>Observability Explorer</strong> — an independent Hugging Face Space.<br>
Collaboration and partnerships: <strong>agenten@magenta.de</strong>
</div>
</footer>
<script>
const topics=[
{
n:"Traces",cat:"Core Signal",
d:"Connect the full execution path of a request across models, tools, retrieval, memory and verification.",
flow:["Request","Router","Model","Tool","Verifier","Response"],
cards:[
["Primary use","End-to-end debugging and root-cause analysis."],
["Best for","Distributed AI workflows and agents."],
["Key question","What happened, in what order, and where did it fail?"],
["Related","Spans, correlation IDs, context propagation."]
],
signals:["trace_id","span_id","duration","status","component","error"],
m:{ProductionImportance:100,DebugValue:100,CorrelationValue:100,ImplementationComplexity:72},
fail:"Important components are not attached to the same trace, so the execution path cannot be reconstructed."
},
{
n:"Logs",cat:"Core Signal",
d:"Record discrete runtime events such as failures, retries, routing decisions and policy outcomes.",
flow:["Event","Record","Store","Search","Investigate"],
cards:[
["Primary use","Event-level debugging and incident analysis."],
["Best for","Errors, retries, policy events and state changes."],
["Key question","Which discrete event occurred at this moment?"],
["Related","Structured logging, audit logs, event streams."]
],
signals:["timestamp","event","severity","component","message","metadata"],
m:{ProductionImportance:94,DebugValue:96,CorrelationValue:78,ImplementationComplexity:52},
fail:"Logs exist, but they are unstructured or cannot be correlated with the request that caused them."
},
{
n:"Metrics",cat:"Core Signal",
d:"Aggregate numerical signals for latency, usage, cost, reliability and system health.",
flow:["Measure","Aggregate","Trend","Alert","Investigate"],
cards:[
["Primary use","Operational monitoring and trend detection."],
["Best for","Latency, cost, error rate and throughput."],
["Key question","How is the system behaving over time?"],
["Related","Dashboards, alerts, SLOs, capacity planning."]
],
signals:["latency","errors","tokens","cost","throughput","success_rate"],
m:{ProductionImportance:96,DebugValue:72,CorrelationValue:62,ImplementationComplexity:45},
fail:"Metrics show a regression, but there is no trace or event context to explain its cause."
},
{
n:"LLM Observability",cat:"Model",
d:"Tracks model calls, prompts, outputs, latency, token usage, cost and provider behavior.",
flow:["Prompt","Model Call","Output","Evaluate","Store"],
cards:[
["Primary use","Inspect model behavior and regressions."],
["Best for","Prompt changes, provider changes and model comparisons."],
["Key question","Which model and prompt produced this result?"],
["Related","Prompt management, cost tracking, evaluation."]
],
signals:["model","provider","prompt_version","input_tokens","output_tokens","latency","cost"],
m:{ProductionImportance:98,DebugValue:94,CorrelationValue:88,ImplementationComplexity:58},
fail:"A quality regression occurs, but prompt version, model version or provider metadata was not recorded."
},
{
n:"Agent Observability",cat:"Agent",
d:"Tracks agent steps, tools, memory, planning, retries, recovery and long-horizon execution.",
flow:["Goal","Plan","Tool","Observe","Memory","Verify","Replan"],
cards:[
["Primary use","Understand agent behavior across many steps."],
["Best for","Long-horizon and autonomous workflows."],
["Key question","Why did the agent choose this sequence of actions?"],
["Related","Tool tracing, memory tracing, checkpointing."]
],
signals:["step","tool","memory_read","memory_write","retry","replan","human_intervention"],
m:{ProductionImportance:100,DebugValue:100,CorrelationValue:98,ImplementationComplexity:88},
fail:"The final agent result is wrong, but intermediate actions, memory changes and replanning decisions are missing."
},
{
n:"Retrieval Observability",cat:"Knowledge",
d:"Makes retrieval queries, candidate documents, reranking and selected context visible.",
flow:["Query","Retrieve","Rerank","Select Context","Generate"],
cards:[
["Primary use","Separate retrieval failures from model failures."],
["Best for","RAG and knowledge-grounded applications."],
["Key question","Did the model receive the right context?"],
["Related","Reranking, source attribution, relevance scoring."]
],
signals:["query","documents","scores","reranker","selected_chunks","latency"],
m:{ProductionImportance:92,DebugValue:96,CorrelationValue:86,ImplementationComplexity:65},
fail:"The answer is poor because weak context was retrieved, but only the final model response was logged."
},
{
n:"Tool Observability",cat:"Action",
d:"Tracks tool selection, arguments, execution, results, errors and permission checks.",
flow:["Select Tool","Authorize","Call","Result","Validate"],
cards:[
["Primary use","Inspect external actions performed by agents."],
["Best for","APIs, code execution, databases and enterprise tools."],
["Key question","What action did the agent actually take?"],
["Related","Permissions, schemas, retries, audit logs."]
],
signals:["tool_name","arguments","duration","result","error","permission","retry"],
m:{ProductionImportance:100,DebugValue:98,CorrelationValue:94,ImplementationComplexity:70},
fail:"A tool modifies an external system, but the arguments and authorization decision were not captured."
},
{
n:"Memory Observability",cat:"State",
d:"Tracks memory reads, writes, freshness, provenance and conflicts in persistent agents.",
flow:["Read","Use","Update","Validate","Persist"],
cards:[
["Primary use","Diagnose stale or incorrect agent state."],
["Best for","Persistent and long-horizon agents."],
["Key question","Which stored information influenced this decision?"],
["Related","State management, provenance, checkpoints."]
],
signals:["memory_id","operation","freshness","provenance","confidence","conflict"],
m:{ProductionImportance:94,DebugValue:97,CorrelationValue:91,ImplementationComplexity:78},
fail:"The agent acts on stale memory, but there is no record of what was read or when it was last updated."
},
{
n:"Cost Observability",cat:"Operations",
d:"Tracks the cost contribution of models, tools, retrieval, embeddings, retries and agent steps.",
flow:["Call","Meter","Attribute","Aggregate","Optimize"],
cards:[
["Primary use","Understand where AI spend originates."],
["Best for","Multi-model and agentic applications."],
["Key question","Which part of the workflow drives cost?"],
["Related","Budgets, routing, model selection, optimization."]
],
signals:["model_cost","tool_cost","retrieval_cost","retry_cost","task_cost","budget"],
m:{ProductionImportance:91,DebugValue:70,CorrelationValue:84,ImplementationComplexity:60},
fail:"Total spend increases, but costs cannot be attributed to individual models, tools or retries."
},
{
n:"Evaluation-Aware Observability",cat:"Quality",
d:"Links runtime execution data with quality, validation and verification signals.",
flow:["Execute","Trace","Evaluate","Diagnose","Improve"],
cards:[
["Primary use","Connect quality changes to concrete runtime causes."],
["Best for","Continuous AI system improvement."],
["Key question","Which execution pattern correlates with better or worse quality?"],
["Related","Validation, verification, benchmarks, feedback."]
],
signals:["quality_score","verifier_result","task_success","policy_status","trace_id","feedback"],
m:{ProductionImportance:97,DebugValue:98,CorrelationValue:100,ImplementationComplexity:84},
fail:"Quality metrics exist separately from runtime telemetry, so regressions cannot be traced back to specific execution behavior."
}
];
let selected=0;
const controls=document.getElementById("controls");
function buildButtons(){
controls.innerHTML=topics.map((x,i)=>`<button class="${i===selected?'on':''}" data-i="${i}">${x.n}</button>`).join("");
controls.querySelectorAll("button").forEach(b=>b.onclick=()=>{
selected=Number(b.dataset.i);buildButtons();render();
});
}
function render(){
const x=topics[selected];
category.textContent=x.cat;
name.textContent=x.n;
desc.textContent=x.d;
trace.innerHTML=x.flow.map((v,i)=>`<span class="node ${i===x.flow.length-1?'focus':''}">${v}</span>${i<x.flow.length-1?'<span class="arrow">→</span>':''}`).join("");
cards.innerHTML=x.cards.map(v=>`<div class="card"><b>${v[0]}</b><span>${v[1]}</span></div>`).join("");
signals.innerHTML=x.signals.map(v=>`<span class="tag">${v}</span>`).join("");
const labels={ProductionImportance:"Production importance",DebugValue:"Debug value",CorrelationValue:"Correlation value",ImplementationComplexity:"Implementation complexity"};
metrics.innerHTML=Object.entries(x.m).map(([k,v])=>`
<div class="metric">
<div class="mt"><span>${labels[k]}</span><span>${v}</span></div>
<div class="bar"><div class="fill" style="width:${v}%"></div></div>
</div>`).join("");
failure.textContent=x.fail;
}
buildButtons();render();
</script>
</body>
</html>