AgentStateGraph / index.html
seonglae's picture
links: point the paper button at arXiv rather than the workshop OpenReview
77adcae
Raw History Blame Contribute Delete
325 kB
<!DOCTYPE html><html lang="en" data-theme="light" data-toc-auto-collapse="1"> <head><meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1"><link rel="icon" type="image/svg+xml" href="favicon.svg"><title>Automata from Agent Traces</title><meta name="description" content="Automata from Agent Traces: compact finite-state machines extracted from LLM agent traces for failure prediction, next-step prediction, and runtime monitoring."><link rel="canonical" href="http://localhost:4321/"><meta property="og:type" content="article"><meta property="og:title" content="Automata from Agent Traces"><meta property="og:description" content="Automata from Agent Traces: compact finite-state machines extracted from LLM agent traces for failure prediction, next-step prediction, and runtime monitoring."><meta property="og:url" content="http://localhost:4321/"><meta property="og:image" content="https://seongland.com/article/asg/og.png"><meta property="article:published_time" content="Jun. 30, 2026"><meta property="article:author" content="Seonglae Cho"><meta property="article:author" content="Franklin Cardenoso Fernandez"><meta property="article:author" content="Umar Mohammed"><meta property="article:author" content="Zekun Wu"><meta property="article:author" content="Kleyton Da Costa"><meta property="article:author" content="Ilham Wicaksono"><meta property="article:author" content="Adriano Koshiyama"><meta name="twitter:card" content="summary_large_image"><meta name="twitter:title" content="Automata from Agent Traces"><meta name="twitter:description" content="Automata from Agent Traces: compact finite-state machines extracted from LLM agent traces for failure prediction, next-step prediction, and runtime monitoring."><meta name="twitter:image" content="https://seongland.com/article/asg/og.png"><script type="application/ld+json">{"@context":"https://schema.org","@type":"Article","headline":"Automata from Agent Traces","description":"Automata from Agent Traces: compact finite-state machines extracted from LLM agent traces for failure prediction, next-step prediction, and runtime monitoring.","datePublished":"Jun. 30, 2026","author":[{"@type":"Person","name":"Seonglae Cho"},{"@type":"Person","name":"Franklin Cardenoso Fernandez"},{"@type":"Person","name":"Umar Mohammed"},{"@type":"Person","name":"Zekun Wu"},{"@type":"Person","name":"Kleyton Da Costa"},{"@type":"Person","name":"Ilham Wicaksono"},{"@type":"Person","name":"Adriano Koshiyama"}],"keywords":"agents, interpretability, finite-state-machines","mainEntityOfPage":"http://localhost:4321/","image":["https://seongland.com/article/asg/og.png"]}</script><script>
(() => {
try {
const saved = localStorage.getItem("theme");
const prefersDark =
window.matchMedia &&
window.matchMedia("(prefers-color-scheme: dark)").matches;
const theme = saved || (prefersDark ? "dark" : "light");
document.documentElement.setAttribute("data-theme", theme);
} catch {}
})();
</script><link rel="stylesheet" href="/_astro/index.DpTQb45s.css"><script type="module" src="/_astro/hoisted.DzT9C6fG.js"></script></head> <body> <button id="theme-toggle" aria-label="Toggle color theme" data-astro-cid-x3pjskd3> <svg class="icon light" width="20" height="20" viewBox="0 0 24 24" aria-hidden="true" focusable="false" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" data-astro-cid-x3pjskd3> <circle cx="12" cy="12" r="5" data-astro-cid-x3pjskd3></circle> <line x1="12" y1="1" x2="12" y2="4" data-astro-cid-x3pjskd3></line> <line x1="12" y1="20" x2="12" y2="23" data-astro-cid-x3pjskd3></line> <line x1="1" y1="12" x2="4" y2="12" data-astro-cid-x3pjskd3></line> <line x1="20" y1="12" x2="23" y2="12" data-astro-cid-x3pjskd3></line> <line x1="4.22" y1="4.22" x2="6.34" y2="6.34" data-astro-cid-x3pjskd3></line> <line x1="17.66" y1="17.66" x2="19.78" y2="19.78" data-astro-cid-x3pjskd3></line> <line x1="4.22" y1="19.78" x2="6.34" y2="17.66" data-astro-cid-x3pjskd3></line> <line x1="17.66" y1="6.34" x2="19.78" y2="4.22" data-astro-cid-x3pjskd3></line> </svg> <svg class="icon dark" width="20" height="20" viewBox="0 0 24 24" aria-hidden="true" focusable="false" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" data-astro-cid-x3pjskd3> <path d="M21 12.79A9 9 0 1 1 11.21 3 7 7 0 0 0 21 12.79z" data-astro-cid-x3pjskd3></path> </svg> </button> <section class="hero" data-astro-cid-bbe6dxrz> <h1 class="hero-title" data-astro-cid-bbe6dxrz>Automata from Agent Traces</h1> <div class="hero-banner" data-astro-cid-bbe6dxrz> <figure class="html-embed"><div class="html-embed__card is-frameless"><div id="frag-hsvhw4lir2a"><!-- Interactive FSM Hero: flat dashboard-style graph with live trace simulation -->
<div class="fsm-hero">
<div class="hero-pills"></div>
<div class="hero-graph"></div>
<div class="hero-trace-bar"></div>
</div>
<style>
.fsm-hero {
position: relative; width: 100%; height: 520px; overflow: hidden;
background: #f5f0e8; border-radius: 12px; border: 1px solid #d9d0c4;
}
.fsm-hero .hero-graph { position: absolute; inset: 0; z-index: 1; }
.fsm-hero .hero-graph svg { width: 100%; height: 100%; }
/* Dataset pills - warm theme */
.fsm-hero .hero-pills {
position: absolute; top: 14px; right: 18px; z-index: 10;
display: flex; gap: 6px; flex-wrap: wrap; justify-content: flex-end;
}
.fsm-hero .pill {
font-size: 10px; padding: 4px 10px; border-radius: 20px;
border: 1px solid #c4b8a8; background: rgba(245,240,232,0.7);
color: #8a8478; cursor: pointer; transition: all 0.2s ease;
font-family: 'IBM Plex Mono', ui-monospace, monospace; font-weight: 500;
letter-spacing: 0.02em;
}
.fsm-hero .pill:hover { border-color: #b5afa5; color: #4a4540; background: rgba(245,240,232,0.9); }
.fsm-hero .pill.active {
border-color: #3d5a80; background: rgba(61, 90, 128,0.06);
color: #3d5a80; font-weight: 600;
}
/* Trace timeline bar - warm theme */
.fsm-hero .hero-trace-bar {
position: absolute; bottom: 18px; left: 50%; transform: translateX(-50%);
z-index: 10; display: flex; align-items: center; gap: 2px;
padding: 5px 10px; border-radius: 10px;
background: rgba(245,240,232,0.85); backdrop-filter: blur(6px);
border: 1px solid #d9d0c4;
max-width: 90%; overflow-x: auto;
}
.fsm-hero .trace-step {
font-size: 8px; padding: 2px 6px; border-radius: 4px;
color: #b5afa5; white-space: nowrap;
font-family: 'IBM Plex Mono', monospace; font-variant-numeric: tabular-nums;
transition: all 0.4s ease;
}
.fsm-hero .trace-step.active {
color: #3d5a80; background: rgba(61, 90, 128,0.08); font-weight: 600;
}
.fsm-hero .trace-step.visited { color: #8a8478; }
.fsm-hero .trace-arrow {
font-size: 7px; color: #c4b8a8; padding: 0 1px;
}
/* SVG styles */
.fsm-hero .edge-line { fill: none; stroke-linecap: round; }
.fsm-hero .node-circle { cursor: grab; }
.fsm-hero .node-circle:active { cursor: grabbing; }
.fsm-hero .node-label {
font-family: 'IBM Plex Mono', ui-monospace, monospace;
font-size: 8.5px; fill: #4a4540;
text-anchor: middle; pointer-events: none;
font-variant-numeric: tabular-nums;
}
.fsm-hero .trace-particle { pointer-events: none; }
.fsm-hero .hero-tip {
position: absolute; top: 0; left: 0; pointer-events: none; padding: 6px 10px; border-radius: 6px;
font-size: 11px; line-height: 1.4; border: 1px solid #d9d0c4;
background: rgba(245,240,232,0.95); color: #4a4540; backdrop-filter: blur(6px);
opacity: 0; transition: opacity 0.15s; z-index: 50;
font-family: 'IBM Plex Mono', monospace; font-variant-numeric: tabular-nums;
max-width: 220px; box-shadow: 0 2px 8px rgba(0,0,0,0.08);
}
[data-theme="dark"] .fsm-hero .hero-tip {
background: rgba(26,24,20,0.95); color: #b5afa5; border-color: #2e2a24;
box-shadow: 0 2px 8px rgba(0,0,0,0.3);
}
@media (max-width: 600px) {
.fsm-hero { height: 420px; }
.fsm-hero .hero-trace-bar { bottom: 12px; }
}
/* Dark theme override */
[data-theme="dark"] .fsm-hero {
background: #1a1814; border-color: #2e2a24;
}
[data-theme="dark"] .fsm-hero .pill {
border-color: #3a3530; background: rgba(26,24,20,0.7); color: #8a8478;
}
[data-theme="dark"] .fsm-hero .pill:hover {
border-color: #4a4540; color: #b5afa5; background: rgba(26,24,20,0.9);
}
[data-theme="dark"] .fsm-hero .pill.active {
border-color: #3d5a80; background: rgba(61, 90, 128,0.12); color: #3d5a80;
}
[data-theme="dark"] .fsm-hero .hero-trace-bar {
background: rgba(26,24,20,0.85); border-color: #2e2a24;
}
[data-theme="dark"] .fsm-hero .trace-step { color: #5a5550; }
[data-theme="dark"] .fsm-hero .trace-step.active { color: #3d5a80; background: rgba(61, 90, 128,0.15); }
[data-theme="dark"] .fsm-hero .trace-step.visited { color: #8a8478; }
[data-theme="dark"] .fsm-hero .trace-arrow { color: #3a3530; }
[data-theme="dark"] .fsm-hero .node-label { fill: #b5afa5; }
</style>
<script>
(() => {
const ensureD3 = (cb) => {
if (window.d3 && typeof window.d3.select === 'function') return cb();
let s = document.getElementById('d3-cdn-script');
if (!s) { s = document.createElement('script'); s.id = 'd3-cdn-script'; s.src = 'https://cdn.jsdelivr.net/npm/d3@7/dist/d3.min.js'; document.head.appendChild(s); }
s.addEventListener('load', () => cb(), { once: true });
};
const bootstrap = () => {
const container = document.querySelector('.fsm-hero:not([data-mounted])');
if (!container) return;
container.dataset.mounted = 'true';
const d3 = window.d3;
// --- Color palette (matches dashboard) ---
const RUST = '#3d5a80';
const EDGE_COLOR = '#b5afa5';
const ARROW_COLOR = '#8a8478';
const TEXT_COLOR = '#4a4540';
const INIT_FILL = '#f5f0e8';
const NODE_STROKE = '#b5afa5';
// --- Short label ---
function shortLabel(id) {
if (id === 'init') return 'q\u2080';
const p = id.split(':');
if (p.length === 3 && p[1] === 'tool') return p[2];
if (p.length === 2 && p[1] === 'text') {
if (['user','assistant','system','tool'].includes(p[0])) return p[0];
return p[0] + '\u21a9';
}
if (p[0] === 'assistant') return p[1];
return p[p.length - 1];
}
// --- Datasets ---
const datasets = {
'Mind2Web': {
states: 8, traces: 500,
nodes: ['init','user:text','assistant:type','assistant:click','assistant:select','assistant:hover','assistant:enter','assistant:text'],
edges: [['init','user:text'],['user:text','assistant:type'],['assistant:type','assistant:click'],['assistant:click','assistant:type'],['assistant:click','assistant:click'],['assistant:click','assistant:select'],['assistant:select','assistant:select'],['assistant:select','assistant:click'],['user:text','assistant:click'],['assistant:click','assistant:hover'],['assistant:type','assistant:type'],['user:text','assistant:hover'],['assistant:hover','assistant:click'],['assistant:hover','assistant:hover'],['assistant:type','assistant:select'],['assistant:type','assistant:hover'],['assistant:type','assistant:enter'],['assistant:enter','assistant:click'],['assistant:select','assistant:type'],['user:text','assistant:select'],['assistant:text','assistant:click']]
},
'tau2-bench': {
states: 6, traces: 1824,
nodes: ['init','system:text','user:text','assistant:text','assistant:tool_call','tool:text'],
edges: [['init','system:text'],['system:text','user:text'],['user:text','assistant:text'],['assistant:text','user:text'],['user:text','assistant:tool_call'],['assistant:tool_call','tool:text'],['tool:text','assistant:tool_call'],['tool:text','assistant:text']]
},
'WebArena': {
states: 25, traces: 8337,
nodes: ['init','assistant:tool:click','click:text','assistant:tool:type','type:text','assistant:tool:select_option','select_option:text','assistant:tool:search','search:text','assistant:tool:hover','hover:text','assistant:tool:scroll_down','scroll_down:text','assistant:tool:quote','quote:text','assistant:tool:go_backward','go_backward:text','assistant:tool:scroll_up','scroll_up:text','assistant:tool:press_enter','press_enter:text','assistant:tool:wait','wait:text','assistant:tool:unknown','unknown:text'],
edges: [['init','assistant:tool:click'],['assistant:tool:click','click:text'],['click:text','assistant:tool:type'],['assistant:tool:type','type:text'],['type:text','assistant:tool:type'],['type:text','assistant:tool:select_option'],['assistant:tool:select_option','select_option:text'],['init','assistant:tool:search'],['assistant:tool:search','search:text'],['search:text','assistant:tool:click'],['click:text','assistant:tool:search'],['click:text','assistant:tool:click'],['type:text','assistant:tool:click'],['init','assistant:tool:type'],['init','assistant:tool:hover'],['assistant:tool:hover','hover:text'],['hover:text','assistant:tool:hover'],['hover:text','assistant:tool:click'],['click:text','assistant:tool:scroll_down'],['assistant:tool:scroll_down','scroll_down:text'],['scroll_down:text','assistant:tool:scroll_down'],['click:text','assistant:tool:quote'],['assistant:tool:quote','quote:text'],['quote:text','assistant:tool:scroll_down'],['scroll_down:text','assistant:tool:quote'],['quote:text','assistant:tool:click'],['select_option:text','assistant:tool:click'],['click:text','assistant:tool:select_option'],['type:text','assistant:tool:scroll_down'],['select_option:text','assistant:tool:type'],['scroll_down:text','assistant:tool:click'],['init','assistant:tool:scroll_down'],['click:text','assistant:tool:go_backward'],['assistant:tool:go_backward','go_backward:text'],['go_backward:text','assistant:tool:click'],['search:text','assistant:tool:scroll_down'],['scroll_down:text','assistant:tool:type'],['go_backward:text','assistant:tool:scroll_down'],['click:text','assistant:tool:scroll_up'],['assistant:tool:scroll_up','scroll_up:text'],['search:text','assistant:tool:select_option'],['select_option:text','assistant:tool:scroll_down'],['select_option:text','assistant:tool:select_option'],['scroll_up:text','assistant:tool:click'],['hover:text','assistant:tool:scroll_down'],['go_backward:text','assistant:tool:scroll_up'],['type:text','assistant:tool:press_enter'],['assistant:tool:press_enter','press_enter:text'],['press_enter:text','assistant:tool:scroll_down'],['scroll_down:text','assistant:tool:scroll_up'],['click:text','assistant:tool:wait'],['assistant:tool:wait','wait:text'],['wait:text','assistant:tool:scroll_down'],['scroll_up:text','assistant:tool:scroll_up'],['search:text','assistant:tool:search'],['scroll_down:text','assistant:tool:unknown'],['assistant:tool:unknown','unknown:text'],['unknown:text','assistant:tool:unknown']]
}
};
let currentDs = 'WebArena';
let traceTimer = null;
// === Pre-computed layout (ported from dashboard computeLayout) ===
function computeLayout(nodes, edges, W, H) {
const n = nodes.length;
if (n === 0) return {};
const cx = W / 2, cy = H / 2;
const pos = {}, vel = {};
const initIdx = nodes.indexOf('init');
const r0 = Math.min(160, n * 22);
nodes.forEach((id, i) => {
const idx = initIdx >= 0 ? (i - initIdx + n) % n : i;
const angle = (idx / n) * 2 * Math.PI - Math.PI / 2;
pos[id] = { x: cx + r0 * Math.cos(angle), y: cy + r0 * Math.sin(angle) };
vel[id] = { x: 0, y: 0 };
});
// Build adjacency
const adj = {};
nodes.forEach(id => adj[id] = new Set());
edges.forEach(([s, t]) => {
if (s !== t) { adj[s].add(t); adj[t].add(s); }
});
// In-degree
const inDeg = {};
edges.forEach(([_, t]) => inDeg[t] = (inDeg[t] || 0) + 1);
// Force simulation - 200 iterations then FREEZE
const REPULSION = 8000, ATTRACTION = 0.008, DAMPING = 0.85, ITERS = 200;
for (let iter = 0; iter < ITERS; iter++) {
const temp = 1 - iter / ITERS;
for (const a of nodes) {
let fx = 0, fy = 0;
for (const b of nodes) {
if (a === b) continue;
const dx = pos[a].x - pos[b].x;
const dy = pos[a].y - pos[b].y;
const d2 = dx * dx + dy * dy + 1;
const f = REPULSION / d2;
const d = Math.sqrt(d2);
fx += f * dx / d;
fy += f * dy / d;
}
const neighbors = adj[a];
if (neighbors) {
for (const bId of neighbors) {
const dx = pos[bId].x - pos[a].x;
const dy = pos[bId].y - pos[a].y;
const d = Math.sqrt(dx * dx + dy * dy);
const idealDist = 100;
const f = ATTRACTION * (d - idealDist);
fx += f * dx / (d + 0.1);
fy += f * dy / (d + 0.1);
}
}
// Center gravity
fx += (cx - pos[a].x) * 0.001;
fy += (cy - pos[a].y) * 0.001;
vel[a].x = (vel[a].x + fx) * DAMPING * temp;
vel[a].y = (vel[a].y + fy) * DAMPING * temp;
}
for (const id of nodes) {
pos[id].x = Math.max(60, Math.min(W - 60, pos[id].x + vel[id].x));
pos[id].y = Math.max(60, Math.min(H - 60, pos[id].y + vel[id].y));
}
}
return pos;
}
// --- Build adjacency for random walks ---
function buildAdj(ds) {
const adj = {};
ds.nodes.forEach(n => adj[n] = []);
ds.edges.forEach(([s, t]) => { if (adj[s]) adj[s].push(t); });
return adj;
}
function randomWalk(ds, len = 16) {
const adj = buildAdj(ds);
const walk = ['init'];
let cur = 'init';
for (let i = 0; i < len; i++) {
const nexts = adj[cur];
if (!nexts || nexts.length === 0) break;
const nonSelf = nexts.filter(n => n !== cur);
if (nonSelf.length > 0 && Math.random() < 0.85) {
cur = nonSelf[Math.floor(Math.random() * nonSelf.length)];
} else {
cur = nexts[Math.floor(Math.random() * nexts.length)];
}
walk.push(cur);
}
return walk;
}
// --- Pills ---
const pillsEl = container.querySelector('.hero-pills');
Object.keys(datasets).forEach(name => {
const pill = document.createElement('div');
pill.className = 'pill' + (name === currentDs ? ' active' : '');
pill.textContent = name;
pill.addEventListener('click', () => {
if (name === currentDs) return;
currentDs = name;
pillsEl.querySelectorAll('.pill').forEach(p => p.classList.remove('active'));
pill.classList.add('active');
renderGraph();
});
pillsEl.appendChild(pill);
});
// --- Trace bar ---
const traceBar = container.querySelector('.hero-trace-bar');
function updateTraceBar(walk, stepIdx) {
traceBar.innerHTML = '';
const maxShow = 14;
const start = Math.max(0, stepIdx - Math.floor(maxShow / 2));
const end = Math.min(walk.length, start + maxShow);
for (let i = start; i < end; i++) {
if (i > start) {
const arrow = document.createElement('span');
arrow.className = 'trace-arrow';
arrow.textContent = '\u2192';
traceBar.appendChild(arrow);
}
const step = document.createElement('span');
step.className = 'trace-step';
step.textContent = shortLabel(walk[i]);
if (i === stepIdx) step.classList.add('active');
else if (i < stepIdx) step.classList.add('visited');
traceBar.appendChild(step);
}
}
// --- Main graph render ---
function renderGraph() {
if (traceTimer) clearTimeout(traceTimer);
const graphEl = container.querySelector('.hero-graph');
graphEl.innerHTML = '';
const ds = datasets[currentDs];
// Use FIXED 600x400 base (matching dashboard exactly), viewBox scales
const W = 600, H = 400;
// Pre-compute layout (STATIC positions - no force physics)
const nodePositions = computeLayout(ds.nodes, ds.edges, W, H);
const svg = d3.select(graphEl).append('svg')
.attr('viewBox', '-20 -10 640 430')
.style('width', '100%').style('height', '100%');
const defs = svg.append('defs');
// Arrow markers (dashboard style)
defs.append('marker').attr('id', 'arrow')
.attr('viewBox', '0 0 10 6').attr('refX', 10).attr('refY', 3)
.attr('markerWidth', 7).attr('markerHeight', 4.5).attr('orient', 'auto')
.append('path').attr('d', 'M0,0 L10,3 L0,6').attr('fill', ARROW_COLOR);
defs.append('marker').attr('id', 'arrow-active')
.attr('viewBox', '0 0 10 6').attr('refX', 10).attr('refY', 3)
.attr('markerWidth', 7).attr('markerHeight', 4.5).attr('orient', 'auto')
.append('path').attr('d', 'M0,0 L10,3 L0,6').attr('fill', RUST);
// Detect bidirectional edges
const edgeSet = new Set(ds.edges.map(([s, t]) => s + '|' + t));
// In-degree for node sizing
const inDeg = {};
ds.edges.forEach(([_, t]) => inDeg[t] = (inDeg[t] || 0) + 1);
const maxDeg = Math.max(1, ...Object.values(inDeg));
function nodeR(id) {
if (id === 'init') return 13;
const visitNorm = (inDeg[id] || 0) / maxDeg;
return 14 + visitNorm * 6;
}
// SVG layers
const edgeLayer = svg.append('g').attr('class', 'edge-layer');
const nodeLayer = svg.append('g').attr('class', 'node-layer');
const particleLayer = svg.append('g').attr('class', 'particle-layer');
// Build link data
const linkData = ds.edges.map(([s, t], i) => ({
source: s, target: t, id: i,
selfLoop: s === t,
bidirectional: edgeSet.has(t + '|' + s),
}));
// Edge path computation (matches dashboard exactly)
function edgePath(d) {
const from = nodePositions[d.source];
const to = nodePositions[d.target];
if (!from || !to) return '';
const r1 = nodeR(d.source);
const r2 = nodeR(d.target);
if (d.selfLoop) {
const r = r1 + 6;
return `M${from.x},${from.y - r} C${from.x - 35},${from.y - r - 40} ${from.x + 35},${from.y - r - 40} ${from.x},${from.y - r}`;
}
const dx = to.x - from.x, dy = to.y - from.y;
const dist = Math.sqrt(dx * dx + dy * dy) || 1;
const nx = dx / dist, ny = dy / dist;
if (d.bidirectional) {
const curve = 25;
const mx = (from.x + to.x) / 2 - ny * curve;
const my = (from.y + to.y) / 2 + nx * curve;
return `M${from.x + nx * (r1 + 2) - ny * 3},${from.y + ny * (r1 + 2) + nx * 3} Q${mx},${my} ${to.x - nx * (r2 + 4) - ny * 3},${to.y - ny * (r2 + 4) + nx * 3}`;
}
return `M${from.x + nx * (r1 + 2)},${from.y + ny * (r1 + 2)} L${to.x - nx * (r2 + 4)},${to.y - ny * (r2 + 4)}`;
}
// Recompute all edges for current positions
function updateEdges() {
edgePaths.attr('d', edgePath);
edgeGlows.attr('d', edgePath);
}
// Draw edges
const edgePaths = edgeLayer.selectAll('.edge-line')
.data(linkData).enter().append('path')
.attr('class', 'edge-line')
.attr('d', edgePath)
.attr('stroke', EDGE_COLOR)
.attr('stroke-width', d => {
const freq = (inDeg[d.target] || 0) / maxDeg;
return 0.5 + freq * 2.5;
})
.attr('stroke-opacity', d => {
const freq = (inDeg[d.target] || 0) / maxDeg;
return 0.3 + freq * 0.5;
})
.attr('marker-end', d => d.selfLoop ? '' : 'url(#arrow)');
// Edge glow overlay (for animation)
const edgeGlows = edgeLayer.selectAll('.edge-glow')
.data(linkData).enter().append('path')
.attr('class', 'edge-glow edge-line')
.attr('d', edgePath)
.attr('stroke', RUST)
.attr('stroke-width', 3)
.attr('opacity', 0)
.attr('marker-end', 'url(#arrow-active)');
// Draw nodes
const nodeGroups = nodeLayer.selectAll('.node-g')
.data(ds.nodes).enter().append('g')
.attr('class', 'node-g')
.attr('transform', id => {
const p = nodePositions[id];
return p ? `translate(${p.x},${p.y})` : '';
});
// Node circles (dashboard style)
nodeGroups.append('circle')
.attr('class', 'node-circle')
.attr('r', id => nodeR(id))
.attr('fill', id => {
if (id === 'init') return INIT_FILL;
const visitNorm = (inDeg[id] || 0) / maxDeg;
return `rgba(61, 90, 128, ${0.04 + visitNorm * 0.12})`;
})
.attr('stroke', id => {
if (id === 'init') return RUST;
const visitNorm = (inDeg[id] || 0) / maxDeg;
return visitNorm > 0.5 ? RUST : NODE_STROKE;
})
.attr('stroke-width', id => {
if (id === 'init') return 2;
const visitNorm = (inDeg[id] || 0) / maxDeg;
return visitNorm > 0.5 ? 1.5 : 1;
});
// Double circle for init
nodeGroups.filter(id => id === 'init').append('circle')
.attr('r', id => nodeR(id) - 3)
.attr('fill', 'none').attr('stroke', RUST).attr('stroke-width', 1);
// Labels (dashboard style)
nodeGroups.append('text')
.attr('class', 'node-label')
.attr('dy', id => nodeR(id) + 12)
.text(id => {
const label = shortLabel(id);
return label.length > 18 ? label.slice(0, 16) + '..' : label;
});
// --- Tooltip ---
let tip = container.querySelector('.hero-tip');
if (!tip) {
tip = document.createElement('div');
tip.className = 'hero-tip';
container.appendChild(tip);
}
tip.style.opacity = '0';
nodeGroups
.on('mouseenter', (ev, id) => {
const deg = inDeg[id] || 0;
const outDeg = ds.edges.filter(([s]) => s === id).length;
tip.innerHTML = '<strong>' + id + '</strong><br/>In: ' + deg + ' | Out: ' + outDeg;
tip.style.opacity = '1';
})
.on('mousemove', (ev) => {
const cr = container.getBoundingClientRect();
const mx = ev.clientX - cr.left + 14;
const my = ev.clientY - cr.top - 14;
tip.style.transform = 'translate(' + mx + 'px, ' + my + 'px)';
})
.on('mouseleave', () => { tip.style.opacity = '0'; });
// --- Drag behavior (reposition nodes, update edges live) ---
const svgEl = svg.node();
const drag = d3.drag()
.on('start', function(ev, id) {
d3.select(this).select('.node-circle').attr('stroke', RUST).attr('stroke-width', 2.5);
})
.on('drag', function(ev, id) {
// Convert screen coords to SVG coords
const pt = svgEl.createSVGPoint();
pt.x = ev.sourceEvent.clientX;
pt.y = ev.sourceEvent.clientY;
const svgPt = pt.matrixTransform(svgEl.getScreenCTM().inverse());
nodePositions[id].x = Math.max(20, Math.min(W - 20, svgPt.x));
nodePositions[id].y = Math.max(20, Math.min(H - 20, svgPt.y));
d3.select(this).attr('transform', 'translate(' + nodePositions[id].x + ',' + nodePositions[id].y + ')');
updateEdges();
})
.on('end', function(ev, id) {
const visitNorm = (inDeg[id] || 0) / maxDeg;
d3.select(this).select('.node-circle')
.attr('stroke', id === 'init' ? RUST : visitNorm > 0.5 ? RUST : NODE_STROKE)
.attr('stroke-width', id === 'init' ? 2 : visitNorm > 0.5 ? 1.5 : 1);
});
nodeGroups.call(drag);
// Trace particle (rust colored)
const particle = particleLayer.append('circle')
.attr('class', 'trace-particle')
.attr('r', 5).attr('fill', RUST)
.attr('opacity', 0);
const particleTrail = particleLayer.append('circle')
.attr('class', 'trace-particle')
.attr('r', 10).attr('fill', RUST)
.attr('opacity', 0);
// --- Trace simulation (50% speed = doubled durations) ---
function runTrace() {
const walk = randomWalk(ds, ds.nodes.length > 15 ? 20 : 14);
let step = 0;
updateTraceBar(walk, 0);
function findEdge(s, t) {
return linkData.find(e => e.source === s && e.target === t);
}
function animateStep() {
if (step >= walk.length) {
particle.attr('opacity', 0);
particleTrail.attr('opacity', 0);
// Reset node strokes
nodeGroups.select('.node-circle')
.transition().duration(400)
.attr('stroke', function(id) {
if (id === 'init') return RUST;
const visitNorm = (inDeg[id] || 0) / maxDeg;
return visitNorm > 0.5 ? RUST : NODE_STROKE;
})
.attr('stroke-width', function(id) {
if (id === 'init') return 2;
const visitNorm = (inDeg[id] || 0) / maxDeg;
return visitNorm > 0.5 ? 1.5 : 1;
});
edgeGlows.transition().duration(600).attr('opacity', 0);
traceTimer = setTimeout(runTrace, 3000);
return;
}
const curId = walk[step];
const curPos = nodePositions[curId];
updateTraceBar(walk, step);
// Highlight current node
nodeGroups.select('.node-circle')
.transition().duration(400)
.attr('stroke', function(id) {
if (id === curId) return RUST;
if (id === 'init') return RUST;
const visitNorm = (inDeg[id] || 0) / maxDeg;
return visitNorm > 0.5 ? RUST : NODE_STROKE;
})
.attr('stroke-width', function(id) {
if (id === curId) return 2.5;
if (id === 'init') return 2;
const visitNorm = (inDeg[id] || 0) / maxDeg;
return visitNorm > 0.5 ? 1.5 : 1;
});
// Animate particle along edge
if (step < walk.length - 1) {
const nextId = walk[step + 1];
const nextPos = nodePositions[nextId];
const edge = findEdge(curId, nextId);
if (edge) {
const edgeIdx = edge.id;
edgeGlows.filter(d => d.id === edgeIdx)
.attr('opacity', 0.6)
.transition().duration(1000).attr('opacity', 0.1);
}
if (curPos && nextPos) {
particle.attr('cx', curPos.x).attr('cy', curPos.y).attr('opacity', 0.9);
particleTrail.attr('cx', curPos.x).attr('cy', curPos.y).attr('opacity', 0.12);
particle.transition().duration(900).ease(d3.easeCubicInOut)
.attr('cx', nextPos.x).attr('cy', nextPos.y);
particleTrail.transition().duration(1000).ease(d3.easeCubicInOut)
.attr('cx', nextPos.x).attr('cy', nextPos.y);
}
}
step++;
traceTimer = setTimeout(animateStep, 1200);
}
traceTimer = setTimeout(animateStep, 800);
}
// Start trace after brief pause
traceTimer = setTimeout(runTrace, 1200);
}
renderGraph();
// Responsive
let resizeTimer;
if (window.ResizeObserver) {
new ResizeObserver(() => {
clearTimeout(resizeTimer);
resizeTimer = setTimeout(renderGraph, 200);
}).observe(container);
}
};
if (document.readyState === 'loading')
document.addEventListener('DOMContentLoaded', () => ensureD3(bootstrap), { once: true });
else ensureD3(bootstrap);
})();
</script>
</div></div></figure> <p class="hero-desc" data-astro-cid-bbe6dxrz>One automaton, built from traces, serves as a memory, prediction, and monitoring prior</p> </div> </section> <header class="meta" aria-label="Article meta information" data-astro-cid-bbe6dxrz> <div class="meta-container" data-astro-cid-bbe6dxrz> <div class="meta-container-cell" data-astro-cid-bbe6dxrz> <h3 data-astro-cid-bbe6dxrz>Authors</h3> <ul class="authors" data-astro-cid-bbe6dxrz> <li data-astro-cid-bbe6dxrz> <a href="https://seongland.com" data-astro-cid-bbe6dxrz>Seonglae Cho</a><sup data-astro-cid-bbe6dxrz>1</sup><span data-astro-cid-bbe6dxrz>, </span> </li><li data-astro-cid-bbe6dxrz> Franklin Cardenoso Fernandez<sup data-astro-cid-bbe6dxrz>1, 3</sup><span data-astro-cid-bbe6dxrz>, </span> </li><li data-astro-cid-bbe6dxrz> Umar Mohammed<sup data-astro-cid-bbe6dxrz>1</sup><span data-astro-cid-bbe6dxrz>, </span> </li><li data-astro-cid-bbe6dxrz> <a href="https://981526092.github.io/zekunwu.github.io/" data-astro-cid-bbe6dxrz>Zekun Wu</a><sup data-astro-cid-bbe6dxrz>2</sup><span data-astro-cid-bbe6dxrz>, </span> </li><li data-astro-cid-bbe6dxrz> <a href="https://kleytoncosta.com" data-astro-cid-bbe6dxrz>Kleyton Da Costa</a><sup data-astro-cid-bbe6dxrz>2</sup><span data-astro-cid-bbe6dxrz>, </span> </li><li data-astro-cid-bbe6dxrz> Ilham Wicaksono<sup data-astro-cid-bbe6dxrz>1</sup><span data-astro-cid-bbe6dxrz>, </span> </li><li data-astro-cid-bbe6dxrz> Adriano Koshiyama<sup data-astro-cid-bbe6dxrz>2</sup> </li> </ul> </div> <div class="meta-container-cell meta-container-cell--affiliations" data-astro-cid-bbe6dxrz> <h3 data-astro-cid-bbe6dxrz>Affiliations</h3> <ol class="affiliations" data-astro-cid-bbe6dxrz> <li value="1" data-astro-cid-bbe6dxrz> <a href="https://www.holisticai.com" target="_blank" rel="noopener noreferrer" data-astro-cid-bbe6dxrz> Holistic AI </a> </li><li value="2" data-astro-cid-bbe6dxrz> <a href="https://www.ucl.ac.uk" target="_blank" rel="noopener noreferrer" data-astro-cid-bbe6dxrz> University College London </a> </li><li value="3" data-astro-cid-bbe6dxrz> <a href="https://www.puc-rio.br" target="_blank" rel="noopener noreferrer" data-astro-cid-bbe6dxrz> PUC-Rio </a> </li> </ol> </div> <div class="meta-container-cell meta-container-cell--published" data-astro-cid-bbe6dxrz> <h3 data-astro-cid-bbe6dxrz>Published</h3> <p data-astro-cid-bbe6dxrz>Jun. 30, 2026</p> </div> <div class="meta-container-cell meta-container-cell--links" data-astro-cid-bbe6dxrz> <h3 data-astro-cid-bbe6dxrz>Links</h3> <p class="hero-links" data-astro-cid-bbe6dxrz> <a class="button lk-paper" href="https://arxiv.org/abs/2608.23670" target="_blank" rel="noopener noreferrer" data-astro-cid-bbe6dxrz>Paper</a> <a class="button lk-demo" href="https://seongland.com/article/asg/browser" target="_blank" rel="noopener noreferrer" data-astro-cid-bbe6dxrz>Demo</a> <a class="button lk-dash" href="https://seongland.com/article/asg/browser?tab=overview" target="_blank" rel="noopener noreferrer" data-astro-cid-bbe6dxrz>Dashboard</a> </p> </div> <!-- {doi && (
<div class="meta-container-cell">
<h3>DOI</h3>
<p><a href={`https://doi.org/${doi}`} target="_blank" rel="noopener noreferrer">{doi}</a></p>
</div>
)} --> </div> </header> <section class="content-grid"> <nav class="table-of-contents" aria-label="Table of Contents" data-auto-collapse="1"> <div class="title">Table of Contents</div> <div id="article-toc-placeholder"></div> </nav> <details class="table-of-contents-mobile"> <summary>Table of Contents</summary> <div id="article-toc-mobile-placeholder"></div> </details> <script>
// Build TOC from article headings (h2/h3/h4) and render into the sticky aside
const buildTOC = () => {
const holder = document.getElementById("article-toc-placeholder");
const holderMobile = document.getElementById(
"article-toc-mobile-placeholder",
);
// Always rebuild TOC to avoid stale entries
if (holder) holder.innerHTML = "";
if (holderMobile) holderMobile.innerHTML = "";
const articleRoot = document.querySelector("section.content-grid main");
if (!articleRoot) return;
const headings = articleRoot.querySelectorAll("h2, h3, h4");
if (!headings.length) return;
// Inclure tous les titres H2/H3/H4 sans filtrer "Table of contents"
const headingsArr = Array.from(headings);
if (!headingsArr.length) return;
// Ensure unique ids for headings (deduplicate duplicates)
const usedIds = new Set();
const slugify = (s) =>
String(s || "")
.toLowerCase()
.trim()
.replace(/\s+/g, "_")
.replace(/[^a-z0-9_\-]/g, "");
headingsArr.forEach((h) => {
let id = (h.id || "").trim();
if (!id) {
const base = slugify(h.textContent || "");
id = base || "section";
}
let candidate = id;
let n = 2;
while (usedIds.has(candidate)) {
candidate = `${id}-${n++}`;
}
if (h.id !== candidate) h.id = candidate;
usedIds.add(candidate);
});
const nav = document.createElement("nav");
let ulStack = [document.createElement("ul")];
nav.appendChild(ulStack[0]);
const levelOf = (tag) => (tag === "H2" ? 2 : tag === "H3" ? 3 : 4);
let prev = 2;
let headingCount = 0;
headingsArr.forEach((h) => {
const lvl = levelOf(h.tagName);
// adjust depth
while (lvl > prev) {
const ul = document.createElement("ul");
ulStack[ulStack.length - 1].lastElementChild?.appendChild(ul);
ulStack.push(ul);
prev++;
}
while (lvl < prev) {
ulStack.pop();
prev--;
}
const li = document.createElement("li");
const a = document.createElement("a");
a.href = "#" + h.id;
a.textContent = h.textContent;
a.target = "_self";
li.appendChild(a);
// Ajouter un index unique à chaque heading pour le tracking
li.setAttribute("data-heading-idx", String(headingCount));
headingCount++;
ulStack[ulStack.length - 1].appendChild(li);
});
if (holder) holder.appendChild(nav);
const navClone = nav.cloneNode(true);
if (holderMobile) holderMobile.appendChild(navClone);
// active link on scroll
const links = [
...(holder ? holder.querySelectorAll("a") : []),
...(holderMobile ? holderMobile.querySelectorAll("a") : []),
];
// Read breakpoint from CSS var and set autoCollapse only on desktop (disabled on mobile)
const getCollapsePx = () => {
const root = document.documentElement;
const raw = getComputedStyle(root)
.getPropertyValue("--bp-content-collapse")
.trim();
return raw || "1100px";
};
const mq = window.matchMedia(`(max-width: ${getCollapsePx()})`);
const attrEnabled =
document
.querySelector(".table-of-contents")
?.getAttribute("data-auto-collapse") === "1";
let autoCollapse = attrEnabled && !mq.matches;
// Inject styles for collapsible & animation (tous les niveaux)
const ensureStyles = () => {
if (document.getElementById("toc-collapse-style")) return;
const style = document.createElement("style");
style.id = "toc-collapse-style";
style.textContent = `
.table-of-contents nav.table-of-contents-collapsible li > ul,
details.table-of-contents-mobile nav.table-of-contents-collapsible li > ul { overflow: hidden; transition: height 200ms ease; }
.table-of-contents nav.table-of-contents-collapsible li.collapsed > ul,
details.table-of-contents-mobile nav.table-of-contents-collapsible li.collapsed > ul { display: block; }
`;
document.head.appendChild(style);
};
ensureStyles();
const getAllItemsWithChildren = () => {
const sideNav = holder ? holder.querySelector("nav") : null;
const mobileNav = holderMobile ? holderMobile.querySelector("nav") : null;
const q = (navEl) =>
navEl
? Array.from(navEl.querySelectorAll("li[data-heading-idx]")).filter(
(li) => li.querySelector(":scope > ul"),
)
: [];
return {
sideNav,
mobileNav,
sideItems: q(sideNav),
mobileItems: q(mobileNav),
};
};
const setNavCollapsible = () => {
const sideNav = holder ? holder.querySelector("nav") : null;
const mobileNav = holderMobile ? holderMobile.querySelector("nav") : null;
if (sideNav) sideNav.classList.add("table-of-contents-collapsible");
if (mobileNav) mobileNav.classList.add("table-of-contents-collapsible");
};
const measure = (el) => {
if (!el) return 0;
// Temporarily set height to auto to measure scrollHeight reliably
const prev = el.style.height;
el.style.height = "auto";
// Force un reflow pour que le navigateur calcule les wraps de texte
void el.offsetHeight;
// Maintenant scrollHeight inclut la vraie hauteur avec tous les line wraps
const h = el.scrollHeight;
el.style.height = prev || "";
return h;
};
// Tracker les animations en cours pour pouvoir les annuler
const activeAnimations = new Map();
const cancelAnimation = (el) => {
if (!el) return;
const animData = activeAnimations.get(el);
if (animData) {
// Nettoyer le listener de l'animation précédente
el.removeEventListener("transitionend", animData.onEnd);
activeAnimations.delete(el);
}
};
const animateTo = (el, target) => {
if (!el) return;
// Annuler toute animation en cours sur cet élément
cancelAnimation(el);
// Obtenir la hauteur ACTUELLE (même si une animation est en cours)
const current = parseFloat(getComputedStyle(el).height) || 0;
// Si on est déjà proche de la cible, pas besoin d'animer
if (Math.abs(current - target) < 1) {
el.style.height = target ? "auto" : "0px";
return;
}
// Démarrer depuis la hauteur actuelle
el.style.height = current + "px";
// Force reflow
void el.offsetHeight;
// Aller vers la cible
el.style.height = target + "px";
// Créer le listener de fin
const onEnd = (e) => {
if (e.propertyName !== "height") return;
el.removeEventListener("transitionend", onEnd);
activeAnimations.delete(el);
if (target > 0) el.style.height = "auto";
};
// Sauvegarder le listener pour pouvoir l'annuler plus tard
activeAnimations.set(el, { onEnd });
el.addEventListener("transitionend", onEnd);
};
let prevActiveIdx = -1;
let prevActiveElements = new Set();
let prevActiveHeadingId = null;
const setCollapsedState = (activeIdx) => {
if (!autoCollapse) return;
if (activeIdx == null || activeIdx < 0) activeIdx = 0;
const { sideItems, mobileItems } = getAllItemsWithChildren();
// Trouver l'élément <li> correspondant au heading actif et tous ses ancêtres
const getActiveAndAncestors = (items, targetIdx) => {
const toExpand = new Set();
// Trouver le <li> qui correspond au targetIdx
const findActiveLi = (li) => {
const idx = Number(li.getAttribute("data-heading-idx") || "-1");
if (idx === targetIdx) {
return li;
}
const childUl = li.querySelector(":scope > ul");
if (!childUl) return null;
const childLis = childUl.querySelectorAll(
":scope > li[data-heading-idx]",
);
for (const child of childLis) {
const found = findActiveLi(child);
if (found) return found;
}
return null;
};
let activeLi = null;
for (const li of items) {
activeLi = findActiveLi(li);
if (activeLi) break;
}
if (!activeLi) return toExpand;
// Collecter l'élément actif lui-même
const activeIdx = Number(
activeLi.getAttribute("data-heading-idx") || "-1",
);
toExpand.add(activeIdx);
// Remonter et collecter TOUS les ancêtres, sans condition
// La structure de la TOC détermine automatiquement qui doit être ouvert
let current = activeLi;
while (current) {
const parent = current.parentElement?.closest("li[data-heading-idx]");
if (parent) {
const parentIdx = Number(
parent.getAttribute("data-heading-idx") || "-1",
);
toExpand.add(parentIdx);
current = parent;
} else {
break;
}
}
return toExpand;
};
const update = (items) => {
const newActiveAncestors = getActiveAndAncestors(items, activeIdx);
// Étape 0 : Annuler TOUTES les animations en cours avant de commencer
// Cela évite les conflits si l'utilisateur scroll rapidement
items.forEach((li) => {
const sub = li.querySelector(":scope > ul");
if (sub) cancelAnimation(sub);
});
// Étape 1 : Identifier TOUS les éléments qui vont changer d'état
const allChanges = [];
items.forEach((li) => {
const sub = li.querySelector(":scope > ul");
if (!sub) return;
const idx = Number(li.getAttribute("data-heading-idx") || "-1");
// Un élément doit être expanded SI il contient (directement ou indirectement) le heading actif
// Donc soit il est dans newActiveAncestors, soit un de ses descendants l'est
let shouldBeExpanded = false;
// Vérifier si cet élément ou un de ses descendants est dans le chemin actif
const allDescendants = li.querySelectorAll("li[data-heading-idx]");
const allRelatedIndices = [
idx,
...Array.from(allDescendants).map((d) =>
Number(d.getAttribute("data-heading-idx") || "-1"),
),
];
// Si au moins un de ces indices est dans newActiveAncestors, garder ouvert
shouldBeExpanded = allRelatedIndices.some((i) =>
newActiveAncestors.has(i),
);
const isCurrentlyCollapsed = li.classList.contains("collapsed");
const isChanging =
(shouldBeExpanded && isCurrentlyCollapsed) ||
(!shouldBeExpanded && !isCurrentlyCollapsed);
if (isChanging) {
allChanges.push({ li, sub, shouldBeExpanded, idx });
}
});
// Étape 2 : Parmi tous les changements, trouver ceux qui sont des "top-level"
// (= n'ont PAS d'ancêtre qui change aussi)
const topLevelChanges = [];
const descendantChanges = [];
allChanges.forEach((change) => {
let hasAncestorChanging = false;
// Remonter l'arbre pour voir si un ancêtre change aussi
let currentLi = change.li;
while (currentLi) {
const parentLi = currentLi.parentElement?.closest(
"li[data-heading-idx]",
);
if (!parentLi) break;
const parentIdx = Number(
parentLi.getAttribute("data-heading-idx") || "-1",
);
// Vérifier si ce parent est dans la liste des changements
const parentIsChanging = allChanges.some(
(c) => c.idx === parentIdx,
);
if (parentIsChanging) {
hasAncestorChanging = true;
break;
}
currentLi = parentLi;
}
if (hasAncestorChanging) {
descendantChanges.push(change);
} else {
topLevelChanges.push(change);
}
});
// Étape 3 : Appliquer TOUS les descendants instantanément (sans animation)
// Ceci doit être fait AVANT toute animation pour que les hauteurs soient correctes
if (descendantChanges.length > 0) {
descendantChanges.forEach(({ li, sub, shouldBeExpanded }) => {
const oldTransition = sub.style.transition;
sub.style.transition = "none";
if (shouldBeExpanded) {
li.classList.remove("collapsed");
sub.style.height = "auto";
} else {
li.classList.add("collapsed");
sub.style.height = "0px";
}
// Forcer un reflow immédiat pour cet élément
void sub.offsetHeight;
sub.style.transition = oldTransition || "";
});
// Forcer un reflow global pour que TOUS les changements soient appliqués
void document.body.offsetHeight;
// IMPORTANT : Attendre un frame pour que le navigateur ait fini tous les calculs
// avant de mesurer les hauteurs des parents
}
// Étape 4 : Animer SEULEMENT les top-level avec requestAnimationFrame
// Les descendants sont déjà dans leur état final, donc la hauteur du parent sera correcte
if (topLevelChanges.length > 0) {
// Double requestAnimationFrame pour être sûr que le DOM est stabilisé
requestAnimationFrame(() => {
requestAnimationFrame(() => {
topLevelChanges.forEach(({ li, sub, shouldBeExpanded }) => {
if (shouldBeExpanded) {
li.classList.remove("collapsed");
// CRITIQUE : Avant de mesurer, mettre ABSOLUMENT TOUS les sous-éléments
// dans leur état final (expanded OU collapsed) de manière synchrone
const allInnerItems = sub.querySelectorAll(
"li[data-heading-idx]",
);
// D'abord, désactiver toutes les transitions
allInnerItems.forEach((innerLi) => {
const innerSub = innerLi.querySelector(":scope > ul");
if (innerSub) {
innerSub.style.transition = "none";
}
});
// Ensuite, mettre chaque élément dans son état final
allInnerItems.forEach((innerLi) => {
const innerIdx = Number(
innerLi.getAttribute("data-heading-idx") || "-1",
);
const innerSub = innerLi.querySelector(":scope > ul");
if (innerSub) {
if (newActiveAncestors.has(innerIdx)) {
// Cet élément devrait être expanded
innerLi.classList.remove("collapsed");
innerSub.style.height = "auto";
} else {
// Cet élément devrait être collapsed
innerLi.classList.add("collapsed");
innerSub.style.height = "0px";
}
}
});
// Forcer un reflow global pour que TOUT soit calculé
void sub.offsetHeight;
// Réactiver les transitions
allInnerItems.forEach((innerLi) => {
const innerSub = innerLi.querySelector(":scope > ul");
if (innerSub) {
innerSub.style.transition = "";
}
});
// Maintenant on peut mesurer avec confiance : tous les éléments
// sont dans leur état final définitif
const target = measure(sub);
animateTo(sub, target);
} else {
li.classList.add("collapsed");
animateTo(sub, 0);
}
});
});
});
}
prevActiveElements = newActiveAncestors;
};
update(sideItems);
update(mobileItems);
setNavCollapsible();
prevActiveIdx = activeIdx;
};
// When switching between desktop/mobile, refresh autoCollapse and expand all on mobile
const expandAll = () => {
const { sideItems, mobileItems } = getAllItemsWithChildren();
const expand = (items) =>
items.forEach((li) => {
li.classList.remove("collapsed");
const sub = li.querySelector(":scope > ul");
if (sub) sub.style.height = "auto";
});
expand(sideItems);
expand(mobileItems);
};
const onMqChange = () => {
autoCollapse = attrEnabled && !mq.matches;
if (!autoCollapse) {
expandAll();
} else {
setCollapsedState(prevActiveIdx);
}
};
if (mq.addEventListener) mq.addEventListener("change", onMqChange);
else if (mq.addListener) mq.addListener(onMqChange);
// Debounce pour traiter la dernière mise à jour après que le scroll se stabilise
let scrollDebounceTimer = null;
let lastRequestedIdx = -1;
let isProcessing = false;
// Fonction pour mettre à jour l'URL avec l'ancre actuelle
const updateURL = (headingId) => {
if (!headingId) return;
const newUrl = `${window.location.pathname}${window.location.search}#${headingId}`;
// Mettre à jour l'URL sans recharger la page
if (window.location.href !== newUrl) {
history.pushState(null, null, newUrl);
// Essayer différentes méthodes pour communiquer avec la fenêtre parente
if (window.parent !== window) {
try {
// Méthode 1: Essayer de modifier directement l'URL du parent (si autorisé)
try {
window.parent.location.hash = headingId;
console.debug("Successfully updated parent URL directly");
return;
} catch (e) {
console.debug("Direct parent URL update blocked:", e.message);
}
// Méthode 2: Essayer avec window.top (fenêtre racine)
try {
if (window.top !== window && window.top !== window.parent) {
window.top.location.hash = headingId;
console.debug("Successfully updated top window URL directly");
return;
}
} catch (e) {
console.debug("Direct top window URL update blocked:", e.message);
}
// Méthode 3: Utiliser postMessage avec différents formats vers parent
const messages = [
{
type: "urlChange",
url: newUrl,
hash: headingId,
},
{
type: "anchorChange",
anchorId: headingId,
url: newUrl,
},
{
type: "HF_SPACE_URL_UPDATE",
hash: headingId,
url: newUrl,
},
// Format officiel Hugging Face
{
queryString: "",
hash: headingId,
},
];
messages.forEach((msg) => {
try {
window.parent.postMessage(msg, "*");
} catch (e) {
console.debug("PostMessage to parent failed:", e.message);
}
});
// Méthode 4: Essayer avec window.top
try {
if (window.top !== window) {
messages.forEach((msg) => {
try {
window.top.postMessage(msg, "*");
} catch (e) {
console.debug("PostMessage to top failed:", e.message);
}
});
}
} catch (e) {
console.debug("PostMessage to top window failed:", e.message);
}
// Méthode 5: Essayer avec l'origine spécifique Hugging Face
try {
window.parent.postMessage(
{
queryString: "",
hash: headingId,
},
"https://huggingface.co",
);
console.debug("Sent HF official format message");
} catch (e) {
console.debug("HF official format failed:", e.message);
}
} catch (e) {
console.debug(
"All parent communication methods failed:",
e.message,
);
}
}
}
};
const onScroll = () => {
// active link highlight
let activeIdx = -1;
let activeHeadingId = null;
for (let i = headingsArr.length - 1; i >= 0; i--) {
const top = headingsArr[i].getBoundingClientRect().top;
if (top - 60 <= 0) {
links.forEach((l) => l.classList.remove("active"));
const id = "#" + headingsArr[i].id;
const actives = Array.from(links).filter(
(l) => l.getAttribute("href") === id,
);
actives.forEach((a) => a.classList.add("active"));
// Utiliser l'index du heading actif (n'importe quel niveau)
activeIdx = i;
activeHeadingId = headingsArr[i].id;
break;
}
}
// Update active heading tracking (but don't touch URL hash on scroll)
if (activeHeadingId && activeHeadingId !== prevActiveHeadingId) {
prevActiveHeadingId = activeHeadingId;
}
if (activeIdx === prevActiveIdx) return;
// Sauvegarder la dernière demande
lastRequestedIdx = activeIdx;
// Si on est en train de traiter, ne rien faire (on traitera la dernière demande après)
if (isProcessing) return;
// Debounce : attendre un peu que le scroll se stabilise
clearTimeout(scrollDebounceTimer);
scrollDebounceTimer = setTimeout(() => {
// Traiter la dernière demande
if (lastRequestedIdx !== prevActiveIdx) {
isProcessing = true;
setCollapsedState(lastRequestedIdx);
// Le processing flag sera réinitialisé après les animations
setTimeout(() => {
isProcessing = false;
// Si une nouvelle demande est arrivée pendant qu'on traitait, la traiter maintenant
if (lastRequestedIdx !== prevActiveIdx) {
onScroll();
}
}, 250); // Attendre que les animations soient lancées
}
}, 100); // Debounce de 100ms
};
// If auto-collapse, collapse immediately (expand first section) before any scroll
if (autoCollapse) setCollapsedState(0);
window.addEventListener("scroll", onScroll);
// Gérer la navigation par ancres au chargement de la page
const handleInitialNavigation = () => {
const hash = window.location.hash;
if (hash) {
const targetElement = document.querySelector(hash);
if (targetElement) {
// Attendre que le DOM soit prêt puis faire défiler vers l'élément
setTimeout(() => {
targetElement.scrollIntoView({ block: "start" });
// Mettre à jour l'URL après le scroll
setTimeout(() => {
updateURL(hash.substring(1)); // Enlever le # du hash
}, 100);
}, 100);
}
} else {
// No anchor - skip auto-setting hash
}
};
// Initialize state
onScroll();
// Gérer la navigation initiale
handleInitialNavigation();
// Gérer les événements de navigation du navigateur (boutons précédent/suivant)
window.addEventListener("popstate", (event) => {
const hash = window.location.hash;
if (hash) {
const targetElement = document.querySelector(hash);
if (targetElement) {
targetElement.scrollIntoView({ block: "start" });
}
} else {
// Si pas d'ancre, aller au début de la page
window.scrollTo({ top: 0 });
}
});
// Close mobile accordion when a link inside it is clicked
if (holderMobile) {
const details = holderMobile.closest("details");
holderMobile.addEventListener("click", (ev) => {
const target = ev.target;
const anchor =
target && "closest" in target ? target.closest("a") : null;
if (anchor instanceof HTMLAnchorElement && details && details.open) {
details.open = false;
}
});
}
};
if (document.readyState === "loading") {
document.addEventListener("DOMContentLoaded", buildTOC, { once: true });
} else {
buildTOC();
}
</script> <main> <h1 id="the-opacity-problem"><a href="#the-opacity-problem">The Opacity Problem</a></h1>
<p>LLM-based <a href="https://texonom.com/5efeaf8cd77f495988083eb2084d7f01">agents</a> execute multi-step tasks, but their <strong>behavioral structure</strong> remains opaque. Existing approaches operate per-trace or success-only, so they miss the cross-run topology (what one run shares with the next) that links next-step and failure prediction. Watch one agent work and you get a wall of text; watch a thousand and you get a machine, a <code>search</code>, <code>edit</code>, <code>execute</code> loop the agent never declares and the system prompt never specifies.</p>
<p>Yang and colleagues put LLM-based agents to work resolving GitHub issues <span class="" id="citation--yang2024sweagent--yang2025swesmith--1">(<a href="#bib-yang2024sweagent" id="refctx-bib-yang2024sweagent-1">Yang et al., 2024</a>, 2025<a href="#bib-yang2025swesmith" id="refctx-bib-yang2025swesmith-1">)</a></span>. Others have them navigating websites <span class="" id="citation--zhou2024webarena--2">(<a href="#bib-zhou2024webarena" id="refctx-bib-zhou2024webarena-1">Zhou et al., 2024</a>)</span>, operating desktops <span class="" id="citation--xie2024osworld--3">(<a href="https://openreview.net/forum?id=tN61DTr4Ed" id="refctx-bib-xie2024osworld-1" data-ref-id="bib-xie2024osworld" target="_blank" rel="noopener noreferrer">Xie et al., 2024</a>)</span>, driving mobile interfaces <span class="" id="citation--lu2024guiodyssey--4">(<a href="#bib-lu2024guiodyssey" id="refctx-bib-lu2024guiodyssey-1">Lu et al., 2025</a>)</span>, managing customer service interactions <span class="" id="citation--yao2024taubench--5">(<a href="#bib-yao2024taubench" id="refctx-bib-yao2024taubench-1">Yao et al., 2024</a>)</span>, orchestrating multi-agent pipelines <span class="" id="citation--wu2023autogen--hong2023metagpt--6">(<a href="#bib-hong2023metagpt" id="refctx-bib-hong2023metagpt-1">Hong et al., 2024</a>; <a href="#bib-wu2023autogen" id="refctx-bib-wu2023autogen-1">Wu et al., 2023</a>)</span>, and attributing the blame when one of those pipelines breaks down <span class="" id="citation--yang2025whoandwhen--7">(<a href="#bib-yang2025whoandwhen" id="refctx-bib-yang2025whoandwhen-1">Zhang et al., 2025</a>)</span>. Following the ReAct pattern <span class="" id="citation--yao2023react--8">(<a href="#bib-yao2023react" id="refctx-bib-yao2023react-1">Yao et al., 2023</a>)</span>, they interleave chain-of-thought reasoning <span class="" id="citation--wei2022cot--9">(<a href="#bib-wei2022cot" id="refctx-bib-wei2022cot-1">Wei et al., 2022</a>)</span> with tool calls — and every run produces an execution trace of tool calls, natural language, and environment feedback.</p>
<p>This article recovers that machine. From nothing but the agent’s own execution traces — no labels, no task descriptions — we extract a compact <a href="https://texonom.com/6ccd60d415e74d07a615b9714ce83510">finite-state machine</a>, or FSM, of 7 to 43 states. That machine does the two things an operator actually needs: predict what the agent will do next, and catch a failing run before it wastes the compute.</p>
<p>The whole article follows one thread: heterogeneous agent traces pass through one deterministic abstraction <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>ϕ</mi></mrow><annotation encoding="application/x-tex">\phi</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.8889em;vertical-align:-0.1944em"></span><span class="mord mathnormal">ϕ</span></span></span></span>, pile up into a prefix tree, and collapse in a single classical merge into one automaton. That one object — not four bespoke pipelines — is then read four different ways: as workflow memory, a next-step predictor, a failure detector, and a runtime monitor.</p>
<div class="wide"> <figure class="html-embed"><figcaption class="html-embed__title" style="text-align:left">How the machine is built, and how it is used</figcaption><div class="html-embed__card"><div id="frag-1qhaojmay9y"><!-- System / data-flow diagram. LEFT->RIGHT: how the machine is BUILT (construction,
offline): raw traces -> phi abstraction -> prefix tree -> one merge -> automaton.
RIGHT: how the machine is USED at runtime by four consumers (what each reads from it).
No result metrics — this shows data flow and execution flow, not numbers.
Self-contained SVG; themed via the article's --asg-* CSS variables (dark-mode safe). -->
<div class="asg-sys" id="asg-sys">
<svg viewBox="0 0 960 460" role="img"
aria-label="Construction pipeline: agent traces are abstracted by phi into activity sequences, accumulated into a prefix tree, and merged once into a single automaton. At runtime the same machine is read by four consumers: workflow memory, next-step prediction, failure prediction, and runtime monitoring.">
<defs>
<marker id="sys-arrow" viewBox="0 0 10 10" refX="8" refY="5" markerWidth="7" markerHeight="7" orient="auto-start-reverse">
<path d="M0 1 L9 5 L0 9 z" class="sys-arrowhead"/>
</marker>
</defs>
<!-- ===================== connectors (behind boxes) ===================== -->
<g class="sys-links" fill="none">
<path class="lnk" d="M150 226 H172" marker-end="url(#sys-arrow)"/>
<path class="lnk" d="M276 226 H300" marker-end="url(#sys-arrow)"/>
<path class="lnk lnk-key" d="M416 226 H466" marker-end="url(#sys-arrow)"/>
<!-- fan-out from FSM hub to four consumers -->
<path class="spoke" data-task="0" d="M622 200 C648 200 642 54 664 54"/>
<path class="spoke" data-task="1" d="M622 216 C648 216 642 166 664 166"/>
<path class="spoke" data-task="2" d="M622 236 C648 236 642 278 664 278"/>
<path class="spoke" data-task="3" d="M622 252 C648 252 642 390 664 390"/>
</g>
<!-- operation labels above the construction arrows (clear of the box band) -->
<text class="op-label" x="288" y="181" text-anchor="middle">build trie</text>
<text class="op-label op-key" x="430" y="181" text-anchor="middle">merge states</text>
<!-- ===================== construction spine ===================== -->
<!-- agent traces (input data) -->
<g class="node node-src" data-stage="src">
<rect x="14" y="170" width="136" height="112" rx="10"/>
<text class="n-title" x="82" y="192" text-anchor="middle">Agent traces</text>
<text class="n-sub" x="82" y="208" text-anchor="middle">messages · tool calls</text>
<g class="chips">
<rect x="28" y="220" width="50" height="17" rx="4" class="chip"/>
<text class="chip-t" x="53" y="232" text-anchor="middle">coding</text>
<rect x="86" y="220" width="50" height="17" rx="4" class="chip"/>
<text class="chip-t" x="111" y="232" text-anchor="middle">web</text>
<rect x="28" y="241" width="50" height="17" rx="4" class="chip"/>
<text class="chip-t" x="53" y="253" text-anchor="middle">GUI</text>
<rect x="86" y="241" width="50" height="17" rx="4" class="chip"/>
<text class="chip-t" x="111" y="253" text-anchor="middle">telecom</text>
</g>
</g>
<!-- phi: the one transform (message -> symbol) -->
<g class="node node-phi" data-stage="phi">
<rect x="174" y="194" width="102" height="64" rx="10"/>
<text class="n-glyph" x="225" y="220" text-anchor="middle">&#966;</text>
<text class="n-sub" x="225" y="238" text-anchor="middle">message &#8594; symbol</text>
<text class="n-foot" x="225" y="250" text-anchor="middle">deterministic</text>
</g>
<!-- prefix tree (intermediate data structure) -->
<g class="node node-pta" data-stage="pta">
<rect x="302" y="194" width="114" height="64" rx="10"/>
<text class="n-title" x="359" y="218" text-anchor="middle">Prefix tree</text>
<text class="n-sub" x="359" y="234" text-anchor="middle">PTA</text>
<text class="n-foot" x="359" y="248" text-anchor="middle">one branch per trace</text>
</g>
<!-- FSM hub (the artifact every consumer reads) -->
<g class="node node-fsm" data-stage="fsm">
<rect x="470" y="158" width="152" height="136" rx="14"/>
<text class="hub-eyebrow" x="546" y="182" text-anchor="middle">ONE PER AGENT</text>
<text class="hub-title" x="546" y="206" text-anchor="middle">Automaton</text>
<g class="hub-mini">
<circle cx="508" cy="232" r="8"/>
<circle cx="546" cy="248" r="8"/>
<circle cx="584" cy="232" r="8" class="hub-mini-accept"/>
<path d="M515 234 L539 245" class="hub-edge"/>
<path d="M553 245 L577 234" class="hub-edge"/>
<path d="M516 229 C530 218 562 218 576 229" class="hub-edge"/>
</g>
<text class="hub-foot" x="546" y="274" text-anchor="middle">states +</text>
<text class="hub-foot" x="546" y="286" text-anchor="middle">transition probabilities</text>
</g>
<!-- ===================== four runtime consumers ===================== -->
<g class="task" data-task="0">
<rect x="664" y="8" width="284" height="92" rx="11"/>
<text class="t-title" x="684" y="40">Workflow memory</text>
<text class="t-sub" x="684" y="62">state &#8594; likely next actions,</text>
<text class="t-sub" x="684" y="80">injected into the agent's prompt</text>
</g>
<g class="task" data-task="1">
<rect x="664" y="120" width="284" height="92" rx="11"/>
<text class="t-title" x="684" y="152">Next-step prediction</text>
<text class="t-sub" x="684" y="174">reads P(next action | current</text>
<text class="t-sub" x="684" y="192">state) off the machine</text>
</g>
<g class="task" data-task="2">
<rect x="664" y="232" width="284" height="92" rx="11"/>
<text class="t-title" x="684" y="264">Failure prediction</text>
<text class="t-sub" x="684" y="286">replay a run &#8594; per-state</text>
<text class="t-sub" x="684" y="304">features &#8594; classifier</text>
</g>
<g class="task" data-task="3">
<rect x="664" y="344" width="284" height="92" rx="11"/>
<text class="t-title" x="684" y="376">Runtime monitor</text>
<text class="t-sub" x="684" y="398">online conformance check,</text>
<text class="t-sub" x="684" y="416">flags drift mid-run</text>
</g>
<!-- phase captions -->
<text class="phase" x="300" y="312" text-anchor="middle">CONSTRUCTION &#183; built once, offline</text>
<text class="phase" x="806" y="452" text-anchor="middle">USED AT RUNTIME &#183; per execution</text>
</svg>
</div>
<style>
.asg-sys { width: 100%; font-family: inherit; padding: 6px 0 2px; }
.asg-sys svg { width: 100%; height: auto; display: block; overflow: visible; }
/* boxes */
.asg-sys .node rect {
fill: var(--surface-bg, #f9f9f9);
stroke: var(--border-color, rgba(0,0,0,0.14));
stroke-width: 1.2;
transition: stroke .2s, fill .2s;
}
.asg-sys .node-fsm rect {
fill: color-mix(in srgb, var(--asg-ours, #3d5a80) 9%, var(--page-bg, #fff));
stroke: var(--asg-ours, #3d5a80);
stroke-width: 1.8;
}
/* text */
.asg-sys text { font-family: inherit; }
.asg-sys .n-title { font-size: 13px; font-weight: 650; fill: var(--text-color, #1a1a2e); }
.asg-sys .n-sub { font-size: 10px; fill: var(--muted-color, rgba(0,0,0,0.55)); }
.asg-sys .n-foot { font-size: 9px; fill: var(--muted-color, rgba(0,0,0,0.5)); font-family: var(--font-mono, ui-monospace, monospace); }
.asg-sys .n-glyph { font-size: 23px; font-style: italic; font-weight: 600; fill: var(--asg-ours, #3d5a80); }
.asg-sys .chip { fill: color-mix(in srgb, var(--asg-ours, #3d5a80) 12%, transparent); stroke: none; }
.asg-sys .chip-t { font-size: 9px; fill: var(--text-color, #1a1a2e); font-family: var(--font-mono, ui-monospace, monospace); }
/* operation labels on construction arrows */
.asg-sys .op-label { font-size: 9px; fill: var(--muted-color, rgba(0,0,0,0.6)); font-family: var(--font-mono, ui-monospace, monospace); letter-spacing: .3px; }
.asg-sys .op-key { fill: var(--asg-ours, #3d5a80); font-weight: 600; }
/* FSM hub */
.asg-sys .hub-eyebrow { font-size: 9px; letter-spacing: 1.6px; fill: var(--asg-ours, #3d5a80); font-weight: 600; }
.asg-sys .hub-title { font-size: 16px; font-weight: 700; fill: var(--text-color, #1a1a2e); }
.asg-sys .hub-foot { font-size: 9px; fill: var(--muted-color, rgba(0,0,0,0.6)); font-family: var(--font-mono, ui-monospace, monospace); }
.asg-sys .hub-mini circle { fill: var(--page-bg, #fff); stroke: var(--asg-ours, #3d5a80); stroke-width: 1.4; }
.asg-sys .hub-mini-accept { fill: color-mix(in srgb, var(--asg-ours, #3d5a80) 22%, var(--page-bg, #fff)) !important; }
.asg-sys .hub-edge { stroke: var(--asg-ours, #3d5a80); stroke-width: 1.2; fill: none; opacity: .7; }
/* connectors */
.asg-sys .lnk { stroke: var(--muted-color, rgba(0,0,0,0.45)); stroke-width: 1.4; }
.asg-sys .lnk-key { stroke: var(--asg-ours, #3d5a80); }
.asg-sys .sys-arrowhead { fill: var(--muted-color, rgba(0,0,0,0.55)); }
.asg-sys .spoke {
stroke: color-mix(in srgb, var(--asg-ours, #3d5a80) 38%, transparent);
stroke-width: 1.6;
transition: stroke .2s, stroke-width .2s;
}
/* consumer cards */
.asg-sys .task rect {
fill: var(--surface-bg, #f9f9f9);
stroke: var(--border-color, rgba(0,0,0,0.12));
stroke-width: 1.2;
transition: stroke .2s, fill .2s;
}
.asg-sys .t-title { font-size: 16px; font-weight: 640; fill: var(--text-color, #1a1a2e); }
.asg-sys .t-sub { font-size: 13px; fill: var(--muted-color, rgba(0,0,0,0.6)); }
/* phase captions */
.asg-sys .phase { font-size: 8.5px; letter-spacing: 1.2px; fill: var(--muted-color, rgba(0,0,0,0.45)); font-family: var(--font-mono, ui-monospace, monospace); }
/* hover: light up a spoke + its consumer card */
.asg-sys .task.is-hot rect { stroke: var(--asg-ours, #3d5a80); fill: color-mix(in srgb, var(--asg-ours, #3d5a80) 7%, var(--page-bg, #fff)); }
.asg-sys .spoke.is-hot { stroke: var(--asg-ours, #3d5a80); stroke-width: 2.4; }
/* entrance */
.asg-sys .node, .asg-sys .task, .asg-sys .sys-links, .asg-sys .op-label, .asg-sys .phase { opacity: 0; animation: sys-in .5s ease forwards; }
.asg-sys .sys-links { animation-delay: .05s; }
.asg-sys .node-src { animation-delay: .04s; }
.asg-sys .node-phi { animation-delay: .12s; }
.asg-sys .node-pta { animation-delay: .20s; }
.asg-sys .op-label { animation-delay: .24s; }
.asg-sys .node-fsm { animation-delay: .30s; }
.asg-sys .task[data-task="0"] { animation-delay: .44s; }
.asg-sys .task[data-task="1"] { animation-delay: .52s; }
.asg-sys .task[data-task="2"] { animation-delay: .60s; }
.asg-sys .task[data-task="3"] { animation-delay: .68s; }
.asg-sys .phase { animation-delay: .72s; }
@keyframes sys-in { to { opacity: 1; } }
@media (prefers-reduced-motion: reduce) {
.asg-sys .node, .asg-sys .task, .asg-sys .sys-links, .asg-sys .op-label, .asg-sys .phase { animation: none; opacity: 1; }
}
</style>
<script>
(function () {
const root = document.getElementById('asg-sys');
if (!root || root.dataset.wired) return;
root.dataset.wired = '1';
const tasks = Array.from(root.querySelectorAll('.task'));
const spokes = Array.from(root.querySelectorAll('.spoke'));
function hot(i, on) {
tasks[i] && tasks[i].classList.toggle('is-hot', on);
const sp = spokes.find(s => s.getAttribute('data-task') === String(i));
sp && sp.classList.toggle('is-hot', on);
}
tasks.forEach((t, i) => {
t.addEventListener('mouseenter', () => hot(i, true));
t.addEventListener('mouseleave', () => hot(i, false));
});
})();
</script>
</div></div><figcaption class="html-embed__desc" style="text-align:left">Data and execution flow. On the left, construction (offline, once per agent): each trace message is mapped to a symbol by φ, the symbol sequences accumulate into a prefix tree, and one state-merge collapses it into a single automaton. On the right, the four runtime consumers and what each reads from that same machine. Hover a consumer to trace it back to the automaton.</figcaption></figure> </div>
<p>A coding agent cycles through <code>search</code>, <code>edit</code>, <code>execute</code>. A customer service agent alternates between database queries and user communication. This structure emerges from the interaction between the system prompt, the available tools, and the task distribution — yet it’s <strong>nowhere written down</strong>.</p>
<p>Understanding this latent structure matters the moment you deploy: safety auditing has to verify that an agent visits the right states and avoids the attack chains that agent security benchmarks enumerate <span class="" id="citation--zhang2025asb--10">(<a href="#bib-zhang2025asb" id="refctx-bib-zhang2025asb-1">H. Zhang et al., 2025</a>)</span>, debugging means locating the bottleneck states where agents get stuck, and production monitoring flags behavioral drift before it costs anything.</p>
<p>Yet current agent analysis works at the level of a single trace <span class="" id="citation--wu2025agentgraph--11">(<a href="#bib-wu2025agentgraph" id="refctx-bib-wu2025agentgraph-1">Z. Wu et al., 2025</a>)</span>. Benchmarks such as AgentBench and AgentBoard score whether a run succeeded <span class="" id="citation--liu2024agentbench--ma2024agentboard--12">(<a href="#bib-liu2024agentbench" id="refctx-bib-liu2024agentbench-1">Liu et al., 2024</a>; <a href="#bib-ma2024agentboard" id="refctx-bib-ma2024agentboard-1">Ma et al., 2024</a>)</span>, and sandboxes such as ToolEmu probe what a run risks <span class="" id="citation--ruan2024toolemu--13">(<a href="#bib-ruan2024toolemu" id="refctx-bib-ruan2024toolemu-1">Ruan et al., 2024</a>)</span>. Neither offers a structural model of the behavior that links one run to the next.</p>
<div class="wide"> <figure class="html-embed"><figcaption class="html-embed__title" style="text-align:left">A trace, read symbol by symbol, drives a finite-state machine</figcaption><div class="html-embed__card"><div id="frag-heb6hyv9u7v"><!-- Tape -> FSM: a trace tape feeds a read-head that drives the FSM state (mini of the dashboard landing) -->
<div class="tape-fsm"></div>
<style>
.tape-fsm { position: relative; width: 100%; font-family: inherit; }
.tape-fsm .tf-top { display: flex; align-items: center; gap: 10px; margin-bottom: 10px; flex-wrap: wrap; }
.tape-fsm .tf-select {
border: 1px solid var(--border-color); background: var(--page-bg); color: var(--text-color);
border-radius: 6px; padding: 4px 8px; font-size: 12px; font-family: inherit; cursor: pointer;
}
.tape-fsm .tf-select:focus { outline: none; border-color: var(--asg-ours, #3d5a80); }
.tape-fsm .tf-meta { font-size: 11px; color: var(--muted-color); }
.tape-fsm .tf-stage {
display: grid; grid-template-columns: 132px 1fr; grid-template-rows: 1fr; gap: 14px;
height: 460px;
}
.tape-fsm .tf-tape-wrap {
position: relative; border: 1px solid var(--border-color); border-radius: 10px;
background: var(--surface-bg); overflow: hidden; cursor: ns-resize; touch-action: none;
min-width: 0; min-height: 0;
}
.tape-fsm .tf-readhead {
position: absolute; left: 0; right: 0; top: 50%; height: 28px; transform: translateY(-50%);
background: color-mix(in srgb, var(--asg-ours, #3d5a80) 14%, transparent);
border-top: 1px solid color-mix(in srgb, var(--asg-ours, #3d5a80) 45%, transparent);
border-bottom: 1px solid color-mix(in srgb, var(--asg-ours, #3d5a80) 45%, transparent);
pointer-events: none; z-index: 2;
}
.tape-fsm .tf-readhead::after {
content: '▸'; position: absolute; left: 6px; top: 50%; transform: translateY(-50%);
color: var(--asg-ours, #3d5a80); font-size: 12px;
}
.tape-fsm .tf-tape { position: absolute; left: 0; right: 0; top: 0; transition: transform 0.3s cubic-bezier(.4,0,.2,1); }
.tape-fsm .tf-tape.tf-dragging { transition: none; }
.tape-fsm .tf-cell {
height: 28px; display: flex; align-items: center; gap: 7px; padding: 0 8px 0 24px;
font-family: var(--font-mono, ui-monospace, monospace); font-size: 11px; color: var(--muted-color);
border-bottom: 1px dashed var(--border-color); white-space: nowrap;
}
.tape-fsm .tf-cell.tf-start { border-top: 2px solid color-mix(in srgb, var(--asg-rpni, #8fa6c4) 60%, transparent); }
.tape-fsm .tf-cell .tf-dot { width: 6px; height: 6px; border-radius: 50%; background: var(--asg-rpni, #8fa6c4); flex: none; }
.tape-fsm .tf-cell.tf-active { color: var(--text-color); font-weight: 600; }
.tape-fsm .tf-cell.tf-active .tf-dot { background: var(--asg-ours, #3d5a80); }
.tape-fsm .tf-graph-wrap {
position: relative; border: 1px solid var(--border-color); border-radius: 10px;
background: var(--surface-bg); overflow: hidden; min-width: 0; min-height: 0;
}
.tape-fsm svg { position: absolute; inset: 0; display: block; width: 100%; height: 100%; }
.tape-fsm .tf-node circle { fill: var(--page-bg); stroke: var(--border-color); stroke-width: 2; transition: all 0.3s ease; }
.tape-fsm .tf-node text { fill: var(--muted-color); font-size: 9px; font-family: var(--font-mono, monospace); text-anchor: middle; transition: fill 0.3s ease; }
.tape-fsm .tf-node.tf-on circle { fill: var(--asg-ours, #3d5a80); stroke: var(--asg-ours, #3d5a80); }
.tape-fsm .tf-node.tf-on text { fill: var(--asg-ours, #3d5a80); font-weight: 700; }
.tape-fsm .tf-edge { stroke: var(--border-color); stroke-width: 1.5; fill: none; transition: stroke 0.3s ease, stroke-width 0.3s ease; marker-end: url(#tf-arrow); }
.tape-fsm .tf-edge.tf-fire { stroke: var(--asg-ours, #3d5a80); stroke-width: 3; }
.tape-fsm .tf-controls { display: flex; align-items: center; gap: 12px; margin-top: 10px; }
.tape-fsm .tf-btn {
border: 1px solid var(--border-color); background: var(--page-bg); color: var(--text-color);
border-radius: 6px; padding: 4px 12px; font-size: 12px; cursor: pointer; font-family: inherit;
}
.tape-fsm .tf-btn:hover { border-color: var(--asg-ours, #3d5a80); color: var(--asg-ours, #3d5a80); }
.tape-fsm .tf-step { padding: 4px 9px; font-size: 10px; line-height: 1; }
.tape-fsm .tf-speedwrap { display: inline-flex; align-items: center; gap: 5px; }
.tape-fsm .tf-speedwrap .tf-select { padding: 3px 6px; }
.tape-fsm .tf-slider { flex: 1; min-width: 80px; accent-color: var(--asg-ours, #3d5a80); }
@media (max-width: 560px) { .tape-fsm .tf-controls { flex-wrap: wrap; } }
.tape-fsm .tf-label { font-size: 11px; color: var(--muted-color); white-space: nowrap; font-variant-numeric: tabular-nums; }
.tape-fsm .tf-label b { color: var(--text-color); }
</style>
<script>
(() => {
const bootstrap = () => {
const root = document.querySelector('.tape-fsm:not([data-mounted])');
if (!root) return;
root.dataset.mounted = 'true';
// ---- Representative datasets (each its own alphabet, FSM layout, and sample traces) ----
const DATASETS = {
sweagent: {
name: 'SWE-agent', domain: 'coding', alphabet: 24,
pos: { init:[58,32], search:[58,108], edit:[178,108], execute:[178,32], submit:[118,180] },
edges: [['init','search'],['search','edit'],['edit','execute'],['execute','search'],['execute','submit'],['execute','edit']],
traces: [
{ ok:true, steps:['init','search','edit','execute','search','edit','execute','submit'] },
{ ok:false, steps:['init','search','edit','execute','edit','execute','edit','execute'] },
],
},
'tau2-telecom': {
name: 'tau2-bench telecom', domain: 'customer service', alphabet: 42,
pos: { init:[44,38], lookup:[140,32], diagnose:[214,92], tool:[176,166], respond:[70,166], resolve:[34,98] },
edges: [['init','lookup'],['lookup','diagnose'],['diagnose','tool'],['tool','respond'],['respond','resolve'],['lookup','respond'],['respond','lookup']],
traces: [
{ ok:true, steps:['init','lookup','diagnose','tool','respond','resolve'] },
{ ok:false, steps:['init','lookup','respond','lookup','respond','lookup','respond'] },
],
},
webarena: {
name: 'WebArena', domain: 'web', alphabet: 24,
pos: { init:[58,32], navigate:[178,32], click:[178,108], type:[58,108], submit:[118,180] },
edges: [['init','navigate'],['navigate','click'],['click','type'],['type','submit'],['type','click'],['click','navigate']],
traces: [
{ ok:true, steps:['init','navigate','click','type','submit'] },
{ ok:false, steps:['init','navigate','click','type','click','type','click','type'] },
],
},
swesmith: {
name: 'SWE-smith', domain: 'coding', alphabet: 9,
pos: { init:[58,32], locate:[58,108], patch:[178,108], test:[178,32], submit:[118,180] },
edges: [['init','locate'],['locate','patch'],['patch','test'],['test','locate'],['test','submit'],['test','patch']],
traces: [
{ ok:true, steps:['init','locate','patch','test','submit'] },
{ ok:false, steps:['init','locate','patch','test','patch','test','patch','test'] },
],
},
mind2web: {
name: 'Mind2Web', domain: 'web', alphabet: 7,
pos: { init:[58,32], click:[178,32], type:[178,108], select:[58,108], submit:[118,180] },
edges: [['init','click'],['click','type'],['type','select'],['select','submit'],['type','click'],['click','select']],
traces: [
{ ok:true, steps:['init','click','type','select','submit'] },
{ ok:false, steps:['init','click','type','click','type','click','type'] },
],
},
agentnet: {
name: 'AgentNet', domain: 'desktop GUI', alphabet: 24,
pos: { init:[44,38], open:[140,32], click:[214,92], type:[176,166], scroll:[70,166], save:[34,98] },
edges: [['init','open'],['open','click'],['click','type'],['type','save'],['click','scroll'],['scroll','click']],
traces: [
{ ok:true, steps:['init','open','click','type','save'] },
{ ok:false, steps:['init','open','click','scroll','click','scroll','click','scroll'] },
],
},
};
root.innerHTML = `
<div class="tf-top">
<label class="tf-meta" for="tf-ds">Dataset</label>
<select class="tf-select" id="tf-ds">
${Object.entries(DATASETS).map(([k,v]) => `<option value="${k}">${v.name}</option>`).join('')}
</select>
<span class="tf-meta tf-domain"></span>
</div>
<div class="tf-stage">
<div class="tf-tape-wrap"><div class="tf-readhead"></div><div class="tf-tape"></div></div>
<div class="tf-graph-wrap"></div>
</div>
<div class="tf-controls">
<button class="tf-btn" data-play>Pause</button>
<button class="tf-btn tf-step" data-prev title="Step back">&#9664;</button>
<button class="tf-btn tf-step" data-next title="Step forward">&#9654;</button>
<input class="tf-slider" type="range" min="0" max="1" value="0" />
<label class="tf-meta tf-speedwrap">Speed
<select class="tf-select" data-speed>
<option value="1500">0.5&times;</option>
<option value="750" selected>1&times;</option>
<option value="375">2&times;</option>
<option value="188">4&times;</option>
</select>
</label>
<span class="tf-label"></span>
</div>`;
const tapeEl = root.querySelector('.tf-tape');
const wrapEl = root.querySelector('.tf-tape-wrap');
const graphEl = root.querySelector('.tf-graph-wrap');
const btn = root.querySelector('[data-play]');
const slider = root.querySelector('.tf-slider');
const label = root.querySelector('.tf-label');
const sel = root.querySelector('#tf-ds');
const domainEl = root.querySelector('.tf-domain');
const prevBtn = root.querySelector('[data-prev]');
const nextBtn = root.querySelector('[data-next]');
const speedSel = root.querySelector('[data-speed]');
const CELL = 28;
const NS = 'http://www.w3.org/2000/svg';
let SEQ = [], STATES = [], EDGES = [], POS = {}, nodeEls = {}, edgeEls = {};
let head = 0, playing = true, timer = null, speed = 750;
function buildDataset(key) {
const ds = DATASETS[key];
POS = ds.pos; EDGES = ds.edges; STATES = Object.keys(ds.pos);
SEQ = [];
ds.traces.forEach((t, ti) => t.steps.forEach((s, si) =>
SEQ.push({ sym:s, traceIdx:ti, ok:t.ok, start: si===0 })));
domainEl.textContent = ds.domain + ' · |A| = ' + ds.alphabet;
slider.max = SEQ.length - 1;
// tape cells
tapeEl.innerHTML = SEQ.map((s, i) =>
`<div class="tf-cell${s.start?' tf-start':''}" data-i="${i}"><span class="tf-dot"></span>${s.sym}</div>`).join('');
// graph
const svg = document.createElementNS(NS, 'svg');
svg.setAttribute('viewBox', '0 0 240 226');
svg.innerHTML = `<defs><marker id="tf-arrow" viewBox="0 0 10 10" refX="9" refY="5" markerWidth="7" markerHeight="7" orient="auto-start-reverse"><path d="M0,0 L10,5 L0,10 z" fill="var(--border-color)"/></marker></defs>`;
edgeEls = {};
EDGES.forEach(([a,b]) => {
const [x1,y1] = POS[a], [x2,y2] = POS[b];
const mx = (x1+x2)/2, my = (y1+y2)/2;
const back = STATES.indexOf(b) < STATES.indexOf(a);
const bend = back ? 26 : 0;
const p = document.createElementNS(NS, 'path');
p.setAttribute('d', `M${x1},${y1} Q${mx+bend},${my+bend} ${x2},${y2}`);
p.setAttribute('class', 'tf-edge');
svg.appendChild(p);
edgeEls[a+'>'+b] = p;
});
nodeEls = {};
STATES.forEach(s => {
const [x,y] = POS[s];
const g = document.createElementNS(NS, 'g');
g.setAttribute('class', 'tf-node');
g.innerHTML = `<circle cx="${x}" cy="${y}" r="13"></circle><text x="${x}" y="${y+25}">${s}</text>`;
svg.appendChild(g);
nodeEls[s] = g;
});
graphEl.innerHTML = '';
graphEl.appendChild(svg);
head = 0;
}
function clamp(v) { return Math.max(0, Math.min(SEQ.length - 1, v)); }
function paint() {
const s = SEQ[head];
const prev = head > 0 && !s.start ? SEQ[head-1].sym : null;
const offset = wrapEl.clientHeight/2 - CELL/2 - head*CELL;
tapeEl.style.transform = `translateY(${offset}px)`;
root.querySelectorAll('.tf-cell').forEach(c => c.classList.toggle('tf-active', +c.dataset.i === head));
Object.values(nodeEls).forEach(n => n.classList.remove('tf-on'));
if (nodeEls[s.sym]) nodeEls[s.sym].classList.add('tf-on');
Object.values(edgeEls).forEach(e => e.classList.remove('tf-fire'));
if (prev && edgeEls[prev+'>'+s.sym]) edgeEls[prev+'>'+s.sym].classList.add('tf-fire');
slider.value = head;
label.innerHTML = `trace <b>#${s.traceIdx+1}</b> ${s.ok?'(success)':'(stuck)'} &middot; step <b>${head+1}/${SEQ.length}</b>`;
}
function step() { head = (head+1) % SEQ.length; paint(); }
function play() { playing=true; btn.textContent='Pause'; clearInterval(timer); timer=setInterval(step, speed); }
function pause() { playing=false; btn.textContent='Play'; clearInterval(timer); }
btn.addEventListener('click', () => playing ? pause() : play());
prevBtn.addEventListener('click', () => { pause(); head = clamp(head - 1); paint(); });
nextBtn.addEventListener('click', () => { pause(); head = clamp(head + 1); paint(); });
slider.addEventListener('input', () => { pause(); head = clamp(+slider.value); paint(); });
speedSel.addEventListener('change', () => { speed = +speedSel.value; if (playing) play(); });
sel.addEventListener('change', () => { buildDataset(sel.value); paint(); play(); });
// ---- Scroll like the dashboard: wheel + drag to scrub the read-head ----
wrapEl.addEventListener('wheel', (e) => {
e.preventDefault(); pause();
head = clamp(head + (e.deltaY > 0 ? 1 : -1));
paint();
}, { passive: false });
let dragging = false, startY = 0, startHead = 0;
wrapEl.addEventListener('pointerdown', (e) => {
dragging = true; startY = e.clientY; startHead = head; pause();
tapeEl.classList.add('tf-dragging'); wrapEl.setPointerCapture(e.pointerId);
});
wrapEl.addEventListener('pointermove', (e) => {
if (!dragging) return;
const delta = Math.round((startY - e.clientY) / CELL);
head = clamp(startHead + delta); paint();
});
const endDrag = () => { dragging = false; tapeEl.classList.remove('tf-dragging'); };
wrapEl.addEventListener('pointerup', endDrag);
wrapEl.addEventListener('pointercancel', endDrag);
buildDataset('sweagent');
paint();
play();
};
if (document.readyState === 'loading') document.addEventListener('DOMContentLoaded', bootstrap, { once:true });
else bootstrap();
})();
</script>
</div></div><figcaption class="html-embed__desc" style="text-align:left">Pick a dataset, then scrub the tape (drag or scroll) or let it play. The read-head advances through the agent's activity tape (left) and lights up the current FSM state and the transition it takes (right). A successful run threads to submit; a stuck run loops and never reaches it. These are representative traces; the full set of twelve datasets with their real extracted FSMs is in the live dashboard.</figcaption></figure> </div>
<h2 id="the-inverse-problem"><a href="#the-inverse-problem">The Inverse Problem</a></h2>
<p>We frame behavioral recovery as an <strong>inverse problem</strong>: given a corpus of execution traces, reconstruct a finite-state machine that explains the observed behavior. Gold posed this as grammatical inference <span class="" id="citation--gold1967language--14">(<a href="#bib-gold1967language" id="refctx-bib-gold1967language-1">Gold, 1967</a>)</span>, Oncina and García later gave it a working algorithm <span class="" id="citation--oncina1992rpni--15">(<a href="#bib-oncina1992rpni" id="refctx-bib-oncina1992rpni-1">Oncina &amp; Garcı́a, 1992</a>)</span>, and de la Higuera surveys what the field settled on <span class="" id="citation--delahiguera2010grammatical--16">(<a href="#bib-delahiguera2010grammatical" id="refctx-bib-delahiguera2010grammatical-1">de la Higuera, 2010</a>)</span>. The classical setting assumes both positive and negative examples. Angluin’s alternative replaces the negatives with an oracle that answers membership queries <span class="" id="citation--angluin1987learning--17">(<a href="#bib-angluin1987learning" id="refctx-bib-angluin1987learning-1">Angluin, 1987</a>)</span>, and an execution log answers nothing. Agent traces give us positive examples only (the runs that happened) with no labeled counter-examples. Gold also proved that identifying the target language from positive examples alone is <strong>impossible in the limit</strong> <span class="" id="citation--gold1967language--18">(<a href="#bib-gold1967language" id="refctx-bib-gold1967language-2">Gold, 1967</a>)</span>.</p>
<p>A property specific to agents rescues the problem: unlike arbitrary regular languages, agent behavior is generated by a <strong>bounded set of tools and actions</strong>, so traces draw from a small activity alphabet, 6 to 42 symbols across our twelve datasets.</p>
<div class="note note--neutral" data-astro-cid-qg6lmfty> <!-- When there's no title, emoji is above content -->
<div class="note__layout" data-astro-cid-qg6lmfty> <div class="note__body" data-astro-cid-qg6lmfty> <div class="note__content" data-astro-cid-qg6lmfty> <p>A small alphabet is the whole reason this works. It makes the compact automaton both small and, as later chapters show, statistically dense enough to predict from. The only modeling choice in the entire pipeline is how a message becomes a symbol.</p> </div> </div> </div> </div>
<h2 id="twelve-datasets-eight-domains"><a href="#twelve-datasets-eight-domains">Twelve Datasets, Eight Domains</a></h2>
<p>We evaluate on twelve public datasets spanning <strong>coding</strong>, <strong>web navigation</strong>, <strong>desktop GUI</strong>, <strong>desktop OS</strong>, <strong>mobile GUI</strong>, <strong>customer service</strong>, <strong>safety</strong>, and <strong>multi-agent coordination</strong>. The trace counts below are the trajectories we use in our experiments (from 184 to 8,337 each), not the size of each source corpus: for the largest sources we draw a fixed slice rather than the whole set.</p>
<div class="table-scroll"><table><thead><tr><th style="text-align:left">Dataset</th><th style="text-align:left">Domain</th><th style="text-align:right">Traces used</th><th style="text-align:right">Actions</th><th style="text-align:right">States</th><th style="text-align:right">Fitness</th></tr></thead><tbody><tr><td style="text-align:left">SWE-smith <span class="" id="citation--yang2025swesmith--19">(<a href="#bib-yang2025swesmith" id="refctx-bib-yang2025swesmith-2">Yang et al., 2025</a>)</span></td><td style="text-align:left">Coding</td><td style="text-align:right">500</td><td style="text-align:right">9</td><td style="text-align:right">10</td><td style="text-align:right">1.000</td></tr><tr><td style="text-align:left">SWE-agent <span class="" id="citation--yang2024sweagent--20">(<a href="#bib-yang2024sweagent" id="refctx-bib-yang2024sweagent-2">Yang et al., 2024</a>)</span></td><td style="text-align:left">Coding</td><td style="text-align:right">2,000</td><td style="text-align:right">24</td><td style="text-align:right">25</td><td style="text-align:right">0.999</td></tr><tr><td style="text-align:left">Mind2Web <span class="" id="citation--deng2024mind2web--21">(<a href="#bib-deng2024mind2web" id="refctx-bib-deng2024mind2web-1">Deng et al., 2023</a>)</span></td><td style="text-align:left">Web</td><td style="text-align:right">500</td><td style="text-align:right">7</td><td style="text-align:right">8</td><td style="text-align:right">1.000</td></tr><tr><td style="text-align:left">WebArena <span class="" id="citation--zhou2024webarena--22">(<a href="#bib-zhou2024webarena" id="refctx-bib-zhou2024webarena-2">Zhou et al., 2024</a>)</span></td><td style="text-align:left">Web</td><td style="text-align:right">8,337</td><td style="text-align:right">24</td><td style="text-align:right">25</td><td style="text-align:right">1.000</td></tr><tr><td style="text-align:left">AgentNet <span class="" id="citation--wang2025opencua--23">(<a href="#bib-wang2025opencua" id="refctx-bib-wang2025opencua-1">Wang et al., 2025</a>)</span></td><td style="text-align:left">Desktop GUI</td><td style="text-align:right">5,000</td><td style="text-align:right">24</td><td style="text-align:right">25</td><td style="text-align:right">1.000</td></tr><tr><td style="text-align:left">GUI-Odyssey <span class="" id="citation--lu2024guiodyssey--24">(<a href="#bib-lu2024guiodyssey" id="refctx-bib-lu2024guiodyssey-2">Lu et al., 2025</a>)</span></td><td style="text-align:left">Mobile GUI</td><td style="text-align:right">7,735</td><td style="text-align:right">6</td><td style="text-align:right">7</td><td style="text-align:right">1.000</td></tr><tr><td style="text-align:left">Who &amp; When <span class="" id="citation--yang2025whoandwhen--25">(<a href="#bib-yang2025whoandwhen" id="refctx-bib-yang2025whoandwhen-2">S. Zhang et al., 2025</a>)</span></td><td style="text-align:left">Multi-agent</td><td style="text-align:right">184</td><td style="text-align:right">8</td><td style="text-align:right">9</td><td style="text-align:right">1.000</td></tr><tr><td style="text-align:left">tau2-bench airline <span class="" id="citation--barres2025tau2--26">(<a href="#bib-barres2025tau2" id="refctx-bib-barres2025tau2-1">Barres et al., 2025</a>)</span></td><td style="text-align:left">Customer service</td><td style="text-align:right">800</td><td style="text-align:right">17</td><td style="text-align:right">18</td><td style="text-align:right">1.000</td></tr><tr><td style="text-align:left">tau2-bench retail <span class="" id="citation--barres2025tau2--27">(<a href="#bib-barres2025tau2" id="refctx-bib-barres2025tau2-2">Barres et al., 2025</a>)</span></td><td style="text-align:left">Customer service</td><td style="text-align:right">1,824</td><td style="text-align:right">18</td><td style="text-align:right">19</td><td style="text-align:right">1.000</td></tr><tr><td style="text-align:left">tau2-bench telecom <span class="" id="citation--barres2025tau2--28">(<a href="#bib-barres2025tau2" id="refctx-bib-barres2025tau2-3">Barres et al., 2025</a>)</span></td><td style="text-align:left">Customer service</td><td style="text-align:right">1,824</td><td style="text-align:right">42</td><td style="text-align:right">43</td><td style="text-align:right">1.000</td></tr><tr><td style="text-align:left">ATBench <span class="" id="citation--li2026atbench--29">(<a href="#bib-li2026atbench" id="refctx-bib-li2026atbench-1">Li et al., 2026</a>)</span></td><td style="text-align:left">Safety</td><td style="text-align:right">1,000</td><td style="text-align:right">14</td><td style="text-align:right">15</td><td style="text-align:right">1.000</td></tr><tr><td style="text-align:left">OSWorld <span class="" id="citation--xie2024osworld--30">(<a href="https://openreview.net/forum?id=tN61DTr4Ed" id="refctx-bib-xie2024osworld-2" data-ref-id="bib-xie2024osworld" target="_blank" rel="noopener noreferrer">Xie et al., 2024</a>)</span></td><td style="text-align:left">Desktop OS</td><td style="text-align:right">2,166</td><td style="text-align:right">26</td><td style="text-align:right">27</td><td style="text-align:right">0.997</td></tr></tbody></table></div>
<p>Every dataset replays held-out traces at fitness of at least 0.997.</p>
<p>The three largest source datasets are subsampled to a fixed slice: SWE-agent uses 2,000 of the 80,036 available trajectories, Mind2Web <span class="" id="citation--deng2024mind2web--31">(<a href="#bib-deng2024mind2web" id="refctx-bib-deng2024mind2web-2">Deng et al., 2023</a>)</span> 500 of 2,350, and AgentNet 5,000 from the OpenCUA Ubuntu subset <span class="" id="citation--wang2025opencua--32">(<a href="#bib-wang2025opencua" id="refctx-bib-wang2025opencua-2">Wang et al., 2025</a>)</span> — the other nine datasets are used in full.</p>
<div class="note note--info" data-astro-cid-qg6lmfty> <!-- When there's a title, emoji is inline with title -->
<div class="note__body" data-astro-cid-qg6lmfty> <div class="note__header" data-astro-cid-qg6lmfty> <div class="note__title" data-astro-cid-qg6lmfty>Every case is in the live dashboard</div> </div> <div class="note__content" data-astro-cid-qg6lmfty> <p>Every figure in this article is a fixed snapshot. The <a href="https://seongland.com/article/asg/browser">interactive dashboard</a> renders all twelve datasets straight from the experiment outputs — the FSMs, the failure predictor, the runtime monitor, and more.</p> </div> </div> </div> <div id="bibliography-references-list" class="references csl-bib-body" data-bibliography-block="true" data-built-refs="1"><ol class="references"><li id="bib-angluin1987learning">Angluin, D. (1987). Learning Regular Sets from Queries and Counterexamples. <i>Information and Computation</i>, <i>75</i>(2), 87–106.<small class="backrefs"><a href="#refctx-bib-angluin1987learning-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-barres2025tau2">Barres, V., Dong, H., Ray, S., Si, X., &amp; Narasimhan, K. (2025). τ<sup>2</sup>-Bench: Evaluating Conversational Agents in a Dual-Control Environment. <i>arXiv Preprint arXiv:2506.07982</i>.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-barres2025tau2-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-barres2025tau2-2" aria-label="Back to citation">2</a>, <a href="#refctx-bib-barres2025tau2-3" aria-label="Back to citation">3</a></small></li><li id="bib-delahiguera2010grammatical">de la Higuera, C. (2010). <i>Grammatical Inference: Learning Automata and Grammars</i>.<small class="backrefs"><a href="#refctx-bib-delahiguera2010grammatical-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-deng2024mind2web">Deng, X., Gu, Y., Zheng, B., Chen, S., Stevens, S., Wang, B., Sun, H., &amp; Su, Y. (2023). Mind2Web: Towards a Generalist Agent for the Web. <i>Advances in Neural Information Processing Systems (NeurIPS)</i>.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-deng2024mind2web-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-deng2024mind2web-2" aria-label="Back to citation">2</a></small></li><li id="bib-gold1967language">Gold, E. M. (1967). Language Identification in the Limit. <i>Information and Control</i>, <i>10</i>(5), 447–474.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-gold1967language-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-gold1967language-2" aria-label="Back to citation">2</a></small></li><li id="bib-hong2023metagpt">Hong, S., Zhuge, M., Chen, J., Zheng, X., Cheng, Y., Zhang, C., Wang, J., Wang, Z., Yau, S. K. S., Lin, Z., &amp; others. (2024). MetaGPT: Meta Programming for A Multi-Agent Collaborative Framework. <i>Proceedings of the International Conference on Learning Representations (ICLR)</i>.<small class="backrefs"><a href="#refctx-bib-hong2023metagpt-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-li2026atbench">Li, Y., Luo, H., Xie, Y., Fu, Y., Yang, Z., Shao, S., Ren, Q., Qu, W., Fu, Y., Yang, Y., Shao, J., Hu, X., &amp; Liu, D. (2026). ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis. <i>arXiv Preprint arXiv:2604.02022</i>.<small class="backrefs"><a href="#refctx-bib-li2026atbench-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-liu2024agentbench">Liu, X., Yu, H., Zhang, H., Xu, Y., Lei, X., Lai, H., Gu, Y., Ding, H., Men, K., Yang, K., &amp; others. (2024). AgentBench: Evaluating LLMs as Agents. <i>Proceedings of the International Conference on Learning Representations (ICLR)</i>.<small class="backrefs"><a href="#refctx-bib-liu2024agentbench-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-lu2024guiodyssey">Lu, Q., Zhao, W., Jia, J., Ren, K., Lu, K., Han, J., Chen, Y., Zheng, J., Zhang, Z., &amp; Ding, L. (2025). GUI-Odyssey: A Comprehensive Dataset for Cross-App GUI Navigation on Mobile Devices. <i>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</i>.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-lu2024guiodyssey-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-lu2024guiodyssey-2" aria-label="Back to citation">2</a></small></li><li id="bib-ma2024agentboard">Ma, C., Zhang, J., Zhu, Z., Yang, C., Yang, Y., Jin, Y., Lan, Z., Kong, L., &amp; He, J. (2024). AgentBoard: An Analytical Evaluation Board of Multi-turn LLM Agents. <i>arXiv Preprint arXiv:2401.13178</i>.<small class="backrefs"><a href="#refctx-bib-ma2024agentboard-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-oncina1992rpni">Oncina, J., &amp; Garcı́a, P. (1992). Inferring Regular Languages in Polynomial Updated Time. <i>Pattern Recognition and Image Analysis</i>, 49–61.<small class="backrefs"><a href="#refctx-bib-oncina1992rpni-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-ruan2024toolemu">Ruan, Y., Dong, H., Wang, A., Pitis, S., Zhou, Y., Ba, J., Dubois, Y., Maddison, C. J., &amp; Hashimoto, T. (2024). Identifying the Risks of LM Agents with an LM-Emulated Sandbox. <i>Proceedings of the International Conference on Learning Representations (ICLR)</i>.<small class="backrefs"><a href="#refctx-bib-ruan2024toolemu-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-wang2025opencua">Wang, X., Wang, B., Lu, D., Yang, J., Xie, T., Wang, J., Deng, J., Guo, X., Xu, Y., Wu, C. H., Shen, Z., Li, Z., Li, R., Li, X., Chen, J., Boyuan, Z., Li, P., Lei, F., Cao, R., … Yu, T. (2025). OpenCUA: Open Foundations for Computer-Use Agents. <i>arXiv Preprint arXiv:2508.09123</i>.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-wang2025opencua-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-wang2025opencua-2" aria-label="Back to citation">2</a></small></li><li id="bib-wei2022cot">Wei, J., Wang, X., Schuurmans, D., Bosma, M., Ichter, B., Xia, F., Chi, E., Le, Q., &amp; Zhou, D. (2022). Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. <i>Advances in Neural Information Processing Systems (NeurIPS)</i>.<small class="backrefs"><a href="#refctx-bib-wei2022cot-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-wu2023autogen">Wu, Q., Bansal, G., Zhang, J., Wu, Y., Li, B., Zhu, E., Jiang, L., Zhang, X., Zhang, S., Liu, J., Awadallah, A. H., White, R. W., Burger, D., &amp; Wang, C. (2023). AutoGen: Enabling Next-Gen LLM Applications via Multi-Agent Conversation. <i>arXiv Preprint arXiv:2308.08155</i>.<small class="backrefs"><a href="#refctx-bib-wu2023autogen-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-wu2025agentgraph">Wu, Z., Cho, S., Munoz, C., King, T., Mohammed, U., Kazim, E., Perez-Ortiz, M., Bulathwela, S., &amp; Koshiyama, A. (2025). AgentGraph: Trace-to-Graph Platform for Interactive Analysis and Robustness Testing in Agentic AI Systems. <i>Proceedings of the AAAI Conference on Artificial Intelligence</i>.<small class="backrefs"><a href="#refctx-bib-wu2025agentgraph-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-xie2024osworld">Xie, T., Zhang, D., Chen, J., Li, X., Zhao, S., Cao, R., Hua, T. J., Cheng, Z., Shin, D., Lei, F., Liu, Y., Xu, Y., Zhou, S., Savarese, S., Xiong, C., Zhong, V., &amp; Yu, T. (2024). OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments. <i>The Thirty-Eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track</i>. <a href="https://openreview.net/forum?id=tN61DTr4Ed" target="_blank" rel="noopener noreferrer">https://openreview.net/forum?id=tN61DTr4Ed</a><small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-xie2024osworld-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-xie2024osworld-2" aria-label="Back to citation">2</a></small></li><li id="bib-yang2024sweagent">Yang, J., Jimenez, C. E., Wettig, A., Lieret, K., Yao, S., Narasimhan, K., &amp; Press, O. (2024). SWE-agent: Agent-Computer Interfaces Enable Automated Software Engineering. <i>Advances in Neural Information Processing Systems (NeurIPS)</i>.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-yang2024sweagent-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-yang2024sweagent-2" aria-label="Back to citation">2</a></small></li><li id="bib-yang2025swesmith">Yang, J., Lieret, K., Jimenez, C. E., Wettig, A., Khandpur, K., Zhang, Y., Hui, B., Press, O., Schmidt, L., &amp; Yang, D. (2025). SWE-smith: Scaling Data for Software Engineering Agents. <i>Proceedings of the Annual Conference on Neural Information Processing Systems (NeurIPS), Datasets &amp; Benchmarks Track</i>.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-yang2025swesmith-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-yang2025swesmith-2" aria-label="Back to citation">2</a></small></li><li id="bib-yao2024taubench">Yao, S., Shinn, N., Razavi, P., &amp; Narasimhan, K. R. (2024). τ-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains. <i>arXiv Preprint arXiv:2406.12045</i>.<small class="backrefs"><a href="#refctx-bib-yao2024taubench-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-yao2023react">Yao, S., Zhao, J., Yu, D., Du, N., Shafran, I., Narasimhan, K., &amp; Cao, Y. (2023). ReAct: Synergizing Reasoning and Acting in Language Models. <i>International Conference on Learning Representations (ICLR)</i>.<small class="backrefs"><a href="#refctx-bib-yao2023react-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-zhang2025asb">Zhang, H., Huang, J., Mei, K., Yao, Y., Wang, Z., Zhan, C., Wang, H., &amp; Zhang, Y. (2025). Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents. <i>Proceedings of the International Conference on Learning Representations (ICLR)</i>.<small class="backrefs"><a href="#refctx-bib-zhang2025asb-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-yang2025whoandwhen">Zhang, S., Yin, M., Zhang, J., Liu, J., Han, Z., Zhang, J., Li, B., Wang, C., Wang, H., Chen, Y., &amp; Wu, Q. (2025). Which Agent Causes Task Failures and When? <i>arXiv Preprint arXiv:2505.00212</i>.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-yang2025whoandwhen-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-yang2025whoandwhen-2" aria-label="Back to citation">2</a></small></li><li id="bib-zhou2024webarena">Zhou, S., Xu, F. F., Zhu, H., Zhou, X., Lo, R., Sridhar, A., Cheng, X., Bisk, Y., Fried, D., Alon, U., &amp; Neubig, G. (2024). WebArena: A Realistic Web Environment for Building Autonomous Agents. <i>Proceedings of the International Conference on Learning Representations (ICLR)</i>.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-zhou2024webarena-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-zhou2024webarena-2" aria-label="Back to citation">2</a></small></li></ol></div>
<h1 id="from-traces-to-symbols"><a href="#from-traces-to-symbols">From Traces to Symbols</a></h1>
<p>An agent execution trace is a sequence of messages <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>τ</mi><mo>=</mo><mo stretchy="false">(</mo><msub><mi>m</mi><mn>1</mn></msub><mo separator="true">,</mo><msub><mi>m</mi><mn>2</mn></msub><mo separator="true">,</mo><mo>…</mo><mo separator="true">,</mo><msub><mi>m</mi><mi>T</mi></msub><mo stretchy="false">)</mo></mrow><annotation encoding="application/x-tex">\tau = (m_1, m_2, \ldots, m_T)</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.4306em"></span><span class="mord mathnormal" style="margin-right:0.1132em">τ</span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">=</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em"></span><span class="mopen">(</span><span class="mord"><span class="mord mathnormal">m</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.3011em"><span style="top:-2.55em;margin-left:0em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight">1</span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.15em"><span></span></span></span></span></span></span><span class="mpunct">,</span><span class="mspace" style="margin-right:0.1667em"></span><span class="mord"><span class="mord mathnormal">m</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.3011em"><span style="top:-2.55em;margin-left:0em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight">2</span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.15em"><span></span></span></span></span></span></span><span class="mpunct">,</span><span class="mspace" style="margin-right:0.1667em"></span><span class="minner">…</span><span class="mspace" style="margin-right:0.1667em"></span><span class="mpunct">,</span><span class="mspace" style="margin-right:0.1667em"></span><span class="mord"><span class="mord mathnormal">m</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.3283em"><span style="top:-2.55em;margin-left:0em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mathnormal mtight" style="margin-right:0.13889em">T</span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.15em"><span></span></span></span></span></span></span><span class="mclose">)</span></span></span></span>, where each message has a <strong>role</strong> (system, user, assistant, tool) and <strong>content</strong>. Step one maps each message to a symbolic activity from a finite alphabet.</p>
<h2 id="activity-extraction"><a href="#activity-extraction">Activity Extraction</a></h2>
<p>An <em>activity extraction function</em> <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>ϕ</mi><mo>:</mo><msub><mi>m</mi><mi>t</mi></msub><mo>↦</mo><msub><mi>a</mi><mi>t</mi></msub><mo>∈</mo><mi mathvariant="script">A</mi></mrow><annotation encoding="application/x-tex">\phi: m_t \mapsto a_t \in \mathcal{A}</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.8889em;vertical-align:-0.1944em"></span><span class="mord mathnormal">ϕ</span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">:</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:0.661em;vertical-align:-0.15em"></span><span class="mord"><span class="mord mathnormal">m</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.2806em"><span style="top:-2.55em;margin-left:0em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mathnormal mtight">t</span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.15em"><span></span></span></span></span></span></span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">↦</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:0.6891em;vertical-align:-0.15em"></span><span class="mord"><span class="mord mathnormal">a</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.2806em"><span style="top:-2.55em;margin-left:0em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mathnormal mtight">t</span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.15em"><span></span></span></span></span></span></span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">∈</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:0.6833em"></span><span class="mord mathcal">A</span></span></span></span> maps each message to a symbol, and it applies three rules in priority order: a message carrying a <code>tool_call</code> field takes the function name as its activity (<code>bash</code>, <code>search_flight</code>, <code>click</code>); content carrying an <code>[ACTION] description</code> tag takes the action label; and for agents that act through code blocks we take the first command token and map it to a semantic category (<code>edit</code>, <code>search</code>, <code>navigate</code>, <code>execute</code>). If no rule matches, the activity defaults to <code>role:content_type</code>, and <code>assistant:text</code> is the common case.</p>
<div class="note note--neutral" data-astro-cid-qg6lmfty> <!-- When there's no title, emoji is above content -->
<div class="note__layout" data-astro-cid-qg6lmfty> <div class="note__body" data-astro-cid-qg6lmfty> <div class="note__content" data-astro-cid-qg6lmfty> <p>The symbol layer is where other work has made different choices. ToolBench takes each API call as the unit <span class="" id="citation--qin2024toolbench--1">(<a href="#bib-qin2024toolbench" id="refctx-bib-qin2024toolbench-1">Qin et al., 2024</a>)</span>, SWE-bench scores the patch an agent produces rather than the path it took <span class="" id="citation--jimenez2024swebench--2">(<a href="#bib-jimenez2024swebench" id="refctx-bib-jimenez2024swebench-1">Jimenez et al., 2024</a>)</span>, and AgentTrek reconstructs trajectories from written tutorials before an agent ever runs <span class="" id="citation--hou2025agenttrek--3">(<a href="#bib-hou2025agenttrek" id="refctx-bib-hou2025agenttrek-1">Xu et al., 2025</a>)</span>. Each fixes its unit in advance and computes it cheaply, as <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>ϕ</mi></mrow><annotation encoding="application/x-tex">\phi</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.8889em;vertical-align:-0.1944em"></span><span class="mord mathnormal">ϕ</span></span></span></span> does, but only <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>ϕ</mi></mrow><annotation encoding="application/x-tex">\phi</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.8889em;vertical-align:-0.1944em"></span><span class="mord mathnormal">ϕ</span></span></span></span> has to survive four trace formats at once.</p><p>The extraction is entirely deterministic and format-specific, with no LLM calls; the whole process completes in milliseconds.</p> </div> </div> </div> </div>
<h2 id="an-example"><a href="#an-example">An Example</a></h2>
<p>Consider a coding agent trace from SWE-agent <span class="" id="citation--yang2024sweagent--4">(<a href="#bib-yang2024sweagent" id="refctx-bib-yang2024sweagent-1">Yang et al., 2024</a>)</span> with 47 messages. The raw trace contains system prompts, file contents, error messages, and tool invocations. After extraction, the activity sequence is:</p>
<p style="text-align:center; line-height:2.2;"><p><code>init</code> → <code>user</code> → <code>search</code> → <code>user</code> → <code>edit</code> → <code>user</code> → <code>execute</code> → <code>user</code> → <code>edit</code> → <code>user</code> → <code>submit</code></p></p>
<p>From 47 messages and thousands of tokens, we get the 11 symbols the FSM will model — drawn from an alphabet of 24 possible activities.</p>
<div class="wide"> <figure class="html-embed"><figcaption class="html-embed__title" style="text-align:left">Trace Explorer</figcaption><div class="html-embed__card"><div id="frag-1js91fhlzu8"><!-- Trace Explorer: Interactive trace-to-symbol mapper -->
<div class="trace-explorer"></div>
<style>
.trace-explorer { position: relative; width: 100%; font-family: inherit; }
.trace-explorer .seg-control {
display: inline-flex; background: var(--surface-bg);
border: 1px solid var(--border-color); border-radius: 8px;
padding: 3px; gap: 2px;
}
.trace-explorer .seg-control button {
padding: 5px 14px; border-radius: 6px; border: none;
background: transparent; font-size: 12px; font-weight: 500;
color: var(--text-color); cursor: pointer; transition: all 0.15s ease;
opacity: 0.6;
}
.trace-explorer .seg-control button:hover { opacity: 0.8; }
.trace-explorer .seg-control button.active {
background: var(--text-color); color: var(--page-bg);
opacity: 1; font-weight: 600;
}
.trace-explorer .top-bar {
display: flex; align-items: center; gap: 12px; margin-bottom: 14px; flex-wrap: wrap;
}
.trace-explorer .step-info {
font-size: 12px; color: var(--text-color); opacity: 0.5; margin-left: auto;
font-variant-numeric: tabular-nums;
}
.trace-explorer .progress-track {
width: 100%; height: 3px; border-radius: 2px;
background: var(--border-color); margin-bottom: 14px; overflow: hidden;
}
.trace-explorer .progress-fill {
height: 100%; border-radius: 2px; background: var(--text-color);
transition: width 0.3s ease; opacity: 0.5;
}
.trace-explorer .panels {
display: grid; grid-template-columns: 1fr 1fr; gap: 16px;
}
@media (max-width: 700px) { .trace-explorer .panels { grid-template-columns: 1fr; } }
.trace-explorer .panel {
border: 1px solid var(--border-color); border-radius: 10px;
background: var(--surface-bg); overflow: hidden;
backdrop-filter: blur(12px); -webkit-backdrop-filter: blur(12px);
}
.trace-explorer .panel-header {
padding: 10px 14px; font-size: 11px; font-weight: 700; text-transform: uppercase;
letter-spacing: 1px; border-bottom: 1px solid var(--border-color);
color: var(--text-color); opacity: 0.5;
}
.trace-explorer .panel-body {
padding: 10px 12px; font-size: 12px; line-height: 1.6; max-height: 340px;
overflow-y: auto; color: var(--text-color);
}
.trace-explorer .msg {
padding: 7px 10px; margin: 2px 0; border-radius: 6px; transition: all .25s ease;
border-left: 3px solid transparent; cursor: default;
}
.trace-explorer .msg.active-system { background: rgba(107, 143, 113, 0.08); border-left-color: #6b8f71; }
.trace-explorer .msg.active-user { background: rgba(61, 90, 128, 0.08); border-left-color: #8fa6c4; }
.trace-explorer .msg.active-assistant { background: rgba(61, 90, 128, 0.08); border-left-color: #3d5a80; }
.trace-explorer .msg.active-tool { background: rgba(196, 154, 58, 0.08); border-left-color: #c49a3a; }
.trace-explorer .msg .role {
font-weight: 700; font-size: 10px; text-transform: uppercase; letter-spacing: 0.6px;
margin-bottom: 2px;
}
.trace-explorer .msg .role.system { color: #6b8f71; }
.trace-explorer .msg .role.user { color: #8fa6c4; }
.trace-explorer .msg .role.assistant { color: #3d5a80; }
.trace-explorer .msg .role.tool { color: #c49a3a; }
.trace-explorer .msg .content {
font-size: 11px; opacity: 0.55; white-space: nowrap; overflow: hidden; text-overflow: ellipsis;
}
.trace-explorer .symbol-flow {
display: flex; flex-wrap: wrap; gap: 4px; align-items: center;
}
.trace-explorer .symbol {
padding: 4px 10px; border-radius: 6px; font-size: 10px; font-weight: 600;
font-family: var(--font-mono, monospace); background: rgba(150,150,150,0.06);
border: 1px solid var(--border-color); opacity: 0.2; transition: all .25s ease;
transform: scale(0.92);
}
.trace-explorer .symbol.revealed { opacity: 1; transform: scale(1); }
.trace-explorer .symbol.current {
border-color: var(--text-color);
box-shadow: 0 0 0 2px rgba(61, 90, 128, 0.15);
}
.trace-explorer .symbol.sym-system { border-color: #6b8f71; }
.trace-explorer .symbol.sym-user { border-color: #8fa6c4; }
.trace-explorer .symbol.sym-assistant { border-color: #3d5a80; }
.trace-explorer .symbol.sym-tool { border-color: #c49a3a; }
.trace-explorer .symbol.revealed.sym-system { background: rgba(107, 143, 113, 0.1); }
.trace-explorer .symbol.revealed.sym-user { background: rgba(61, 90, 128, 0.1); }
.trace-explorer .symbol.revealed.sym-assistant { background: rgba(61, 90, 128, 0.1); }
.trace-explorer .symbol.revealed.sym-tool { background: rgba(196, 154, 58, 0.1); }
.trace-explorer .arrow {
color: var(--text-color); opacity: 0.15; font-size: 12px;
transition: opacity .25s ease;
}
.trace-explorer .arrow.revealed { opacity: 0.4; }
.trace-explorer .rule-callout {
margin-top: 12px; padding: 10px 14px; border-radius: 8px;
background: var(--surface-bg); border: 1px solid var(--border-color);
font-size: 11px; line-height: 1.5; color: var(--text-color);
backdrop-filter: blur(12px); -webkit-backdrop-filter: blur(12px);
transition: opacity .2s ease;
}
.trace-explorer .rule-callout .rule-icon {
font-weight: 700; margin-right: 6px; opacity: 0.6;
}
</style>
<script>
(() => {
const container = document.querySelector('.trace-explorer:not([data-mounted])');
if (!container) return;
container.dataset.mounted = 'true';
const trace = [
{ role: 'system', content: 'You are a coding assistant. Use bash and str_replace_editor tools...', activity: 'system:text', rule: 'Default: role:content_type' },
{ role: 'user', content: 'Fix the failing test in tests/test_parser.py. The error is IndexError on line 42.', activity: 'user:text', rule: 'Default: role:content_type' },
{ role: 'assistant', content: 'tool_call: bash({"command": "cat tests/test_parser.py"})', activity: 'assistant:tool:bash', rule: 'Tool call: function name = bash' },
{ role: 'tool', content: '# test_parser.py\ndef test_parse_input():\n result = parse("hello world")\n assert result[5] == ...', activity: 'tool:text', rule: 'Default: role:content_type' },
{ role: 'assistant', content: 'tool_call: str_replace_editor({"file": "parser.py", "old": "result[5]", "new": "result[min(5,len(result)-1)]"})', activity: 'assistant:tool:str_replace_editor', rule: 'Tool call: function name = str_replace_editor' },
{ role: 'tool', content: 'File edited successfully.', activity: 'tool:text', rule: 'Default: role:content_type' },
{ role: 'assistant', content: 'tool_call: bash({"command": "python -m pytest tests/test_parser.py -v"})', activity: 'assistant:tool:bash', rule: 'Tool call: function name = bash' },
{ role: 'tool', content: 'tests/test_parser.py::test_parse_input PASSED\n1 passed in 0.02s', activity: 'tool:text', rule: 'Default: role:content_type' },
{ role: 'assistant', content: 'tool_call: submit()', activity: 'assistant:tool:submit', rule: 'Tool call: function name = submit' },
{ role: 'tool', content: 'Submission accepted.', activity: 'tool:text', rule: 'Default: role:content_type' },
];
let step = -1;
// Top bar with segmented control
const topBar = document.createElement('div');
topBar.className = 'top-bar';
const seg = document.createElement('div');
seg.className = 'seg-control';
const prevBtn = document.createElement('button');
prevBtn.textContent = '\u2190 Prev';
const nextBtn = document.createElement('button');
nextBtn.textContent = 'Next \u2192';
nextBtn.className = 'active';
const playBtn = document.createElement('button');
playBtn.textContent = '\u25B6 Play';
const resetBtn = document.createElement('button');
resetBtn.textContent = 'Reset';
seg.append(prevBtn, nextBtn, playBtn, resetBtn);
const stepInfo = document.createElement('span');
stepInfo.className = 'step-info';
topBar.append(seg, stepInfo);
container.appendChild(topBar);
// Progress bar
const progressTrack = document.createElement('div');
progressTrack.className = 'progress-track';
const progressFill = document.createElement('div');
progressFill.className = 'progress-fill';
progressFill.style.width = '0%';
progressTrack.appendChild(progressFill);
container.appendChild(progressTrack);
// Panels
const panels = document.createElement('div');
panels.className = 'panels';
// Left: Raw trace
const leftPanel = document.createElement('div');
leftPanel.className = 'panel';
leftPanel.innerHTML = '<div class="panel-header">Raw Trace Messages</div>';
const leftBody = document.createElement('div');
leftBody.className = 'panel-body';
trace.forEach((msg, i) => {
const div = document.createElement('div');
div.className = 'msg';
div.dataset.idx = i;
div.innerHTML = '<div class="role ' + msg.role + '">' + msg.role + '</div><div class="content">' + msg.content.replace(/</g,'&lt;').slice(0, 80) + (msg.content.length > 80 ? '...' : '') + '</div>';
leftBody.appendChild(div);
});
leftPanel.appendChild(leftBody);
// Right: Symbol sequence + rule
const rightPanel = document.createElement('div');
rightPanel.className = 'panel';
rightPanel.innerHTML = '<div class="panel-header">Activity Sequence</div>';
const rightBody = document.createElement('div');
rightBody.className = 'panel-body';
const flow = document.createElement('div');
flow.className = 'symbol-flow';
trace.forEach((msg, i) => {
if (i > 0) {
const arrow = document.createElement('span');
arrow.className = 'arrow';
arrow.textContent = '\u2192';
arrow.dataset.idx = i;
flow.appendChild(arrow);
}
const sym = document.createElement('span');
sym.className = 'symbol sym-' + msg.role;
sym.textContent = msg.activity;
sym.dataset.idx = i;
flow.appendChild(sym);
});
rightBody.appendChild(flow);
const ruleBox = document.createElement('div');
ruleBox.className = 'rule-callout';
ruleBox.innerHTML = '<span class="rule-icon">f(x)</span> Click "Next" to step through the trace';
rightBody.appendChild(ruleBox);
rightPanel.appendChild(rightBody);
panels.append(leftPanel, rightPanel);
container.appendChild(panels);
function update() {
const pct = step < 0 ? 0 : Math.max(0, ((step + 1) / trace.length) * 100);
progressFill.style.width = pct + '%';
stepInfo.textContent = step < 0 ? 'Ready' : 'Step ' + (step + 1) + ' / ' + trace.length;
// Update active button highlights
prevBtn.classList.toggle('active', false);
nextBtn.classList.toggle('active', step < trace.length - 1);
// Highlight messages with role-specific accent
leftBody.querySelectorAll('.msg').forEach(m => {
const idx = parseInt(m.dataset.idx);
m.className = 'msg';
if (idx === step) {
m.classList.add('active-' + trace[idx].role);
}
});
// Reveal symbols
flow.querySelectorAll('.symbol').forEach(s => {
const idx = parseInt(s.dataset.idx);
s.classList.toggle('revealed', idx <= step);
s.classList.toggle('current', idx === step);
});
flow.querySelectorAll('.arrow').forEach(a => {
const idx = parseInt(a.dataset.idx);
a.classList.toggle('revealed', idx <= step);
});
// Rule callout
if (step >= 0 && step < trace.length) {
ruleBox.innerHTML = '<span class="rule-icon">f(x)</span> ' + trace[step].rule;
ruleBox.style.opacity = '1';
const activeMsg = leftBody.querySelector('.msg[class*="active-"]');
if (activeMsg) {
const msgTop = activeMsg.offsetTop - leftBody.offsetTop;
const msgBot = msgTop + activeMsg.offsetHeight;
const viewTop = leftBody.scrollTop;
const viewBot = viewTop + leftBody.clientHeight;
if (msgTop < viewTop || msgBot > viewBot) {
leftBody.scrollTo({ top: msgTop - leftBody.clientHeight / 2 + activeMsg.offsetHeight / 2, behavior: 'smooth' });
}
}
} else if (step >= trace.length - 1 && step >= 0) {
ruleBox.innerHTML = '<span class="rule-icon">\u2713</span> Extraction complete: 10 messages \u2192 10 symbols from alphabet of 5';
ruleBox.style.opacity = '1';
} else {
ruleBox.innerHTML = '<span class="rule-icon">f(x)</span> Click "Next" to step through the trace';
ruleBox.style.opacity = '0.6';
}
}
prevBtn.addEventListener('click', () => { if (step > -1) { step--; update(); } });
nextBtn.addEventListener('click', () => { if (step < trace.length - 1) { step++; update(); } });
resetBtn.addEventListener('click', () => { step = -1; update(); clearInterval(playTimer); playBtn.textContent = '\u25B6 Play'; playBtn.classList.remove('active'); });
let playTimer = null;
playBtn.addEventListener('click', () => {
if (playTimer) { clearInterval(playTimer); playTimer = null; playBtn.textContent = '\u25B6 Play'; playBtn.classList.remove('active'); return; }
playBtn.textContent = '\u23F8 Pause';
playBtn.classList.add('active');
playTimer = setInterval(() => {
if (step < trace.length - 1) { step++; update(); }
else { clearInterval(playTimer); playTimer = null; playBtn.textContent = '\u25B6 Play'; playBtn.classList.remove('active'); }
}, 1000);
});
update();
})();
</script>
</div></div><figcaption class="html-embed__desc" style="text-align:left">Step through an agent trace to see how raw messages map to symbolic activities. Each message is classified by the extraction rules into one of the agent's activity symbols.</figcaption></figure> </div>
<p>Replay any real trace symbol by symbol in the <a href="https://seongland.com/article/asg/browser?tab=traces&dataset=sweagent">trace view</a>.</p>
<h2 id="why-this-works"><a href="#why-this-works">Why This Works</a></h2>
<p>These alphabets are small by construction, against the tens of thousands of symbols in natural language — that’s what keeps FSM extraction tractable.</p>
<p>Even a 42-tool telecom customer service agent (tau2-bench telecom <span class="" id="citation--barres2025tau2--5">(<a href="#bib-barres2025tau2" id="refctx-bib-barres2025tau2-1">Barres et al., 2025</a>)</span>) needs only 43 states to capture its behavioral structure — the agent can only call the tools we gave it. So its capabilities stay bounded and so does its alphabet.</p>
<div class="note note--info" data-astro-cid-qg6lmfty> <!-- When there's a title, emoji is inline with title -->
<div class="note__body" data-astro-cid-qg6lmfty> <div class="note__header" data-astro-cid-qg6lmfty> <div class="note__title" data-astro-cid-qg6lmfty>How sensitive is the result to this choice?</div> </div> <div class="note__content" data-astro-cid-qg6lmfty> <p><span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>ϕ</mi></mrow><annotation encoding="application/x-tex">\phi</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.8889em;vertical-align:-0.1944em"></span><span class="mord mathnormal">ϕ</span></span></span></span> is the one design decision in the pipeline — so we stress-test it. Fitness stays above 0.999 across all four granularities, from role-only (two to four symbols) to full tool-level — failure prediction stays within 0.03 <a href="https://texonom.com/c3fea8b9caa445768eb529fb629818c7">AUROC</a> (area under the ROC curve) on any dataset — the rules above are one valid setting, not the only one.</p> </div> </div> </div> <div id="bibliography-references-list" class="references csl-bib-body" data-bibliography-block="true" data-built-refs="1"><ol class="references"><li id="bib-barres2025tau2">Barres, V., Dong, H., Ray, S., Si, X., &amp; Narasimhan, K. (2025). τ<sup>2</sup>-Bench: Evaluating Conversational Agents in a Dual-Control Environment. <i>arXiv Preprint arXiv:2506.07982</i>.<small class="backrefs"><a href="#refctx-bib-barres2025tau2-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-jimenez2024swebench">Jimenez, C. E., Yang, J., Wettig, A., Yao, S., Pei, K., Press, O., &amp; Narasimhan, K. (2024). SWE-bench: Can Language Models Resolve Real-World GitHub Issues? <i>Proceedings of the International Conference on Learning Representations (ICLR)</i>.<small class="backrefs"><a href="#refctx-bib-jimenez2024swebench-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-qin2024toolbench">Qin, Y., Liang, S., Ye, Y., Zhu, K., Yan, L., Lu, Y., Lin, Y., Cong, X., Tang, X., Qian, B., &amp; others. (2024). ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs. <i>Proceedings of the International Conference on Learning Representations (ICLR)</i>.<small class="backrefs"><a href="#refctx-bib-qin2024toolbench-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-hou2025agenttrek">Xu, Y., Lu, D., Shen, Z., Wang, J., Wang, Z., Mao, Y., Xiong, C., &amp; Yu, T. (2025). AgentTrek: Agent Trajectory Synthesis via Guiding Replay with Web Tutorials. <i>Proceedings of the International Conference on Learning Representations (ICLR)</i>.<small class="backrefs"><a href="#refctx-bib-hou2025agenttrek-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-yang2024sweagent">Yang, J., Jimenez, C. E., Wettig, A., Lieret, K., Yao, S., Narasimhan, K., &amp; Press, O. (2024). SWE-agent: Agent-Computer Interfaces Enable Automated Software Engineering. <i>Advances in Neural Information Processing Systems (NeurIPS)</i>.<small class="backrefs"><a href="#refctx-bib-yang2024sweagent-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li></ol></div>
<h1 id="building-the-machine"><a href="#building-the-machine">Building the Machine</a></h1>
<p>Given a corpus of activity sequences, FSM construction is two steps and nothing else: build a prefix tree, then merge structurally equivalent states. There are no thresholds, no number of clusters, no learning rate.</p>
<h2 id="step-1-prefix-tree"><a href="#step-1-prefix-tree">Step 1: Prefix Tree</a></h2>
<p>Insert all activity sequences into a <a href="https://texonom.com/8b0d3da022e64279a5127e872536d7ca">trie</a> — each unique prefix becomes a distinct state. The prefix tree has <strong>perfect training fitness</strong>, replaying every training trace exactly. But it can have tens of thousands of states.</p>
<p>For SWE-agent (2,000 traces), the prefix tree has <strong>59,510 states</strong>. Most are visited once and represent memorized suffixes rather than reusable transition patterns.</p>
<h2 id="step-2-structural-merging"><a href="#step-2-structural-merging">Step 2: Structural Merging</a></h2>
<p>Two states are <em>structurally equivalent</em> if for every activity <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>a</mi><mo>∈</mo><mi mathvariant="script">A</mi></mrow><annotation encoding="application/x-tex">a \in \mathcal{A}</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.5782em;vertical-align:-0.0391em"></span><span class="mord mathnormal">a</span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">∈</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:0.6833em"></span><span class="mord mathcal">A</span></span></span></span>: (i) <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>δ</mi><mo stretchy="false">(</mo><mi>q</mi><mo separator="true">,</mo><mi>a</mi><mo stretchy="false">)</mo></mrow><annotation encoding="application/x-tex">\delta(q, a)</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em"></span><span class="mord mathnormal" style="margin-right:0.03785em">δ</span><span class="mopen">(</span><span class="mord mathnormal" style="margin-right:0.03588em">q</span><span class="mpunct">,</span><span class="mspace" style="margin-right:0.1667em"></span><span class="mord mathnormal">a</span><span class="mclose">)</span></span></span></span> is defined exactly when <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>δ</mi><mo stretchy="false">(</mo><msup><mi>q</mi><mo mathvariant="normal" lspace="0em" rspace="0em">′</mo></msup><mo separator="true">,</mo><mi>a</mi><mo stretchy="false">)</mo></mrow><annotation encoding="application/x-tex">\delta(q&#39;, a)</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:1.0019em;vertical-align:-0.25em"></span><span class="mord mathnormal" style="margin-right:0.03785em">δ</span><span class="mopen">(</span><span class="mord"><span class="mord mathnormal" style="margin-right:0.03588em">q</span><span class="msupsub"><span class="vlist-t"><span class="vlist-r"><span class="vlist" style="height:0.7519em"><span style="top:-3.063em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mord mtight">′</span></span></span></span></span></span></span></span></span><span class="mpunct">,</span><span class="mspace" style="margin-right:0.1667em"></span><span class="mord mathnormal">a</span><span class="mclose">)</span></span></span></span> is defined, and (ii) the targets are themselves equivalent — this recursion is computed bottom-up in a single pass.</p>
<div class="note note--neutral" data-astro-cid-qg6lmfty> <!-- When there's no title, emoji is above content -->
<div class="note__layout" data-astro-cid-qg6lmfty> <div class="note__body" data-astro-cid-qg6lmfty> <div class="note__content" data-astro-cid-qg6lmfty> <p>Structural equivalence is exactly the <a href="https://texonom.com/ab9c3c96247d8389af2a8156427385b4">Myhill-Nerode</a> equivalence on the observed prefix language, and by that theorem the quotient is the <strong>unique minimal <a href="https://texonom.com/f08438340ef447f89eade6aac52f2cbb">deterministic finite automaton (DFA)</a></strong>, so no smaller automaton can reproduce the observed behavior <span class="" id="citation--hopcroft2006automata--1">(<a href="#bib-hopcroft2006automata" id="refctx-bib-hopcroft2006automata-1">Hopcroft et al., 2006</a>)</span>.</p> </div> </div> </div> </div>
<p>The SWE-agent prefix tree with 59,510 states collapses to just <strong>25 states</strong> — a 2,380<span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mo>×</mo></mrow><annotation encoding="application/x-tex">\times</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6667em;vertical-align:-0.0833em"></span><span class="mord">×</span></span></span></span> compression — and the resulting FSM still replays held-out traces at 0.999 fitness.</p>
<figure class="html-embed"><figcaption class="html-embed__title" style="text-align:left">Prefix Tree to FSM Collapse</figcaption><div class="html-embed__card"><div id="frag-qmk1rbd7hyb"><!-- Prefix Tree Collapse: Animated tree → FSM visualization -->
<div class="prefix-collapse"></div>
<style>
.prefix-collapse { position: relative; width: 100%; min-height: 480px; }
.prefix-collapse svg { display: block; width: 100%; }
.prefix-collapse .top-bar {
display: flex; align-items: center; gap: 12px; margin-bottom: 14px; flex-wrap: wrap;
}
.prefix-collapse .seg-control {
display: inline-flex; background: var(--surface-bg);
border: 1px solid var(--border-color); border-radius: 8px;
padding: 3px; gap: 2px;
}
.prefix-collapse .seg-control button {
padding: 5px 14px; border-radius: 6px; border: none;
background: transparent; font-size: 12px; font-weight: 500;
color: var(--text-color); cursor: pointer; transition: all 0.15s ease;
opacity: 0.6;
}
.prefix-collapse .seg-control button:hover { opacity: 0.8; }
.prefix-collapse .seg-control button.active {
background: var(--text-color); color: var(--page-bg);
opacity: 1; font-weight: 600;
}
.prefix-collapse .stat-pills {
display: flex; gap: 8px; margin-left: auto; flex-wrap: wrap;
}
.prefix-collapse .stat-pill {
padding: 4px 12px; border-radius: 20px; font-size: 11px; font-weight: 600;
border: 1px solid var(--border-color); background: var(--surface-bg);
color: var(--text-color); font-variant-numeric: tabular-nums;
}
.prefix-collapse .stat-pill.highlight {
border-color: #3d5a80; color: #3d5a80;
}
.prefix-collapse .node-label-tree {
font-size: 9px; fill: var(--text-color); pointer-events: none; text-anchor: middle;
opacity: 0.7;
}
</style>
<script>
(() => {
const ensureD3 = (cb) => {
if (window.d3 && typeof window.d3.select === 'function') return cb();
let s = document.getElementById('d3-cdn-script');
if (!s) { s = document.createElement('script'); s.id = 'd3-cdn-script'; s.src = 'https://cdn.jsdelivr.net/npm/d3@7/dist/d3.min.js'; document.head.appendChild(s); }
s.addEventListener('load', () => cb(), { once: true });
};
const bootstrap = () => {
const container = document.querySelector('.prefix-collapse:not([data-mounted])');
if (!container) return;
container.dataset.mounted = 'true';
const d3 = window.d3;
// --- Palette (ported from browser/) ---
const RUST = '#3d5a80';
const traces = [
['init','sys','usr','bash','tool','edit','tool','bash','tool','submit','tool'],
['init','sys','usr','bash','tool','bash','tool','edit','tool','submit','tool'],
['init','sys','usr','edit','tool','bash','tool','submit','tool'],
['init','sys','usr','bash','tool','edit','tool','edit','tool','bash','tool','submit','tool'],
['init','sys','usr','bash','tool','bash','tool','bash','tool','submit','tool'],
];
// Build prefix tree
let nextId = 0;
const root = { id: nextId++, label: 'init', children: {}, depth: 0 };
traces.forEach(trace => {
let node = root;
for (let i = 1; i < trace.length; i++) {
const sym = trace[i];
if (!node.children[sym]) {
node.children[sym] = { id: nextId++, label: sym, children: {}, depth: i };
}
node = node.children[sym];
}
});
function flattenTree(node, parent) {
const result = [{ id: node.id, label: node.label, parent: parent ? parent.id : null, depth: node.depth }];
Object.values(node.children).forEach(child => {
result.push(...flattenTree(child, node));
});
return result;
}
const treeNodes = flattenTree(root, null);
const treeEdges = treeNodes.filter(n => n.parent !== null).map(n => ({ source: n.parent, target: n.id }));
const fsmNodes = [
{ id: 'f-init', label: 'init' },
{ id: 'f-sys', label: 'sys' },
{ id: 'f-usr', label: 'usr' },
{ id: 'f-bash', label: 'bash' },
{ id: 'f-tool', label: 'tool' },
{ id: 'f-edit', label: 'edit' },
{ id: 'f-submit', label: 'submit' },
];
const fsmEdges = [
{ source: 'f-init', target: 'f-sys' },
{ source: 'f-sys', target: 'f-usr' },
{ source: 'f-usr', target: 'f-bash' },
{ source: 'f-usr', target: 'f-edit' },
{ source: 'f-bash', target: 'f-tool' },
{ source: 'f-edit', target: 'f-tool' },
{ source: 'f-tool', target: 'f-bash' },
{ source: 'f-tool', target: 'f-edit' },
{ source: 'f-tool', target: 'f-submit' },
{ source: 'f-submit', target: 'f-tool' },
{ source: 'f-tool', target: 'f-usr' },
];
// Tree node colors (warm, subtle)
const treeNodeColor = '#3d5a80';
let showFSM = true;
// Top bar
const topBar = document.createElement('div');
topBar.className = 'top-bar';
const seg = document.createElement('div');
seg.className = 'seg-control';
const treeBtn = document.createElement('button');
treeBtn.textContent = 'Prefix Tree';
const mergeBtn = document.createElement('button');
mergeBtn.textContent = 'Merged FSM';
mergeBtn.className = 'active';
seg.append(treeBtn, mergeBtn);
const statPills = document.createElement('div');
statPills.className = 'stat-pills';
topBar.append(seg, statPills);
container.prepend(topBar);
function updateStats() {
if (!showFSM) {
statPills.innerHTML = '<span class="stat-pill">' + treeNodes.length + ' nodes</span><span class="stat-pill">' + treeEdges.length + ' edges</span>';
} else {
const ratio = Math.round(treeNodes.length / fsmNodes.length);
statPills.innerHTML = '<span class="stat-pill">' + fsmNodes.length + ' states</span><span class="stat-pill">' + fsmEdges.length + ' edges</span><span class="stat-pill highlight">' + treeNodes.length + ' \u2192 ' + fsmNodes.length + ' (' + ratio + '\u00d7)</span>';
}
}
function render() {
const oldSvg = container.querySelector('svg');
if (oldSvg) {
oldSvg.style.transition = 'opacity 0.2s ease';
oldSvg.style.opacity = '0';
setTimeout(() => oldSvg.remove(), 200);
}
const rect = container.getBoundingClientRect();
const W = Math.max(400, Math.round(rect.width));
const H = 420;
updateStats();
setTimeout(() => {
const svg = d3.select(container).append('svg')
.attr('width', W).attr('height', H)
.style('opacity', '0')
.style('transition', 'opacity 0.25s ease');
if (!showFSM) {
// ========== Prefix Tree ==========
// Arrow marker for tree
const defs = svg.append('defs');
defs.append('marker')
.attr('id', 'pc-tree-arrow').attr('viewBox', '0 0 10 6')
.attr('refX', 10).attr('refY', 3)
.attr('markerWidth', 5).attr('markerHeight', 3.5)
.attr('orient', 'auto')
.append('path').attr('d', 'M0,0 L10,3 L0,6').attr('fill', '#b5afa5');
const treeData = d3.stratify()
.id(d => d.id)
.parentId(d => d.parent)(treeNodes);
const treeLayout = d3.tree().size([W - 60, H - 80]);
const layoutData = treeLayout(treeData);
// Edges - warm taupe
svg.append('g').selectAll('path').data(layoutData.links()).enter().append('path')
.attr('d', d => {
const sx = d.source.x + 30, sy = d.source.y + 35;
const tx = d.target.x + 30, ty = d.target.y + 35;
const my = (sy + ty) / 2;
return 'M' + sx + ',' + sy + ' C' + sx + ',' + my + ' ' + tx + ',' + my + ' ' + tx + ',' + ty;
})
.attr('stroke', '#b5afa5')
.attr('stroke-opacity', 0.35).attr('stroke-width', 1.2)
.attr('fill', 'none');
// Nodes - rust tinted by depth
svg.append('g').selectAll('circle').data(layoutData.descendants()).enter().append('circle')
.attr('cx', d => d.x + 30).attr('cy', d => d.y + 35)
.attr('r', d => d.depth === 0 ? 7 : 5)
.attr('fill', d => {
if (d.depth === 0) return 'var(--surface-bg)';
const t = Math.min(1, d.depth / 8);
return 'rgba(61, 90, 128, ' + (0.15 + t * 0.35) + ')';
})
.attr('stroke', d => d.depth === 0 ? RUST : '#b5afa5')
.attr('stroke-width', d => d.depth === 0 ? 2 : 0.8);
// Init double circle
const initNode = layoutData.descendants().find(d => d.depth === 0);
if (initNode) {
svg.append('circle')
.attr('cx', initNode.x + 30).attr('cy', initNode.y + 35)
.attr('r', 4).attr('fill', 'none').attr('stroke', RUST).attr('stroke-width', 1);
}
// Labels for shallow nodes
svg.append('g').selectAll('text').data(layoutData.descendants().filter(d => d.depth <= 2))
.enter().append('text').attr('class', 'node-label-tree')
.attr('x', d => d.x + 30).attr('y', d => d.y + 23)
.attr('font-family', 'IBM Plex Mono, ui-monospace, monospace')
.attr('font-size', '8.5px')
.text(d => d.data.label);
} else {
// ========== Merged FSM (browser-style) ==========
// Arrow marker - warm taupe
const defs = svg.append('defs');
defs.append('marker')
.attr('id', 'pc-fsm-arrow').attr('viewBox', '0 0 10 6')
.attr('refX', 10).attr('refY', 3)
.attr('markerWidth', 7).attr('markerHeight', 4.5)
.attr('orient', 'auto')
.append('path').attr('d', 'M0,0 L10,3 L0,6').attr('fill', '#8a8478');
// Pre-compute degree
const deg = {};
fsmNodes.forEach(n => { deg[n.id] = 0; });
fsmEdges.forEach(e => { deg[e.source]++; deg[e.target]++; });
const maxDeg = Math.max(1, ...Object.values(deg));
// Edge set for bidirectional detection
const edgeSet = new Set();
fsmEdges.forEach(e => edgeSet.add(e.source + '|' + e.target));
function nodeR(id) {
if (id === 'f-init') return 13;
return 12 + (deg[id] / maxDeg) * 6;
}
// Force layout - run to completion, then static drag
const nodes = fsmNodes.map(n => ({ ...n }));
const links = fsmEdges.map(e => ({ source: e.source, target: e.target }));
const simulation = d3.forceSimulation(nodes)
.force('link', d3.forceLink(links).id(d => d.id).distance(85))
.force('charge', d3.forceManyBody().strength(-400))
.force('center', d3.forceCenter(W / 2, H / 2))
.force('collision', d3.forceCollide().radius(d => nodeR(d.id) + 6))
.stop();
// Run simulation to completion (static layout)
for (let i = 0; i < 300; i++) simulation.tick();
const linkGroup = svg.append('g');
const nodeGroup = svg.append('g');
// Edges
const linkSel = linkGroup.selectAll('path').data(links).enter().append('path')
.attr('fill', 'none')
.attr('stroke', '#b5afa5')
.attr('stroke-width', d => {
return 0.5 + ((deg[d.source.id] + deg[d.target.id]) / (maxDeg * 2)) * 2.5;
})
.attr('stroke-opacity', d => {
return 0.3 + ((deg[d.source.id] + deg[d.target.id]) / (maxDeg * 2)) * 0.5;
})
.attr('marker-end', 'url(#pc-fsm-arrow)');
// Nodes
const nodeGs = nodeGroup.selectAll('g').data(nodes).enter().append('g')
.attr('cursor', 'grab');
nodeGs.each(function(d) {
const g = d3.select(this);
const isInit = d.id === 'f-init';
const r = nodeR(d.id);
const degNorm = deg[d.id] / maxDeg;
if (isInit) {
g.append('circle').attr('r', r)
.attr('fill', 'var(--surface-bg)').attr('stroke', RUST).attr('stroke-width', 2);
g.append('circle').attr('r', r - 3)
.attr('fill', 'none').attr('stroke', RUST).attr('stroke-width', 1);
} else {
g.append('circle').attr('r', r)
.attr('fill', 'rgba(61, 90, 128, ' + (0.06 + degNorm * 0.14) + ')')
.attr('stroke', degNorm > 0.5 ? RUST : '#b5afa5')
.attr('stroke-width', degNorm > 0.5 ? 1.5 : 1);
}
g.append('text')
.attr('y', r + 12)
.attr('text-anchor', 'middle')
.attr('fill', 'var(--text-color)').attr('opacity', 0.7)
.attr('font-size', '9px')
.attr('font-family', 'IBM Plex Mono, ui-monospace, monospace')
.attr('font-weight', '600')
.attr('pointer-events', 'none')
.text(d.label);
});
// Update positions helper (no simulation, just re-render)
function updatePositions() {
linkSel.attr('d', d => {
const s = d.source, t = d.target;
const rS = nodeR(s.id), rT = nodeR(t.id);
if (s.id === t.id) {
return 'M' + s.x + ',' + (s.y - rS) +
' C' + (s.x - 30) + ',' + (s.y - rS - 35) +
' ' + (s.x + 30) + ',' + (s.y - rS - 35) +
' ' + s.x + ',' + (s.y - rS);
}
const dx = t.x - s.x, dy = t.y - s.y;
const dist = Math.sqrt(dx * dx + dy * dy) || 1;
const nx = dx / dist, ny = dy / dist;
if (edgeSet.has(t.id + '|' + s.id)) {
const curve = 22;
const mx = (s.x + t.x) / 2 - ny * curve;
const my = (s.y + t.y) / 2 + nx * curve;
const x1 = s.x + nx * (rS + 2) - ny * 3;
const y1 = s.y + ny * (rS + 2) + nx * 3;
const x2 = t.x - nx * (rT + 4) - ny * 3;
const y2 = t.y - ny * (rT + 4) + nx * 3;
return 'M' + x1 + ',' + y1 + ' Q' + mx + ',' + my + ' ' + x2 + ',' + y2;
}
const x1 = s.x + nx * (rS + 2);
const y1 = s.y + ny * (rS + 2);
const x2 = t.x - nx * (rT + 4);
const y2 = t.y - ny * (rT + 4);
return 'M' + x1 + ',' + y1 + ' L' + x2 + ',' + y2;
});
nodeGs.attr('transform', d => {
d.x = Math.max(40, Math.min(W - 40, d.x));
d.y = Math.max(40, Math.min(H - 40, d.y));
return 'translate(' + d.x + ',' + d.y + ')';
});
}
// Initial render from pre-computed positions
updatePositions();
// Static drag - only moves the dragged node, no forces
nodeGs.call(d3.drag()
.on('start', function() { d3.select(this).attr('cursor', 'grabbing'); })
.on('drag', (ev, d) => { d.x = ev.x; d.y = ev.y; updatePositions(); })
.on('end', function() { d3.select(this).attr('cursor', 'grab'); })
);
}
// Fade in
requestAnimationFrame(() => { svg.style('opacity', '1'); });
}, oldSvg ? 200 : 0);
}
treeBtn.addEventListener('click', () => {
if (!showFSM) return;
showFSM = false;
treeBtn.className = 'active';
mergeBtn.className = '';
render();
});
mergeBtn.addEventListener('click', () => {
if (showFSM) return;
showFSM = true;
mergeBtn.className = 'active';
treeBtn.className = '';
render();
});
render();
let resizeTimer;
if (window.ResizeObserver) new ResizeObserver(() => {
clearTimeout(resizeTimer);
resizeTimer = setTimeout(() => render(), 150);
}).observe(container);
};
if (document.readyState === 'loading') document.addEventListener('DOMContentLoaded', () => ensureD3(bootstrap), { once: true });
else ensureD3(bootstrap);
})();
</script>
</div></div><figcaption class="html-embed__desc" style="text-align:left">Watch a prefix tree collapse into a compact FSM through structural merging. States with identical outgoing transition patterns are merged iteratively until no further merge is possible.</figcaption></figure>
<h2 id="the-resulting-fsm"><a href="#the-resulting-fsm">The Resulting FSM</a></h2>
<p>The FSM <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi mathvariant="script">M</mi><mo>=</mo><mo stretchy="false">(</mo><mi>Q</mi><mo separator="true">,</mo><mi mathvariant="script">A</mi><mo separator="true">,</mo><mi>δ</mi><mo separator="true">,</mo><msub><mi>q</mi><mn>0</mn></msub><mo stretchy="false">)</mo></mrow><annotation encoding="application/x-tex">\mathcal{M} = (Q, \mathcal{A}, \delta, q_0)</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6833em"></span><span class="mord mathcal">M</span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">=</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em"></span><span class="mopen">(</span><span class="mord mathnormal">Q</span><span class="mpunct">,</span><span class="mspace" style="margin-right:0.1667em"></span><span class="mord mathcal">A</span><span class="mpunct">,</span><span class="mspace" style="margin-right:0.1667em"></span><span class="mord mathnormal" style="margin-right:0.03785em">δ</span><span class="mpunct">,</span><span class="mspace" style="margin-right:0.1667em"></span><span class="mord"><span class="mord mathnormal" style="margin-right:0.03588em">q</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.3011em"><span style="top:-2.55em;margin-left:-0.0359em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight">0</span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.15em"><span></span></span></span></span></span></span><span class="mclose">)</span></span></span></span> encodes the agent’s <strong>behavioral topology</strong>. Recurring patterns become loops. And the state count tracks the number of distinct behavioral modes.</p>
<p>In the tau2-bench retail and telecom customer service agents, a tool-call loop (<code>assistant:tool_call</code> to <code>tool:text</code>) dominates execution, with the conversational path through <code>assistant:text</code> as a separate branch. We find that in a coding agent the <code>search</code>, <code>edit</code>, <code>execute</code> cycle accounts for most of the trace.</p>
<figure class="html-embed"><figcaption class="html-embed__title" style="text-align:left">Explorable FSM Graphs</figcaption><div class="html-embed__card"><div id="frag-gaz7qnjakj7"><!-- FSM Force Graph: Explorable FSM - browser-style warm rendering -->
<div class="fsm-force-graph"></div>
<style>
.fsm-force-graph { position: relative; width: 100%; min-height: 500px; }
.fsm-force-graph svg { display: block; width: 100%; }
.fsm-force-graph .top-bar {
display: flex; align-items: center; gap: 12px; margin-bottom: 14px; flex-wrap: wrap;
}
.fsm-force-graph .seg-control {
display: inline-flex; background: var(--surface-bg);
border: 1px solid var(--border-color); border-radius: 8px;
padding: 3px; gap: 2px;
}
.fsm-force-graph .seg-control button {
padding: 5px 14px; border-radius: 6px; border: none;
background: transparent; font-size: 12px; font-weight: 500;
color: var(--text-color); cursor: pointer; transition: all 0.15s ease;
opacity: 0.6;
}
.fsm-force-graph .seg-control button:hover { opacity: 0.8; }
.fsm-force-graph .seg-control button.active {
background: var(--text-color); color: var(--page-bg);
opacity: 1; font-weight: 600;
}
.fsm-force-graph .stat-pills {
display: flex; gap: 8px; margin-left: auto; flex-wrap: wrap;
}
.fsm-force-graph .stat-pill {
padding: 4px 12px; border-radius: 20px; font-size: 11px; font-weight: 600;
border: 1px solid var(--border-color); background: var(--surface-bg);
color: var(--text-color); font-variant-numeric: tabular-nums;
}
.fsm-force-graph .action-bar {
display: flex; align-items: center; gap: 8px; margin-bottom: 10px;
}
.fsm-force-graph .action-bar button {
padding: 4px 12px; border-radius: 6px; font-size: 11px; font-weight: 500;
border: 1px solid var(--border-color); background: var(--surface-bg);
color: var(--text-color); cursor: pointer; transition: all 0.15s ease; opacity: 0.6;
}
.fsm-force-graph .action-bar button:hover { opacity: 1; }
.fsm-force-graph .tooltip {
position: absolute; top: 0; left: 0; pointer-events: none; padding: 10px 14px; border-radius: 8px;
font-size: 12px; line-height: 1.6; border: 1px solid var(--border-color);
background: var(--surface-bg); color: var(--text-color);
box-shadow: 0 4px 20px rgba(0,0,0,0.12), 0 0 0 1px rgba(0,0,0,0.04);
backdrop-filter: blur(12px); -webkit-backdrop-filter: blur(12px);
opacity: 0; transition: opacity 0.15s ease;
z-index: 100; max-width: 260px;
font-variant-numeric: tabular-nums;
}
</style>
<script>
(() => {
const ensureD3 = (cb) => {
if (window.d3 && typeof window.d3.select === 'function') return cb();
let s = document.getElementById('d3-cdn-script');
if (!s) { s = document.createElement('script'); s.id = 'd3-cdn-script'; s.src = 'https://cdn.jsdelivr.net/npm/d3@7/dist/d3.min.js'; document.head.appendChild(s); }
s.addEventListener('load', () => cb(), { once: true });
};
const bootstrap = () => {
const container = document.querySelector('.fsm-force-graph:not([data-mounted])');
if (!container) return;
container.dataset.mounted = 'true';
const d3 = window.d3;
// --- Palette (ported from browser/) ---
const RUST = '#3d5a80';
const graphs = {
'ATBench': {
states: ['init','user:text','assistant:tool:extract','tool:text','assistant:tool:read','assistant:tool:create','assistant:text','assistant:tool:delete','assistant:tool:update','assistant:tool:comm','assistant:tool:auth','assistant:tool:search','assistant:tool:misc','assistant:tool:compute','assistant:tool:exec'],
edges: [['init','user:text'],['user:text','assistant:tool:extract'],['assistant:tool:extract','tool:text'],['tool:text','assistant:tool:read'],['assistant:tool:read','tool:text'],['tool:text','assistant:tool:create'],['assistant:tool:create','tool:text'],['tool:text','assistant:text'],['user:text','assistant:tool:read'],['tool:text','assistant:tool:delete'],['assistant:tool:delete','tool:text'],['tool:text','assistant:tool:extract'],['tool:text','assistant:tool:update'],['assistant:tool:update','tool:text'],['assistant:text','user:text'],['user:text','assistant:tool:comm'],['assistant:tool:comm','tool:text'],['user:text','assistant:tool:auth'],['assistant:tool:auth','tool:text'],['user:text','assistant:tool:search'],['assistant:tool:search','tool:text'],['user:text','assistant:tool:misc'],['assistant:tool:misc','tool:text'],['tool:text','assistant:tool:misc'],['user:text','assistant:tool:create'],['user:text','assistant:tool:delete'],['tool:text','assistant:tool:comm'],['user:text','assistant:tool:update'],['tool:text','assistant:tool:search'],['user:text','assistant:text'],['tool:text','assistant:tool:auth'],['tool:text','assistant:tool:compute'],['assistant:tool:compute','tool:text'],['user:text','assistant:tool:exec'],['assistant:tool:exec','tool:text'],['user:text','assistant:tool:compute'],['tool:text','assistant:tool:exec']]
},
'SWE-smith': {
states: ['init','system:text','user:text','assistant:tool:bash','tool:text','assistant:tool:str_replace_editor','assistant:tool:submit','assistant:text','tool:tool:str','tool:tool_call'],
edges: [['init','system:text'],['system:text','user:text'],['user:text','assistant:tool:bash'],['assistant:tool:bash','tool:text'],['tool:text','assistant:tool:str_replace_editor'],['assistant:tool:str_replace_editor','tool:text'],['tool:text','assistant:tool:bash'],['tool:text','assistant:tool:submit'],['assistant:tool:submit','tool:text'],['tool:text','assistant:text'],['assistant:tool:str_replace_editor','tool:tool:str'],['tool:tool:str','assistant:tool:str_replace_editor'],['tool:tool:str','assistant:tool:bash'],['tool:tool_call','assistant:tool:bash'],['assistant:tool:str_replace_editor','tool:tool_call'],['tool:tool_call','assistant:tool:str_replace_editor'],['assistant:tool:bash','tool:tool_call']]
},
'SWE-agent': {
states: ['init','user:text','edit:text','execute:text','navigate:text','search:text','submit:text','user:tool:of','user:cli','assistant:text','user:tool:dict','user:tool_call','user:metadata','user:tool:132','user:tool:os','user:tool:filename','user:jupytext','user:versioneer','user:0','navigate:tool_call','search:tool_call','user:tool:node','user:tool:not','user:tool:None','user:influx2'],
edges: [['init','user:text'],['user:text','edit:text'],['edit:text','user:text'],['user:text','execute:text'],['execute:text','user:text'],['user:text','navigate:text'],['navigate:text','user:text'],['user:text','search:text'],['search:text','user:text'],['user:text','submit:text'],['edit:text','user:tool:of'],['user:tool:of','execute:text'],['execute:text','user:cli'],['user:text','assistant:text'],['assistant:text','user:text'],['user:tool:dict','edit:text'],['search:text','user:tool_call'],['user:tool_call','search:text'],['execute:text','user:tool_call'],['user:tool_call','navigate:text'],['user:tool_call','edit:text'],['user:metadata','edit:text'],['navigate:text','user:tool:132'],['user:tool:132','navigate:text'],['user:tool:os','edit:text'],['edit:text','user:tool:filename'],['user:tool:filename','execute:text'],['submit:text','user:text'],['execute:text','user:jupytext'],['user:jupytext','execute:text'],['user:versioneer','edit:text'],['execute:text','user:0'],['navigate:text','user:tool_call'],['navigate:tool_call','user:text'],['search:tool_call','user:text'],['user:tool:node','execute:text'],['edit:text','user:tool:not'],['user:tool:not','execute:text'],['edit:text','user:tool:None'],['user:tool:None','execute:text'],['execute:text','user:influx2'],['user:influx2','execute:text'],['user:influx2','navigate:text']]
}
};
const datasetKeys = Object.keys(graphs);
let currentKey = 'SWE-agent';
function shortLabel(id) {
if (id === 'init') return 'init';
const parts = id.split(':');
if (parts.length >= 3) return parts[parts.length - 1].slice(0, 12);
if (parts.length === 2) return parts[1].slice(0, 12);
return id.slice(0, 12);
}
// --- UI ---
const topBar = document.createElement('div');
topBar.className = 'top-bar';
const seg = document.createElement('div');
seg.className = 'seg-control';
const buttons = {};
datasetKeys.forEach(key => {
const btn = document.createElement('button');
const g = graphs[key];
btn.textContent = key + ' (' + g.states.length + ')';
btn.dataset.key = key;
if (key === currentKey) btn.className = 'active';
btn.addEventListener('click', () => {
if (key === currentKey) return;
currentKey = key;
Object.values(buttons).forEach(b => b.className = '');
btn.className = 'active';
renderGraph(key);
});
buttons[key] = btn;
seg.appendChild(btn);
});
const statPills = document.createElement('div');
statPills.className = 'stat-pills';
topBar.append(seg, statPills);
container.prepend(topBar);
const actionBar = document.createElement('div');
actionBar.className = 'action-bar';
const resetBtn = document.createElement('button');
resetBtn.textContent = 'Reset Layout';
actionBar.appendChild(resetBtn);
container.appendChild(actionBar);
const tip = document.createElement('div');
tip.className = 'tooltip';
container.appendChild(tip);
let currentSimulation = null;
resetBtn.addEventListener('click', () => { renderGraph(currentKey); });
function renderGraph(name) {
const graph = graphs[name];
statPills.innerHTML = '<span class="stat-pill">' + graph.states.length + ' states</span><span class="stat-pill">' + graph.edges.length + ' transitions</span>';
if (currentSimulation) { currentSimulation.stop(); currentSimulation = null; }
container.querySelectorAll('svg').forEach(s => s.remove());
const rect = container.getBoundingClientRect();
const W = Math.max(400, Math.round(rect.width));
const H = 460;
const svg = d3.select(container).insert('svg', '.tooltip')
.attr('viewBox', '-20 -10 ' + (W + 40) + ' ' + (H + 20))
.attr('width', W).attr('height', H)
.style('overflow', 'hidden')
.style('cursor', 'grab')
.style('touch-action', 'none');
// Arrow marker - warm taupe (ported from browser/)
const defs = svg.append('defs');
defs.append('marker')
.attr('id', 'fg-arrow').attr('viewBox', '0 0 10 6')
.attr('refX', 10).attr('refY', 3)
.attr('markerWidth', 7).attr('markerHeight', 4.5)
.attr('orient', 'auto')
.append('path').attr('d', 'M0,0 L10,3 L0,6').attr('fill', '#8a8478');
defs.append('marker')
.attr('id', 'fg-arrow-rust').attr('viewBox', '0 0 10 6')
.attr('refX', 10).attr('refY', 3)
.attr('markerWidth', 7).attr('markerHeight', 4.5)
.attr('orient', 'auto')
.append('path').attr('d', 'M0,0 L10,3 L0,6').attr('fill', RUST);
// --- Pre-compute graph metrics ---
const deg = {};
graph.states.forEach(s => { deg[s] = 0; });
graph.edges.forEach(([s, t]) => { deg[s]++; deg[t]++; });
const maxDeg = Math.max(1, ...Object.values(deg));
const edgeSet = new Set();
graph.edges.forEach(([s, t]) => edgeSet.add(s + '|' + t));
function nodeR(id) {
if (id === 'init') return 13;
return 10 + (deg[id] / maxDeg) * 8;
}
// --- D3 force simulation - run to completion, then static ---
const nodes = graph.states.map(id => ({ id }));
const links = graph.edges.map(([s, t]) => ({ source: s, target: t }));
const simulation = d3.forceSimulation(nodes)
.force('link', d3.forceLink(links).id(d => d.id).distance(55 + graph.states.length * 1.5))
.force('charge', d3.forceManyBody().strength(-180 - graph.states.length * 3))
.force('center', d3.forceCenter(W / 2, H / 2))
.force('collision', d3.forceCollide().radius(d => nodeR(d.id) + 4))
.force('x', d3.forceX(W / 2).strength(0.03))
.force('y', d3.forceY(H / 2).strength(0.03))
.stop();
// Run simulation to completion (static layout)
for (let i = 0; i < 300; i++) simulation.tick();
currentSimulation = simulation;
const zoomLayer = svg.append('g').attr('class', 'zoom-layer');
const linkGroup = zoomLayer.append('g');
const nodeGroup = zoomLayer.append('g');
// Pan + zoom across the whole canvas (scroll to zoom, drag background to pan).
const zoom = d3.zoom().scaleExtent([0.3, 4])
.on('zoom', (ev) => zoomLayer.attr('transform', ev.transform));
svg.call(zoom);
// Double-click resets the view instead of zooming in.
svg.on('dblclick.zoom', () => svg.transition().duration(300).call(zoom.transform, d3.zoomIdentity));
// --- Edges ---
const linkSel = linkGroup.selectAll('path').data(links).enter().append('path')
.attr('fill', 'none')
.attr('stroke', '#b5afa5')
.attr('stroke-width', d => {
// Thicker edges for high-degree endpoints
const sId = typeof d.source === 'object' ? d.source.id : d.source;
const tId = typeof d.target === 'object' ? d.target.id : d.target;
return 0.5 + ((deg[sId] + deg[tId]) / (maxDeg * 2)) * 2.5;
})
.attr('stroke-opacity', d => {
const sId = typeof d.source === 'object' ? d.source.id : d.source;
const tId = typeof d.target === 'object' ? d.target.id : d.target;
return 0.3 + ((deg[sId] + deg[tId]) / (maxDeg * 2)) * 0.5;
})
.attr('marker-end', 'url(#fg-arrow)');
// --- Nodes (groups with circles + labels) ---
const nodeGs = nodeGroup.selectAll('g').data(nodes).enter().append('g')
.attr('cursor', 'grab');
// Render node circles
nodeGs.each(function(d) {
const g = d3.select(this);
const isInit = d.id === 'init';
const r = nodeR(d.id);
const degNorm = deg[d.id] / maxDeg;
if (isInit) {
g.append('circle').attr('r', r)
.attr('fill', 'var(--surface-bg)').attr('stroke', RUST).attr('stroke-width', 2)
.attr('class', 'node-circle');
g.append('circle').attr('r', r - 3)
.attr('fill', 'none').attr('stroke', RUST).attr('stroke-width', 1);
} else {
g.append('circle').attr('r', r)
.attr('fill', 'rgba(61, 90, 128, ' + (0.06 + degNorm * 0.14) + ')')
.attr('stroke', degNorm > 0.5 ? RUST : '#b5afa5')
.attr('stroke-width', degNorm > 0.5 ? 1.5 : 1)
.attr('class', 'node-circle');
}
// Label below node
g.append('text')
.attr('y', r + 12)
.attr('text-anchor', 'middle')
.attr('fill', 'var(--text-color)').attr('opacity', 0.7)
.attr('font-size', '8.5px')
.attr('font-family', 'IBM Plex Mono, ui-monospace, monospace')
.attr('pointer-events', 'none')
.text(shortLabel(d.id));
});
// --- Hover interactions ---
nodeGs
.on('mouseenter', function(ev, d) {
const r = nodeR(d.id);
d3.select(this).select('.node-circle')
.transition().duration(100).attr('r', r + 3).attr('stroke', RUST).attr('stroke-width', 2.5);
// Highlight connected edges
linkSel.transition().duration(100)
.attr('stroke', l => {
const sId = l.source.id, tId = l.target.id;
return (sId === d.id || tId === d.id) ? RUST : '#b5afa5';
})
.attr('stroke-opacity', l => {
const sId = l.source.id, tId = l.target.id;
return (sId === d.id || tId === d.id) ? 0.7 : 0.08;
})
.attr('stroke-width', l => {
const sId = l.source.id, tId = l.target.id;
return (sId === d.id || tId === d.id) ? 2 : 0.8;
})
.attr('marker-end', l => {
const sId = l.source.id, tId = l.target.id;
return (sId === d.id || tId === d.id) ? 'url(#fg-arrow-rust)' : 'url(#fg-arrow)';
});
// Dim other nodes
nodeGs.transition().duration(100).attr('opacity', n => {
if (n.id === d.id) return 1;
const connected = graph.edges.some(e => (e[0] === d.id && e[1] === n.id) || (e[1] === d.id && e[0] === n.id));
return connected ? 1 : 0.2;
});
// Tooltip
const outEdges = graph.edges.filter(e => e[0] === d.id);
const inEdges = graph.edges.filter(e => e[1] === d.id);
tip.innerHTML = '<strong style="color:' + RUST + '">' + d.id + '</strong><br/>Out: ' + outEdges.length + ' &nbsp; In: ' + inEdges.length;
tip.style.opacity = '1';
})
.on('mousemove', function(ev) {
const [mx, my] = d3.pointer(ev, container);
const tipW = tip.offsetWidth || 120;
const tipH = tip.offsetHeight || 40;
const cRect = container.getBoundingClientRect();
let tx = mx + 14, ty = my - 14;
if (tx + tipW > cRect.width) tx = mx - tipW - 10;
if (ty + tipH > cRect.height) ty = my - tipH - 10;
tip.style.transform = 'translate(' + Math.max(0, tx) + 'px,' + Math.max(0, ty) + 'px)';
})
.on('mouseleave', function(ev, d) {
const r = nodeR(d.id);
const degNorm = deg[d.id] / maxDeg;
d3.select(this).select('.node-circle')
.transition().duration(150)
.attr('r', r)
.attr('stroke', d.id === 'init' ? RUST : degNorm > 0.5 ? RUST : '#b5afa5')
.attr('stroke-width', d.id === 'init' ? 2 : degNorm > 0.5 ? 1.5 : 1);
linkSel.transition().duration(200)
.attr('stroke', '#b5afa5')
.attr('stroke-opacity', l => {
const sId = l.source.id, tId = l.target.id;
return 0.3 + ((deg[sId] + deg[tId]) / (maxDeg * 2)) * 0.5;
})
.attr('stroke-width', l => {
const sId = l.source.id, tId = l.target.id;
return 0.5 + ((deg[sId] + deg[tId]) / (maxDeg * 2)) * 2.5;
})
.attr('marker-end', 'url(#fg-arrow)');
nodeGs.transition().duration(200).attr('opacity', 1);
tip.style.opacity = '0';
})
.call(d3.drag()
.on('start', function(ev) { if (ev.sourceEvent) ev.sourceEvent.stopPropagation(); d3.select(this).attr('cursor', 'grabbing'); })
.on('drag', (ev, d) => { d.x = ev.x; d.y = ev.y; updatePositions(); })
.on('end', function() { d3.select(this).attr('cursor', 'grab'); })
);
// --- Position update helper (static, no forces) ---
function updatePositions() {
linkSel.attr('d', d => {
const s = d.source, t = d.target;
const rS = nodeR(s.id), rT = nodeR(t.id);
// Self-loop (cubic bezier above node)
if (s.id === t.id) {
return 'M' + s.x + ',' + (s.y - rS) +
' C' + (s.x - 30) + ',' + (s.y - rS - 35) +
' ' + (s.x + 30) + ',' + (s.y - rS - 35) +
' ' + s.x + ',' + (s.y - rS);
}
const dx = t.x - s.x, dy = t.y - s.y;
const dist = Math.sqrt(dx * dx + dy * dy) || 1;
const nx = dx / dist, ny = dy / dist;
// Bidirectional (quadratic bezier with perpendicular offset)
if (edgeSet.has(t.id + '|' + s.id)) {
const curve = 22;
const mx = (s.x + t.x) / 2 - ny * curve;
const my = (s.y + t.y) / 2 + nx * curve;
const x1 = s.x + nx * (rS + 2) - ny * 3;
const y1 = s.y + ny * (rS + 2) + nx * 3;
const x2 = t.x - nx * (rT + 4) - ny * 3;
const y2 = t.y - ny * (rT + 4) + nx * 3;
return 'M' + x1 + ',' + y1 + ' Q' + mx + ',' + my + ' ' + x2 + ',' + y2;
}
// Unidirectional (straight line)
const x1 = s.x + nx * (rS + 2);
const y1 = s.y + ny * (rS + 2);
const x2 = t.x - nx * (rT + 4);
const y2 = t.y - ny * (rT + 4);
return 'M' + x1 + ',' + y1 + ' L' + x2 + ',' + y2;
});
nodeGs.attr('transform', d => 'translate(' + d.x + ',' + d.y + ')');
}
// Initial render from pre-computed positions
updatePositions();
}
renderGraph(currentKey);
};
if (document.readyState === 'loading') document.addEventListener('DOMContentLoaded', () => ensureD3(bootstrap), { once: true });
else ensureD3(bootstrap);
})();
</script>
</div></div><figcaption class="html-embed__desc" style="text-align:left">Select a dataset to explore its extracted FSM. Node size reflects visit frequency; edge thickness reflects transition frequency. Drag nodes to rearrange, scroll to zoom, drag the background to pan, and double-click to reset the view.</figcaption></figure>
<p>Open the FSM explorer for any of the twelve datasets in the <a href="https://seongland.com/article/asg/browser?tab=graph&dataset=sweagent">live dashboard</a>.</p>
<h2 id="theoretical-properties"><a href="#theoretical-properties">Theoretical Properties</a></h2>
<p>The construction guarantees three properties.</p>
<p><strong>Fitness preservation.</strong> Structural merging preserves training fitness: if a trace is accepted by the prefix tree, it’s accepted by the merged FSM, because merging only adds out-edges — each state carries the union of its merged transitions.</p>
<p><strong>Compactness.</strong> We find the merged FSM is a compact directly-follows automaton: one state per activity, deterministic, accepting every observed trace. We recover this — not the <em>generating</em> automaton, which is impossible to identify from positive examples alone <span class="" id="citation--gold1967language--2">(<a href="#bib-gold1967language" id="refctx-bib-gold1967language-1">Gold, 1967</a>)</span>. But it’s enough for faithful replay and prediction.</p>
<p><strong>Linear runtime.</strong> Prefix tree construction is <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>O</mi><mo stretchy="false">(</mo><msub><mo>∑</mo><mi>i</mi></msub><msub><mi>T</mi><mi>i</mi></msub><mo stretchy="false">)</mo></mrow><annotation encoding="application/x-tex">O(\sum_i T_i)</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:1.0497em;vertical-align:-0.2997em"></span><span class="mord mathnormal" style="margin-right:0.02778em">O</span><span class="mopen">(</span><span class="mop"><span class="mop op-symbol small-op" style="position:relative;top:0em">∑</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.162em"><span style="top:-2.4003em;margin-left:0em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mathnormal mtight">i</span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.2997em"><span></span></span></span></span></span></span><span class="mspace" style="margin-right:0.1667em"></span><span class="mord"><span class="mord mathnormal" style="margin-right:0.13889em">T</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.3117em"><span style="top:-2.55em;margin-left:-0.1389em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mathnormal mtight">i</span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.15em"><span></span></span></span></span></span></span><span class="mclose">)</span></span></span></span>, and structural merging is a partition refinement <span class="" id="citation--hopcroft2006automata--3">(<a href="#bib-hopcroft2006automata" id="refctx-bib-hopcroft2006automata-2">Hopcroft et al., 2006</a>)</span> in <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>O</mi><mo stretchy="false">(</mo><mi mathvariant="normal">∣</mi><msub><mi>Q</mi><mi mathvariant="script">P</mi></msub><mi mathvariant="normal">∣</mi><mo>⋅</mo><mi mathvariant="normal">∣</mi><mi mathvariant="script">A</mi><mi mathvariant="normal">∣</mi><mo stretchy="false">)</mo></mrow><annotation encoding="application/x-tex">O(|Q_\mathcal{P}| \cdot |\mathcal{A}|)</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em"></span><span class="mord mathnormal" style="margin-right:0.02778em">O</span><span class="mopen">(</span><span class="mord">∣</span><span class="mord"><span class="mord mathnormal">Q</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.3283em"><span style="top:-2.55em;margin-left:0em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mathcal mtight" style="margin-right:0.08222em">P</span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.15em"><span></span></span></span></span></span></span><span class="mord">∣</span><span class="mspace" style="margin-right:0.2222em"></span><span class="mbin">⋅</span><span class="mspace" style="margin-right:0.2222em"></span></span><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em"></span><span class="mord">∣</span><span class="mord mathcal">A</span><span class="mord">∣</span><span class="mclose">)</span></span></span></span>. In practice all twelve datasets complete in under one second on a single CPU core.</p>
<div class="note note--info" data-astro-cid-qg6lmfty> <!-- When there's no title, emoji is above content -->
<div class="note__layout" data-astro-cid-qg6lmfty> <div class="note__body" data-astro-cid-qg6lmfty> <div class="note__content" data-astro-cid-qg6lmfty> <p>The construction itself is classical <span class="" id="citation--daciuk2000incremental--4">(<a href="#bib-daciuk2000incremental" id="refctx-bib-daciuk2000incremental-1">Daciuk et al., 2000</a>)</span>. Bounded agent alphabets are new: they make the resulting compact automaton small enough — and dense enough per state — to be useful for the prediction and monitoring tasks that follow.</p> </div> </div> </div> </div> <div id="bibliography-references-list" class="references csl-bib-body" data-bibliography-block="true" data-built-refs="1"><ol class="references"><li id="bib-daciuk2000incremental">Daciuk, J., Mihov, S., Watson, B. W., &amp; Watson, R. E. (2000). Incremental Construction of Minimal Acyclic Finite-State Automata. <i>Computational Linguistics</i>, <i>26</i>(1), 3–16.<small class="backrefs"><a href="#refctx-bib-daciuk2000incremental-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-gold1967language">Gold, E. M. (1967). Language Identification in the Limit. <i>Information and Control</i>, <i>10</i>(5), 447–474.<small class="backrefs"><a href="#refctx-bib-gold1967language-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-hopcroft2006automata">Hopcroft, J. E., Motwani, R., &amp; Ullman, J. D. (2006). <i>Introduction to Automata Theory, Languages, and Computation</i> (3rd ed.). Pearson.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-hopcroft2006automata-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-hopcroft2006automata-2" aria-label="Back to citation">2</a></small></li></ol></div>
<h1 id="why-the-machine-stays-small"><a href="#why-the-machine-stays-small">Why the Machine Stays Small</a></h1>
<p>A small state count alone doesn’t make a machine useful; an automaton can still be huge if the language is complex. And a huge automaton spreads its observations thinly (leaving every per-state statistic noisy). The FSM works as a substrate because it’s small, stable, and converges fast, so each state pools enough traces to estimate from. The prediction and monitoring chapters depend on that property, not the exact count.</p>
<h2 id="it-converges-on-a-few-percent-of-the-data"><a href="#it-converges-on-a-few-percent-of-the-data">It converges on a few percent of the data</a></h2>
<p>Across five random train/test splits the extracted state count is <strong>identical every time</strong>, so the topology is a property of the agent, not of which traces you happened to sample. Replay fitness plateaus just as fast: within the first <strong>1 to 10% of the training traces</strong> every dataset clears 0.95 fitness, and SWE-smith holds 0.9996 from the first 1%.</p>
<div class="sidenote-container"> <aside class="sidenote"> <p>Motwani and Raghavan <span class="" id="citation--motwani1995randomized--1">(<a href="#bib-motwani1995randomized" id="refctx-bib-motwani1995randomized-1">Motwani &amp; Raghavan, 1995</a>)</span> give the formal version: if traces are i.i.d. and each transition appears with probability at least <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><msub><mi>p</mi><mi>min</mi><mo>⁡</mo></msub></mrow><annotation encoding="application/x-tex">p_{\min}</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.625em;vertical-align:-0.1944em"></span><span class="mord"><span class="mord mathnormal">p</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.3175em"><span style="top:-2.55em;margin-left:0em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mop mtight"><span class="mtight">m</span><span class="mtight">i</span><span class="mtight">n</span></span></span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.15em"><span></span></span></span></span></span></span></span></span></span>, the extracted FSM equals the population FSM after <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>N</mi><mo>≥</mo><mfrac><mn>1</mn><msub><mi>p</mi><mi>min</mi><mo>⁡</mo></msub></mfrac><mi>ln</mi><mo>⁡</mo><mo stretchy="false">(</mo><mi>k</mi><mi mathvariant="normal">/</mi><mi>δ</mi><mo stretchy="false">)</mo></mrow><annotation encoding="application/x-tex">N \geq \frac{1}{p_{\min}} \ln(k/\delta)</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.8193em;vertical-align:-0.136em"></span><span class="mord mathnormal" style="margin-right:0.10903em">N</span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">≥</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:1.3262em;vertical-align:-0.4811em"></span><span class="mord"><span class="mopen nulldelimiter"></span><span class="mfrac"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.8451em"><span style="top:-2.655em"><span class="pstrut" style="height:3em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mord mtight"><span class="mord mathnormal mtight">p</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.334em"><span style="top:-2.357em;margin-left:0em;margin-right:0.0714em"><span class="pstrut" style="height:2.5em"></span><span class="sizing reset-size3 size1 mtight"><span class="mord mtight"><span class="mop mtight"><span class="mtight">m</span><span class="mtight">i</span><span class="mtight">n</span></span></span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.143em"><span></span></span></span></span></span></span></span></span></span><span style="top:-3.23em"><span class="pstrut" style="height:3em"></span><span class="frac-line" style="border-bottom-width:0.04em"></span></span><span style="top:-3.394em"><span class="pstrut" style="height:3em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mord mtight">1</span></span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.4811em"><span></span></span></span></span></span><span class="mclose nulldelimiter"></span></span><span class="mspace" style="margin-right:0.1667em"></span><span class="mop">ln</span><span class="mopen">(</span><span class="mord mathnormal" style="margin-right:0.03148em">k</span><span class="mord">/</span><span class="mord mathnormal" style="margin-right:0.03785em">δ</span><span class="mclose">)</span></span></span></span> traces. For SWE-agent, where <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>k</mi><mo>=</mo><mn>43</mn></mrow><annotation encoding="application/x-tex">k=43</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6944em"></span><span class="mord mathnormal" style="margin-right:0.03148em">k</span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">=</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:0.6444em"></span><span class="mord">43</span></span></span></span> transitions and <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><msub><mi>p</mi><mi>min</mi><mo>⁡</mo></msub><mo>≈</mo><mn>0.01</mn></mrow><annotation encoding="application/x-tex">p_{\min} \approx 0.01</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6776em;vertical-align:-0.1944em"></span><span class="mord"><span class="mord mathnormal">p</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.3175em"><span style="top:-2.55em;margin-left:0em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mop mtight"><span class="mtight">m</span><span class="mtight">i</span><span class="mtight">n</span></span></span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.15em"><span></span></span></span></span></span></span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">≈</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:0.6444em"></span><span class="mord">0.01</span></span></span></span>, that is at most 676 traces at <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>δ</mi><mo>=</mo><mn>0.05</mn></mrow><annotation encoding="application/x-tex">\delta=0.05</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6944em"></span><span class="mord mathnormal" style="margin-right:0.03785em">δ</span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">=</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:0.6444em"></span><span class="mord">0.05</span></span></span></span>, comfortably under the 2,000 we train on.</p> </aside> </div>
<h2 id="three-different-methods-agree"><a href="#three-different-methods-agree">Three different methods agree</a></h2>
<p>The compactness isn’t an artifact of our particular merge rule: three fundamentally different algorithms converge to nearly the same state count on every dataset: <strong>structural merging</strong>, ours, a structural partition; <strong>Alergia</strong> <span class="" id="citation--carrasco1994alergia--2">(<a href="#bib-carrasco1994alergia" id="refctx-bib-carrasco1994alergia-1">Carrasco &amp; Oncina, 1994</a>)</span>, a statistical merge that lands within 1.0 to 6.0<span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mo>×</mo></mrow><annotation encoding="application/x-tex">\times</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6667em;vertical-align:-0.0833em"></span><span class="mord">×</span></span></span></span> of our own state count; and a <strong><a href="https://texonom.com/126c3c96247d80e0b880c38e5f3bb924">hidden Markov model</a></strong> (HMM) <span class="" id="citation--rabiner1989hmm--3">(<a href="#bib-rabiner1989hmm" id="refctx-bib-rabiner1989hmm-1">Rabiner, 1989</a>)</span>, a probabilistic latent-state model that matches the count, though its states aren’t interpretable.</p>
<p>A structural, a statistical, and a probabilistic method all agree — which rules out an algorithmic coincidence.</p><div id="bibliography-references-list" class="references csl-bib-body" data-bibliography-block="true" data-built-refs="1"><ol class="references"><li id="bib-carrasco1994alergia">Carrasco, R. C., &amp; Oncina, J. (1994). Learning Stochastic Regular Grammars by Means of a State Merging Method. <i>International Colloquium on Grammatical Inference</i>, 139–152.<small class="backrefs"><a href="#refctx-bib-carrasco1994alergia-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-motwani1995randomized">Motwani, R., &amp; Raghavan, P. (1995). <i>Randomized Algorithms</i>. Cambridge University Press.<small class="backrefs"><a href="#refctx-bib-motwani1995randomized-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-rabiner1989hmm">Rabiner, L. R. (1989). A Tutorial on Hidden Markov Models and Selected Applications in Speech Recognition. <i>Proceedings of the IEEE</i>, <i>77</i>(2), 257–286.<small class="backrefs"><a href="#refctx-bib-rabiner1989hmm-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li></ol></div>
<h1 id="compression-and-comparison"><a href="#compression-and-comparison">Compression and Comparison</a></h1>
<p>We compare against nine baselines: RPNI <span class="" id="citation--oncina1992rpni--1">(<a href="#bib-oncina1992rpni" id="refctx-bib-oncina1992rpni-1">Oncina &amp; Garcı́a, 1992</a>)</span>, EDSM <span class="" id="citation--lang1998edsm--2">(<a href="#bib-lang1998edsm" id="refctx-bib-lang1998edsm-1">Lang et al., 1998</a>)</span>, Alergia <span class="" id="citation--carrasco1994alergia--3">(<a href="#bib-carrasco1994alergia" id="refctx-bib-carrasco1994alergia-1">Carrasco &amp; Oncina, 1994</a>)</span> and k-Tails <span class="" id="citation--biermann1972ktails--4">(<a href="#bib-biermann1972ktails" id="refctx-bib-biermann1972ktails-1">Biermann &amp; Feldman, 1972</a>)</span> from <a href="https://texonom.com/37bc3c96247d803b8156ee3fcfdd4556">automata learning</a>, run through AALpy <span class="" id="citation--muskardin2022aalpy--5">(<a href="#bib-muskardin2022aalpy" id="refctx-bib-muskardin2022aalpy-1">Muškardin et al., 2022</a>)</span>; the HMM <span class="" id="citation--rabiner1989hmm--6">(<a href="#bib-rabiner1989hmm" id="refctx-bib-rabiner1989hmm-1">Rabiner, 1989</a>)</span>; the Alpha, Inductive and Heuristic miners from process mining <span class="" id="citation--vanderaalst2016process--7">(<a href="#bib-vanderaalst2016process" id="refctx-bib-vanderaalst2016process-1">van der Aalst, 2016</a>)</span>, run through PM4Py <span class="" id="citation--berti2019pm4py--8">(<a href="https://arxiv.org/abs/1905.06169" id="refctx-bib-berti2019pm4py-1" data-ref-id="bib-berti2019pm4py" target="_blank" rel="noopener noreferrer">Berti et al., 2019</a>)</span>; and AWM, agent workflow extraction <span class="" id="citation--wang2024agent_workflow_memory--9">(<a href="#bib-wang2024agent_workflow_memory" id="refctx-bib-wang2024agent_workflow_memory-1">Wang et al., 2024</a>)</span>. All receive only the same positive training sequences, no failure labels.</p>
<h2 id="how-much-smaller"><a href="#how-much-smaller">How Much Smaller</a></h2>
<p>Our FSMs achieve <strong>15 to 3,036<span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mo>×</mo></mrow><annotation encoding="application/x-tex">\times</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6667em;vertical-align:-0.0833em"></span><span class="mord">×</span></span></span></span> compression</strong> over RPNI while replaying held-out traces at fitness of at least 0.997. The ratio grows with trace length and branching: 15<span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mo>×</mo></mrow><annotation encoding="application/x-tex">\times</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6667em;vertical-align:-0.0833em"></span><span class="mord">×</span></span></span></span> on WebArena (short web traces, where RPNI succeeds) up to 3,036<span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mo>×</mo></mrow><annotation encoding="application/x-tex">\times</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6667em;vertical-align:-0.0833em"></span><span class="mord">×</span></span></span></span> on GUI-Odyssey <span class="" id="citation--lu2024guiodyssey--10">(<a href="#bib-lu2024guiodyssey" id="refctx-bib-lu2024guiodyssey-1">Lu et al., 2025</a>)</span> (long, repetitive mobile-GUI traces, where RPNI’s prefix tree explodes to 21,255 states against our 7).</p>
<div class="wide"> <figure class="html-embed"><figcaption class="html-embed__title" style="text-align:left">Compression Ratios</figcaption><div class="html-embed__card"><div id="frag-0f6dvhcb8vn5"><!-- Compression: log-scale dumbbell of ours vs RPNI state counts across 12 datasets -->
<div class="compression-bars"></div>
<style>
.compression-bars { position: relative; width: 100%; min-height: 440px; }
.compression-bars svg { display: block; width: 100%; }
.compression-bars .tick text {
fill: var(--text-color); font-size: 11px; opacity: 0.6;
font-variant-numeric: tabular-nums;
}
.compression-bars .tick line { display: none; }
.compression-bars .domain { stroke: var(--border-color); opacity: 0.3; }
.compression-bars .y-axis .domain { display: none; }
.compression-bars .axis-label {
fill: var(--text-color); font-size: 12px; font-weight: 500;
}
.compression-bars .cmp-row { transition: opacity 0.15s ease; }
.compression-bars .tooltip {
position: absolute; top: 0; left: 0; pointer-events: none; padding: 10px 14px; border-radius: 8px;
font-size: 12px; line-height: 1.6; border: 1px solid var(--border-color);
background: var(--surface-bg); color: var(--text-color);
box-shadow: 0 4px 20px rgba(0,0,0,0.12), 0 0 0 1px rgba(0,0,0,0.04);
backdrop-filter: blur(12px); -webkit-backdrop-filter: blur(12px);
opacity: 0; transition: opacity 0.15s ease;
z-index: 100; max-width: 240px;
font-variant-numeric: tabular-nums;
}
</style>
<script>
(() => {
const ensureD3 = (cb) => {
if (window.d3 && typeof window.d3.select === 'function') return cb();
let s = document.getElementById('d3-cdn-script');
if (!s) { s = document.createElement('script'); s.id = 'd3-cdn-script'; s.src = 'https://cdn.jsdelivr.net/npm/d3@7/dist/d3.min.js'; document.head.appendChild(s); }
s.addEventListener('load', () => cb(), { once: true });
};
const bootstrap = () => {
const container = document.querySelector('.compression-bars:not([data-mounted])');
if (!container) return;
container.dataset.mounted = 'true';
const d3 = window.d3;
const data = [
{ name: 'WebArena', ours: 25, rpni: 382, compression: 15, fitness: 1.000 },
{ name: 'ATBench', ours: 15, rpni: 899, compression: 60, fitness: 1.000 },
{ name: 'Mind2Web', ours: 8, rpni: 476, compression: 60, fitness: 1.000 },
{ name: 'Who & When', ours: 9, rpni: 971, compression: 108, fitness: 1.000 },
{ name: 'tau2-bench air', ours: 18, rpni: 6506, compression: 361, fitness: 1.000 },
{ name: 'tau2-bench ret', ours: 19, rpni: 14249, compression: 750, fitness: 1.000 },
{ name: 'SWE-smith', ours: 10, rpni: 11631, compression: 1163, fitness: 1.000 },
{ name: 'OSWorld', ours: 27, rpni: 38232, compression: 1416, fitness: 0.997 },
{ name: 'tau2-bench tel', ours: 43, rpni: 63897, compression: 1486, fitness: 1.000 },
{ name: 'SWE-agent', ours: 25, rpni: 59510, compression: 2380, fitness: 0.999 },
{ name: 'AgentNet', ours: 25, rpni: 62495, compression: 2500, fitness: 1.000 },
{ name: 'GUI-Odyssey', ours: 7, rpni: 21255, compression: 3036, fitness: 1.000 },
].sort((a, b) => a.compression - b.compression);
const C_OURS = '#3d5a80'; // strong slate -- ours
const C_RPNI = '#8fa6c4'; // light slate -- RPNI
const margin = { top: 36, right: 78, bottom: 44, left: 120 };
const tip = document.createElement('div');
tip.className = 'tooltip';
container.appendChild(tip);
let firstRender = true;
function render() {
container.querySelectorAll('svg').forEach(s => s.remove());
const rect = container.getBoundingClientRect();
const W = Math.max(300, Math.round(rect.width));
const rowH = 30;
const H = margin.top + margin.bottom + data.length * rowH;
const w = Math.max(0, W - margin.left - margin.right);
const h = Math.max(0, H - margin.top - margin.bottom);
const svg = d3.select(container).insert('svg', '.tooltip')
.attr('width', W).attr('height', H);
const g = svg.append('g').attr('transform', `translate(${margin.left},${margin.top})`);
const x = d3.scaleLog().domain([5, 90000]).range([0, Math.max(0, w)]).clamp(true);
const y = d3.scaleBand().domain(data.map(d => d.name)).range([0, Math.max(0, h)]).padding(0.32);
// decade grid lines
const ticks = [10, 100, 1000, 10000];
g.append('g').selectAll('line').data(ticks).enter().append('line')
.attr('x1', d => x(d)).attr('x2', d => x(d))
.attr('y1', -6).attr('y2', Math.max(0, h))
.attr('stroke', 'var(--border-color)').attr('stroke-opacity', 0.25)
.attr('stroke-dasharray', '2,4');
// x axis
g.append('g').attr('transform', `translate(0,${Math.max(0, h)})`)
.call(d3.axisBottom(x).tickValues(ticks).tickFormat(d => d >= 1000 ? (d / 1000) + 'k' : d + ''))
.call(s => s.select('.domain').attr('opacity', 0.3))
.selectAll('.tick text').attr('opacity', 0.6);
g.append('text').attr('class', 'axis-label')
.attr('x', Math.max(0, w) / 2).attr('y', Math.max(0, h) + 36).attr('text-anchor', 'middle')
.text('State count, log scale (gap = compression)');
// y labels
g.append('g').attr('class', 'y-axis')
.call(d3.axisLeft(y).tickSize(0).tickPadding(10))
.call(s => s.select('.domain').remove())
.selectAll('.tick text').attr('font-size', '11px').attr('opacity', 0.7);
// legend
const leg = g.append('g').attr('transform', 'translate(0, -22)');
leg.append('circle').attr('cx', 0).attr('cy', 0).attr('r', 5).attr('fill', C_OURS);
leg.append('text').attr('x', 10).attr('y', 4).attr('fill', 'var(--text-color)').attr('font-size', '11px').attr('opacity', 0.75).text('Ours (FSM)');
leg.append('circle').attr('cx', 96).attr('cy', 0).attr('r', 5).attr('fill', C_RPNI);
leg.append('text').attr('x', 106).attr('y', 4).attr('fill', 'var(--text-color)').attr('font-size', '11px').attr('opacity', 0.75).text('RPNI');
const rows = g.selectAll('.cmp-row').data(data).enter().append('g')
.attr('class', 'cmp-row')
.attr('transform', d => `translate(0, ${y(d.name) + y.bandwidth() / 2})`)
.style('cursor', 'pointer');
const lines = rows.append('line')
.attr('x1', d => x(d.ours)).attr('y1', 0).attr('y2', 0)
.attr('stroke', C_RPNI).attr('stroke-width', 2).attr('stroke-opacity', 0.5)
.attr('stroke-linecap', 'round');
rows.append('circle').attr('class', 'dot-rpni')
.attr('cx', d => x(d.rpni)).attr('cy', 0).attr('r', 5).attr('fill', C_RPNI);
rows.append('circle').attr('class', 'dot-ours')
.attr('cx', d => x(d.ours)).attr('cy', 0).attr('r', 5).attr('fill', C_OURS);
const ratioText = rows.append('text')
.attr('x', d => x(d.rpni) + 9).attr('y', 4)
.attr('fill', 'var(--text-color)').attr('font-size', '10px').attr('font-weight', 600)
.attr('font-variant-numeric', 'tabular-nums').attr('opacity', 0.85)
.text(d => d.compression.toLocaleString() + '×');
if (firstRender) {
lines.attr('x2', d => x(d.ours))
.transition().duration(650).delay((d, i) => i * 45).ease(d3.easeCubicOut)
.attr('x2', d => x(d.rpni));
rows.select('.dot-rpni').attr('cx', d => x(d.ours))
.transition().duration(650).delay((d, i) => i * 45).ease(d3.easeCubicOut)
.attr('cx', d => x(d.rpni));
ratioText.attr('opacity', 0)
.transition().duration(300).delay((d, i) => i * 45 + 550).attr('opacity', 0.85);
} else {
lines.attr('x2', d => x(d.rpni));
}
rows
.on('mouseenter', function(ev, d) {
d3.select(this).selectAll('circle').attr('r', 6.5);
tip.innerHTML = [
'<strong>' + d.name + '</strong>',
'Ours: ' + d.ours + ' states',
'RPNI: ' + d.rpni.toLocaleString() + ' states',
'Compression: <strong>' + d.compression.toLocaleString() + '×</strong>',
'Fitness: ' + d.fitness.toFixed(3)
].join('<br/>');
tip.style.opacity = '1';
})
.on('mousemove', function(ev) {
const [mx, my] = d3.pointer(ev, container);
const tipW = tip.offsetWidth || 180, tipH = tip.offsetHeight || 80;
const cRect = container.getBoundingClientRect();
let tx = mx + 14, ty = my - 14;
if (tx + tipW > cRect.width) tx = mx - tipW - 14;
if (ty + tipH > cRect.height) ty = my - tipH - 14;
tip.style.transform = `translate(${tx}px, ${ty}px)`;
})
.on('mouseleave', function() {
d3.select(this).selectAll('circle').attr('r', 5);
tip.style.opacity = '0';
});
firstRender = false;
}
render();
if (window.ResizeObserver) new ResizeObserver(() => render()).observe(container);
};
if (document.readyState === 'loading') document.addEventListener('DOMContentLoaded', () => ensureD3(bootstrap), { once: true });
else ensureD3(bootstrap);
})();
</script>
</div></div><figcaption class="html-embed__desc" style="text-align:left">Each dataset's state count on a log scale: ours (dark slate) versus RPNI (light slate). The distance between the two dots is the compression, from 15x on WebArena to 3,036x on GUI-Odyssey, all at replay fitness of at least 0.997.</figcaption></figure> </div>
<p>Compare all eight methods interactively in the <a href="https://seongland.com/article/asg/browser?tab=baselines">baselines view</a>, or watch structure stabilize in the <a href="https://seongland.com/article/asg/browser?tab=convergence">convergence view</a>.</p>
<h2 id="convergence-and-stability"><a href="#convergence-and-stability">Convergence and Stability</a></h2>
<p>Replay fitness reaches its plateau well before the training set is exhausted: on SWE-agent it’s already at 0.985 within 1% of the training traces and settles at 0.996 by 10%, and the state count keeps inching up as rare command patterns appear. Structured tool-call domains converge fastest: SWE-smith holds 0.9996 from the first 1% of data. Open web and delegation traces take longer, with Mind2Web <span class="" id="citation--deng2024mind2web--11">(<a href="#bib-deng2024mind2web" id="refctx-bib-deng2024mind2web-1">Deng et al., 2023</a>)</span> needing 5% and Who&amp;When <span class="" id="citation--yang2025whoandwhen--12">(<a href="#bib-yang2025whoandwhen" id="refctx-bib-yang2025whoandwhen-1">Zhang et al., 2025</a>)</span> 10% of their traces to clear 0.95 fitness.</p>
<div class="wide"> <figure class="html-embed"><figcaption class="html-embed__title" style="text-align:left">Convergence Curves</figcaption><div class="html-embed__card"><div id="frag-ujn7qtudt98"><!-- Convergence Curves: Small-multiples sparkline cards with convergence gauges -->
<div class="convergence-curves"></div>
<style>
.convergence-curves { position: relative; width: 100%; }
.convergence-curves .cc-grid {
display: grid;
grid-template-columns: repeat(auto-fit, minmax(180px, 1fr));
gap: 12px;
}
.convergence-curves .cc-card {
border: 1px solid var(--border-color);
border-radius: 12px;
padding: 16px 18px 14px;
background: var(--surface-bg);
position: relative;
overflow: hidden;
transition: border-color 0.2s ease, box-shadow 0.2s ease;
cursor: default;
}
.convergence-curves .cc-card:hover {
border-color: var(--text-color);
box-shadow: 0 2px 12px rgba(0,0,0,0.06);
}
.convergence-curves .cc-name {
font-size: 12px; font-weight: 600; color: var(--text-color);
margin-bottom: 2px; letter-spacing: 0.01em;
}
.convergence-curves .cc-meta {
font-size: 10px; color: var(--text-color); opacity: 0.4;
font-variant-numeric: tabular-nums; margin-bottom: 10px;
}
.convergence-curves .cc-big {
font-size: 28px; font-weight: 800; line-height: 1;
font-variant-numeric: tabular-nums;
margin-bottom: 2px;
}
.convergence-curves .cc-sub {
font-size: 10px; color: var(--text-color); opacity: 0.45;
margin-bottom: 10px;
}
.convergence-curves .cc-spark { display: block; width: 100%; }
.convergence-curves .cc-bar-track {
height: 4px; border-radius: 2px; background: var(--border-color);
opacity: 0.3; margin-top: 10px; overflow: hidden;
}
.convergence-curves .cc-bar-fill {
height: 100%; border-radius: 2px;
transition: width 1.2s cubic-bezier(0.22, 1, 0.36, 1);
}
.convergence-curves .cc-fitness {
font-size: 10px; color: var(--text-color); opacity: 0.5;
font-variant-numeric: tabular-nums; margin-top: 6px;
text-align: right;
}
.convergence-curves .tooltip {
position: absolute; top: 0; left: 0; pointer-events: none; padding: 10px 14px; border-radius: 8px;
font-size: 12px; line-height: 1.6; border: 1px solid var(--border-color);
background: var(--surface-bg); color: var(--text-color);
box-shadow: 0 4px 20px rgba(0,0,0,0.12);
backdrop-filter: blur(12px); -webkit-backdrop-filter: blur(12px);
opacity: 0; transition: opacity 0.15s ease;
z-index: 100; max-width: 220px;
font-variant-numeric: tabular-nums;
}
@media (max-width: 600px) {
.convergence-curves .cc-grid { grid-template-columns: repeat(2, 1fr); gap: 8px; }
.convergence-curves .cc-card { padding: 12px 14px 10px; }
.convergence-curves .cc-big { font-size: 22px; }
}
</style>
<script>
(() => {
const ensureD3 = (cb) => {
if (window.d3 && typeof window.d3.select === 'function') return cb();
let s = document.getElementById('d3-cdn-script');
if (!s) { s = document.createElement('script'); s.id = 'd3-cdn-script'; s.src = 'https://cdn.jsdelivr.net/npm/d3@7/dist/d3.min.js'; document.head.appendChild(s); }
s.addEventListener('load', () => cb(), { once: true });
};
const bootstrap = () => {
const container = document.querySelector('.convergence-curves:not([data-mounted])');
if (!container) return;
container.dataset.mounted = 'true';
const d3 = window.d3;
// Real incremental-convergence runs (experiments/convergence-rate/convergence-results.json).
// convPct = smallest % of training data at which replay fitness first reaches 0.95.
// curve = the experiment's actual fitnessTrajectory: [pctData, fitness].
const datasets = [
{ name: 'SWE-agent', traces: '2,000', states: 25, finalFitness: 0.996,
convPct: 1, color: '#3d5a80',
curve: [[0.01,0.9851],[0.05,0.9851],[0.10,0.9961],[0.20,0.9961],[0.50,0.9961],[1.0,0.9961]] },
{ name: 'SWE-smith', traces: '500', states: 10, finalFitness: 1.000,
convPct: 1, color: '#9e5e5a',
curve: [[0.01,0.9996],[0.05,0.9996],[0.10,0.9996],[0.20,0.9996],[0.50,1.000],[1.0,1.000]] },
{ name: 'Mind2Web', traces: '500', states: 8, finalFitness: 0.999,
convPct: 5, color: '#4d6278',
curve: [[0.01,0.7921],[0.05,0.9503],[0.10,0.9503],[0.20,0.9665],[0.50,0.999],[1.0,0.999]] },
{ name: 'Who & When', traces: '184', states: 9, finalFitness: 1.000,
convPct: 10, color: '#8fa6c4',
curve: [[0.01,0.1709],[0.05,0.7945],[0.10,0.9776],[0.20,0.9797],[0.50,0.9797],[1.0,1.000]] },
];
// Build grid
const grid = document.createElement('div');
grid.className = 'cc-grid';
container.appendChild(grid);
const tip = document.createElement('div');
tip.className = 'tooltip';
container.appendChild(tip);
datasets.forEach((ds, idx) => {
const card = document.createElement('div');
card.className = 'cc-card';
card.innerHTML = `
<div class="cc-name">${ds.name}</div>
<div class="cc-meta">${ds.traces} traces &middot; ${ds.states} states</div>
<div class="cc-big" style="color:${ds.color}">${ds.convPct}%</div>
<div class="cc-sub">of training data to reach 0.95 fitness</div>
<svg class="cc-spark" data-idx="${idx}"></svg>
<div class="cc-bar-track"><div class="cc-bar-fill" style="background:${ds.color};width:0%"></div></div>
<div class="cc-fitness">final fitness ${ds.finalFitness.toFixed(3)}</div>
`;
grid.appendChild(card);
// Hover tooltip
card.addEventListener('mouseenter', (ev) => {
const convTraces = Math.round(ds.convPct / 100 * parseInt(ds.traces.replace(/,/g, '')));
tip.innerHTML = `<strong>${ds.name}</strong><br/>
Converges at <strong>${ds.convPct}%</strong> (${convTraces.toLocaleString()} traces)<br/>
Final fitness: <strong>${ds.finalFitness.toFixed(3)}</strong><br/>
FSM states: <strong>${ds.states}</strong>`;
tip.style.opacity = '1';
});
card.addEventListener('mousemove', (ev) => {
const cr = container.getBoundingClientRect();
const mx = ev.clientX - cr.left;
const my = ev.clientY - cr.top;
const tw = tip.offsetWidth || 200;
let tx = mx + 14;
if (tx + tw > cr.width) tx = mx - tw - 14;
tip.style.transform = `translate(${tx}px, ${my - 14}px)`;
});
card.addEventListener('mouseleave', () => { tip.style.opacity = '0'; });
});
// Draw sparklines
function drawSparklines() {
container.querySelectorAll('.cc-spark').forEach(svgEl => {
const idx = +svgEl.dataset.idx;
const ds = datasets[idx];
const rect = svgEl.parentElement.getBoundingClientRect();
const W = Math.max(100, Math.round(rect.width - 36));
const H = 48;
svgEl.setAttribute('width', W);
svgEl.setAttribute('height', H);
svgEl.setAttribute('viewBox', `0 0 ${W} ${H}`);
svgEl.innerHTML = '';
const svg = d3.select(svgEl);
// Full trajectory on x (0-100% of data); y adapts to each curve's real floor.
const fitVals = ds.curve.map(p => p[1]);
const yFloor = Math.max(0, Math.min(...fitVals) - 0.03);
const x = d3.scaleLinear().domain([0, 1]).range([0, W]);
const y = d3.scaleLinear().domain([yFloor, 1.005]).range([H, 0]);
const clipped = ds.curve;
// Area fill
const areaGen = d3.area()
.x(d => x(d[0])).y0(H).y1(d => y(d[1]))
.curve(d3.curveMonotoneX);
svg.append('path')
.datum(clipped)
.attr('d', areaGen)
.attr('fill', ds.color)
.attr('opacity', 0.08);
// Line
const lineGen = d3.line()
.x(d => x(d[0])).y(d => y(d[1]))
.curve(d3.curveMonotoneX);
svg.append('path')
.datum(clipped)
.attr('fill', 'none')
.attr('stroke', ds.color)
.attr('stroke-width', 2)
.attr('opacity', 0.7)
.attr('d', lineGen);
// 0.999 reference line
if (y(0.999) > 2 && y(0.999) < H - 2) {
svg.append('line')
.attr('x1', 0).attr('x2', W)
.attr('y1', y(0.999)).attr('y2', y(0.999))
.attr('stroke', 'var(--text-color)')
.attr('stroke-width', 0.5)
.attr('stroke-dasharray', '2,3')
.attr('opacity', 0.2);
}
// Convergence marker
const convX = x(ds.convPct / 100);
if (convX >= 0 && convX <= W) {
// Vertical line at convergence
svg.append('line')
.attr('x1', convX).attr('x2', convX)
.attr('y1', 0).attr('y2', H)
.attr('stroke', ds.color)
.attr('stroke-width', 1)
.attr('stroke-dasharray', '2,2')
.attr('opacity', 0.35);
// Dot
const convY = y(ds.finalFitness);
svg.append('circle')
.attr('cx', convX).attr('cy', convY)
.attr('r', 3.5)
.attr('fill', ds.color)
.attr('stroke', 'var(--surface-bg)')
.attr('stroke-width', 1.5);
}
});
}
drawSparklines();
// Animate progress bars on scroll into view
let animated = false;
const animateBars = () => {
if (animated) return;
const rect = container.getBoundingClientRect();
if (rect.top < window.innerHeight * 0.85) {
animated = true;
container.querySelectorAll('.cc-bar-fill').forEach((bar, i) => {
setTimeout(() => {
bar.style.width = datasets[i].convPct + '%';
}, i * 120);
});
}
};
animateBars();
window.addEventListener('scroll', animateBars, { passive: true });
if (window.ResizeObserver) new ResizeObserver(() => drawSparklines()).observe(container);
};
if (document.readyState === 'loading') document.addEventListener('DOMContentLoaded', () => ensureD3(bootstrap), { once: true });
else ensureD3(bootstrap);
})();
</script>
</div></div><figcaption class="html-embed__desc" style="text-align:left">Replay fitness as training traces accumulate, for the four datasets with incremental-convergence runs. The dashed marker shows where each first reaches 0.95 fitness.</figcaption></figure> </div>
<h2 id="baselines-at-a-glance"><a href="#baselines-at-a-glance">Baselines at a Glance</a></h2>
<div class="wide"> <figure class="html-embed"><figcaption class="html-embed__title" style="text-align:left">Method Comparison</figcaption><div class="html-embed__card"><div id="frag-daq4o60t5z7"><!-- Baseline Heatmap: Interactive comparison table of methods x datasets -->
<div class="baseline-heatmap"></div>
<style>
.baseline-heatmap { position: relative; width: 100%; }
.baseline-heatmap .controls { display: flex; gap: 12px; margin-bottom: 12px; align-items: center; flex-wrap: wrap; }
.baseline-heatmap .seg-control {
display: inline-flex; background: var(--surface-bg);
border: 1px solid var(--border-color); border-radius: 8px;
padding: 3px; gap: 2px;
}
.baseline-heatmap .seg-control button {
padding: 5px 14px; border-radius: 6px; border: none;
background: transparent; font-size: 12px; font-weight: 500;
color: var(--text-color); cursor: pointer; transition: all 0.15s ease;
opacity: 0.6;
}
.baseline-heatmap .seg-control button:hover { opacity: 0.8; }
.baseline-heatmap .seg-control button.active {
background: var(--text-color); color: var(--page-bg);
opacity: 1; font-weight: 600;
}
.baseline-heatmap .table-wrap {
overflow-x: auto; border: 1px solid var(--border-color); border-radius: 10px;
}
.baseline-heatmap table {
border-collapse: separate; border-spacing: 0; width: 100%;
font-size: 12px; line-height: 1.4; font-variant-numeric: tabular-nums;
}
.baseline-heatmap thead th {
background: var(--surface-bg); color: var(--text-color); font-weight: 600;
padding: 10px 10px; text-align: center; white-space: nowrap;
position: sticky; top: 0; z-index: 2;
border-bottom: 1px solid var(--border-color);
}
.baseline-heatmap thead th:first-child {
text-align: left; position: sticky; left: 0; z-index: 3;
border-right: 1px solid var(--border-color);
}
.baseline-heatmap tbody td {
padding: 8px 10px; text-align: center;
border-bottom: 1px solid color-mix(in srgb, var(--border-color) 40%, transparent);
transition: background 0.15s ease, opacity 0.25s ease;
}
.baseline-heatmap tbody td:first-child {
text-align: left; font-weight: 600; background: var(--surface-bg);
position: sticky; left: 0; z-index: 1; white-space: nowrap;
border-right: 1px solid var(--border-color);
}
.baseline-heatmap tbody tr:last-child td { border-bottom: none; }
.baseline-heatmap tbody tr:hover td { background: color-mix(in srgb, var(--text-color) 6%, transparent); }
.baseline-heatmap tbody tr:hover td:first-child { background: color-mix(in srgb, var(--text-color) 8%, var(--surface-bg)); }
.baseline-heatmap td.best { font-weight: 700; }
.baseline-heatmap td.col-hover { background: color-mix(in srgb, var(--text-color) 4%, transparent) !important; }
.baseline-heatmap td.fade-in { animation: bh-fade 0.3s ease forwards; }
@keyframes bh-fade { from { opacity: 0; transform: translateY(2px); } to { opacity: 1; transform: translateY(0); } }
</style>
<script>
(() => {
const container = document.querySelector('.baseline-heatmap:not([data-mounted])');
if (!container) return;
container.dataset.mounted = 'true';
const datasets = ['SWE-smith','SWE-agent','WebArena','AgentNet','ATBench','OSWorld','tau2-air','tau2-ret','tau2-tel','W&W','M2W','GUI-Ody'];
const statesData = {
'Ours': [10, 25, 25, 25, 15, 27, 18, 19, 43, 9, 8, 7],
'RPNI': [11631, 59510, 382, 62495, 899, 38232, 6506, 14249, 63897, 971, 476, 21255],
'Alergia': [10, 35, 149, 45, 15, 31, 23, 25, 75, 12, 8, 24],
'HMM': [10, 25, 25, 25, 15, 27, 18, 19, 43, 9, 8, 7],
'EDSM': [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1],
};
const fitnessData = {
'Ours': [1.000, 0.999, 1.000, 1.000, 1.000, 0.997, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000],
'RPNI': [0.761, 0.646, 1.000, 0.742, 0.984, 0.706, 0.844, 0.837, 0.491, 0.984, 0.970, 0.929],
'Alergia': [1.000, 0.999, 1.000, 1.000, 1.000, 0.999, 0.999, 1.000, 0.999, 1.000, 1.000, 1.000],
'HMM': [1.000, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000],
'EDSM': [1.000, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000, 1.000],
};
const methods = Object.keys(statesData);
let metric = 'states';
// --- Controls ---
const controls = document.createElement('div');
controls.className = 'controls';
const seg = document.createElement('div');
seg.className = 'seg-control';
const btnStates = document.createElement('button');
btnStates.textContent = 'State Count'; btnStates.className = 'active';
const btnFitness = document.createElement('button');
btnFitness.textContent = 'Fitness';
seg.appendChild(btnStates); seg.appendChild(btnFitness);
controls.appendChild(seg);
container.prepend(controls);
btnStates.addEventListener('click', () => { metric = 'states'; btnStates.className = 'active'; btnFitness.className = ''; renderTable(); });
btnFitness.addEventListener('click', () => { metric = 'fitness'; btnFitness.className = 'active'; btnStates.className = ''; renderTable(); });
// --- Table wrapper ---
const wrap = document.createElement('div');
wrap.className = 'table-wrap';
container.appendChild(wrap);
// --- Best value per column (returns the value, so EVERY tied cell can be bolded) ---
function findBestVal(colIndex, data) {
let bestVal = null;
methods.forEach((m) => {
const v = data[m][colIndex];
if (v === '-') return;
if (metric === 'states') {
// Best = lowest non-degenerate (>1)
if (v > 1 && (bestVal === null || v < bestVal)) bestVal = v;
} else {
// Best = highest
if (bestVal === null || v > bestVal) bestVal = v;
}
});
return bestVal;
}
// A cell is "best" if it equals the column's best value (ties all qualify).
function isBest(val, bestVal) {
if (bestVal === null || val === '-') return false;
if (metric === 'states') return typeof val === 'number' && val > 1 && val === bestVal;
return val === bestVal;
}
// --- Color logic ---
function cellBg(val, m) {
if (val === '-') return 'transparent';
if (metric === 'states') {
if (m === 'EDSM') return 'rgba(200,200,200,0.12)';
const logVal = Math.log10(Math.max(1, val));
const t = Math.min(1, logVal / 5);
// low states = teal tint, high = rust tint
if (t < 0.3) return `rgba(42,157,143,${0.08 + t * 0.15})`;
return `rgba(61, 90, 128,${0.06 + (t - 0.3) * 0.2})`;
} else {
const t = Math.max(0, (val - 0.5) / 0.5);
return `rgba(42,157,143,${t * 0.35})`;
}
}
// --- Column hover ---
let hoveredCol = -1;
function updateColHover(colIdx) {
if (hoveredCol === colIdx) return;
hoveredCol = colIdx;
const cells = wrap.querySelectorAll('td, th');
cells.forEach(c => c.classList.remove('col-hover'));
if (colIdx < 0) return;
const rows = wrap.querySelectorAll('tr');
rows.forEach(row => {
const cell = row.children[colIdx];
if (cell) cell.classList.add('col-hover');
});
}
function renderTable() {
wrap.innerHTML = '';
const currentData = metric === 'states' ? statesData : fitnessData;
// Precompute best per column
const bestPerCol = datasets.map((_, ci) => findBestVal(ci, currentData));
const table = document.createElement('table');
// --- Header ---
const thead = document.createElement('thead');
const hr = document.createElement('tr');
const th0 = document.createElement('th');
th0.textContent = 'Method';
hr.appendChild(th0);
datasets.forEach((d, ci) => {
const th = document.createElement('th');
th.textContent = d;
th.addEventListener('mouseenter', () => updateColHover(ci + 1));
th.addEventListener('mouseleave', () => updateColHover(-1));
hr.appendChild(th);
});
thead.appendChild(hr);
table.appendChild(thead);
// --- Body ---
const tbody = document.createElement('tbody');
methods.forEach((method, mi) => {
const tr = document.createElement('tr');
const td0 = document.createElement('td');
td0.textContent = method;
tr.appendChild(td0);
currentData[method].forEach((val, ci) => {
const td = document.createElement('td');
td.style.background = cellBg(val, method);
td.className = 'fade-in';
td.style.animationDelay = `${mi * 20 + ci * 10}ms`;
if (val === '-') {
td.textContent = '-';
td.style.opacity = '0.35';
} else if (metric === 'states') {
td.textContent = typeof val === 'number' ? val.toLocaleString() : val;
} else {
td.textContent = typeof val === 'number' ? val.toFixed(3) : val;
}
if (isBest(val, bestPerCol[ci])) td.classList.add('best');
td.addEventListener('mouseenter', () => updateColHover(ci + 1));
td.addEventListener('mouseleave', () => updateColHover(-1));
tr.appendChild(td);
});
tbody.appendChild(tr);
});
table.appendChild(tbody);
wrap.appendChild(table);
}
renderTable();
})();
</script>
</div></div><figcaption class="html-embed__desc" style="text-align:left">State count and fitness across five methods and twelve datasets. Our FSM holds the smallest state count among non-degenerate methods while keeping the highest fitness.</figcaption></figure> </div>
<ul>
<li><strong>RPNI</strong> <span class="" id="citation--oncina1992rpni--13">(<a href="#bib-oncina1992rpni" id="refctx-bib-oncina1992rpni-2">Oncina &amp; Garcı́a, 1992</a>)</span> without negative examples keeps large portions of the prefix tree (382 to 63,897 states) at degraded fitness.</li>
<li><strong>Alergia</strong> <span class="" id="citation--carrasco1994alergia--14">(<a href="#bib-carrasco1994alergia" id="refctx-bib-carrasco1994alergia-2">Carrasco &amp; Oncina, 1994</a>)</span>, the strongest competitor, matches our fitness but uses 1.0 to 6.0<span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mo>×</mo></mrow><annotation encoding="application/x-tex">\times</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6667em;vertical-align:-0.0833em"></span><span class="mord">×</span></span></span></span> more states.</li>
<li><strong>HMM</strong> <span class="" id="citation--rabiner1989hmm--15">(<a href="#bib-rabiner1989hmm" id="refctx-bib-rabiner1989hmm-2">Rabiner, 1989</a>)</span> matches our state count but produces non-interpretable latent states.</li>
<li><strong>EDSM</strong> (evidence-driven state merging) <span class="" id="citation--lang1998edsm--16">(<a href="#bib-lang1998edsm" id="refctx-bib-lang1998edsm-2">Lang et al., 1998</a>)</span> without negatives collapses to a trivial 1-state acceptor.</li>
<li><strong>k-Tails</strong> made us pick <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>k</mi></mrow><annotation encoding="application/x-tex">k</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6944em"></span><span class="mord mathnormal" style="margin-right:0.03148em">k</span></span></span></span> ourselves, and at <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>k</mi><mo>=</mo><mn>1</mn></mrow><annotation encoding="application/x-tex">k=1</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6944em"></span><span class="mord mathnormal" style="margin-right:0.03148em">k</span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">=</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:0.6444em"></span><span class="mord">1</span></span></span></span> it produced 1.4 to 10<span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mo>×</mo></mrow><annotation encoding="application/x-tex">\times</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6667em;vertical-align:-0.0833em"></span><span class="mord">×</span></span></span></span> more states than ours (with state counts exploding past <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>k</mi><mo>=</mo><mn>2</mn></mrow><annotation encoding="application/x-tex">k=2</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6944em"></span><span class="mord mathnormal" style="margin-right:0.03148em">k</span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">=</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:0.6444em"></span><span class="mord">2</span></span></span></span>).</li>
<li><strong>Process mining</strong> <span class="" id="citation--vanderaalst2016process--17">(<a href="#bib-vanderaalst2016process" id="refctx-bib-vanderaalst2016process-2">van der Aalst, 2016</a>)</span> miners reach high fitness but precision 0.00 to 0.80, the “flower model” problem where every activity is reachable from every state.</li>
</ul>
<h2 id="precision"><a href="#precision">Precision</a></h2>
<p>The FSM is more than a vocabulary: it rejects every random trace, and at least 99.9% of permuted traces that keep the activity set but scramble the order. Even single-symbol mutations — a substitution or an insertion or an adjacent swap — are blocked 77 to 100% of the time. RPNI — with its thousands of states — accepts 75% of those same permuted traces on WebArena.</p><div id="bibliography-references-list" class="references csl-bib-body" data-bibliography-block="true" data-built-refs="1"><ol class="references"><li id="bib-berti2019pm4py">Berti, A., van Zelst, S. J., &amp; van der Aalst, W. (2019). <i>Process Mining for Python (PM4Py): Bridging the Gap Between Process- and Data Science</i>. <a href="https://arxiv.org/abs/1905.06169" target="_blank" rel="noopener noreferrer">https://arxiv.org/abs/1905.06169</a><small class="backrefs"><a href="#refctx-bib-berti2019pm4py-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-biermann1972ktails">Biermann, A. W., &amp; Feldman, J. A. (1972). On the Synthesis of Finite-State Machines from Samples of Their Behavior. <i>IEEE Transactions on Computers</i>, <i>C–21</i>(6), 592–597.<small class="backrefs"><a href="#refctx-bib-biermann1972ktails-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-carrasco1994alergia">Carrasco, R. C., &amp; Oncina, J. (1994). Learning Stochastic Regular Grammars by Means of a State Merging Method. <i>International Colloquium on Grammatical Inference</i>, 139–152.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-carrasco1994alergia-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-carrasco1994alergia-2" aria-label="Back to citation">2</a></small></li><li id="bib-deng2024mind2web">Deng, X., Gu, Y., Zheng, B., Chen, S., Stevens, S., Wang, B., Sun, H., &amp; Su, Y. (2023). Mind2Web: Towards a Generalist Agent for the Web. <i>Advances in Neural Information Processing Systems (NeurIPS)</i>.<small class="backrefs"><a href="#refctx-bib-deng2024mind2web-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-lang1998edsm">Lang, K. J., Pearlmutter, B. A., &amp; Price, R. A. (1998). Results of the Abbadingo One DFA Learning Competition and a New Evidence-Driven State Merging Algorithm. <i>International Colloquium on Grammatical Inference (ICGI)</i>, 1–12.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-lang1998edsm-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-lang1998edsm-2" aria-label="Back to citation">2</a></small></li><li id="bib-lu2024guiodyssey">Lu, Q., Zhao, W., Jia, J., Ren, K., Lu, K., Han, J., Chen, Y., Zheng, J., Zhang, Z., &amp; Ding, L. (2025). GUI-Odyssey: A Comprehensive Dataset for Cross-App GUI Navigation on Mobile Devices. <i>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</i>.<small class="backrefs"><a href="#refctx-bib-lu2024guiodyssey-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-muskardin2022aalpy">Muškardin, E., Aichernig, B. K., Pill, I., Pferscher, A., &amp; Tappler, M. (2022). AALpy: An Active Automata Learning Library. <i>Innovations in Systems and Software Engineering</i>, <i>18</i>, 417–426.<small class="backrefs"><a href="#refctx-bib-muskardin2022aalpy-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-oncina1992rpni">Oncina, J., &amp; Garcı́a, P. (1992). Inferring Regular Languages in Polynomial Updated Time. <i>Pattern Recognition and Image Analysis</i>, 49–61.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-oncina1992rpni-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-oncina1992rpni-2" aria-label="Back to citation">2</a></small></li><li id="bib-rabiner1989hmm">Rabiner, L. R. (1989). A Tutorial on Hidden Markov Models and Selected Applications in Speech Recognition. <i>Proceedings of the IEEE</i>, <i>77</i>(2), 257–286.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-rabiner1989hmm-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-rabiner1989hmm-2" aria-label="Back to citation">2</a></small></li><li id="bib-vanderaalst2016process">van der Aalst, W. M. P. (2016). <i>Process Mining: Data Science in Action</i> (2nd ed.). Springer.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-vanderaalst2016process-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-vanderaalst2016process-2" aria-label="Back to citation">2</a></small></li><li id="bib-wang2024agent_workflow_memory">Wang, Z. Z., Mao, J., Fried, D., &amp; Neubig, G. (2024). Agent Workflow Memory. <i>arXiv Preprint arXiv:2409.07429</i>.<small class="backrefs"><a href="#refctx-bib-wang2024agent_workflow_memory-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-yang2025whoandwhen">Zhang, S., Yin, M., Zhang, J., Liu, J., Han, Z., Zhang, J., Li, B., Wang, C., Wang, H., Chen, Y., &amp; Wu, Q. (2025). Which Agent Causes Task Failures and When? <i>arXiv Preprint arXiv:2505.00212</i>.<small class="backrefs"><a href="#refctx-bib-yang2025whoandwhen-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li></ol></div>
<h1 id="prediction-next-step-and-failure"><a href="#prediction-next-step-and-failure">Prediction: Next Step and Failure</a></h1>
<p>We find the same FSM state answers both questions an operator asks: what the agent will do next, and whether this run is heading for failure. Both read off the per-state transition distribution, which compactness makes reliable.</p>
<h2 id="next-step-prediction"><a href="#next-step-prediction">Next-Step Prediction</a></h2>
<p>At each step the predictor estimates <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>P</mi><mo stretchy="false">(</mo><msub><mi>a</mi><mi>t</mi></msub><mo>∣</mo><mtext>context</mtext><mo stretchy="false">)</mo></mrow><annotation encoding="application/x-tex">P(a_t \mid \text{context})</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em"></span><span class="mord mathnormal" style="margin-right:0.13889em">P</span><span class="mopen">(</span><span class="mord"><span class="mord mathnormal">a</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.2806em"><span style="top:-2.55em;margin-left:0em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mathnormal mtight">t</span></span></span></span><span class="vlist-s">​</span></span><span class="vlist-r"><span class="vlist" style="height:0.15em"><span></span></span></span></span></span></span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">∣</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em"></span><span class="mord text"><span class="mord">context</span></span><span class="mclose">)</span></span></span></span>, scored by <a href="https://texonom.com/fefc71fd930a41a7842f39bccb3abcc9">cross-entropy</a> in bits via 5<span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mo>×</mo></mrow><annotation encoding="application/x-tex">\times</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.6667em;vertical-align:-0.0833em"></span><span class="mord">×</span></span></span></span>5-fold <a href="https://texonom.com/184e7cfc80d6410188922d27ef4ef52c">cross-validation</a> (CV). Conditioning on FSM state alone, with no learning beyond an <a href="https://texonom.com/3bde15938b6b4c79981a3e0523b9877d">order-1 Markov model</a>, accounts for 83 to 99% of the total cross-entropy improvement on each dataset. We find the average drops to <strong>0.93 bits</strong>, a <strong>62% cut</strong> from the unigram baseline at 2.44 bits.</p>
<div class="sidenote-container"> <aside class="sidenote"> <p>We find RPNI goes the other way: 3.40 bits, worse than a unigram. Its state count, from hundreds to tens of thousands, spreads each transition estimate too thin, so the per-state distributions are noise.</p> </aside> </div>
<p>The cleanest test holds the predictor fixed and adds FSM state as a feature: under absolute discounting, FSM state conditioning adds <strong>+0.155 bits on average</strong> (0.580 vs 0.735) and helps on every one of six datasets, from +0.016 on SWE-agent to +0.364 on Mind2Web, whose branching web-action vocabulary gains most from knowing where in the workflow it is. We find that combining FSM state with a small learned model gives the best predictor in the study, at <strong>0.73 bits</strong>.</p>
<h2 id="workflow-memory"><a href="#workflow-memory">Workflow Memory</a></h2>
<p>The gain generalizes to the agent’s own LLM: feeding the current FSM state as context for choosing the next action beats Agent Workflow Memory (AWM) <span class="" id="citation--wang2024agent_workflow_memory--1">(<a href="#bib-wang2024agent_workflow_memory" id="refctx-bib-wang2024agent_workflow_memory-1">Wang et al., 2024</a>)</span> on <strong>all eight</strong> ground-truth datasets, with six gaps significant at <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>p</mi><mo>&lt;</mo><msup><mn>10</mn><mrow><mo>−</mo><mn>8</mn></mrow></msup></mrow><annotation encoding="application/x-tex">p &lt; 10^{-8}</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.7335em;vertical-align:-0.1944em"></span><span class="mord mathnormal">p</span><span class="mspace" style="margin-right:0.2778em"></span><span class="mrel">&lt;</span><span class="mspace" style="margin-right:0.2778em"></span></span><span class="base"><span class="strut" style="height:0.8141em"></span><span class="mord">1</span><span class="mord"><span class="mord">0</span><span class="msupsub"><span class="vlist-t"><span class="vlist-r"><span class="vlist" style="height:0.8141em"><span style="top:-3.063em;margin-right:0.05em"><span class="pstrut" style="height:2.7em"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mord mtight">−</span><span class="mord mtight">8</span></span></span></span></span></span></span></span></span></span></span></span>.</p>
<div class="table-scroll"><table><thead><tr><th style="text-align:left">Dataset</th><th style="text-align:right">N</th><th style="text-align:right">AWM</th><th style="text-align:right">Ours</th><th style="text-align:right">Δ</th></tr></thead><tbody><tr><td style="text-align:left">WebArena <span class="" id="citation--zhou2024webarena--2">(<a href="#bib-zhou2024webarena" id="refctx-bib-zhou2024webarena-1">Zhou et al., 2024</a>)</span></td><td style="text-align:right">4,800</td><td style="text-align:right">65.5</td><td style="text-align:right"><strong>81.2</strong></td><td style="text-align:right">+15.7</td></tr><tr><td style="text-align:left">SWE-smith <span class="" id="citation--yang2025swesmith--3">(<a href="#bib-yang2025swesmith" id="refctx-bib-yang2025swesmith-1">Yang et al., 2025</a>)</span></td><td style="text-align:right">300</td><td style="text-align:right">74.7</td><td style="text-align:right"><strong>100.0</strong></td><td style="text-align:right">+25.3</td></tr><tr><td style="text-align:left">SWE-agent <span class="" id="citation--yang2024sweagent--4">(<a href="#bib-yang2024sweagent" id="refctx-bib-yang2024sweagent-1">Yang et al., 2024</a>)</span></td><td style="text-align:right">1,200</td><td style="text-align:right">67.7</td><td style="text-align:right"><strong>70.5</strong></td><td style="text-align:right">+2.8</td></tr><tr><td style="text-align:left">tau2-bench telecom <span class="" id="citation--barres2025tau2--5">(<a href="#bib-barres2025tau2" id="refctx-bib-barres2025tau2-1">Barres et al., 2025</a>)</span></td><td style="text-align:right">1,095</td><td style="text-align:right">28.5</td><td style="text-align:right"><strong>45.6</strong></td><td style="text-align:right">+17.1</td></tr><tr><td style="text-align:left">tau2-bench retail <span class="" id="citation--barres2025tau2--6">(<a href="#bib-barres2025tau2" id="refctx-bib-barres2025tau2-2">Barres et al., 2025</a>)</span></td><td style="text-align:right">1,095</td><td style="text-align:right">52.9</td><td style="text-align:right"><strong>65.1</strong></td><td style="text-align:right">+12.2</td></tr><tr><td style="text-align:left">tau2-bench airline <span class="" id="citation--barres2025tau2--7">(<a href="#bib-barres2025tau2" id="refctx-bib-barres2025tau2-3">Barres et al., 2025</a>)</span></td><td style="text-align:right">480</td><td style="text-align:right">56.5</td><td style="text-align:right"><strong>57.3</strong></td><td style="text-align:right">+0.8</td></tr><tr><td style="text-align:left">ATBench</td><td style="text-align:right">600</td><td style="text-align:right">47.8</td><td style="text-align:right"><strong>62.5</strong></td><td style="text-align:right">+14.7</td></tr><tr><td style="text-align:left">OSWorld <span class="" id="citation--xie2024osworld--8">(<a href="https://openreview.net/forum?id=tN61DTr4Ed" id="refctx-bib-xie2024osworld-1" data-ref-id="bib-xie2024osworld" target="_blank" rel="noopener noreferrer">Xie et al., 2024</a>)</span></td><td style="text-align:right">1,286</td><td style="text-align:right">55.0</td><td style="text-align:right"><strong>70.7</strong></td><td style="text-align:right">+15.7</td></tr></tbody></table></div>
<p>The margin runs from +0.8 points on tau2-bench airline to +25.3 on SWE-smith — and it never reverses: on none of the eight does AWM come out ahead. Three of the eight gaps clear 15 points, and the two narrowest — airline at +0.8 and SWE-agent at +2.8 — are the two datasets where the agent already succeeds most often.</p>
<p>AWM extracts workflows from successful traces only, so on low-success datasets it has little to say — that’s exactly where the gap is widest.</p>
<div class="note note--danger" data-astro-cid-qg6lmfty> <!-- When there's a title, emoji is inline with title -->
<div class="note__body" data-astro-cid-qg6lmfty> <div class="note__header" data-astro-cid-qg6lmfty> <div class="note__title" data-astro-cid-qg6lmfty>The win is not automatic</div> </div> <div class="note__content" data-astro-cid-qg6lmfty> <p>Handing the LLM the full FSM, every state and transition, loses to AWM: 52.2% against 52.9% on tau2-bench retail. What wins, at <strong>65.1%</strong>, is a minimal format: next-action probabilities plus a few top continuations from the current state. Finding that minimal context is itself part of the contribution, as AWM’s linear-workflow format was for AWM.</p> </div> </div> </div>
<div class="table-scroll"><table><thead><tr><th style="text-align:left">Context given to the LLM (tau2-bench retail)</th><th style="text-align:right">Top-1 %</th></tr></thead><tbody><tr><td style="text-align:left">No memory, just the trace so far</td><td style="text-align:right">27.6</td></tr><tr><td style="text-align:left">Linear workflows from successful runs (AWM)</td><td style="text-align:right">52.9</td></tr><tr><td style="text-align:left">Full machine: current state, every transition, the whole graph</td><td style="text-align:right">52.2</td></tr><tr><td style="text-align:left">Full machine, plus multi-step continuations</td><td style="text-align:right">49.2</td></tr><tr><td style="text-align:left">Full machine, from successful traces only</td><td style="text-align:right">50.3</td></tr><tr><td style="text-align:left"><strong>Minimal: next-action probabilities and a few likely continuations</strong></td><td style="text-align:right"><strong>65.1</strong></td></tr></tbody></table></div>
<p>Step through the workflow-memory comparison per dataset in the <a href="https://seongland.com/article/asg/browser?tab=memory">memory view</a>.</p>
<h2 id="predicting-failure"><a href="#predicting-failure">Predicting Failure</a></h2>
<p>Failure has mostly been studied after the fact: Who&amp;When asks which agent and which step were to blame once a multi-agent run has already gone wrong <span class="" id="citation--yang2025whoandwhen--9">(<a href="#bib-yang2025whoandwhen" id="refctx-bib-yang2025whoandwhen-1">Zhang et al., 2025</a>)</span>, and Trace treats the whole workflow as a differentiable graph to be optimised offline <span class="" id="citation--cheng2024trace_autodiff--10">(<a href="#bib-cheng2024trace_autodiff" id="refctx-bib-cheng2024trace_autodiff-1">Cheng et al., 2024</a>)</span>. Ours is the earlier and cheaper question: with the run still going, does the machine already know?</p>
<p>Replay a trace through the FSM and read off <strong>per-state behavioral features</strong> (visit frequency, message-length statistics, error rate, early/late entropy drift) plus five cross-entropy anomaly features. A single <a href="https://texonom.com/b841ada1c339440eb058a4d3654cf5a7">gradient-boosted</a> classifier on a fixed 80/20 split reaches held-out AUROC <strong>up to 0.94</strong>.</p>
<div class="note note--neutral" data-astro-cid-qg6lmfty> <!-- When there's no title, emoji is above content -->
<div class="note__layout" data-astro-cid-qg6lmfty> <div class="note__body" data-astro-cid-qg6lmfty> <div class="note__content" data-astro-cid-qg6lmfty> <p>Raw fitness is useless here (AUROC near 0.50): successful and failed traces both replay perfectly. But the signal is in the <strong>per-state decomposition</strong> and in <em>surprise</em>: failing traces take low-probability transitions under the FSM.</p> </div> </div> </div> </div>
<p>Failure prediction scales with machine size: more states give a finer map of where a run can go wrong. The 43-state telecom agent tops out at <strong>0.941</strong>; WebArena <span class="" id="citation--zhou2024webarena--11">(<a href="#bib-zhou2024webarena" id="refctx-bib-zhou2024webarena-2">Zhou et al., 2024</a>)</span> (0.903) and AgentNet <span class="" id="citation--wang2025opencua--12">(<a href="#bib-wang2025opencua" id="refctx-bib-wang2025opencua-1">X. Wang et al., 2025</a>)</span> (0.890) follow; SWE-agent <span class="" id="citation--yang2024sweagent--13">(<a href="#bib-yang2024sweagent" id="refctx-bib-yang2024sweagent-2">Yang et al., 2024</a>)</span> — with 25 states — reaches 0.799. ATBench, the only safety-labeled benchmark, reaches 0.894 (0.864 ± 0.024 under repeated CV). We see across all eight real-trace datasets the CV standard deviation stays in 0.012 to 0.031 — so these aren’t single-split artifacts.</p>
<div class="wide"> <figure class="html-embed"><figcaption class="html-embed__title" style="text-align:left">Failure Prediction</figcaption><div class="html-embed__card"><div id="frag-71atzyto7i6"><!-- Failure Features: Dumbbell chart with CV + Holdout AUROC and top features -->
<div class="failure-features"></div>
<style>
.failure-features { position: relative; width: 100%; min-height: 420px; }
.failure-features svg { display: block; width: 100%; }
.failure-features .tick text { fill: var(--text-color); font-size: 11px; }
.failure-features .tick line, .failure-features .domain { stroke: var(--border-color); }
.failure-features .grid line { stroke: var(--border-color); opacity: 0.15; }
.failure-features .grid .domain { display: none; }
.failure-features .axis-label { fill: var(--text-color); font-size: 13px; font-weight: 500; }
.failure-features .tooltip {
position: absolute; top: 0; left: 0; pointer-events: none; padding: 10px 14px; border-radius: 8px;
font-size: 12px; line-height: 1.6; border: 1px solid var(--border-color);
background: var(--surface-bg); color: var(--text-color);
box-shadow: 0 4px 20px rgba(0,0,0,0.12), 0 0 0 1px rgba(0,0,0,0.04);
backdrop-filter: blur(12px); -webkit-backdrop-filter: blur(12px);
opacity: 0; transition: opacity 0.15s ease;
z-index: 100; max-width: 260px;
font-variant-numeric: tabular-nums;
}
.failure-features .ref-line { stroke-dasharray: 6,4; }
.failure-features .feature-pill {
font-family: 'SF Mono', 'Fira Code', 'Consolas', monospace;
font-size: 10px;
}
.failure-features .legend-item text { fill: var(--text-color); font-size: 11px; }
@media (max-width: 600px) {
.failure-features { min-height: 340px; }
.failure-features .tooltip { max-width: 200px; }
.failure-features .feature-pill { display: none; }
}
</style>
<script>
(() => {
const ensureD3 = (cb) => {
if (window.d3 && typeof window.d3.select === 'function') return cb();
let s = document.getElementById('d3-cdn-script');
if (!s) { s = document.createElement('script'); s.id = 'd3-cdn-script'; s.src = 'https://cdn.jsdelivr.net/npm/d3@7/dist/d3.min.js'; document.head.appendChild(s); }
s.addEventListener('load', () => cb(), { once: true });
};
const bootstrap = () => {
const container = document.querySelector('.failure-features:not([data-mounted])');
if (!container) return;
container.dataset.mounted = 'true';
const d3 = window.d3;
const data = [
{ name: 'tau2 telecom', states: 43, cvAuroc: 0.946, holdout: 0.941, topFeature: 'message length before human handoff', features: 129 },
{ name: 'WebArena', states: 25, cvAuroc: 0.882, holdout: 0.903, topFeature: 'whether the run ends on a click', features: 50 },
{ name: 'ATBench', states: 15, cvAuroc: 0.864, holdout: 0.894, topFeature: 'whether the run ends on an assistant turn', features: 36 },
{ name: 'AgentNet', states: 25, cvAuroc: 0.886, holdout: 0.890, topFeature: 'a write-then-click step', features: 73 },
{ name: 'tau2 airline', states: 18, cvAuroc: 0.826, holdout: 0.864, topFeature: 'a cancel-reservation step', features: 70 },
{ name: 'SWE-agent', states: 25, cvAuroc: 0.806, holdout: 0.799, topFeature: 'edit message length', features: 39 },
{ name: 'tau2 retail', states: 19, cvAuroc: 0.752, holdout: 0.779, topFeature: 'longest user message', features: 52 },
{ name: 'OSWorld', states: 27, cvAuroc: 0.779, holdout: 0.774, topFeature: 'trace length', features: 79 },
{ name: 'SWE-smith', states: 10, cvAuroc: 0.636, holdout: 0.703, topFeature: 'longest tool message', features: 54 },
].sort((a, b) => b.holdout - a.holdout);
const ours = '#3d5a80';
const slate = '#8fa6c4';
const margin = { top: 38, right: 150, bottom: 48, left: 110 };
const tip = document.createElement('div');
tip.className = 'tooltip';
container.appendChild(tip);
let hasAnimated = false;
function render() {
container.querySelectorAll('svg').forEach(s => s.remove());
const rect = container.getBoundingClientRect();
const W = Math.max(400, Math.round(rect.width));
const barH = 34;
const H = margin.top + margin.bottom + data.length * (barH + 5);
const w = Math.max(0, W - margin.left - margin.right);
const h = Math.max(0, H - margin.top - margin.bottom);
const svg = d3.select(container).insert('svg', '.tooltip')
.attr('width', W).attr('height', Math.max(0, H));
const g = svg.append('g').attr('transform', `translate(${margin.left},${margin.top})`);
const x = d3.scaleLinear().domain([0.5, 1.0]).range([0, w]);
const y = d3.scaleBand().domain(data.map(d => d.name)).range([0, h]).padding(0.3);
// Grid
g.append('g').attr('class', 'grid')
.attr('transform', `translate(0,${h})`)
.call(d3.axisBottom(x).ticks(5).tickSize(Math.max(0, -h)).tickFormat(''));
// Random baseline (0.5) dashed line
g.append('line').attr('class', 'ref-line')
.attr('x1', x(0.5)).attr('x2', x(0.5)).attr('y1', -8).attr('y2', h)
.attr('stroke', 'var(--text-color)').attr('stroke-width', 1).attr('stroke-opacity', 0.3);
g.append('text')
.attr('x', x(0.5)).attr('y', -12)
.attr('fill', 'var(--text-color)').attr('font-size', '10px')
.attr('text-anchor', 'middle').attr('opacity', 0.45).text('random (0.5)');
// Bottom axis
g.append('g').attr('transform', `translate(0,${h})`).call(d3.axisBottom(x).ticks(5));
// Left axis
g.append('g').call(d3.axisLeft(y).tickSize(0)).select('.domain').remove();
g.append('text').attr('class', 'axis-label')
.attr('x', w / 2).attr('y', h + 40).attr('text-anchor', 'middle')
.text('AUROC');
// --- Horizontal legend at top ---
const leg = g.append('g').attr('transform', `translate(0, -26)`);
// Holdout
leg.append('circle').attr('cx', 0).attr('cy', 0).attr('r', 5).attr('fill', ours);
leg.append('text').attr('class', 'legend-item').attr('x', 10).attr('y', 4)
.attr('fill', 'var(--text-color)').attr('font-size', '11px').text('Holdout AUROC');
// CV
leg.append('circle').attr('cx', 112).attr('cy', 0).attr('r', 5)
.attr('fill', 'none').attr('stroke', slate).attr('stroke-width', 2);
leg.append('text').attr('class', 'legend-item').attr('x', 122).attr('y', 4)
.attr('fill', 'var(--text-color)').attr('font-size', '11px').text('5-fold CV AUROC');
// Connector
leg.append('line').attr('x1', 230).attr('x2', 254).attr('y1', 0).attr('y2', 0)
.attr('stroke', slate).attr('stroke-width', 2).attr('stroke-opacity', 0.4);
leg.append('text').attr('class', 'legend-item').attr('x', 260).attr('y', 4)
.attr('fill', 'var(--text-color)').attr('font-size', '11px').attr('opacity', 0.55).text('gap');
const animate = !hasAnimated;
const dur = 500;
// --- Dumbbell rows ---
data.forEach((d, idx) => {
const cy = y(d.name) + y.bandwidth() / 2;
const delay = idx * 60;
const xCV = x(d.cvAuroc);
const xHO = x(d.holdout);
const x0 = x(0.5);
// Connecting line (CV to Holdout)
const line = g.append('line')
.attr('y1', cy).attr('y2', cy)
.attr('stroke', slate).attr('stroke-width', 2).attr('stroke-opacity', 0.35)
.attr('stroke-linecap', 'round');
if (animate) {
line.attr('x1', x0).attr('x2', x0)
.transition().duration(dur).delay(delay).ease(d3.easeCubicOut)
.attr('x1', Math.min(xCV, xHO)).attr('x2', Math.max(xCV, xHO));
} else {
line.attr('x1', Math.min(xCV, xHO)).attr('x2', Math.max(xCV, xHO));
}
// CV AUROC (open circle)
const cvCircle = g.append('circle')
.attr('cx', animate ? x0 : xCV).attr('cy', cy)
.attr('fill', 'var(--page-bg)').attr('stroke', slate).attr('stroke-width', 2)
.attr('cursor', 'pointer');
if (animate) {
cvCircle.attr('r', 0)
.transition().duration(dur).delay(delay).ease(d3.easeCubicOut)
.attr('cx', xCV).attr('r', 5);
} else {
cvCircle.attr('r', 5);
}
// Holdout AUROC (filled circle)
const hoCircle = g.append('circle')
.attr('cx', animate ? x0 : xHO).attr('cy', cy)
.attr('fill', ours).attr('stroke', 'var(--page-bg)').attr('stroke-width', 2)
.attr('cursor', 'pointer');
if (animate) {
hoCircle.attr('r', 0)
.transition().duration(dur).delay(delay + 80).ease(d3.easeCubicOut)
.attr('cx', xHO).attr('r', 6);
} else {
hoCircle.attr('r', 6);
}
// Hover target (invisible wide rect)
const hitArea = g.append('rect')
.attr('x', 0).attr('y', cy - y.bandwidth() / 2)
.attr('width', Math.max(0, w)).attr('height', Math.max(0, y.bandwidth()))
.attr('fill', 'transparent').attr('cursor', 'pointer');
hitArea
.on('mouseenter', function(ev) {
cvCircle.attr('r', 7); hoCircle.attr('r', 8);
line.attr('stroke-opacity', 0.6).attr('stroke-width', 3);
tip.innerHTML = `<strong>${d.name}</strong><br/>States: ${d.states} | Features: ${d.features}<br/>CV AUROC: ${d.cvAuroc.toFixed(3)}<br/>Holdout AUROC: <strong>${d.holdout.toFixed(3)}</strong><br/>Top feature: <code style="font-size:11px;background:color-mix(in srgb, var(--text-color) 8%, transparent);padding:2px 5px;border-radius:4px">${d.topFeature}</code>`;
tip.style.opacity = '1';
})
.on('mousemove', function(ev) {
const [mx, my] = d3.pointer(ev, container);
const flipX = mx > rect.width * 0.6;
tip.style.transform = `translate(${flipX ? mx - 200 : mx + 14}px, ${my - 14}px)`;
})
.on('mouseleave', function() {
cvCircle.attr('r', 5); hoCircle.attr('r', 6);
line.attr('stroke-opacity', 0.35).attr('stroke-width', 2);
tip.style.opacity = '0';
});
// Feature pill on right
const pillG = g.append('g')
.attr('transform', `translate(${w + 10}, ${cy})`);
const pillText = pillG.append('text')
.attr('class', 'feature-pill')
.attr('x', 6).attr('y', 4)
.attr('fill', 'var(--text-color)').attr('opacity', 0.65)
.text(d.topFeature);
// Compute pill background after text renders
setTimeout(() => {
const bbox = pillText.node().getBBox();
pillG.insert('rect', 'text')
.attr('x', bbox.x - 5).attr('y', bbox.y - 2)
.attr('width', Math.max(0, bbox.width + 10)).attr('height', Math.max(0, bbox.height + 4))
.attr('rx', 4).attr('fill', 'var(--text-color)').attr('opacity', 0.05);
}, 0);
});
hasAnimated = true;
}
render();
if (window.ResizeObserver) {
let resizeTimer;
new ResizeObserver(() => {
clearTimeout(resizeTimer);
resizeTimer = setTimeout(render, 80);
}).observe(container);
}
};
if (document.readyState === 'loading') document.addEventListener('DOMContentLoaded', () => ensureD3(bootstrap), { once: true });
else ensureD3(bootstrap);
})();
</script>
</div></div><figcaption class="html-embed__desc" style="text-align:left">Held-out AUROC across the nine labeled datasets, with the top predictive feature for each. Larger FSMs (more states, more tools) predict better.</figcaption></figure> </div>
<p>Inspect per-state feature importances and failure modes in the <a href="https://seongland.com/article/asg/browser?tab=failure">failure view</a>.</p>
<p>And the predictors are interpretable. On SWE-agent the single strongest feature is whether the trace reaches the <code>submit</code> state: 94.8% of successes get there, only 55.7% of failures do. And we find this isn’t a length proxy — structural features score 0.790 against 0.659 for trace length alone. Successful runs touch only 9 of 25 states along a focused <code>search</code>, <code>edit</code>, <code>submit</code> path — and failures spread across all 25 (Jaccard overlap 0.206).</p>
<h2 id="runtime-monitor"><a href="#runtime-monitor">Runtime Monitor</a></h2>
<p>Deployed online <span class="" id="citation--wang2026agentspec--zhang2026agentracer--14">(H. <a href="#bib-wang2026agentspec" id="refctx-bib-wang2026agentspec-1">Wang et al., 2026</a>; G. <a href="#bib-zhang2026agentracer" id="refctx-bib-zhang2026agentracer-1">Zhang et al., 2026</a>)</span>, a two-rule monitor fires when the cycle-rate exceeds 0.778 and the unique-state count clears a warm-up floor. On all four evaluated datasets it reaches <strong>rank-AUROC 0.66 at the 25% trace checkpoint</strong>, against 0.5 for a flag-everything baseline by construction. On SWE-agent it fires at <strong>32% of trace completion</strong>, stopping the run before two-thirds of its remaining compute is spent, at precision 85.9% and recall 95.5%. By the halfway checkpoint, FSM features alone already recover 92% of the full-trace signal.</p>
<div class="note note--info" data-astro-cid-qg6lmfty> <!-- When there's a title, emoji is inline with title -->
<div class="note__body" data-astro-cid-qg6lmfty> <div class="note__header" data-astro-cid-qg6lmfty> <div class="note__title" data-astro-cid-qg6lmfty>Why rank-AUROC, not F1</div> </div> <div class="note__content" data-astro-cid-qg6lmfty> <p>When 84% of runs fail, flagging everything scores a high <a href="https://texonom.com/10b78f54e2d1464ba65e8709f2679ff2">F1</a> by default (0.914, versus the monitor’s 0.904). The point of a monitor isn’t whether to flag but <em>when</em>. Rank-AUROC measures exactly that early-warning utility. The pipeline is FSM replay only, 0.006 ms per step, with no ML model in the loop.</p> </div> </div> </div>
<p>Watch the monitor flag a failing run in real time in the <a href="https://seongland.com/article/asg/browser?tab=monitor">monitor view</a>.</p><div id="bibliography-references-list" class="references csl-bib-body" data-bibliography-block="true" data-built-refs="1"><ol class="references"><li id="bib-barres2025tau2">Barres, V., Dong, H., Ray, S., Si, X., &amp; Narasimhan, K. (2025). τ<sup>2</sup>-Bench: Evaluating Conversational Agents in a Dual-Control Environment. <i>arXiv Preprint arXiv:2506.07982</i>.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-barres2025tau2-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-barres2025tau2-2" aria-label="Back to citation">2</a>, <a href="#refctx-bib-barres2025tau2-3" aria-label="Back to citation">3</a></small></li><li id="bib-cheng2024trace_autodiff">Cheng, C.-A., Nie, A., &amp; Swaminathan, A. (2024). Trace is the New AutoDiff: Unlocking Efficient Optimization of Computational Workflows. <i>arXiv Preprint arXiv:2406.16218</i>.<small class="backrefs"><a href="#refctx-bib-cheng2024trace_autodiff-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-wang2026agentspec">Wang, H., Poskitt, C. M., &amp; Sun, J. (2026). AgentSpec: Customizable Runtime Enforcement for Safe and Reliable LLM Agents. <i>Proceedings of the International Conference on Software Engineering (ICSE)</i>.<small class="backrefs"><a href="#refctx-bib-wang2026agentspec-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-wang2025opencua">Wang, X., Wang, B., Lu, D., Yang, J., Xie, T., Wang, J., Deng, J., Guo, X., Xu, Y., Wu, C. H., Shen, Z., Li, Z., Li, R., Li, X., Chen, J., Boyuan, Z., Li, P., Lei, F., Cao, R., … Yu, T. (2025). OpenCUA: Open Foundations for Computer-Use Agents. <i>arXiv Preprint arXiv:2508.09123</i>.<small class="backrefs"><a href="#refctx-bib-wang2025opencua-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-wang2024agent_workflow_memory">Wang, Z. Z., Mao, J., Fried, D., &amp; Neubig, G. (2024). Agent Workflow Memory. <i>arXiv Preprint arXiv:2409.07429</i>.<small class="backrefs"><a href="#refctx-bib-wang2024agent_workflow_memory-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-xie2024osworld">Xie, T., Zhang, D., Chen, J., Li, X., Zhao, S., Cao, R., Hua, T. J., Cheng, Z., Shin, D., Lei, F., Liu, Y., Xu, Y., Zhou, S., Savarese, S., Xiong, C., Zhong, V., &amp; Yu, T. (2024). OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments. <i>The Thirty-Eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track</i>. <a href="https://openreview.net/forum?id=tN61DTr4Ed" target="_blank" rel="noopener noreferrer">https://openreview.net/forum?id=tN61DTr4Ed</a><small class="backrefs"><a href="#refctx-bib-xie2024osworld-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-yang2024sweagent">Yang, J., Jimenez, C. E., Wettig, A., Lieret, K., Yao, S., Narasimhan, K., &amp; Press, O. (2024). SWE-agent: Agent-Computer Interfaces Enable Automated Software Engineering. <i>Advances in Neural Information Processing Systems (NeurIPS)</i>.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-yang2024sweagent-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-yang2024sweagent-2" aria-label="Back to citation">2</a></small></li><li id="bib-yang2025swesmith">Yang, J., Lieret, K., Jimenez, C. E., Wettig, A., Khandpur, K., Zhang, Y., Hui, B., Press, O., Schmidt, L., &amp; Yang, D. (2025). SWE-smith: Scaling Data for Software Engineering Agents. <i>Proceedings of the Annual Conference on Neural Information Processing Systems (NeurIPS), Datasets &amp; Benchmarks Track</i>.<small class="backrefs"><a href="#refctx-bib-yang2025swesmith-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-zhang2026agentracer">Zhang, G., Wang, J., Chen, J., Zhou, W., Wang, K., &amp; Yan, S. (2026). AgenTracer: Who Is Inducing Failure in the LLM Agentic Systems? <i>Proceedings of the International Conference on Learning Representations (ICLR)</i>.<small class="backrefs"><a href="#refctx-bib-zhang2026agentracer-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-yang2025whoandwhen">Zhang, S., Yin, M., Zhang, J., Liu, J., Han, Z., Zhang, J., Li, B., Wang, C., Wang, H., Chen, Y., &amp; Wu, Q. (2025). Which Agent Causes Task Failures and When? <i>arXiv Preprint arXiv:2505.00212</i>.<small class="backrefs"><a href="#refctx-bib-yang2025whoandwhen-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-zhou2024webarena">Zhou, S., Xu, F. F., Zhu, H., Zhou, X., Lo, R., Sridhar, A., Cheng, X., Bisk, Y., Fried, D., Alon, U., &amp; Neubig, G. (2024). WebArena: A Realistic Web Environment for Building Autonomous Agents. <i>Proceedings of the International Conference on Learning Representations (ICLR)</i>.<small class="backrefs"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg> back: <a href="#refctx-bib-zhou2024webarena-1" aria-label="Back to citation">1</a>, <a href="#refctx-bib-zhou2024webarena-2" aria-label="Back to citation">2</a></small></li></ol></div>
<h1 id="discussion"><a href="#discussion">Discussion</a></h1>
<h2 id="when-does-it-work"><a href="#when-does-it-work">When Does It Work?</a></h2>
<p>A system’s action vocabulary is bounded, so its behavioral topology is bounded too, and a compact, stable automaton is the natural summary; beating four bespoke pipelines with it is a consequence, not a design goal.</p>
<p>The topology is also model-invariant: a single FSM achieves perfect fitness across four large language models on the same task, so what shapes it is the <strong>system</strong>, the tools and prompts and task distribution, more than the model driving it. It stays stable across extraction granularities too, shifting failure-prediction AUROC by less than 0.03 over four levels of <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>ϕ</mi></mrow><annotation encoding="application/x-tex">\phi</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.8889em;vertical-align:-0.1944em"></span><span class="mord mathnormal">ϕ</span></span></span></span>.</p>
<div class="note note--info" data-astro-cid-qg6lmfty> <!-- When there's no title, emoji is above content -->
<div class="note__layout" data-astro-cid-qg6lmfty> <div class="note__body" data-astro-cid-qg6lmfty> <div class="note__content" data-astro-cid-qg6lmfty> <p>A 25-state machine can be read and checked by a person; the 59,510-state prefix tree it came from can’t. That auditability is a direct dividend of minimality.</p> </div> </div> </div> </div>
<h2 id="related-work"><a href="#related-work">Related Work</a></h2>
<p>Three lines of work touch this one.</p>
<p><strong>State machines placed around agents.</strong> StateFlow runs an agent through a state machine somebody wrote by hand <span class="" id="citation--wu2024stateflow--1">(<a href="#bib-wu2024stateflow" id="refctx-bib-wu2024stateflow-1">Wu et al., 2024</a>)</span>, a definite finite automaton can be bolted onto a chatbot pipeline the same way <span class="" id="citation--sun2024dfa_llm--2">(<a href="#bib-sun2024dfa_llm" id="refctx-bib-sun2024dfa_llm-1">Sun et al., 2024</a>)</span>, FlowMind builds a workflow from an API surface <span class="" id="citation--jin2024flowmind--3">(<a href="#bib-jin2024flowmind" id="refctx-bib-jin2024flowmind-1">Zeng et al., 2024</a>)</span>, and AFlow and ADAS search over agentic workflow designs <span class="" id="citation--zhang2025aflow--hu2024adas--4">(<a href="#bib-hu2024adas" id="refctx-bib-hu2024adas-1">Hu et al., 2024</a>; <a href="#bib-zhang2025aflow" id="refctx-bib-zhang2025aflow-1">Zhang et al., 2025</a>)</span>, while MetaAgent assembles a whole multi-agent system out of one <span class="" id="citation--chen2025metaagent--5">(<a href="#bib-chen2025metaagent" id="refctx-bib-chen2025metaagent-1">Chen et al., 2025</a>)</span>. In every one of them the machine is an input — here it is an output, recovered from traces the agent had already produced.
<strong>Experience kept as memory.</strong> Agent Workflow Memory induces reusable routines from past trajectories <span class="" id="citation--wang2024agent_workflow_memory--6">(<a href="#bib-wang2024agent_workflow_memory" id="refctx-bib-wang2024agent_workflow_memory-1">Wang et al., 2024</a>)</span>, Experience-to-Strategy trains a graph of them <span class="" id="citation--xia2025experience_to_strategy--7">(<a href="#bib-xia2025experience_to_strategy" id="refctx-bib-xia2025experience_to_strategy-1">Xia et al., 2025</a>)</span>, ExpeL distils insights out of them <span class="" id="citation--zhao2024expel--8">(<a href="#bib-zhao2024expel" id="refctx-bib-zhao2024expel-1">Zhao et al., 2024</a>)</span>, and Voyager accumulates a skill library <span class="" id="citation--wang2023voyager--9">(<a href="#bib-wang2023voyager" id="refctx-bib-wang2023voyager-1">G. Wang et al., 2023</a>)</span>. Each of those keeps fragments of behavior. The automaton keeps the topology instead.</p>
<p><strong>Automata learned from observation.</strong> Model learning turns a black-box system into a state machine <span class="" id="citation--vaandrager2017model--10">(<a href="#bib-vaandrager2017model" id="refctx-bib-vaandrager2017model-1">Vaandrager, 2017</a>)</span>, Weiss and colleagues pull one out of a recurrent network with membership and equivalence queries <span class="" id="citation--weiss2018extracting--11">(<a href="#bib-weiss2018extracting" id="refctx-bib-weiss2018extracting-1">Weiss et al., 2018</a>)</span>, DeepDFA learns one by gradient <span class="" id="citation--umili2024deepdfa--12">(<a href="#bib-umili2024deepdfa" id="refctx-bib-umili2024deepdfa-1">Umili &amp; Capobianco, 2024</a>)</span>, prompt chaining extracts one from a flow description <span class="" id="citation--ali2024flowfsm--13">(<a href="#bib-ali2024flowfsm" id="refctx-bib-ali2024flowfsm-1">Wael et al., 2025</a>)</span>, and AALpy packages the classical algorithms <span class="" id="citation--muskardin2022aalpy--14">(<a href="#bib-muskardin2022aalpy" id="refctx-bib-muskardin2022aalpy-1">Muškardin et al., 2022</a>)</span>. Process mining, which discovers process models from event logs, is the closest classical neighbour: van der Aalst’s textbook treatment <span class="" id="citation--vanderaalst2016process--15">(<a href="#bib-vanderaalst2016process" id="refctx-bib-vanderaalst2016process-1">van der Aalst, 2016</a>)</span>, its re-thinking for the agent era <span class="" id="citation--berti2024processmining--16">(<a href="#bib-berti2024processmining" id="refctx-bib-berti2024processmining-1">Berti et al., 2024</a>)</span>, an evaluation of large language models at the task <span class="" id="citation--grohs2024process_mining_llm--17">(<a href="#bib-grohs2024process_mining_llm" id="refctx-bib-grohs2024process_mining_llm-1">Berti, Kourani, et al., 2024</a>)</span>, and skill learning from mined processes <span class="" id="citation--chen2024skill_process_mining--18">(<a href="#bib-chen2024skill_process_mining" id="refctx-bib-chen2024skill_process_mining-1">Redis et al., 2024</a>)</span>. This one works from a different input: positive traces only, no oracle to query, and an alphabet small enough that one merge suffices.</p>
<h2 id="limitations"><a href="#limitations">Limitations</a></h2>
<p>The FSM accepts the observed prefix language, not the agent’s true generating language: like any trace-replay method, it can’t tell a trace that stays within the observed transition patterns from a legitimate one. The extraction function <span class="katex"><span class="katex-mathml"><math xmlns="http://www.w3.org/1998/Math/MathML"><semantics><mrow><mi>ϕ</mi></mrow><annotation encoding="application/x-tex">\phi</annotation></semantics></math></span><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.8889em;vertical-align:-0.1944em"></span><span class="mord mathnormal">ϕ</span></span></span></span> needs a small amount of per-domain knowledge — and fully automatic discovery of it is future work — failure prediction degrades on simpler machines: AUROC falls to 0.799 on SWE-agent and 0.70 on the 10-state SWE-smith — both are smaller task spaces, with less structure to exploit. Extending the workflow-memory comparison beyond AWM to other memory-injection methods is future work.</p>
<p>For agents with much larger action spaces or weaker sequential structure, the construction stays minimal but stops being compact — and the per-state observation density that drives every result above would degrade with it.</p>
<h2 id="broader-impact"><a href="#broader-impact">Broader Impact</a></h2>
<blockquote class="quote" data-astro-cid-arj5dyob> <div class="quote__text" data-astro-cid-arj5dyob> <p>Compact FSM representations make agent behavioral structure inspectable — and that supports safety auditing. The same analysis could be misused to find exploitable behavioral patterns, so deployment should restrict FSM analysis to authorized auditing.</p> </div> </blockquote>
<h2 id="conclusion"><a href="#conclusion">Conclusion</a></h2>
<p>A finite-state machine — built in milliseconds from positive examples with one classical merge — does the work of four bespoke learned pipelines. The same 7-to-43-state object serves all four.</p><div id="bibliography-references-list" class="references csl-bib-body" data-bibliography-block="true" data-built-refs="1"><ol class="references"><li id="bib-grohs2024process_mining_llm">Berti, A., Kourani, H., Hafke, H., Li, C.-Y., &amp; Schuster, D. (2024). Evaluating Large Language Models in Process Mining: Capabilities, Benchmarks, Evaluation Strategies, and Future Challenges. <i>arXiv Preprint arXiv:2403.06749</i>.<small class="backrefs"><a href="#refctx-bib-grohs2024process_mining_llm-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-berti2024processmining">Berti, A., Maatallah, M., Jessen, U., Sroka, M., &amp; Ghannouchi, S. A. (2024). Re-Thinking Process Mining in the AI-Based Agents Era. <i>arXiv Preprint arXiv:2408.07720</i>.<small class="backrefs"><a href="#refctx-bib-berti2024processmining-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-chen2025metaagent">Chen, Z., Wu, Y., Li, Z., &amp; Ji, H. (2025). MetaAgent: Automatically Constructing Multi-Agent Systems Based on Finite State Machines. <i>Proceedings of the International Conference on Machine Learning (ICML)</i>.<small class="backrefs"><a href="#refctx-bib-chen2025metaagent-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-hu2024adas">Hu, S., Lu, C., &amp; Clune, J. (2024). ADAS: Automated Design of Agentic Systems. <i>arXiv Preprint arXiv:2408.08435</i>.<small class="backrefs"><a href="#refctx-bib-hu2024adas-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-muskardin2022aalpy">Muškardin, E., Aichernig, B. K., Pill, I., Pferscher, A., &amp; Tappler, M. (2022). AALpy: An Active Automata Learning Library. <i>Innovations in Systems and Software Engineering</i>, <i>18</i>, 417–426.<small class="backrefs"><a href="#refctx-bib-muskardin2022aalpy-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-chen2024skill_process_mining">Redis, A. C., Fani Sani, M., Zarrin, B., &amp; Burattin, A. (2024). Skill Learning Using Process Mining for Large Language Model Plan Generation. <i>International Conference on Process Mining (ICPM)</i>.<small class="backrefs"><a href="#refctx-bib-chen2024skill_process_mining-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-sun2024dfa_llm">Sun, Y., Hu, J., Cheng, W., &amp; Chen, H. (2024). Chatbot Meets Pipeline: Augment Large Language Model with Definite Finite Automaton. <i>arXiv Preprint arXiv:2402.04411</i>.<small class="backrefs"><a href="#refctx-bib-sun2024dfa_llm-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-umili2024deepdfa">Umili, E., &amp; Capobianco, R. (2024). DeepDFA: Automata Learning through Neural Probabilistic Relaxations. <i>Proceedings of the European Conference on Artificial Intelligence (ECAI)</i>.<small class="backrefs"><a href="#refctx-bib-umili2024deepdfa-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-vaandrager2017model">Vaandrager, F. (2017). Model Learning. <i>Communications of the ACM</i>, <i>60</i>(2), 86–95.<small class="backrefs"><a href="#refctx-bib-vaandrager2017model-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-vanderaalst2016process">van der Aalst, W. M. P. (2016). <i>Process Mining: Data Science in Action</i> (2nd ed.). Springer.<small class="backrefs"><a href="#refctx-bib-vanderaalst2016process-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-ali2024flowfsm">Wael, F., Maklad, Y., Hamdi, A., &amp; Elsersy, W. (2025). An Agentic Flow for Finite State Machine Extraction using Prompt Chaining. <i>arXiv Preprint arXiv:2507.11222</i>.<small class="backrefs"><a href="#refctx-bib-ali2024flowfsm-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-wang2023voyager">Wang, G., Xie, Y., Jiang, Y., Mandlekar, A., Xiao, C., Zhu, Y., Fan, L., &amp; Anandkumar, A. (2023). Voyager: An Open-Ended Embodied Agent with Large Language Models. <i>arXiv Preprint arXiv:2305.16291</i>.<small class="backrefs"><a href="#refctx-bib-wang2023voyager-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-wang2024agent_workflow_memory">Wang, Z. Z., Mao, J., Fried, D., &amp; Neubig, G. (2024). Agent Workflow Memory. <i>arXiv Preprint arXiv:2409.07429</i>.<small class="backrefs"><a href="#refctx-bib-wang2024agent_workflow_memory-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-weiss2018extracting">Weiss, G., Goldberg, Y., &amp; Yahav, E. (2018). Extracting Automata from Recurrent Neural Networks Using Queries and Counterexamples. <i>Proceedings of the International Conference on Machine Learning (ICML)</i>.<small class="backrefs"><a href="#refctx-bib-weiss2018extracting-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-wu2024stateflow">Wu, Y., Yue, T., Zhang, S., Wang, C., &amp; Wu, Q. (2024). StateFlow: Enhancing LLM Task-Solving through State-Driven Workflows. <i>arXiv Preprint arXiv:2403.11322</i>.<small class="backrefs"><a href="#refctx-bib-wu2024stateflow-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-xia2025experience_to_strategy">Xia, S., Xu, Z., Chai, J., Fan, W., Song, Y., Wang, X., Yin, G., Lin, W., Zhang, H., &amp; Wang, J. (2025). From Experience to Strategy: Empowering LLM Agents with Trainable Graph Memory. <i>arXiv Preprint arXiv:2511.07800</i>.<small class="backrefs"><a href="#refctx-bib-xia2025experience_to_strategy-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-jin2024flowmind">Zeng, Z., Watson, W., Cho, N., Rahimi, S., Reynolds, S., Balch, T., &amp; Veloso, M. (2024). FlowMind: Automatic Workflow Generation with LLMs. <i>arXiv Preprint</i>.<small class="backrefs"><a href="#refctx-bib-jin2024flowmind-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-zhang2025aflow">Zhang, J., Xiang, J., Yu, Z., Teng, F., Chen, X., Lu, J., Zhong, M., Zhang, M., Wang, Y., Li, Q., &amp; Hong, H. (2025). AFlow: Automating Agentic Workflow Generation. <i>Proceedings of the International Conference on Learning Representations (ICLR)</i>.<small class="backrefs"><a href="#refctx-bib-zhang2025aflow-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li><li id="bib-zhao2024expel">Zhao, A., Huang, D., Xu, Q., Lin, M., Liu, Y.-J., &amp; Huang, G. (2024). ExpeL: LLM Agents Are Experiential Learners. <i>Proceedings of the AAAI Conference on Artificial Intelligence</i>.<small class="backrefs"><a href="#refctx-bib-zhao2024expel-1" aria-label="Back to citation"><svg class="back-icon" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true" focusable="false"><line x1="12" y1="19" x2="12" y2="5"></line><polyline points="5 12 12 5 19 12"></polyline></svg></a></small></li></ol></div> </main> </section> <footer class="footer"> <div class="footer-inner"> <section class="citation-block"> <h3>Citation</h3> <p>For attribution in academic contexts, please cite this work as</p> <pre class="citation short">Seonglae Cho, Franklin Cardenoso Fernandez, Umar Mohammed, Zekun Wu, Kleyton Da Costa, Ilham Wicaksono, Adriano Koshiyama (2026). &quot;Automata from Agent Traces: Failure and Next-Step Prediction&quot;. Second Workshop on Agents in the Wild: Safety, Security, and Beyond.</pre> <p>BibTeX citation</p> <pre class="citation long">@inproceedings{cho2026automata,
title={Automata from Agent Traces: Failure and Next-Step Prediction},
author={Seonglae Cho and Franklin Cardenoso Fernandez and Umar Mohammed and Zekun Wu and Kleyton Da Costa and Ilham Wicaksono and Adriano Koshiyama},
booktitle={Second Workshop on Agents in the Wild: Safety, Security, and Beyond},
year={2026},
url={https://arxiv.org/abs/2608.23670},
howpublished={\url{https://seongland.com/article/asg}}
}</pre> </section> <section class="reuse-block"> <h3>Reuse</h3> <p>Diagrams and text are licensed under <a href="https://creativecommons.org/licenses/by/4.0/" target="_blank" rel="noopener noreferrer">CC-BY 4.0</a>.
</p> </section> <section class="references-block"> </section> <div class="template-credit"> <p>
Made with ❤️ with <a href="https://huggingface.co/spaces/tfrere/research-article-template" target="_blank" rel="noopener noreferrer">research article template</a> </p> </div> </div> </footer> <script>
(() => {
const getFooter = () =>
document.currentScript?.closest("footer") ||
document.querySelector("footer.footer");
const footer = getFooter();
if (!footer) return;
const target = footer.querySelector(".references-block");
if (!target) return;
const contentRoot =
document.querySelector("section.content-grid main") ||
document.querySelector("main") ||
document.body;
const ensureHeading = (text) => {
const exists = Array.from(target.children).some(
(c) =>
c.tagName === "H3" &&
c.textContent.trim().toLowerCase() === text.toLowerCase(),
);
if (!exists) {
const h = document.createElement("h3");
h.textContent = text;
target.appendChild(h);
}
};
const moveIntoFooter = (element, headingText) => {
if (!element) return false;
// Remove an eventual heading already included inside the block (avoid duplicates)
const firstHeading = element.querySelector(
":scope > h1, :scope > h2, :scope > h3",
);
if (firstHeading) {
const txt = (firstHeading.textContent || "").trim().toLowerCase();
const targetTxt = headingText.trim().toLowerCase();
if (
txt === targetTxt ||
txt.includes("reference") ||
txt.includes("bibliograph")
) {
firstHeading.remove();
}
}
// Move footnote backref links inside paragraphs
if (element.classList && element.classList.contains("footnotes")) {
const footnoteItems = element.querySelectorAll("li");
footnoteItems.forEach((item) => {
const backrefContainer = item.querySelector("small.backrefs");
const lastP = item.querySelector("p:last-of-type");
if (backrefContainer && lastP && !lastP.contains(backrefContainer)) {
lastP.appendChild(document.createTextNode(" "));
lastP.appendChild(backrefContainer);
}
});
}
ensureHeading(headingText);
target.appendChild(element);
return true;
};
const run = () => {
const findFirstOutsideFooter = (selectors) => {
for (const sel of selectors) {
const el = contentRoot.querySelector(sel);
if (el && !footer.contains(el)) return el;
}
return null;
};
// rehype-citation emits one bibliography per MDX file, so a chapter-split
// article produces one per chapter. Moving only the first left the rest
// stranded mid-article, printed as a bare author list between two
// sections, and left six elements sharing one id.
const findAllOutsideFooter = (selectors) => {
const found = [];
for (const sel of selectors) {
for (const el of contentRoot.querySelectorAll(sel)) {
if (footer.contains(el) || found.includes(el)) continue;
found.push(el);
}
}
// .references matches the block and the <ol> inside it; keep the outermost
return found.filter((el) => !found.some((o) => o !== el && o.contains(el)));
};
const refBlocks = findAllOutsideFooter([
"#bibliography-references-list",
"[data-bibliography-block]",
"#references",
"#refs",
".references",
".bibliography",
]);
const referencesEl = refBlocks[0] || null;
if (referencesEl) {
const list = referencesEl.querySelector("ol, ul") || referencesEl;
const seenIds = new Set(
Array.from(list.children)
.map((li) => li.id)
.filter(Boolean),
);
for (const block of refBlocks.slice(1)) {
block
.querySelectorAll(":scope > ol > li, :scope > ul > li, :scope > li")
.forEach((li) => {
if (li.id && seenIds.has(li.id)) return;
if (li.id) seenIds.add(li.id);
list.appendChild(li);
});
block.remove();
}
// one alphabetical list, not six alphabetical lists end to end
Array.from(list.children)
.sort((a, b) =>
(a.textContent || "")
.trim()
.localeCompare((b.textContent || "").trim()),
)
.forEach((li) => list.appendChild(li));
}
const footnotesEl = findFirstOutsideFooter([".footnotes"]);
const movedRefs = moveIntoFooter(referencesEl, "References");
const movedNotes = moveIntoFooter(footnotesEl, "Footnotes");
return movedRefs || movedNotes;
};
// Try now; if not found yet, try again on DOM ready
const done = run();
if (!done) {
const onReady = () => run();
if (document.readyState === "loading") {
document.addEventListener("DOMContentLoaded", onReady, { once: true });
} else {
setTimeout(onReady, 0);
}
}
// Resize on window changes (e.g., fonts, layout)
// No textarea auto-resize needed for <pre> blocks
})();
</script> </body> </html>