import { useState, useEffect, useMemo, useCallback, useRef, lazy, Suspense, Fragment } from 'react' import { BarChart, Bar, LineChart, Line, XAxis, YAxis, CartesianGrid, Tooltip, ResponsiveContainer, Cell, Legend, } from 'recharts' import { cn, fmt, fmtInt, pct, compression } from './lib/utils' const LandingView = lazy(() => import('./LandingView')) // ---------- Types ---------- interface Baseline { method: string label: string trainFitness: number testFitness: number states: number | null transitions: number | null randomAcceptRate: number | null permutedAcceptRate: number | null pmPrecision: number | null category: string note: string } interface TraceStep { idx: number role: string activity: string fromState: string toState: string | null consumed: boolean } interface Trace { id: string split: 'train' | 'test' success: boolean | null fullFitness: number | null successFitness: number | null steps: TraceStep[] stateSequence: string[] fitness: number consumed: number total: number firstFailIdx: number stateVisits: Record } interface Dataset { id: string meta: { name: string shortName: string venue: string traces: number trainSize: number testSize: number activities: number description: string } baselines: Baseline[] traces: Trace[] convergence: { stateConvergenceAt: number transitionConvergenceAt: number fitnessConvergenceAt: number curve: Array<{ n: number; states: number; transitions: number; fitness: number }> } downstream: { fitnessProfile: { mean: number; std: number; min: number; max: number anomalyCount: number; anomalies: string[] histogram: Array<{ min: number; max: number; count: number }> } | null behavioralProfiling: { stateVisitDistribution: Array<{ state: string; count: number; fraction: number }> transitionFrequency: Array<{ transition: string; count: number; fraction: number }> deadTransitions: string[] transitionCoverage: number avgUniqueStatesPerTrace: number } | null structuralAnalysis: { density: number hubs: Array<{ id: string; outDegree: number; inDegree: number }> sinks: string[] sources: string[] selfLoops: string[] nonTrivialSCCs: string[][] } | null differentialFSM: { successOnlyTransitions: string[] failureOnlyTransitions: string[] sharedTransitions: string[] } | null failurePrediction: Record | null mistakeLocalization: Record | null neuralProbe: { seq_mlp: { cv: number; std: number; holdout: number } fsm_mlp: { cv: number; std: number; holdout: number } seq_gru: { cv: number; std: number; holdout: number } fsm_gru: { cv: number; std: number; holdout: number } seq_transf: { cv: number; std: number; holdout: number } fsm_transf: { cv: number; std: number; holdout: number } } | null workflowMemory: { n: number noMemory: { accuracy: number; correct: number } awm: { accuracy: number; correct: number } fsm: { accuracy: number; correct: number } fsmGain: number } | null crossModel: { models: Array<{ name: string; shortName: string; traces: number; successRate: number }> matrices: { fitness: number[][] auroc: number[][] stateCount: number[][] } summary: { diagonalAUROC: { mean: number; std: number } offDiagonalAUROC: { mean: number; std: number } aurocTransferGap: number } } | null monitor: { trainSize: number; testSize: number; succSize: number; failSize: number configs: Array<{ config: string; gamma: number; window: number monitoringCurve: Array<{ point: number; threshold: number; f1: number; precision: number; recall: number }> }> baselines: Record }> } | null counterfactual: { nSuccess: number; nFailure: number decisionPoints: Array<{ state: string; branching: number; targets: string[] }> pathDiversity: { uniqueSuccPaths: number; uniqueFailPaths: number; overlap: number } dpDivergence: Array<{ state: string; jsd: number }> editDistance: { meanInternal?: number; meanCross?: number } | null } | null } multiseed: { numSeeds: number ourFSM: { testFitness: { mean: number; std: number }; states: { mean: number; std: number } } rpni: { testFitness: { mean: number; std: number }; states: { mean: number; std: number } } awm: { testFitness: { mean: number; std: number } } } | null graph: { states: Array<{ id: string }> transitions: Array<{ source: string; target: string; subject: string }> } } type ViewId = 'landing' | 'overview' | 'baselines' | 'convergence' | 'graph' | 'traces' | 'downstream' | 'failure' | 'memory' | 'crossmodel' | 'monitor' | 'counterfactual' | 'about' // All datasets (passed into sub-views that need cross-dataset comparisons) type AllDatasets = Dataset[] // ---------- Palette (warm, no green/purple) ---------- const PALETTE = { rust: '#c4553a', slate: '#3d5a80', ochre: '#c49a3a', charcoal: '#2d2926', clay: '#a66e4e', iron: '#5c6b73', teal: '#2a8f82', } const METHOD_COLORS: Record = { stateGraph: PALETTE.rust, rpni: PALETTE.slate, heuristic: PALETTE.ochre, inductive: PALETTE.clay, alpha: PALETTE.iron, awm: PALETTE.charcoal, awm_all: '#7a7068', } function methodLabel(b: Baseline): string { if (b.method === 'stateGraph') return 'Agent State Graph' return b.label } // ---------- Chart theme ---------- const CHART_GRID = '#d5cfc5' const CHART_TICK = { fill: '#8a8478', fontSize: 11, fontFamily: 'IBM Plex Mono, monospace' } const CHART_AXIS = { stroke: '#d5cfc5' } const CHART_TOOLTIP = { contentStyle: { background: '#f5f0e8', border: '1px solid #d5cfc5', borderRadius: 0, fontSize: 12, fontFamily: 'IBM Plex Mono, monospace', boxShadow: '2px 2px 0 rgba(0,0,0,0.04)', }, } // ---------- URL State ---------- const VALID_TABS: ViewId[] = ['landing', 'overview', 'baselines', 'convergence', 'graph', 'traces', 'downstream', 'failure', 'memory', 'crossmodel', 'monitor', 'counterfactual', 'about'] function readParams(): { dataset?: string; tab?: ViewId; trace?: string; method?: string } { const p = new URLSearchParams(window.location.search) const tab = p.get('tab') as ViewId | null return { dataset: p.get('dataset') || undefined, tab: tab && VALID_TABS.includes(tab) ? tab : undefined, trace: p.get('trace') || undefined, method: p.get('method') || undefined, } } function pushParams(params: Record) { const p = new URLSearchParams(window.location.search) for (const [k, v] of Object.entries(params)) p.set(k, v) const url = `${window.location.pathname}?${p.toString()}` window.history.pushState(null, '', url) } // ---------- Main App ---------- export default function App() { const [datasets, setDatasets] = useState([]) const [activeDs, setActiveDs] = useState(0) const [view, setView] = useState(() => { const tab = new URLSearchParams(window.location.search).get('tab') as ViewId | null return tab && VALID_TABS.includes(tab) ? tab : 'landing' }) const [selectedTrace, setSelectedTrace] = useState(null) const [selectedMethod, setSelectedMethod] = useState(null) const [loadProgress, setLoadProgress] = useState(0) // On load: fetch slim index first (~3MB), then load per-dataset shards in parallel. // Falls back to monolithic data.json if index isn't published yet. // Paths are base-relative (import.meta.env.BASE_URL) so the dashboard works both at // the dev root and when served under /article/asg/browser/ in production. useEffect(() => { const BASE = import.meta.env.BASE_URL const loadShards = async (): Promise => { const idxRes = await fetch(`${BASE}index.json`) if (!idxRes.ok) throw new Error('no-index') const index = await idxRes.json() as Array<{ id: string }> const shards = await Promise.all(index.map(async (entry, i) => { const r = await fetch(`${BASE}datasets/${entry.id}.json`) const ds = await r.json() as Dataset setLoadProgress(Math.round(((i + 1) / index.length) * 100)) return ds })) return shards } const loadMonolith = async (): Promise => { const r = await fetch(`${BASE}data.json`) setLoadProgress(50) const data = await r.json() as Dataset[] setLoadProgress(100) return data } loadShards() .catch(() => loadMonolith()) .then(data => { setDatasets(data) setLoadProgress(100) const { dataset, tab, trace, method } = readParams() if (dataset) { const idx = data.findIndex(d => d.id === dataset) if (idx >= 0) setActiveDs(idx) } if (tab) setView(tab) if (trace) setSelectedTrace(trace) if (method) setSelectedMethod(method) }) }, []) // Sync URL when state changes const navigate = useCallback((dsIndex: number, tab: ViewId, data?: Dataset[]) => { const ds = (data || datasets)[dsIndex] if (ds) pushParams({ dataset: ds.id, tab }) }, [datasets]) const handleSetDataset = useCallback((i: number) => { setActiveDs(i) setSelectedTrace(null) setSelectedMethod(null) // Keep current tab when switching datasets setView(prev => { navigate(i, prev) return prev }) }, [navigate]) const handleSetView = useCallback((v: ViewId) => { setView(v) setSelectedTrace(null) setSelectedMethod(null) const p = new URLSearchParams(window.location.search) p.set('tab', v) p.delete('trace') p.delete('method') window.history.pushState(null, '', `${window.location.pathname}?${p.toString()}`) }, []) const handleSelectTrace = useCallback((traceId: string | null) => { setSelectedTrace(traceId) const p = new URLSearchParams(window.location.search) if (traceId) p.set('trace', traceId); else p.delete('trace') window.history.pushState(null, '', `${window.location.pathname}?${p.toString()}`) }, []) const handleSelectMethod = useCallback((method: string | null) => { setSelectedMethod(method) const p = new URLSearchParams(window.location.search) if (method) p.set('method', method); else p.delete('method') window.history.pushState(null, '', `${window.location.pathname}?${p.toString()}`) }, []) // Handle browser back/forward useEffect(() => { const onPop = () => { const { dataset, tab, trace, method } = readParams() if (dataset && datasets.length) { const idx = datasets.findIndex(d => d.id === dataset) if (idx >= 0) setActiveDs(idx) } if (tab) setView(tab) setSelectedTrace(trace || null) setSelectedMethod(method || null) } window.addEventListener('popstate', onPop) return () => window.removeEventListener('popstate', onPop) }, [datasets]) if (!datasets.length) { return (
Loading...
{loadProgress > 0 && (
)}
) } const ds = datasets[activeDs] const dashboardTabs: [ViewId, string][] = [ ['overview', 'Overview'], ['baselines', 'Baselines'], ['convergence', 'Convergence'], ['graph', 'Graph'], ['traces', 'Traces'], ['downstream', 'Downstream'], ['failure', 'Failure'], ['memory', 'Memory'], ['crossmodel', 'Cross-Model'], ['monitor', 'Monitor'], ['counterfactual', 'Counterfactual'], ['about', 'About'], ] // Landing page: full-screen, no header/nav if (view === 'landing') { return (
Loading...
}> handleSetView('overview')} />
) } return (
{/* Masthead */}

Experiment Results

{datasets.map((d, i) => ( ))}
{/* Nav */} {/* Content */}
{view === 'overview' && } {view === 'baselines' && } {view === 'convergence' && } {view === 'graph' && } {view === 'traces' && } {view === 'downstream' && } {view === 'failure' && } {view === 'memory' && } {view === 'crossmodel' && } {view === 'monitor' && } {view === 'counterfactual' && } {view === 'about' && }
{/* Footer rule */}
Agent State Graph / ICML 2026 AIWILD Workshop {new Date().toLocaleDateString('en-US', { year: 'numeric', month: 'short' })}
) } // ---------- Overview ---------- function MiniGraph({ states, transitions }: { states: { id: string }[]; transitions: { source: string; target: string; subject: string }[] }) { const initialPos = useMemo(() => computeLayout(states, transitions), [states, transitions]) const [pos, setPos] = useState(initialPos) useEffect(() => setPos(initialPos), [initialPos]) const edgeSet = useMemo(() => { const s = new Set() for (const t of transitions) s.add(`${t.source}|${t.target}`) return s }, [transitions]) // Zoom + pan const [zoom, setZoom] = useState(1) const [pan, setPan] = useState({ x: 0, y: 0 }) const containerDrag = useRef(false) const containerLast = useRef({ x: 0, y: 0 }) // Node drag const svgRef = useRef(null) const dragNode = useRef(null) const toSVG = useCallback((clientX: number, clientY: number) => { const svg = svgRef.current if (!svg) return { x: 0, y: 0 } const pt = svg.createSVGPoint() pt.x = clientX; pt.y = clientY const ctm = svg.getScreenCTM() if (!ctm) return { x: 0, y: 0 } const svgPt = pt.matrixTransform(ctm.inverse()) return { x: svgPt.x, y: svgPt.y } }, []) const handleWheel = useCallback((e: React.WheelEvent) => { e.preventDefault() setZoom(z => Math.max(0.3, Math.min(5, z * (e.deltaY > 0 ? 0.9 : 1.1)))) }, []) const handlePointerDown = useCallback((e: React.PointerEvent) => { containerDrag.current = true containerLast.current = { x: e.clientX, y: e.clientY } }, []) const handlePointerMove = useCallback((e: React.PointerEvent) => { if (dragNode.current) { const { x, y } = toSVG(e.clientX, e.clientY) setPos(prev => ({ ...prev, [dragNode.current!]: { x, y } })) return } if (!containerDrag.current) return const dx = e.clientX - containerLast.current.x const dy = e.clientY - containerLast.current.y containerLast.current = { x: e.clientX, y: e.clientY } setPan(p => ({ x: p.x + dx, y: p.y + dy })) }, [toSVG]) const handlePointerUp = useCallback(() => { containerDrag.current = false dragNode.current = null }, []) const handleNodeDown = useCallback((e: React.PointerEvent, id: string) => { e.stopPropagation() dragNode.current = id ;(e.target as Element).setPointerCapture(e.pointerId) }, []) const handleDoubleClick = useCallback(() => { setZoom(1); setPan({ x: 0, y: 0 }) }, []) return (
{transitions.map((t, i) => { const from = pos[t.source], to = pos[t.target] if (!from || !to) return null const al=10, r=14 if (t.source === t.target) { const loopR=30, cx1=from.x-loopR, cy1=from.y-loopR*1.5, cx2=from.x+loopR, cy2=from.y-loopR*1.5 const sa=Math.atan2(cy1-from.y,cx1-from.x), ea=Math.atan2(cy2-from.y,cx2-from.x) const epx=from.x+r*Math.cos(ea), epy=from.y+r*Math.sin(ea) const tdx=epx-cx2, tdy=epy-cy2, td=Math.sqrt(tdx*tdx+tdy*tdy) return } const dx = to.x-from.x, dy = to.y-from.y, d = Math.sqrt(dx*dx+dy*dy), nx=dx/d, ny=dy/d if (edgeSet.has(`${t.target}|${t.source}`)) { const c=20, mx=(from.x+to.x)/2-ny*c, my=(from.y+to.y)/2+nx*c const sa=Math.atan2(my-from.y,mx-from.x), ea=Math.atan2(my-to.y,mx-to.x) const epx=to.x+r*Math.cos(ea), epy=to.y+r*Math.sin(ea) const tdx=epx-mx, tdy=epy-my, td=Math.sqrt(tdx*tdx+tdy*tdy) return } return })} {states.map(s => { const p = pos[s.id] if (!p) return null const isInit = s.id === 'init' const label = s.id.length > 14 ? s.id.slice(0,12)+'..' : s.id return ( handleNodeDown(e, s.id)}> {isInit && } {label} ) })}

{states.length} states, {transitions.length} transitions

) } function OverviewView({ ds, datasets, onSelectDataset }: { ds: Dataset; datasets: Dataset[]; onSelectDataset: (i: number) => void }) { const ours = ds.baselines.find(b => b.category === 'ours')! const rpni = ds.baselines.find(b => b.method === 'rpni')! return (
{/* Title block */}

{ds.meta.name}

{ds.meta.venue}

{ds.meta.description}

{/* Key figures - newspaper style */}
{/* Final graph preview */}
{/* Cross-dataset table */}
{datasets.map((d, i) => { const o = d.baselines.find(b => b.category === 'ours')! const r = d.baselines.find(b => b.method === 'rpni')! const isActive = d.id === ds.id return ( onSelectDataset(i)} className={cn('ruled cursor-pointer transition-colors', isActive ? 'bg-rust-dim' : 'hover:bg-ochre-dim')} > ) })}
Dataset N |A| Fitness ASG |Q| RPNI |Q| Ratio Rand. Rej.
{d.meta.name} {isActive && current} {fmtInt(d.meta.traces)} {o.states ? o.states - 1 : '–'} {fmt(o.testFitness)} {o.states} {fmtInt(r.states || 0)} {compression(o.states || 1, r.states || 1)} {o.randomAcceptRate !== null ? pct(1 - o.randomAcceptRate) : '–'}
{/* Bar chart */}
b.testFitness > 0).map(b => ({ ...b, displayLabel: methodLabel(b) }))} barCategoryGap="16%"> v.toFixed(1)} /> [fmt(v ?? 0, 4), 'Fitness']} /> {ds.baselines.filter(b => b.testFitness > 0).map((b) => ( ))}
) } // ---------- Baselines ---------- function BaselinesView({ ds, datasets, selectedMethod, onSelectMethod }: { ds: Dataset; datasets: AllDatasets; selectedMethod: string | null; onSelectMethod: (m: string | null) => void }) { const sorted = useMemo(() => [...ds.baselines].sort((a, b) => b.testFitness - a.testFitness), [ds.baselines] ) const ours = ds.baselines.find(b => b.category === 'ours') const rpni = ds.baselines.find(b => b.method === 'rpni') // Cross-dataset data for the selected method const methodCrossData = useMemo(() => { if (!selectedMethod) return null return datasets.map(d => { const b = d.baselines.find(b => b.method === selectedMethod) return { dataset: d.meta.shortName, datasetName: d.meta.name, baseline: b || null } }) }, [selectedMethod, datasets]) const selBaseline = selectedMethod ? ds.baselines.find(b => b.method === selectedMethod) : null return (

Baseline Comparison

All methods evaluated on the same 80/20 train/test split with identical activity extraction. {' '}Click a method for cross-dataset comparison.

{sorted.map((b) => { const isSelected = selectedMethod === b.method return ( onSelectMethod(isSelected ? null : b.method)} className={cn( 'ruled cursor-pointer transition-colors', isSelected ? 'bg-slate-dim' : b.category === 'ours' ? 'bg-rust-dim hover:bg-ochre-dim' : 'hover:bg-ochre-dim' )} > ) })}
Method Type Test Fit. |Q| |delta| Rand. Perm. PM Prec. Train Fit.
{methodLabel(b)} {isSelected && selected}
{b.category === 'ours' ? 'Ours' : b.category === 'process-mining' ? 'PM' : b.category} = 0.99 ? 'text-ink' : b.testFitness >= 0.5 ? 'text-ink-2' : 'text-rust' )}> {fmt(b.testFitness)} {b.states !== null ? fmtInt(b.states) : '–'} {b.transitions !== null ? fmtInt(b.transitions) : '–'} {b.randomAcceptRate !== null ? pct(b.randomAcceptRate) : '–'} {b.permutedAcceptRate !== null ? pct(b.permutedAcceptRate) : '–'} {b.pmPrecision !== null ? fmt(b.pmPrecision) : '–'} {fmt(b.trainFitness)}
{/* Method detail panel */} {selectedMethod && selBaseline && methodCrossData && (

{methodLabel(selBaseline)}

{selBaseline.category}
{selBaseline.note && (

{selBaseline.note}

)} {methodCrossData.map(({ dataset, datasetName, baseline: b }) => ( {b ? ( <> ) : ( )} ))}
Dataset Test Fit. Train Fit. |Q| |delta| Rand. Perm.
{datasetName}{fmt(b.testFitness)} {fmt(b.trainFitness)} {b.states !== null ? fmtInt(b.states) : '–'} {b.transitions !== null ? fmtInt(b.transitions) : '–'} {b.randomAcceptRate !== null ? pct(b.randomAcceptRate) : '–'} {b.permutedAcceptRate !== null ? pct(b.permutedAcceptRate) : '–'}Not available
)} {/* Observations */}
b.category === 'ours')?.states ?? 1) - 1}, most methods achieve near-perfect replay fitness, so the alphabet alone rarely discriminates on fitness.`} />
) } // ---------- Convergence ---------- function ConvergenceView({ ds, datasets }: { ds: Dataset; datasets: AllDatasets }) { // Not every dataset has an incremental-convergence run; fall back to one that does // so the tab always shows a real curve instead of an empty placeholder. const hasOwn = ds.convergence.curve.length > 0 const active = hasOwn ? ds : (datasets.find(d => (d.convergence?.curve?.length ?? 0) > 0) ?? ds) const conv = active.convergence const curve = conv.curve.map((p: { n: number; states: number; transitions: number; fitness: number }) => ({ ...p, transPerState: p.states > 0 ? +(p.transitions / p.states).toFixed(2) : 0, })) return (

Convergence

Structure stabilization as training traces increase.

{!hasOwn && active !== ds && (

{ds.meta.name} has no incremental-convergence run; showing {active.meta.name}. The paper's convergence analysis covers SWE-agent, SWE-smith, Mind2Web, and Who&When.

)}
{curve.length > 0 ? (
) : (
Incremental-convergence data is not available for this dataset. See SWE-agent, SWE-smith, Mind2Web and Who&When in the paper for the convergence analysis.
)} {curve.length > 0 && ( <>
v.toFixed(3)} />
States Trans/State Fitness
v.toFixed(3)} /> [fmt(v ?? 0, 4), 'Fitness']} />
)}
) } // ---------- Graph View ---------- function GraphView({ ds }: { ds: Dataset }) { const { states, transitions } = ds.graph const initialPositions = useMemo(() => computeLayout(states, transitions), [states, transitions]) const [nodePositions, setNodePositions] = useState(initialPositions) useEffect(() => setNodePositions(initialPositions), [initialPositions]) const svgRef = useRef(null) const dragNode = useRef(null) const [zoom, setZoom] = useState(1) const [pan, setPan] = useState({ x: 0, y: 0 }) const containerDrag = useRef(false) const containerLast = useRef({ x: 0, y: 0 }) const toSVG = useCallback((clientX: number, clientY: number) => { const svg = svgRef.current if (!svg) return { x: 0, y: 0 } const pt = svg.createSVGPoint() pt.x = clientX; pt.y = clientY const ctm = svg.getScreenCTM() if (!ctm) return { x: 0, y: 0 } const svgPt = pt.matrixTransform(ctm.inverse()) return { x: svgPt.x, y: svgPt.y } }, []) const handleWheel = useCallback((e: React.WheelEvent) => { e.preventDefault() setZoom(z => Math.max(0.3, Math.min(5, z * (e.deltaY > 0 ? 0.9 : 1.1)))) }, []) const handleContainerDown = useCallback((e: React.PointerEvent) => { containerDrag.current = true containerLast.current = { x: e.clientX, y: e.clientY } }, []) const handleNodePointerDown = useCallback((e: React.PointerEvent, id: string) => { e.stopPropagation() dragNode.current = id ;(e.target as Element).setPointerCapture(e.pointerId) }, []) const handlePointerMove = useCallback((e: React.PointerEvent) => { if (dragNode.current) { const { x, y } = toSVG(e.clientX, e.clientY) setNodePositions(prev => ({ ...prev, [dragNode.current!]: { x, y } })) return } if (!containerDrag.current) return const dx = e.clientX - containerLast.current.x const dy = e.clientY - containerLast.current.y containerLast.current = { x: e.clientX, y: e.clientY } setPan(p => ({ x: p.x + dx, y: p.y + dy })) }, [toSVG]) const handlePointerUp = useCallback(() => { dragNode.current = null; containerDrag.current = false }, []) const handleDoubleClick = useCallback(() => { setZoom(1); setPan({ x: 0, y: 0 }) }, []) // Build edge index for detecting bidirectional pairs const edgeSet = useMemo(() => { const s = new Set() for (const t of transitions) s.add(`${t.source}|${t.target}`) return s }, [transitions]) // Get transition frequencies from downstream data const transFreq = useMemo(() => { const freq = new Map() const bp = ds.downstream.behavioralProfiling if (bp) { let maxCount = 1 for (const tf of bp.transitionFrequency) { const parts = tf.transition.split('\u2192') if (parts.length === 2) { freq.set(`${parts[0]}|${parts[1]}`, tf.count) maxCount = Math.max(maxCount, tf.count) } } // Normalize to 0-1 for (const [k, v] of freq) freq.set(k, v / maxCount) } return freq }, [ds.downstream.behavioralProfiling]) // State visit counts for node sizing const stateVisits = useMemo(() => { const visits = new Map() const bp = ds.downstream.behavioralProfiling if (bp) { let maxCount = 1 for (const sv of bp.stateVisitDistribution) { visits.set(sv.state, sv.count) maxCount = Math.max(maxCount, sv.count) } for (const [k, v] of visits) visits.set(k, v / maxCount) } return visits }, [ds.downstream.behavioralProfiling]) return (

Learned Automaton

{states.length} states, {transitions.length} transitions. Built from {ds.meta.trainSize} training traces. {transFreq.size > 0 && ' Edge thickness shows relative frequency.'}

{/* Edges */} {transitions.map((t, i) => { const from = nodePositions[t.source] const to = nodePositions[t.target] if (!from || !to) return null const freq = transFreq.get(`${t.source}|${t.target}`) ?? 0.1 const sw = 0.5 + freq * 2.5 // stroke width 0.5-3 const opacity = 0.3 + freq * 0.7 const al = 10 if (t.source === t.target) { const r = 14 + (stateVisits.get(t.source) || 0) * 6 const loopR = 35, cx1 = from.x - loopR, cy1 = from.y - loopR * 1.5 const cx2 = from.x + loopR, cy2 = from.y - loopR * 1.5 const sa = Math.atan2(cy1 - from.y, cx1 - from.x) const ea = Math.atan2(cy2 - from.y, cx2 - from.x) const epx = from.x + r * Math.cos(ea), epy = from.y + r * Math.sin(ea) const tdx = epx - cx2, tdy = epy - cy2, td = Math.sqrt(tdx * tdx + tdy * tdy) return ( ) } const dx = to.x - from.x, dy = to.y - from.y const dist = Math.sqrt(dx * dx + dy * dy) const nx = dx / dist, ny = dy / dist const isBidirectional = edgeSet.has(`${t.target}|${t.source}`) const nodeR = 14 + (stateVisits.get(t.target) || 0) * 6 if (isBidirectional) { const curve = 25 const mx = (from.x + to.x) / 2 - ny * curve const my = (from.y + to.y) / 2 + nx * curve const sa = Math.atan2(my - from.y, mx - from.x) const ea = Math.atan2(my - to.y, mx - to.x) const srcR = 14 + (stateVisits.get(t.source) || 0) * 6 const epx = to.x + nodeR * Math.cos(ea), epy = to.y + nodeR * Math.sin(ea) const tdx = epx - mx, tdy = epy - my, td = Math.sqrt(tdx * tdx + tdy * tdy) return ( ) } return ( ) })} {/* Nodes */} {states.map((s) => { const pos = nodePositions[s.id] if (!pos) return null const isInit = s.id === 'init' const visitNorm = stateVisits.get(s.id) || 0 const r = isInit ? 13 : 14 + visitNorm * 6 const label = s.id.length > 20 ? s.id.slice(0, 18) + '..' : s.id return ( handleNodePointerDown(e, s.id)}> 0.5 ? PALETTE.rust : '#b5afa5'} strokeWidth={isInit ? 2 : visitNorm > 0.5 ? 1.5 : 1} /> {isInit && ( )} {label} ) })}
{states.map((s) => ( {s.id} ))}
{/* Transition table */}
{transitions.map((t, i) => ( ))}
From To Type
{t.source} {'\u2192'} {t.target} {t.source === t.target ? 'self-loop' : t.subject || 'transition'}
) } // ---------- Traces View ---------- // Break a long trace ID into readable parts: model tag, persona, trailing index, // and the remaining scenario core. Falls back gracefully for non-tau2 ids. function parseTraceId(id: string, datasetId: string) { let rest = id if (rest.startsWith(datasetId + '-')) rest = rest.slice(datasetId.length + 1) let index: string | null = null const im = rest.match(/-(\d+)$/) if (im) { index = im[1]; rest = rest.slice(0, rest.length - im[0].length) } let model: string | null = null // Only tau2-bench and OSWorld ids embed a model tag (model_scenario...); other // datasets (sweagent repo__task, webarena-idx, uuids) have no model prefix. if (datasetId.startsWith('tau2bench-') || datasetId === 'osworld') { const um = rest.match(/^([^_[]+)_/) if (um) { model = um[1].replace(/-\d{4}-\d{2}-\d{2}$/, ''); rest = rest.slice(um[0].length) } } let persona: string | null = null const pm = rest.match(/\[PERSONA:([^\]]+)\]/i) if (pm) { persona = pm[1]; rest = rest.slice(0, pm.index) + rest.slice((pm.index ?? 0) + pm[0].length) } return { model, persona, index, rest: rest.replace(/^[-_\s]+|[-_\s]+$/g, '') } } function TracesView({ ds, selectedTrace, onSelectTrace }: { ds: Dataset; selectedTrace: string | null; onSelectTrace: (id: string | null) => void }) { const [splitFilter, setSplitFilter] = useState<'all' | 'train' | 'test'>('all') const [successFilter, setSuccessFilter] = useState<'all' | 'success' | 'failure'>('all') const [search, setSearch] = useState('') const [page, setPage] = useState(0) const PAGE_SIZE = 50 // Reset page on filter change useEffect(() => { setPage(0) }, [splitFilter, successFilter, search, ds.id]) const filtered = useMemo(() => { let t = ds.traces if (splitFilter !== 'all') t = t.filter(tr => tr.split === splitFilter) if (successFilter === 'success') t = t.filter(tr => tr.success === true) if (successFilter === 'failure') t = t.filter(tr => tr.success === false) if (search) { const q = search.toLowerCase() t = t.filter(tr => tr.id.toLowerCase().includes(q)) } return t }, [ds.traces, splitFilter, successFilter, search]) const totalPages = Math.ceil(filtered.length / PAGE_SIZE) const pageTraces = filtered.slice(page * PAGE_SIZE, (page + 1) * PAGE_SIZE) const trainCount = ds.traces.filter(t => t.split === 'train').length const testCount = ds.traces.filter(t => t.split === 'test').length const successCount = ds.traces.filter(t => t.success === true).length const failureCount = ds.traces.filter(t => t.success === false).length const unknownCount = ds.traces.filter(t => t.success === null).length // Navigate to the page containing the selected trace useEffect(() => { if (selectedTrace) { const idx = filtered.findIndex(t => t.id === selectedTrace) if (idx >= 0) setPage(Math.floor(idx / PAGE_SIZE)) } }, [selectedTrace, filtered]) return (

Traces

{ds.traces.length} traces: {trainCount} train, {testCount} test. {successCount > 0 && ` ${successCount} success, ${failureCount} failure.`} {unknownCount > 0 && ` ${unknownCount} unlabeled.`} {' '}Click a trace for details.

{/* Filters */}
{(['all', 'train', 'test'] as const).map(v => ( ))}
{successCount + failureCount > 0 && (
{(['all', 'success', 'failure'] as const).map(v => ( ))}
)} setSearch(e.target.value)} placeholder="Search trace ID..." className="px-3 py-1 text-[12px] bg-transparent border border-rule mono text-ink placeholder:text-ink-4 outline-none focus:border-ink-3 w-[280px]" /> {filtered.length} results
{/* Table */}
{pageTraces.map((t, i) => { const isSelected = selectedTrace === t.id return ( onSelectTrace(isSelected ? null : t.id)} className={cn( 'ruled cursor-pointer transition-colors', isSelected ? 'bg-slate-dim' : 'hover:bg-ochre-dim' )} > {isSelected && ( )} ) })}
# Trace ID Split Label Full Fit. Succ. Fit.
{page * PAGE_SIZE + i + 1} {(() => { const p = parseTraceId(t.id, ds.id) return (
{p.model && {p.model}} {p.rest || t.id} {p.persona && {p.persona}} {p.index != null && #{p.index}} {isSelected && ▾}
) })()}
{t.split} {t.success === true && pass} {t.success === false && fail} {t.success === null && --} {t.fullFitness !== null ? fmt(t.fullFitness) : ''} {t.successFitness !== null ? fmt(t.successFitness) : ''}
onSelectTrace(null)} />
{/* Pagination */} {totalPages > 1 && (
{page + 1} / {totalPages}
)}
) } // ---------- Trace Detail Panel ---------- function TraceDetailPanel({ trace, onClose }: { trace: Trace; onClose: () => void }) { const [showAll, setShowAll] = useState(false) const STEP_LIMIT = 40 const failSteps = trace.steps.filter(s => !s.consumed) const uniqueStates = [...new Set(trace.stateSequence.filter(s => !s.startsWith('FAIL:')))] const maxVisits = Math.max(1, ...Object.values(trace.stateVisits)) // Steps to display (truncated or all) const displaySteps = showAll ? trace.steps : trace.steps.slice(0, STEP_LIMIT) const hasMore = trace.steps.length > STEP_LIMIT return (
{/* Header */}

Trace Detail

{trace.split} {trace.success === true && pass} {trace.success === false && fail}

{trace.id}

{/* Key metrics */}
= 0.99} /> 0} warn /> = 0 ? `Step ${trace.firstFailIdx}` : 'None'} />
{trace.fullFitness !== null && (
)} {/* State visits heatmap */} {uniqueStates.length > 0 && (
State Visits
{uniqueStates.map(state => { const count = trace.stateVisits[state] || 0 const intensity = count / maxVisits return (
0.6 ? `rgba(196, 85, 58, ${0.08 + intensity * 0.18})` : intensity > 0.3 ? `rgba(196, 154, 58, ${0.06 + intensity * 0.12})` : `rgba(61, 90, 128, ${0.04 + intensity * 0.08})`, }} > {state.length > 20 ? state.slice(0, 18) + '..' : state}
{count}
) })}
)} {/* State sequence flow */} {trace.stateSequence.length > 0 && (
State Sequence
{trace.stateSequence.slice(0, 60).map((state, i) => { const isFail = state.startsWith('FAIL:') return ( {isFail ? state.replace('FAIL:', '') : state} {i < trace.stateSequence.length - 1 && i < 59 && ( {isFail ? '!' : '\u2192'} )} ) })} {trace.stateSequence.length > 60 && ( +{trace.stateSequence.length - 60} more )}
)} {/* Step-by-step FSM traversal */} {trace.steps.length > 0 && (
FSM Traversal ({trace.steps.length} steps)
{hasMore && ( )}
{displaySteps.map((step) => { const isFail = !step.consumed const isFirstFail = step.idx === trace.firstFailIdx return ( ) })}
Step Role Activity From To Status
{step.idx} {step.role} {step.activity} {step.fromState} {isFail ? ! : {'\u2192'} } {step.toState || 'REJECTED'} {isFirstFail ? ( 1st fail ) : isFail ? ( fail ) : ( ok )}
{!showAll && hasMore && (
)}
)} {trace.steps.length === 0 && (
No replay data available for this trace
)}
) } function MiniStat({ label, value, accent, warn }: { label: string; value: string; accent?: boolean; warn?: boolean }) { return (
{label}
{value}
) } // ---------- Downstream View ---------- function DownstreamView({ ds, datasets }: { ds: Dataset; datasets: AllDatasets }) { const down = ds.downstream const bp = down.behavioralProfiling const sa = down.structuralAnalysis const fp = down.fitnessProfile const df = down.differentialFSM return (

Downstream Analysis

Behavioral profiling, structural analysis, and anomaly detection from the learned FSM.

{/* Fitness profile */} {fp && (
{fp.histogram.length > 0 && (
({ range: `${h.min.toFixed(2)}`, count: h.count }))}>
Fitness distribution (bin lower bound)
)} {fp.anomalies.length > 0 && (
Anomalous Traces
{fp.anomalies.map(id => ( {id.length > 40 ? id.slice(0, 38) + '..' : id} ))}
)}
)} {/* Behavioral profiling */} {bp && (
{/* State visits */}
State Visit Distribution
{bp.stateVisitDistribution.map(sv => ( ))}
State Count Per Trace
{sv.state} {fmtInt(sv.count)} {fmt(sv.fraction, 2)}
{/* Top transitions */}
Top Transitions
{bp.transitionFrequency.slice(0, 10).map(tf => ( ))}
Transition Count Per Trace
{tf.transition} {fmtInt(tf.count)} {fmt(tf.fraction, 2)}
{bp.deadTransitions.length > 0 && (
Dead Transitions (in FSM, never observed in data)
{bp.deadTransitions.map(dt => ( {dt} ))}
)}
)} {/* Structural analysis */} {sa && (
{/* Hubs */}
Hub States
{sa.hubs.map(h => ( ))}
State In-Degree Out-Degree
{h.id} {h.inDegree} {h.outDegree}
{/* Sources, sinks, SCCs */}
{sa.sources.length > 0 && (
Source States (no incoming)
{sa.sources.map(s => ( {s} ))}
)} {sa.sinks.length > 0 && (
Sink States (no outgoing)
{sa.sinks.map(s => ( {s} ))}
)} {sa.selfLoops.length > 0 && (
Self-Loops
{sa.selfLoops.map(s => ( {s} ))}
)} {sa.nonTrivialSCCs.length > 0 && (
Strongly Connected Components
{sa.nonTrivialSCCs.map((scc, i) => (
SCC {i + 1}: {scc.map(s => ( {s} ))}
))}
)}
)} {/* Differential FSM */} {df && (
Success-Only ({df.successOnlyTransitions.length})
{df.successOnlyTransitions.length > 0 ? (
{df.successOnlyTransitions.map(t => ( {t} ))}
) : ( None )}
Failure-Only ({df.failureOnlyTransitions.length})
{df.failureOnlyTransitions.length > 0 ? (
{df.failureOnlyTransitions.map(t => ( {t} ))}
) : ( None )}
Shared ({df.sharedTransitions.length})
{df.sharedTransitions.length > 0 ? (
{df.sharedTransitions.map(t => ( {t} ))}
) : ( None )}
)} {/* Failure prediction lives in the dedicated Failure tab (deduplicated). */} {/* Cross-dataset downstream comparison */}
{datasets.map(d => { const fp2 = d.downstream.fitnessProfile const bp2 = d.downstream.behavioralProfiling const sa2 = d.downstream.structuralAnalysis const isActive = d.id === ds.id return ( ) })}
Dataset Fitness u Fitness sigma Anomalies Trans. Cov. Density Dead Trans.
{d.meta.name} {isActive && current} {fp2 ? fmt(fp2.mean) : '–'} {fp2 ? fmt(fp2.std, 4) : '–'} {fp2 ? fp2.anomalyCount : '–'} {bp2 ? pct(bp2.transitionCoverage) : '–'} {sa2 ? fmt(sa2.density) : '–'} {bp2 ? bp2.deadTransitions.length : '–'}
) } // ---------- Failure Prediction View ---------- function FailureView({ ds, datasets }: { ds: Dataset; datasets: AllDatasets }) { const fp = ds.downstream.failurePrediction as { fitnessOnly?: { auroc?: number; f1?: number; accuracy?: number; bestThreshold?: number } featureBased?: { holdoutAUROC?: number; holdoutAUROC_LR?: number; holdoutAUROC_GBT?: number holdoutAUROC_RF?: number; holdoutAUROC_Ensemble?: number bestModel?: string; numFeatures?: number; numSelectedFeatures?: number kfoldAUROC?: { mean?: number; std?: number; k?: number; repeats?: number } f1?: number; accuracy?: number; precision?: number; recall?: number topFeatures?: Array<{ name: string; weight: number; absWeight: number }> } } | null const np = ds.downstream.neuralProbe // Cross-dataset AUROC comparison (fitness-only vs ensemble) const crossData = datasets .map(d => { const dfp = d.downstream.failurePrediction as typeof fp if (!dfp?.fitnessOnly?.auroc || !dfp?.featureBased?.holdoutAUROC) return null return { name: d.meta.shortName, fitnessOnly: dfp.fitnessOnly.auroc, ensemble: dfp.featureBased.holdoutAUROC_Ensemble ?? dfp.featureBased.holdoutAUROC, cvMean: dfp.featureBased.kfoldAUROC?.mean ?? 0, cvStd: dfp.featureBased.kfoldAUROC?.std ?? 0, } }) .filter((x): x is NonNullable => x !== null) // Neural Seq vs FSM cross-dataset const neuralCross = datasets .map(d => { const n = d.downstream.neuralProbe if (!n) return null return { name: d.meta.shortName, seq_mlp: n.seq_mlp.cv, fsm_mlp: n.fsm_mlp.cv, seq_gru: n.seq_gru.cv, fsm_gru: n.fsm_gru.cv, seq_transf: n.seq_transf.cv, fsm_transf: n.fsm_transf.cv, } }) .filter((x): x is NonNullable => x !== null) return (

Failure Prediction

Per-state FSM features predict trace-level success/failure. Compared against a fitness-only baseline and against neural models (MLP / GRU / Transformer) trained on sequence features vs FSM-derived features.

{/* This-dataset headline */} {fp?.featureBased && (
)} {/* Cross-dataset AUROC */} {crossData.length > 0 && (
(typeof v === 'number' ? v.toFixed(3) : String(v))} />
Holdout AUROC, 80/20 split. FSM ensemble = LR + GBT + RF on per-state features.
)} {/* Neural probe comparison */} {neuralCross.length > 0 && (
{neuralCross.map(r => { const winCell = (seq: number, fsm: number, isFsm: boolean) => { const fsmWins = fsm > seq const style = isFsm && fsmWins ? { color: PALETTE.rust, fontWeight: 600 } : {} return } return ( {winCell(r.seq_mlp, r.fsm_mlp, false)} {winCell(r.seq_mlp, r.fsm_mlp, true)} {winCell(r.seq_gru, r.fsm_gru, false)} {winCell(r.seq_gru, r.fsm_gru, true)} {winCell(r.seq_transf, r.fsm_transf, false)} {winCell(r.seq_transf, r.fsm_transf, true)} ) })}
Dataset MLP Seq MLP FSM GRU Seq GRU FSM Transf Seq Transf FSM
{fmt(isFsm ? fsm : seq)}
{r.name}
10×5-fold CV AUROC. Bold rust = FSM features beat sequence features for that model.
)} {/* Neural probe for current dataset only */} {np && (
{(['mlp', 'gru', 'transf'] as const).map(model => { const seq = np[`seq_${model}` as const] const fsm = np[`fsm_${model}` as const] const delta = fsm.cv - seq.cv return (
{model}
Seq {fmt(seq.cv)} ±{fmt(seq.std, 3)}
FSM {fmt(fsm.cv)} ±{fmt(fsm.std, 3)}
Δ = 0 ? PALETTE.rust : PALETTE.iron }}> {delta >= 0 ? '+' : ''}{fmt(delta * 100, 1)}pp
) })}
)} {/* Top features */} {fp?.featureBased?.topFeatures && fp.featureBased.topFeatures.length > 0 && (
{fp.featureBased.topFeatures.slice(0, 12).map(f => (
{f.name} = 0 ? PALETTE.rust : PALETTE.slate }}> {f.weight >= 0 ? '+' : ''}{fmt(f.weight, 3)}
))}
LR coefficients on per-state features. Rust = failure-indicative, slate = success-indicative.
)} {(!fp && !np) && (
No failure prediction data for {ds.meta.name}.
This dataset lacks success/failure labels or hasn't been run through the downstream pipeline.
)}
) } // ---------- Workflow Memory View (FSM vs AWM) ---------- function MemoryView({ ds, datasets }: { ds: Dataset; datasets: AllDatasets }) { const wm = ds.downstream.workflowMemory const cross = datasets .map(d => { const w = d.downstream.workflowMemory if (!w) return null return { name: d.meta.shortName, noMemory: w.noMemory.accuracy, awm: w.awm.accuracy, fsm: w.fsm.accuracy, gain: w.fsmGain, n: w.n, } }) .filter((x): x is NonNullable => x !== null) const fsmWins = cross.filter(r => r.fsm > r.awm).length const meanGain = cross.length > 0 ? cross.reduce((s, r) => s + r.gain, 0) / cross.length : 0 return (

Workflow Memory (FSM vs AWM)

LLM next-action prediction (gpt-4.1-mini, top-1 accuracy) with three context conditions: no memory, Agent Workflow Memory (linear workflows), and FSM-state context.

{/* Summary */} {cross.length > 0 && (
s + r.n, 0))} />
)} {/* Cross-dataset bar chart */} {cross.length > 0 && (
`${Math.round(v * 100)}%`} /> (typeof v === 'number' ? `${(v * 100).toFixed(1)}%` : String(v))} />
LLM-judged top-1 next-action accuracy, full validation split per dataset.
)} {/* Table with deltas */} {cross.length > 0 && (
{cross.map(r => ( ))}
Dataset N No Mem AWM FSM FSM − AWM
{r.name} {fmtInt(r.n)} {pct(r.noMemory)} {pct(r.awm)} {pct(r.fsm)} = 0 ? PALETTE.rust : PALETTE.iron }}> {r.gain >= 0 ? '+' : ''}{fmt(r.gain * 100, 1)}pp
Highest gains: WebArena +15.7pp, SWE-smith +25.3pp. 6/8 statsig at p<1e-8 (per paper).
)} {!wm && (
No workflow memory data for {ds.meta.name}.
Run poc-fsm-as-memory.ts with LLM judging enabled.
)}
) } // ---------- Cross-Model FSM View ---------- function CrossModelView({ ds, datasets }: { ds: Dataset; datasets: AllDatasets }) { const cm = ds.downstream.crossModel const available = datasets.filter(d => d.downstream.crossModel) // Auto-fallback to first available if current dataset has no cross-model data const active = cm ? ds : available[0] const acm = active?.downstream.crossModel if (!acm) { return (

Cross-Model FSM Transfer

No cross-model data available. tau2-bench airline/retail/telecom only.

) } const heatColor = (v: number, min: number, max: number) => { if (max === min) return PALETTE.iron const t = (v - min) / (max - min) const r = Math.round(196 - t * 60) const g = Math.round(85 - t * 20) const b = Math.round(58 - t * 15) return `rgb(${r},${g},${b})` } // Transfer matrices may arrive as a raw number[][] or as a {labels, values} object // (the experiment JSON uses the latter); normalize to a 2-D number array. const toMatrix = (m: unknown): number[][] => Array.isArray(m) ? (m as number[][]) : (((m as { values?: number[][] })?.values) ?? []) const fitMatrix = toMatrix(acm.matrices.fitness) const aurocMatrix = toMatrix(acm.matrices.auroc) const fitnessFlat = fitMatrix.flat() const aurocFlat = aurocMatrix.flat() const fitMin = Math.min(...fitnessFlat), fitMax = Math.max(...fitnessFlat) const aurocMin = Math.min(...aurocFlat), aurocMax = Math.max(...aurocFlat) // Paper headline = the combined figure across ALL tau2-bench suites (36 off-diagonal // pairs), not any single suite. Single-suite telecom (0.66/0.80) understates it. const suiteSummaries = available.map(d => d.downstream.crossModel!.summary) const combinedSelf = suiteSummaries.reduce((s, x) => s + x.diagonalAUROC.mean, 0) / suiteSummaries.length const combinedCross = suiteSummaries.reduce((s, x) => s + x.offDiagonalAUROC.mean, 0) / suiteSummaries.length return (

Cross-Model FSM Transfer

tau2-bench traces collected across 4 LLMs. Train FSM on one model's traces, evaluate on another. Diagonal = self-test, off-diagonal = cross-model transfer.

{available.length > 1 && (
{available.map(d => ( {d.meta.name} ))} (showing: {active.meta.name})
)}
{/* Combined headline across all suites (paper figure) */} {available.length > 1 && (
Mean over the {available.length} tau2-bench suites (36 off-diagonal model pairs). Per-suite breakdown below.
)} {/* Per-suite summary */}
{/* Models */}
{acm.models.map(m => (
{m.name}
Traces
{fmtInt(m.traces)}
Success
{pct(m.successRate)}
))}
{/* Transfer matrix: Fitness */}
{acm.models.map(m => ( ))} {fitMatrix.map((row, i) => ( {row.map((v, j) => ( ))} ))}
FSM Source ↓ / Target →{m.shortName}
{acm.models[i].shortName} {fmt(v)}
Bold diagonal = self (train and test on same model). Darker = lower fitness.
{/* Transfer matrix: AUROC */}
{acm.models.map(m => ( ))} {aurocMatrix.map((row, i) => ( {row.map((v, j) => ( ))} ))}
FSM Source ↓ / Target →{m.shortName}
{acm.models[i].shortName} {fmt(v)}
Per-state failure features transfer at {fmt(acm.summary.offDiagonalAUROC.mean)} mean cross-AUROC vs {fmt(acm.summary.diagonalAUROC.mean)} self.
) } // ---------- Runtime Monitor View ---------- function MonitorView({ ds, datasets }: { ds: Dataset; datasets: AllDatasets }) { const available = datasets.filter(d => d.downstream.monitor) const active = ds.downstream.monitor ? ds : available[0] const m = active?.downstream.monitor if (!m) { return (

Runtime Monitor

No monitor data available. Evaluated on SWE-agent, SWE-smith, tau2-bench retail, and tau2-bench airline.

) } // Build chart data: one row per trace-completion point, one column per config const points = m.configs[0]?.monitoringCurve.map(c => c.point) || [] const curveData = points.map(pt => { const row: Record = { point: pt } m.configs.forEach(c => { const cp = c.monitoringCurve.find(x => x.point === pt) row[c.config] = cp?.f1 ?? 0 }) return row }) const finalF1 = (cfg: typeof m.configs[number]) => cfg.monitoringCurve[cfg.monitoringCurve.length - 1]?.f1 ?? 0 const bestCfg = m.configs.reduce((a, b) => (finalF1(b) > finalF1(a) ? b : a)) const earliestStrong = bestCfg.monitoringCurve.find(c => c.f1 >= 0.9) return (

Runtime Monitor

Online failure detection from FSM conformance during trace execution. Decay parameter γ controls how quickly past conformance is forgotten; W is a sliding window.

{available.length > 1 && (
{available.map(d => ( {d.meta.name} ))} (showing: {active.meta.name})
)}
{/* Summary */}
{/* Curves */}
`${Math.round(v * 100)}%`} /> (typeof v === 'number' ? v.toFixed(3) : String(v))} labelFormatter={(v) => `Completion: ${Math.round(Number(v) * 100)}%`} /> {m.configs.map((c, i) => ( ))}
X-axis: fraction of trace observed (online setting). Y-axis: F1 at the threshold tuned for that point.
{/* Baselines */} {Object.keys(m.baselines).length > 0 && (
{Object.entries(m.baselines).map(([name, b]) => (
{name}
= 0.9 ? PALETTE.ochre : PALETTE.iron }}> {fmt(b.auroc)}
))}
selfLoopRate AUROC = 1.0 on SWE-agent is an artifact of failure traces always containing self-loops — not a meaningful signal in isolation.
)}
) } // ---------- Counterfactual Paths View ---------- function CounterfactualView({ ds, datasets }: { ds: Dataset; datasets: AllDatasets }) { const available = datasets.filter(d => d.downstream.counterfactual) const active = ds.downstream.counterfactual ? ds : available[0] const c = active?.downstream.counterfactual if (!c) { return (

Counterfactual Paths

No counterfactual data. Available for SWE-agent, SWE-smith, tau2-bench retail, and tau2-bench airline.

) } const pathRatio = c.pathDiversity.uniqueSuccPaths > 0 ? c.pathDiversity.uniqueFailPaths / c.pathDiversity.uniqueSuccPaths : 0 const overlapPct = (c.pathDiversity.uniqueSuccPaths + c.pathDiversity.uniqueFailPaths - c.pathDiversity.overlap) > 0 ? c.pathDiversity.overlap / (c.pathDiversity.uniqueSuccPaths + c.pathDiversity.uniqueFailPaths - c.pathDiversity.overlap) : 0 return (

Counterfactual Path Analysis

Where success and failure trajectories diverge through the FSM. Decision points are states whose outgoing transition distributions differ between success and failure traces (Jensen–Shannon divergence).

{available.length > 1 && (
{available.map(d => ( {d.meta.name} ))} (showing: {active.meta.name})
)}
{/* Summary */}
{/* Decision points */} {c.dpDivergence.length > 0 && (
{c.dpDivergence .slice() .sort((a, b) => b.jsd - a.jsd) .slice(0, 15) .map((dp) => { const maxJsd = Math.max(...c.dpDivergence.map(x => x.jsd), 1e-9) return ( ) })}
State JSD Bar
{dp.state} {fmt(dp.jsd, 4)}
Higher JSD = stronger divergence between success and failure transition distributions at that state.
)} {/* Branching */} {c.decisionPoints.length > 0 && (
{c.decisionPoints .slice() .sort((a, b) => b.branching - a.branching) .slice(0, 8) .map((dp) => (
{dp.state} {dp.branching} branches
{dp.targets.map(t => ( {t} ))}
))}
)}
) } // ---------- About / Paper View ---------- function AboutView({ datasets }: { datasets: AllDatasets }) { const totalTraces = datasets.reduce((s, d) => s + d.meta.traces, 0) const compressionExtremes = datasets .map(d => { const ours = d.baselines.find(b => b.category === 'ours') const rpni = d.baselines.find(b => b.method === 'rpni') if (!ours?.states || !rpni?.states) return null return { name: d.meta.shortName, ratio: rpni.states / ours.states } }) .filter((x): x is { name: string; ratio: number } => x !== null) const compMin = compressionExtremes.length ? Math.min(...compressionExtremes.map(c => c.ratio)) : 0 const compMax = compressionExtremes.length ? Math.max(...compressionExtremes.map(c => c.ratio)) : 0 return (

Automata from Agent Traces

Failure and next-step prediction for LLM-based agents via minimal finite-state machines extracted from execution traces.

LLM-based agents execute multi-step tasks, but their behavioural structure remains opaque: long unstructured traces resist the safety auditing and runtime monitoring that deployment requires. We extract finite-state machines (FSMs) that are provably minimal for the observed prefix language via prefix-tree construction and structural state merging. Across {datasets.length} public datasets, the FSMs are compact (7–43 states), achieve ≥0.993 replay fitness on held-out data with zero structural variance across splits, and build in milliseconds. The same FSM unifies workflow memory, next-step prediction, failure prediction, and runtime monitoring.

{datasets.map(d => { const ours = d.baselines.find(b => b.category === 'ours') return ( ) })}
Dataset Venue Traces |A| |Q|
{d.meta.name} {d.meta.venue} {fmtInt(d.meta.traces)} {ours?.states ? ours.states - 1 : '—'} {ours?.states ?? '—'}
) } // ---------- Shared Components ---------- function Figure({ label, value, note, accent }: { label: string; value: string; note?: string; accent?: boolean }) { return (
{label}
{value}
{note &&
{note}
}
) } function SectionRule({ title }: { title: string }) { return (

{title}

) } function Observation({ num, title, body }: { num: string; title: string; body: string }) { return (
{num}. {title}

{body}

) } // ---------- Layout (force-directed simulation) ---------- function computeLayout( states: Array<{ id: string }>, transitions: Array<{ source: string; target: string }> ): Record { const n = states.length if (n === 0) return {} const W = 600, H = 400 const cx = W / 2, cy = H / 2 // Init: circular layout as starting point const pos: Record = {} const vel: Record = {} const initIdx = states.findIndex(s => s.id === 'init') const r0 = Math.min(160, n * 22) states.forEach((s, i) => { const idx = initIdx >= 0 ? (i - initIdx + n) % n : i const angle = (idx / n) * 2 * Math.PI - Math.PI / 2 pos[s.id] = { x: cx + r0 * Math.cos(angle), y: cy + r0 * Math.sin(angle) } vel[s.id] = { x: 0, y: 0 } }) // Build adjacency for attraction const adj = new Map>() for (const s of states) adj.set(s.id, new Set()) for (const t of transitions) { if (t.source !== t.target) { adj.get(t.source)?.add(t.target) adj.get(t.target)?.add(t.source) } } // In-degree for sizing const inDeg: Record = {} for (const t of transitions) inDeg[t.target] = (inDeg[t.target] || 0) + 1 // Simulate const REPULSION = 8000 const ATTRACTION = 0.008 const DAMPING = 0.85 const ITERS = 200 for (let iter = 0; iter < ITERS; iter++) { const temp = 1 - iter / ITERS // cooling for (const a of states) { let fx = 0, fy = 0 // Repulsion from all other nodes for (const b of states) { if (a.id === b.id) continue const dx = pos[a.id].x - pos[b.id].x const dy = pos[a.id].y - pos[b.id].y const d2 = dx * dx + dy * dy + 1 const f = REPULSION / d2 fx += f * dx / Math.sqrt(d2) fy += f * dy / Math.sqrt(d2) } // Attraction along edges const neighbors = adj.get(a.id) if (neighbors) { for (const bId of neighbors) { const dx = pos[bId].x - pos[a.id].x const dy = pos[bId].y - pos[a.id].y const d = Math.sqrt(dx * dx + dy * dy) const idealDist = 100 const f = ATTRACTION * (d - idealDist) fx += f * dx / (d + 0.1) fy += f * dy / (d + 0.1) } } // Center gravity fx += (cx - pos[a.id].x) * 0.001 fy += (cy - pos[a.id].y) * 0.001 vel[a.id].x = (vel[a.id].x + fx) * DAMPING * temp vel[a.id].y = (vel[a.id].y + fy) * DAMPING * temp } // Apply velocities + clamp for (const s of states) { pos[s.id].x = Math.max(60, Math.min(W - 60, pos[s.id].x + vel[s.id].x)) pos[s.id].y = Math.max(60, Math.min(H - 60, pos[s.id].y + vel[s.id].y)) } } return pos }