// Decode throughput over a rolling second. The first token establishes the // baseline, so prompt processing and first-token latency are not counted. export class TokenRateWindow { constructor(windowMs = 1000, minimumMs = 250) { this.windowMs = windowMs; this.minimumMs = minimumMs; this.total = 0; this.firstAt = null; this.points = []; } add(count, at) { if (!(count > 0)) return; this.firstAt ??= at; this.total += count; this.points.push({ at, total: this.total }); this.prune(at); } prune(at) { const cutoff = at - this.windowMs; while (this.points.length > 1 && this.points[1].at <= cutoff) this.points.shift(); } rate(at) { if (this.firstAt === null || at - this.firstAt < this.minimumMs) return null; this.prune(at); const elapsed = Math.min(this.windowMs, at - this.firstAt); return (this.total - this.points[0].total) * 1000 / elapsed; } }