hermes-agent/apps/desktop/scripts/perf/scenarios/stream.mjs
brooklyn! dd418284db
bench(desktop): trustworthy --spawn stream numbers + real baseline (#67694)
Chased the "stream frame p95 = 60ms with ZERO longtasks" mystery to its actual
cause: the default stream chunk had no paragraph breaks, so it grew into one
giant ~22KB block that re-rendered fully every flush — defeating the block
memoization real streaming relies on. Plain text = 21ms; realistic chunk with
`\n\n` breaks (blocks settle, only the tail re-renders) = 23ms. Fixed the
default chunk to model real LLM output; a break-less `--chunk` remains available
as a single-block worst-case stress.

Also hardened the isolated instance so measurements reflect real cost:
- Wait for the gateway socket to actually connect before measuring (a booting/
  absent backend's reconnect backoff churns the main thread). Exposed via a new
  __PERF_DRIVE__.connected() probe reading $gateway.connectionState.
- Focus emulation + anti-throttle/occlusion flags so a backgrounded perf window
  isn't frame-throttled (no OS focus stealing).
- Generation-guarded the rAF frame recorder so repeated runs don't leave
  overlapping recorders polluting frame intervals.

Baseline re-captured as the median of 5 --spawn runs (darwin-arm64); all three
CI scenarios now green and stable. Absolute values are dev-build (noted in
_meta) — regression guards, not shipped numbers.
2026-07-19 21:30:40 +00:00

191 lines
7 KiB
JavaScript

// Streaming render cost. Subsumes measure-synthetic-stream, profile-synth-stream,
// profile-long-stream (synthetic) and measure-real-stream / profile-real-stream
// (via --real). CPU profiling is provided by the runner's --cpuprofile flag.
//
// Metrics (lower is better): longtask count + max, frame p95/p99, slow-frame
// count, inter-mutation p95. These are what "is streaming smooth?" reduces to.
import { SELECTORS, sleep, typeIntoComposer } from '../lib/cdp.mjs'
import { frameHistogram, percentile } from '../lib/stats.mjs'
const RECORDERS = `
(() => {
// Generation guard: a prior run's rAF loop re-reads window.__FT__ each frame,
// so simply reassigning it would leave the old loop running and pushing into
// the new array (overlapping recorders inflate frame intervals on run 2+).
// Bumping the generation makes every stale loop exit on its next tick.
window.__FT_GEN__ = (window.__FT_GEN__ || 0) + 1
const ftGen = window.__FT_GEN__
window.__FT__ = { times: [], stop: false }
let last = performance.now()
const tick = () => {
if (window.__FT_GEN__ !== ftGen || window.__FT__.stop) return
const now = performance.now()
window.__FT__.times.push(now - last)
last = now
requestAnimationFrame(tick)
}
requestAnimationFrame(tick)
window.__LT__ = { entries: [], stop: false }
try {
const po = new PerformanceObserver((list) => {
if (window.__LT__.stop) return
for (const e of list.getEntries()) window.__LT__.entries.push({ duration: e.duration, startTime: e.startTime })
})
po.observe({ entryTypes: ['longtask'] })
window.__LT__.po = po
} catch {}
window.__MO__ = { mutations: [], stop: false, current: null }
window.__MO__.arm = () => {
const all = document.querySelectorAll(${JSON.stringify(SELECTORS.assistantMessage)})
const last = all[all.length - 1]
if (!last || last === window.__MO__.current) return
window.__MO__.current = last
window.__MO__.obs && window.__MO__.obs.disconnect()
const obs = new MutationObserver(() => {
if (window.__MO__.stop) return
window.__MO__.mutations.push({ t: performance.now(), len: last.textContent.length })
})
obs.observe(last, { childList: true, subtree: true, characterData: true })
window.__MO__.obs = obs
}
return 'armed'
})()
`
const COLLECT = `
(() => {
window.__FT__.stop = true
window.__LT__.stop = true
window.__MO__.stop = true
try { window.__LT__.po && window.__LT__.po.disconnect() } catch {}
try { window.__MO__.obs && window.__MO__.obs.disconnect() } catch {}
return JSON.stringify({
frames: window.__FT__.times,
longtasks: window.__LT__.entries,
mutations: window.__MO__.mutations
})
})()
`
function analyze(data, warmupMs) {
// Drop warm-up frames (recorder installs before the stream starts).
const frames = []
let acc = 0
for (const f of data.frames) {
acc += f
if (acc >= warmupMs) {
frames.push(f)
}
}
const interMut = []
for (let i = 1; i < data.mutations.length; i++) {
interMut.push(data.mutations[i].t - data.mutations[i - 1].t)
}
const ltDurations = data.longtasks.map(e => e.duration)
const windowS = frames.reduce((a, b) => a + b, 0) / 1000
return {
metrics: {
longtasks_n: data.longtasks.length,
longtask_max_ms: Math.round((ltDurations.length ? Math.max(...ltDurations) : 0) * 10) / 10,
frame_p95_ms: Math.round(percentile(frames, 0.95) * 10) / 10,
frame_p99_ms: Math.round(percentile(frames, 0.99) * 10) / 10,
slow_frames_33: frames.filter(f => f > 33).length,
intermut_p95_ms: Math.round(percentile(interMut, 0.95) * 10) / 10
},
detail: {
windowS: Math.round(windowS * 10) / 10,
avgFps: windowS ? Math.round((frames.length / windowS) * 10) / 10 : 0,
frameHistogram: frameHistogram(frames),
mutations: data.mutations.length,
finalLen: data.mutations.at(-1)?.len ?? 0
}
}
}
export default {
name: 'stream',
tier: 'ci',
description: 'Assistant-message streaming: longtasks, frame pacing, mutation cadence.',
async run(cdp, opts = {}) {
const tokens = Number(opts.tokens ?? 400)
const intervalMs = Number(opts.intervalMs ?? 16)
const flushMinMs = Number(opts.flushMinMs ?? 33)
// Realistic default: a short markdown paragraph ending in a blank line, so
// blocks SETTLE as they stream — exactly how real LLM output behaves, and
// what block-memoization is designed for (only the growing tail re-renders).
// A chunk with NO paragraph break (e.g. `--chunk 'word '`) instead grows one
// ever-larger block that re-renders fully every flush — a useful worst-case
// stress, but not the typical number. No raw autolink (avoids DNS/link-embed
// noise unrelated to render cost).
const chunk = opts.chunk ?? 'A streamed sentence with **bold**, `code`, and ordinary prose like a normal reply.\n\n'
const real = Boolean(opts.real)
await cdp.send('Runtime.enable')
await cdp.eval(RECORDERS)
if (real) {
// Backend path: fire a real prompt and wait for the stream to appear.
const baseCount = await cdp.eval(`document.querySelectorAll(${JSON.stringify(SELECTORS.assistantMessage)}).length`)
await typeIntoComposer(cdp, opts.prompt ?? 'count from 1 to 80, one number per line', { cps: 40 })
await cdp.eval(`(() => {
const el = document.querySelector(${JSON.stringify(SELECTORS.composer)})
el && el.dispatchEvent(new KeyboardEvent('keydown', { key: 'Enter', code: 'Enter', bubbles: true, cancelable: true }))
})()`)
const deadline = Date.now() + Number(opts.timeoutMs ?? 60000)
let started = false
while (Date.now() < deadline) {
await sleep(50)
const n = await cdp.eval(`document.querySelectorAll(${JSON.stringify(SELECTORS.assistantMessage)}).length`)
if (n > baseCount) {
started = true
break
}
}
if (!started) {
throw new Error('real stream never started (no LLM credit / backend?)')
}
await cdp.eval('window.__MO__.arm()')
// Let it run to completion or timeout.
const runDeadline = Date.now() + Number(opts.runMs ?? 30000)
while (Date.now() < runDeadline) {
await sleep(250)
const busy = await cdp.eval(`!!document.querySelector('[data-status="running"], [data-busy="true"]')`)
if (!busy) {
break
}
}
} else {
// Synthetic path: drive $messages directly. No LLM, no credits.
await cdp.eval(
`window.__PERF_DRIVE__.stream({ chunk: ${JSON.stringify(chunk)}, intervalMs: ${intervalMs}, totalTokens: ${tokens}, flushMinMs: ${flushMinMs} })`
)
await sleep(200)
await cdp.eval('window.__MO__.arm()')
await sleep(tokens * intervalMs + 1500)
}
const data = JSON.parse(await cdp.eval(COLLECT))
if (!real) {
await cdp.eval('window.__PERF_DRIVE__.reset()')
}
return analyze(data, real ? 0 : 500)
}
}