hermes-agent/apps/desktop/scripts/perf/scenarios/stream.mjs
brooklyn! d1c455acf7
bench(desktop): systematized perf harness; sunset 12 one-off scripts (#67466)
Replaces the dozen ad-hoc measure-*/profile-* scripts (each reinventing the
CDP client — 4 different copies — plus its own arg parsing, stats, output
path, and none with a baseline) with one framework under scripts/perf/:

- lib/cdp.mjs      one CDP client + target discovery + typing + CPU-profile wrapper + DOM selectors
- lib/stats.mjs    percentiles, histograms, CPU-profile self-time ranking
- lib/baseline.mjs load/compare/update baseline + regression gate (new capability)
- lib/launch.mjs   attach, OR spawn a fully ISOLATED instance
- scenarios/*      one module per measurement, registered in scenarios/index.mjs
- run.mjs / serve.mjs, baseline.json, README.md

Isolation solves the long-standing measurement blocker: a running `hgui` held
the Electron single-instance lock, so a second instance quit. `--spawn` /
`perf:serve` launch with their own --user-data-dir (separate lock scope), their
own HERMES_HOME (separate backend/sessions, config seeded from ~/.hermes so it
reaches a chat view without onboarding), and their own --remote-debugging-port.
Synthetic scenarios drive $messages via window.__PERF_DRIVE__, so no LLM credits.

Scenario -> sunset script mapping:
  stream            <- measure-synthetic-stream, profile-synth-stream, profile-long-stream
  stream --real     <- measure-real-stream, profile-real-stream
  keystroke         <- measure-latency, profile-typing, leak-typing
  transcript        <- (new: long-transcript mount cost)
  submit            <- measure-submit, measure-jump
  session-switch    <- profile-session-switch
  profile-switch    <- measure-profile-switch
CPU profiling is now a cross-cutting --cpuprofile flag, not 5 separate scripts.

CI-tier scenarios (stream, keystroke, transcript) need no backend/credits and
are gated against baseline.json (seed values; re-capture with --update-baseline
on a reference device). Backend-tier scenarios are report-only.

perf-probe.tsx gains loadTranscript() for the transcript scenario. No core
files touched; isolation is via CLI args, not env-gated app changes.

Verified: node --check all modules, tsc, eslint, and a unit smoke of the
stats + regression-gate logic. The end-to-end GUI run (which opens a window)
is left to run interactively via `npm run perf -- --spawn`.
2026-07-19 07:41:00 -04:00

178 lines
6 KiB
JavaScript

// Streaming render cost. Subsumes measure-synthetic-stream, profile-synth-stream,
// profile-long-stream (synthetic) and measure-real-stream / profile-real-stream
// (via --real). CPU profiling is provided by the runner's --cpuprofile flag.
//
// Metrics (lower is better): longtask count + max, frame p95/p99, slow-frame
// count, inter-mutation p95. These are what "is streaming smooth?" reduces to.
import { SELECTORS, sleep, typeIntoComposer } from '../lib/cdp.mjs'
import { frameHistogram, percentile } from '../lib/stats.mjs'
const RECORDERS = `
(() => {
window.__FT__ = { times: [], stop: false }
let last = performance.now()
const tick = () => {
if (window.__FT__.stop) return
const now = performance.now()
window.__FT__.times.push(now - last)
last = now
requestAnimationFrame(tick)
}
requestAnimationFrame(tick)
window.__LT__ = { entries: [], stop: false }
try {
const po = new PerformanceObserver((list) => {
if (window.__LT__.stop) return
for (const e of list.getEntries()) window.__LT__.entries.push({ duration: e.duration, startTime: e.startTime })
})
po.observe({ entryTypes: ['longtask'] })
window.__LT__.po = po
} catch {}
window.__MO__ = { mutations: [], stop: false, current: null }
window.__MO__.arm = () => {
const all = document.querySelectorAll(${JSON.stringify(SELECTORS.assistantMessage)})
const last = all[all.length - 1]
if (!last || last === window.__MO__.current) return
window.__MO__.current = last
window.__MO__.obs && window.__MO__.obs.disconnect()
const obs = new MutationObserver(() => {
if (window.__MO__.stop) return
window.__MO__.mutations.push({ t: performance.now(), len: last.textContent.length })
})
obs.observe(last, { childList: true, subtree: true, characterData: true })
window.__MO__.obs = obs
}
return 'armed'
})()
`
const COLLECT = `
(() => {
window.__FT__.stop = true
window.__LT__.stop = true
window.__MO__.stop = true
try { window.__LT__.po && window.__LT__.po.disconnect() } catch {}
try { window.__MO__.obs && window.__MO__.obs.disconnect() } catch {}
return JSON.stringify({
frames: window.__FT__.times,
longtasks: window.__LT__.entries,
mutations: window.__MO__.mutations
})
})()
`
function analyze(data, warmupMs) {
// Drop warm-up frames (recorder installs before the stream starts).
const frames = []
let acc = 0
for (const f of data.frames) {
acc += f
if (acc >= warmupMs) {
frames.push(f)
}
}
const interMut = []
for (let i = 1; i < data.mutations.length; i++) {
interMut.push(data.mutations[i].t - data.mutations[i - 1].t)
}
const ltDurations = data.longtasks.map(e => e.duration)
const windowS = frames.reduce((a, b) => a + b, 0) / 1000
return {
metrics: {
longtasks_n: data.longtasks.length,
longtask_max_ms: Math.round((ltDurations.length ? Math.max(...ltDurations) : 0) * 10) / 10,
frame_p95_ms: Math.round(percentile(frames, 0.95) * 10) / 10,
frame_p99_ms: Math.round(percentile(frames, 0.99) * 10) / 10,
slow_frames_33: frames.filter(f => f > 33).length,
intermut_p95_ms: Math.round(percentile(interMut, 0.95) * 10) / 10
},
detail: {
windowS: Math.round(windowS * 10) / 10,
avgFps: windowS ? Math.round((frames.length / windowS) * 10) / 10 : 0,
frameHistogram: frameHistogram(frames),
mutations: data.mutations.length,
finalLen: data.mutations.at(-1)?.len ?? 0
}
}
}
export default {
name: 'stream',
tier: 'ci',
description: 'Assistant-message streaming: longtasks, frame pacing, mutation cadence.',
async run(cdp, opts = {}) {
const tokens = Number(opts.tokens ?? 400)
const intervalMs = Number(opts.intervalMs ?? 16)
const flushMinMs = Number(opts.flushMinMs ?? 33)
const chunk = opts.chunk ?? '**word** in _italic_ with `code`, a [link](https://x.dev) and prose. '
const real = Boolean(opts.real)
await cdp.send('Runtime.enable')
await cdp.eval(RECORDERS)
if (real) {
// Backend path: fire a real prompt and wait for the stream to appear.
const baseCount = await cdp.eval(`document.querySelectorAll(${JSON.stringify(SELECTORS.assistantMessage)}).length`)
await typeIntoComposer(cdp, opts.prompt ?? 'count from 1 to 80, one number per line', { cps: 40 })
await cdp.eval(`(() => {
const el = document.querySelector(${JSON.stringify(SELECTORS.composer)})
el && el.dispatchEvent(new KeyboardEvent('keydown', { key: 'Enter', code: 'Enter', bubbles: true, cancelable: true }))
})()`)
const deadline = Date.now() + Number(opts.timeoutMs ?? 60000)
let started = false
while (Date.now() < deadline) {
await sleep(50)
const n = await cdp.eval(`document.querySelectorAll(${JSON.stringify(SELECTORS.assistantMessage)}).length`)
if (n > baseCount) {
started = true
break
}
}
if (!started) {
throw new Error('real stream never started (no LLM credit / backend?)')
}
await cdp.eval('window.__MO__.arm()')
// Let it run to completion or timeout.
const runDeadline = Date.now() + Number(opts.runMs ?? 30000)
while (Date.now() < runDeadline) {
await sleep(250)
const busy = await cdp.eval(`!!document.querySelector('[data-status="running"], [data-busy="true"]')`)
if (!busy) {
break
}
}
} else {
// Synthetic path: drive $messages directly. No LLM, no credits.
await cdp.eval(
`window.__PERF_DRIVE__.stream({ chunk: ${JSON.stringify(chunk)}, intervalMs: ${intervalMs}, totalTokens: ${tokens}, flushMinMs: ${flushMinMs} })`
)
await sleep(200)
await cdp.eval('window.__MO__.arm()')
await sleep(tokens * intervalMs + 1500)
}
const data = JSON.parse(await cdp.eval(COLLECT))
if (!real) {
await cdp.eval('window.__PERF_DRIVE__.reset()')
}
return analyze(data, real ? 0 : 500)
}
}