mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-21 16:18:55 +00:00
Replaces the dozen ad-hoc measure-*/profile-* scripts (each reinventing the CDP client — 4 different copies — plus its own arg parsing, stats, output path, and none with a baseline) with one framework under scripts/perf/: - lib/cdp.mjs one CDP client + target discovery + typing + CPU-profile wrapper + DOM selectors - lib/stats.mjs percentiles, histograms, CPU-profile self-time ranking - lib/baseline.mjs load/compare/update baseline + regression gate (new capability) - lib/launch.mjs attach, OR spawn a fully ISOLATED instance - scenarios/* one module per measurement, registered in scenarios/index.mjs - run.mjs / serve.mjs, baseline.json, README.md Isolation solves the long-standing measurement blocker: a running `hgui` held the Electron single-instance lock, so a second instance quit. `--spawn` / `perf:serve` launch with their own --user-data-dir (separate lock scope), their own HERMES_HOME (separate backend/sessions, config seeded from ~/.hermes so it reaches a chat view without onboarding), and their own --remote-debugging-port. Synthetic scenarios drive $messages via window.__PERF_DRIVE__, so no LLM credits. Scenario -> sunset script mapping: stream <- measure-synthetic-stream, profile-synth-stream, profile-long-stream stream --real <- measure-real-stream, profile-real-stream keystroke <- measure-latency, profile-typing, leak-typing transcript <- (new: long-transcript mount cost) submit <- measure-submit, measure-jump session-switch <- profile-session-switch profile-switch <- measure-profile-switch CPU profiling is now a cross-cutting --cpuprofile flag, not 5 separate scripts. CI-tier scenarios (stream, keystroke, transcript) need no backend/credits and are gated against baseline.json (seed values; re-capture with --update-baseline on a reference device). Backend-tier scenarios are report-only. perf-probe.tsx gains loadTranscript() for the transcript scenario. No core files touched; isolation is via CLI args, not env-gated app changes. Verified: node --check all modules, tsc, eslint, and a unit smoke of the stats + regression-gate logic. The end-to-end GUI run (which opens a window) is left to run interactively via `npm run perf -- --spawn`.
80 lines
2.3 KiB
JavaScript
80 lines
2.3 KiB
JavaScript
// Session-switch latency. Subsumes profile-session-switch. Backend tier: needs
|
|
// two real stored session ids and a live backend. Report-only.
|
|
//
|
|
// node scripts/perf/run.mjs session-switch --a <sidA> --b <sidB> [--rounds 2]
|
|
|
|
import { SELECTORS, sleep } from '../lib/cdp.mjs'
|
|
import { summarize } from '../lib/stats.mjs'
|
|
|
|
export default {
|
|
name: 'session-switch',
|
|
tier: 'backend',
|
|
description: 'Route to a session and wait for first-paint + settle of its transcript.',
|
|
requiredOpts: ['a', 'b'],
|
|
async run(cdp, opts = {}) {
|
|
const { a, b } = opts
|
|
const rounds = Number(opts.rounds ?? 2)
|
|
const settleTimeoutMs = Number(opts.settleTimeoutMs ?? 30000)
|
|
|
|
if (!a || !b) {
|
|
throw new Error('session-switch needs --a <sessionId> --b <sessionId>')
|
|
}
|
|
|
|
await cdp.send('Runtime.enable')
|
|
|
|
const switchTo = async sid => {
|
|
const t0 = await cdp.eval(`(() => { location.hash = '#/' + ${JSON.stringify(sid)}; return performance.now() })()`)
|
|
const deadline = Date.now() + settleTimeoutMs
|
|
let firstPaint = null
|
|
let stable = 0
|
|
let lastCount = -1
|
|
|
|
while (Date.now() < deadline) {
|
|
await sleep(50)
|
|
const s = await cdp.eval(`({
|
|
t: performance.now(),
|
|
route: location.hash,
|
|
msgs: document.querySelectorAll(${JSON.stringify(SELECTORS.assistantMessage)}).length
|
|
})`)
|
|
|
|
if (!String(s.route).includes(sid)) {
|
|
continue
|
|
}
|
|
|
|
if (s.msgs > 0 && firstPaint === null) {
|
|
firstPaint = s.t - t0
|
|
}
|
|
|
|
stable = s.msgs === lastCount && s.msgs > 0 ? stable + 1 : 0
|
|
lastCount = s.msgs
|
|
|
|
if (stable >= 3) {
|
|
return { firstPaint, settled: s.t - t0 }
|
|
}
|
|
}
|
|
|
|
return { firstPaint, settled: null }
|
|
}
|
|
|
|
const firstPaints = []
|
|
const settles = []
|
|
|
|
for (let round = 0; round < rounds; round++) {
|
|
for (const sid of [a, b]) {
|
|
const r = await switchTo(sid)
|
|
|
|
if (typeof r.firstPaint === 'number') firstPaints.push(r.firstPaint)
|
|
if (typeof r.settled === 'number') settles.push(r.settled)
|
|
await sleep(800)
|
|
}
|
|
}
|
|
|
|
return {
|
|
metrics: {
|
|
switch_first_paint_p95_ms: summarize(firstPaints).p95,
|
|
switch_settled_p95_ms: summarize(settles).p95
|
|
},
|
|
detail: { rounds, firstPaint: summarize(firstPaints), settled: summarize(settles) }
|
|
}
|
|
}
|
|
}
|