diff --git a/src/main/daemon/session-ingest-throughput.bench.test.ts b/src/main/daemon/session-ingest-throughput.bench.test.ts new file mode 100644 index 00000000000..e226ddf869d --- /dev/null +++ b/src/main/daemon/session-ingest-throughput.bench.test.ts @@ -0,0 +1,130 @@ +import { describe, expect, it } from 'vitest' +import { performance } from 'node:perf_hooks' +import { Session, type SubprocessHandle } from './session' + +// Benchmark harness for the terminal performance initiative: measures the +// daemon-side ingest rate (Session.handleSubprocessData -> HeadlessEmulator +// write + pending-output recording + client fanout) for the same workload +// shapes as tools/benchmarks/terminal-pipeline-bench.mjs. Bare headless +// xterm parses these at ~80-100 MB/s; the end-to-end Orca pipeline measured +// 2-15 MB/s (baseline-jul02) — this isolates the daemon layer's share. +// Run with: +// ORCA_TERMINAL_PERF_BENCH=1 pnpm vitest run \ +// src/main/daemon/session-ingest-throughput.bench.test.ts \ +// --config config/vitest.config.ts +const benchEnabled = process.env.ORCA_TERMINAL_PERF_BENCH === '1' + +const COLS = 114 +const ROWS = 85 +const TARGET_BYTES = 10 * 1024 * 1024 +const CHUNK = 64 * 1024 + +function asciiLog(targetBytes: number): string { + const parts: string[] = [] + let bytes = 0 + let line = 0 + while (bytes < targetBytes) { + line++ + const s = `\x1b[32m[build ${String(line).padStart(6, '0')}]\x1b[0m compile transform resolve bundle emit chunk module (${line % 5000}ms)\r\n` + parts.push(s) + bytes += s.length + } + return parts.join('') +} + +function agentTui(targetBytes: number): string { + const statusRows = 10 + const parts: string[] = [] + let bytes = 0 + let frame = 0 + let painted = false + const push = (s: string): void => { + parts.push(s) + bytes += Buffer.byteLength(s, 'utf8') + } + while (bytes < targetBytes) { + frame++ + push('\x1b[?2026h') + if (painted) { + push(`\x1b[${statusRows}A\x1b[0J`) + } + push(`\x1b[2m●\x1b[0m transcript line for frame ${frame} with some words\r\n`) + for (let r = 0; r < statusRows; r++) { + push( + `\x1b[38;5;${33 + (r % 6)}m⠼ task ${frame % 100}·${r}\x1b[0m ${'▇'.repeat((frame + r) % 40)}\r\n` + ) + } + painted = true + push('\x1b[?2026l') + } + return parts.join('') +} + +function makeSubprocess(): SubprocessHandle & { emit: (data: string) => void } { + let onData: ((data: string) => void) | null = null + return { + pid: 4242, + getForegroundProcess: () => 'bench', + write: () => {}, + resize: () => {}, + kill: () => {}, + forceKill: () => {}, + signal: () => {}, + onData: (cb) => { + onData = cb + }, + onExit: () => {}, + dispose: () => {}, + emit: (data: string) => onData?.(data) + } +} + +function ingest(fixture: string, drainPendingEveryChunks: number | null): number { + const subprocess = makeSubprocess() + const session = new Session({ + sessionId: 'bench', + cols: COLS, + rows: ROWS, + subprocess, + shellReadySupported: false + }) + session.attachClient({ onData: () => {}, onExit: () => {} }) + // Warmup primes JIT paths. + subprocess.emit(fixture.slice(0, 256 * 1024)) + session.takePendingOutput(false) + const start = performance.now() + let chunks = 0 + for (let i = 0; i < fixture.length; i += CHUNK) { + subprocess.emit(fixture.slice(i, i + CHUNK)) + chunks++ + // Why: without periodic takes the 2MB pending cap overflows and recording + // short-circuits, understating the real steady-state cost. The 5s adapter + // tick drains in production; drain per ~1.5MB approximates a hot session. + if (drainPendingEveryChunks && chunks % drainPendingEveryChunks === 0) { + session.takePendingOutput(false) + } + } + const ms = performance.now() - start + session.dispose() + return ms +} + +describe.skipIf(!benchEnabled)('daemon session ingest throughput', () => { + it('measures MB/s per workload shape', () => { + const rows: string[] = [] + for (const [name, fixture] of [ + ['ascii-log', asciiLog(TARGET_BYTES)], + ['agent-tui', agentTui(TARGET_BYTES)] + ] as const) { + const bytes = Buffer.byteLength(fixture, 'utf8') + const ms = ingest(fixture, 24) + const rate = bytes / 1024 / 1024 / (ms / 1000) + rows.push( + `${name}: ${rate.toFixed(1)} MB/s (${ms.toFixed(0)}ms for ${(bytes / 1024 / 1024).toFixed(1)}MB)` + ) + } + // eslint-disable-next-line no-console -- bench harness output + console.log(`\n[session-ingest] ${COLS}x${ROWS}\n ${rows.join('\n ')}`) + expect(rows.length).toBe(2) + }) +}) diff --git a/tools/benchmarks/terminal-headless-parse-bench.mjs b/tools/benchmarks/terminal-headless-parse-bench.mjs new file mode 100644 index 00000000000..694593cfb3c --- /dev/null +++ b/tools/benchmarks/terminal-headless-parse-bench.mjs @@ -0,0 +1,70 @@ +#!/usr/bin/env node +/** + * Decomposes cross-terminal pipeline results: feeds the same fixtures from + * terminal-pipeline-bench through a bare @xterm/headless Terminal — no Orca + * layers, no IPC, no rendering — to locate where throughput is lost. + * + * If headless xterm parses a fixture near the plain-text rate, the pipeline + * gap for that fixture lives in Orca's layers (delivery, side-effect + * scanning, renderer paint). If headless collapses too, the cost is intrinsic + * to xterm.js's parser/buffer for that byte pattern. + * + * Usage: + * node tools/benchmarks/terminal-headless-parse-bench.mjs + * [--size-mb 10] [--cols 114] [--rows 85] [--scrollback 5000] + */ +import { performance } from 'node:perf_hooks' +import xterm from '@xterm/headless' +import { buildFixture } from './terminal-pipeline-bench.mjs' + +const { Terminal } = xterm +const CHUNK = 64 * 1024 + +function arg(name, fallback) { + const i = process.argv.indexOf(name) + return i === -1 ? fallback : Number(process.argv[i + 1]) +} + +const sizeMb = arg('--size-mb', 10) +const cols = arg('--cols', 114) +const rows = arg('--rows', 85) +const scrollback = arg('--scrollback', 5000) +const targetBytes = Math.floor(sizeMb * 1024 * 1024) + +function writeAll(term, data) { + return new Promise((resolve) => { + let offset = 0 + const next = () => { + if (offset >= data.length) { + resolve() + return + } + const chunk = data.slice(offset, offset + CHUNK) + offset += CHUNK + // write callback fires after the chunk is parsed — same "fully parsed" + // fence semantics as the DSR fence in the pipeline bench. + term.write(chunk, next) + } + next() + }) +} + +const FIXTURES = ['ascii-log', 'cjk-emoji', 'agent-tui', 'styles-stress'] + +console.log( + `headless xterm ${cols}x${rows} scrollback=${scrollback}, ${sizeMb}MB per fixture (parse-only, no render)` +) +for (const name of FIXTURES) { + const fixture = buildFixture(name, targetBytes, cols, rows) + const bytes = Buffer.byteLength(fixture, 'utf8') + const term = new Terminal({ cols, rows, scrollback, allowProposedApi: true }) + // Warmup primes JIT so the first fixture isn't penalized. + await writeAll(term, fixture.slice(0, 256 * 1024)) + const start = performance.now() + await writeAll(term, fixture) + const ms = performance.now() - start + console.log( + `${name.padEnd(15)} ${(bytes / 1024 / 1024 / (ms / 1000)).toFixed(1).padStart(7)} MB/s (${ms.toFixed(0)}ms)` + ) + term.dispose() +} diff --git a/tools/benchmarks/terminal-pipeline-bench.mjs b/tools/benchmarks/terminal-pipeline-bench.mjs index c7660807902..6274befd9ae 100644 --- a/tools/benchmarks/terminal-pipeline-bench.mjs +++ b/tools/benchmarks/terminal-pipeline-bench.mjs @@ -218,7 +218,7 @@ function stylesStressFixture(targetBytes, cols) { return parts.join('') } -function buildFixture(name, targetBytes, cols, rows) { +export function buildFixture(name, targetBytes, cols, rows) { switch (name) { case 'ascii-log': return asciiLogFixture(targetBytes) @@ -551,8 +551,14 @@ async function main() { process.exit(0) } -main().catch((err) => { - restoreTerminal() - console.error(err) - process.exit(1) -}) +// Why the guard: buildFixture is imported by sibling analysis scripts +// (headless parse decomposition); importing must not start a benchmark run. +const invokedDirectly = + process.argv[1] && import.meta.url === new URL(`file://${process.argv[1]}`).href +if (invokedDirectly) { + main().catch((err) => { + restoreTerminal() + console.error(err) + process.exit(1) + }) +}