Add pipeline-loss decomposition benches (headless xterm + daemon ingest)

Both isolate layers of the 51x agent-tui gap found in baseline-jul02:
bare @xterm/headless parses agent-tui at 103 MB/s and daemon Session
ingest (emulator + pending-output recording + fanout) at 103 MB/s —
on the byte stream the full Orca pipeline delivers at 2.0 MB/s.
Parser and daemon are exonerated; the loss is in main per-chunk
processing, delivery/ACK pacing, or renderer layers above xterm.

Co-authored-by: Orca <help@stably.ai>
This commit is contained in:
Jinwoo-H
2026-07-02 21:22:50 -04:00
co-authored by Orca
parent a793c39e18
commit e5511d87b8
3 changed files with 212 additions and 6 deletions
@@ -0,0 +1,130 @@
import { describe, expect, it } from 'vitest'
import { performance } from 'node:perf_hooks'
import { Session, type SubprocessHandle } from './session'
// Benchmark harness for the terminal performance initiative: measures the
// daemon-side ingest rate (Session.handleSubprocessData -> HeadlessEmulator
// write + pending-output recording + client fanout) for the same workload
// shapes as tools/benchmarks/terminal-pipeline-bench.mjs. Bare headless
// xterm parses these at ~80-100 MB/s; the end-to-end Orca pipeline measured
// 2-15 MB/s (baseline-jul02) — this isolates the daemon layer's share.
// Run with:
// ORCA_TERMINAL_PERF_BENCH=1 pnpm vitest run \
// src/main/daemon/session-ingest-throughput.bench.test.ts \
// --config config/vitest.config.ts
const benchEnabled = process.env.ORCA_TERMINAL_PERF_BENCH === '1'
const COLS = 114
const ROWS = 85
const TARGET_BYTES = 10 * 1024 * 1024
const CHUNK = 64 * 1024
function asciiLog(targetBytes: number): string {
const parts: string[] = []
let bytes = 0
let line = 0
while (bytes < targetBytes) {
line++
const s = `\x1b[32m[build ${String(line).padStart(6, '0')}]\x1b[0m compile transform resolve bundle emit chunk module (${line % 5000}ms)\r\n`
parts.push(s)
bytes += s.length
}
return parts.join('')
}
function agentTui(targetBytes: number): string {
const statusRows = 10
const parts: string[] = []
let bytes = 0
let frame = 0
let painted = false
const push = (s: string): void => {
parts.push(s)
bytes += Buffer.byteLength(s, 'utf8')
}
while (bytes < targetBytes) {
frame++
push('\x1b[?2026h')
if (painted) {
push(`\x1b[${statusRows}A\x1b[0J`)
}
push(`\x1b[2m●\x1b[0m transcript line for frame ${frame} with some words\r\n`)
for (let r = 0; r < statusRows; r++) {
push(
`\x1b[38;5;${33 + (r % 6)}m⠼ task ${frame % 100}·${r}\x1b[0m ${'▇'.repeat((frame + r) % 40)}\r\n`
)
}
painted = true
push('\x1b[?2026l')
}
return parts.join('')
}
function makeSubprocess(): SubprocessHandle & { emit: (data: string) => void } {
let onData: ((data: string) => void) | null = null
return {
pid: 4242,
getForegroundProcess: () => 'bench',
write: () => {},
resize: () => {},
kill: () => {},
forceKill: () => {},
signal: () => {},
onData: (cb) => {
onData = cb
},
onExit: () => {},
dispose: () => {},
emit: (data: string) => onData?.(data)
}
}
function ingest(fixture: string, drainPendingEveryChunks: number | null): number {
const subprocess = makeSubprocess()
const session = new Session({
sessionId: 'bench',
cols: COLS,
rows: ROWS,
subprocess,
shellReadySupported: false
})
session.attachClient({ onData: () => {}, onExit: () => {} })
// Warmup primes JIT paths.
subprocess.emit(fixture.slice(0, 256 * 1024))
session.takePendingOutput(false)
const start = performance.now()
let chunks = 0
for (let i = 0; i < fixture.length; i += CHUNK) {
subprocess.emit(fixture.slice(i, i + CHUNK))
chunks++
// Why: without periodic takes the 2MB pending cap overflows and recording
// short-circuits, understating the real steady-state cost. The 5s adapter
// tick drains in production; drain per ~1.5MB approximates a hot session.
if (drainPendingEveryChunks && chunks % drainPendingEveryChunks === 0) {
session.takePendingOutput(false)
}
}
const ms = performance.now() - start
session.dispose()
return ms
}
describe.skipIf(!benchEnabled)('daemon session ingest throughput', () => {
it('measures MB/s per workload shape', () => {
const rows: string[] = []
for (const [name, fixture] of [
['ascii-log', asciiLog(TARGET_BYTES)],
['agent-tui', agentTui(TARGET_BYTES)]
] as const) {
const bytes = Buffer.byteLength(fixture, 'utf8')
const ms = ingest(fixture, 24)
const rate = bytes / 1024 / 1024 / (ms / 1000)
rows.push(
`${name}: ${rate.toFixed(1)} MB/s (${ms.toFixed(0)}ms for ${(bytes / 1024 / 1024).toFixed(1)}MB)`
)
}
// eslint-disable-next-line no-console -- bench harness output
console.log(`\n[session-ingest] ${COLS}x${ROWS}\n ${rows.join('\n ')}`)
expect(rows.length).toBe(2)
})
})
@@ -0,0 +1,70 @@
#!/usr/bin/env node
/**
* Decomposes cross-terminal pipeline results: feeds the same fixtures from
* terminal-pipeline-bench through a bare @xterm/headless Terminal — no Orca
* layers, no IPC, no rendering — to locate where throughput is lost.
*
* If headless xterm parses a fixture near the plain-text rate, the pipeline
* gap for that fixture lives in Orca's layers (delivery, side-effect
* scanning, renderer paint). If headless collapses too, the cost is intrinsic
* to xterm.js's parser/buffer for that byte pattern.
*
* Usage:
* node tools/benchmarks/terminal-headless-parse-bench.mjs
* [--size-mb 10] [--cols 114] [--rows 85] [--scrollback 5000]
*/
import { performance } from 'node:perf_hooks'
import xterm from '@xterm/headless'
import { buildFixture } from './terminal-pipeline-bench.mjs'
const { Terminal } = xterm
const CHUNK = 64 * 1024
function arg(name, fallback) {
const i = process.argv.indexOf(name)
return i === -1 ? fallback : Number(process.argv[i + 1])
}
const sizeMb = arg('--size-mb', 10)
const cols = arg('--cols', 114)
const rows = arg('--rows', 85)
const scrollback = arg('--scrollback', 5000)
const targetBytes = Math.floor(sizeMb * 1024 * 1024)
function writeAll(term, data) {
return new Promise((resolve) => {
let offset = 0
const next = () => {
if (offset >= data.length) {
resolve()
return
}
const chunk = data.slice(offset, offset + CHUNK)
offset += CHUNK
// write callback fires after the chunk is parsed — same "fully parsed"
// fence semantics as the DSR fence in the pipeline bench.
term.write(chunk, next)
}
next()
})
}
const FIXTURES = ['ascii-log', 'cjk-emoji', 'agent-tui', 'styles-stress']
console.log(
`headless xterm ${cols}x${rows} scrollback=${scrollback}, ${sizeMb}MB per fixture (parse-only, no render)`
)
for (const name of FIXTURES) {
const fixture = buildFixture(name, targetBytes, cols, rows)
const bytes = Buffer.byteLength(fixture, 'utf8')
const term = new Terminal({ cols, rows, scrollback, allowProposedApi: true })
// Warmup primes JIT so the first fixture isn't penalized.
await writeAll(term, fixture.slice(0, 256 * 1024))
const start = performance.now()
await writeAll(term, fixture)
const ms = performance.now() - start
console.log(
`${name.padEnd(15)} ${(bytes / 1024 / 1024 / (ms / 1000)).toFixed(1).padStart(7)} MB/s (${ms.toFixed(0)}ms)`
)
term.dispose()
}
+12 -6
View File
@@ -218,7 +218,7 @@ function stylesStressFixture(targetBytes, cols) {
return parts.join('')
}
function buildFixture(name, targetBytes, cols, rows) {
export function buildFixture(name, targetBytes, cols, rows) {
switch (name) {
case 'ascii-log':
return asciiLogFixture(targetBytes)
@@ -551,8 +551,14 @@ async function main() {
process.exit(0)
}
main().catch((err) => {
restoreTerminal()
console.error(err)
process.exit(1)
})
// Why the guard: buildFixture is imported by sibling analysis scripts
// (headless parse decomposition); importing must not start a benchmark run.
const invokedDirectly =
process.argv[1] && import.meta.url === new URL(`file://${process.argv[1]}`).href
if (invokedDirectly) {
main().catch((err) => {
restoreTerminal()
console.error(err)
process.exit(1)
})
}