From 4ac6395caf47d8934eebe3d56fed791be7aa58f2 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:31:00 -0700 Subject: [PATCH] perf(terminal): bound the Codex backfill scan instead of rewriting it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The streaming detector was 40-70% slower in CPU than the regex it replaced and had a pinned semantic difference for aborted escapes. Codex prints the backfill timeout from its state-db startup gate, before the TUI draws, so the signature can only appear in a pane's first few KB; the original detector is restored verbatim and simply disarms after 256 KB of raw output. This is faster than either version (66 ms -> 3.3 ms over the 23.5 MB corpus) and removes 95% of the allocation, with no scanner rewrite on the hot path. Also pins three classifier edge cases the corpus tests did not bite on: 38;2 parameter skip width, the 38:… colon guard, and astral advance. --- .../codex-backfill-error-detector.test.ts | 165 ++----------- .../codex-backfill-error-detector.ts | 220 ++---------------- .../terminal-complex-script.test.ts | 7 + 3 files changed, 46 insertions(+), 346 deletions(-) diff --git a/src/renderer/src/components/terminal-pane/codex-backfill-error-detector.test.ts b/src/renderer/src/components/terminal-pane/codex-backfill-error-detector.test.ts index c1a92e6e627..38022cafebc 100644 --- a/src/renderer/src/components/terminal-pane/codex-backfill-error-detector.test.ts +++ b/src/renderer/src/components/terminal-pane/codex-backfill-error-detector.test.ts @@ -1,64 +1,10 @@ import { describe, expect, it } from 'vitest' import { CODEX_BACKFILL_RECOVERY_NOTICE, - CODEX_BACKFILL_TIMEOUT_SIGNATURE, + CODEX_BACKFILL_SCAN_BUDGET_CHARS, createCodexBackfillErrorDetector } from './codex-backfill-error-detector' -const ANSI_ESCAPE_PATTERN = - // eslint-disable-next-line no-control-regex -- terminal escape sequences contain control bytes - /\u001b(?:\[[0-9;?]*[ -/]*[@-~]|\][^\u0007\u001b]*(?:\u0007|\u001b\\)?)/g -const DETECTOR_BUFFER_MAX_CHARS = 4096 - -/** The pre-optimization detector, kept verbatim as the differential oracle. */ -function createReferenceDetector(): { observe(chunk: string): string | null } { - let tail = '' - let armed = true - return { - observe(chunk: string): string | null { - if (!armed) { - return null - } - const normalized = (tail + chunk).replace(ANSI_ESCAPE_PATTERN, '').replace(/\r/g, '') - tail = normalized.slice(-DETECTOR_BUFFER_MAX_CHARS) - if (!tail.toLowerCase().includes(CODEX_BACKFILL_TIMEOUT_SIGNATURE)) { - return null - } - armed = false - return CODEX_BACKFILL_RECOVERY_NOTICE - } - } -} - -function splitAt(text: string, offsets: readonly number[]): string[] { - const chunks: string[] = [] - let previous = 0 - for (const offset of offsets) { - chunks.push(text.slice(previous, offset)) - previous = offset - } - chunks.push(text.slice(previous)) - return chunks -} - -function observeAll( - detector: { observe(chunk: string): string | null }, - chunks: readonly string[] -): (string | null)[] { - return chunks.map((chunk) => detector.observe(chunk)) -} - -function expectMatchesReference(chunks: readonly string[]): (string | null)[] { - const actual = observeAll(createCodexBackfillErrorDetector(), chunks) - const expected = observeAll(createReferenceDetector(), chunks) - expect(actual).toEqual(expected) - return actual -} - -const ESC = String.fromCharCode(0x1b) - -const SIGNATURE_LINE = '\u001b[31mError: timed out waiting for state DB backfill\u001b[0m\r\n' - describe('Codex backfill error detector', () => { it('recognizes the timeout across ANSI-decorated chunks once', () => { const detector = createCodexBackfillErrorDetector() @@ -74,103 +20,28 @@ describe('Codex backfill error detector', () => { expect(detector.observe('local database appears to be damaged')).toBeNull() }) - it('detects the signature at every chunk-boundary split, including inside escapes', () => { - for (let offset = 0; offset <= SIGNATURE_LINE.length; offset++) { - const results = expectMatchesReference(splitAt(SIGNATURE_LINE, [offset])) - expect(results.some((result) => result === CODEX_BACKFILL_RECOVERY_NOTICE)).toBe(true) - } - }) - - it('detects a signature interleaved with escape sequences at every split', () => { - const interleaved = - 'timed \u001b[1mout \u001b[0;32mwaiting for state \u001b[Kdb \u001b[38;5;214mbackfill\u001b[0m\r\n' - for (let offset = 0; offset <= interleaved.length; offset++) { - const results = expectMatchesReference(splitAt(interleaved, [offset])) - expect(results.some((result) => result === CODEX_BACKFILL_RECOVERY_NOTICE)).toBe(true) - } - }) - - it('keeps a literal escape that never completes breaking the signature', () => { - // The unmatched ESC stays in the normalized stream, so the halves do not join. - expectMatchesReference(['timed out waiting for state db back\u001b', 'fill\n']) + it('still fires when the signature lands inside the scan budget', () => { const detector = createCodexBackfillErrorDetector() - expect(detector.observe('timed out waiting for state db back\u001b')).toBeNull() - expect(detector.observe('fill\n')).toBeNull() + const filler = 'x'.repeat(1024) + const chunksBeforeSignature = Math.floor(CODEX_BACKFILL_SCAN_BUDGET_CHARS / filler.length) - 1 + for (let index = 0; index < chunksBeforeSignature; index++) { + expect(detector.observe(filler)).toBeNull() + } + expect(detector.observe('Error: timed out waiting for state DB backfill\r\n')).toBe( + CODEX_BACKFILL_RECOVERY_NOTICE + ) }) - it('ignores a signature pushed out of the trailing normalized window by one chunk', () => { - const chunk = `${CODEX_BACKFILL_TIMEOUT_SIGNATURE}${'x'.repeat(DETECTOR_BUFFER_MAX_CHARS)}` - expect(createCodexBackfillErrorDetector().observe(chunk)).toBeNull() - expect(createReferenceDetector().observe(chunk)).toBeNull() - }) - - it('still fires when the signature ends inside the trailing window', () => { - const padding = DETECTOR_BUFFER_MAX_CHARS - CODEX_BACKFILL_TIMEOUT_SIGNATURE.length - const chunk = `${CODEX_BACKFILL_TIMEOUT_SIGNATURE}${'x'.repeat(padding)}` - expectMatchesReference([chunk]) - expect(createCodexBackfillErrorDetector().observe(chunk)).toBe(CODEX_BACKFILL_RECOVERY_NOTICE) - }) - - it('reads an aborted escape sequence once, the way a terminal parser does', () => { - // Documented difference from the pre-optimization detector: that one re-stripped - // its own retained output every chunk, so an aborted CSI prefix could fuse with a - // byte that arrived in a LATER chunk and swallow it. Reaching this needs an escape - // sequence aborted mid-sequence by another ESC, which no terminal renders as text. - const aborted = `${ESC}[31${ESC}[0mtimed out waiting for state d` + it('disarms once the budget is spent so a later signature is ignored', () => { const detector = createCodexBackfillErrorDetector() - expect(detector.observe(aborted)).toBeNull() - expect(detector.observe('b backfill')).toBe(CODEX_BACKFILL_RECOVERY_NOTICE) - - const reference = createReferenceDetector() - expect(reference.observe(aborted)).toBeNull() - // The reference fuses the retained `ESC [ 3 1` with the following 't' on its next - // pass and eats the signature's first character. - expect(reference.observe('b backfill')).toBeNull() + expect(detector.observe('x'.repeat(CODEX_BACKFILL_SCAN_BUDGET_CHARS))).toBeNull() + expect(detector.observe('Error: timed out waiting for state DB backfill\r\n')).toBeNull() }) - it('matches the reference detector over randomized agent-like output', () => { - const alphabet = [ - 'timed out waiting for state db backfill', - 'Timed Out Waiting For State DB Backfill', - 'timed out waiting for state db back', - 'fill', - '\u001b[0m', - '\u001b[38;5;214m', - '\u001b[?25l', - '\u001b[2K', - '\u001b]0;codex build\u0007', - '\u001b]8;;https://example.com\u001b\\', - '\u001b]0;unterminated', - '\u001b', - '', - '\u001bM', - '\u001b[<0;1;2M', - '\r', - '\r\n', - '\n', - 'working spinner ', - 'local database appears to be damaged', - 'x'.repeat(97), - 'timed out\u001b[0m waiting for state db backfill' - ] - let seed = 0x2f6e2b1 - const random = (): number => { - seed = (seed * 1103515245 + 12345) & 0x7fffffff - return seed / 0x7fffffff - } - for (let iteration = 0; iteration < 400; iteration++) { - let text = '' - const pieces = 1 + Math.floor(random() * 14) - for (let piece = 0; piece < pieces; piece++) { - text += alphabet[Math.floor(random() * alphabet.length)] ?? '' - } - const offsets: number[] = [] - const splits = Math.floor(random() * 5) - for (let split = 0; split < splits; split++) { - offsets.push(Math.floor(random() * (text.length + 1))) - } - offsets.sort((a, b) => a - b) - expectMatchesReference(splitAt(text, offsets)) - } + it('charges raw chunk length, so escape-heavy output cannot extend the scan', () => { + const detector = createCodexBackfillErrorDetector() + const escapes = '\u001b[0m'.repeat(CODEX_BACKFILL_SCAN_BUDGET_CHARS / 4) + expect(detector.observe(escapes)).toBeNull() + expect(detector.observe('timed out waiting for state db backfill')).toBeNull() }) }) diff --git a/src/renderer/src/components/terminal-pane/codex-backfill-error-detector.ts b/src/renderer/src/components/terminal-pane/codex-backfill-error-detector.ts index 528c855a58f..3d448eafbfc 100644 --- a/src/renderer/src/components/terminal-pane/codex-backfill-error-detector.ts +++ b/src/renderer/src/components/terminal-pane/codex-backfill-error-detector.ts @@ -5,220 +5,42 @@ export const CODEX_BACKFILL_RECOVERY_NOTICE = [ 'Keep Orca open for a few minutes, then retry this pane. Orca attempts background recovery for managed local and WSL homes.' ].join('\n') -/** - * Width of the normalized-output window the signature must land inside. Kept - * from the original `normalized.slice(-4096)` detector so a signature followed - * by more than this many normalized characters in one chunk still does not fire. - */ +const ANSI_ESCAPE_PATTERN = + // eslint-disable-next-line no-control-regex -- terminal escape sequences contain control bytes + /\u001b(?:\[[0-9;?]*[ -/]*[@-~]|\][^\u0007\u001b]*(?:\u0007|\u001b\\)?)/g const DETECTOR_BUFFER_MAX_CHARS = 4096 - -const SIGNATURE_CODES = Uint16Array.from(CODEX_BACKFILL_TIMEOUT_SIGNATURE, (character) => - character.charCodeAt(0) -) - -/** KMP prefix table so the streaming match needs no per-chunk buffer or backtracking. */ -function buildSignatureFailureTable(codes: Uint16Array): Int32Array { - const failure = new Int32Array(codes.length) - let matched = 0 - for (let index = 1; index < codes.length; index++) { - while (matched > 0 && codes[index] !== codes[matched]) { - matched = failure[matched - 1] - } - if (codes[index] === codes[matched]) { - matched += 1 - } - failure[index] = matched - } - return failure -} - -const SIGNATURE_FAILURE = buildSignatureFailureTable(SIGNATURE_CODES) - -const STATE_TEXT = 0 -const STATE_ESCAPE = 1 -const STATE_CSI_PARAMS = 2 -const STATE_CSI_INTERMEDIATE = 3 -const STATE_OSC = 4 -const STATE_OSC_ESCAPE = 5 - -const ESC = 0x1b -const CARRIAGE_RETURN = 0x0d -const BEL = 0x07 -const LEFT_BRACKET = 0x5b -const RIGHT_BRACKET = 0x5d -const BACKSLASH = 0x5c - -function isCsiParameterByte(code: number): boolean { - // Mirrors the old [0-9;?] class exactly — ':' and '<' '=' '>' are deliberately excluded. - return (code >= 0x30 && code <= 0x39) || code === 0x3b || code === 0x3f -} - -function isCsiIntermediateByte(code: number): boolean { - return code >= 0x20 && code <= 0x2f -} - -function isCsiFinalByte(code: number): boolean { - return code >= 0x40 && code <= 0x7e -} +/** + * Raw output after which the scan disarms. Codex prints the timeout from its + * state-db startup gate, before the TUI ever draws, so it can only appear in + * the first few KB of a pane's output; without a budget every later chunk is + * stripped and copied for the pane's whole life. + */ +export const CODEX_BACKFILL_SCAN_BUDGET_CHARS = 256 * 1024 export type CodexBackfillErrorDetector = { observe(chunk: string): string | null } -/** - * Scans one Codex pane's output once for the unambiguous backfill timeout. - * - * Streams the chunk through an ANSI/CR state machine feeding a KMP matcher, so - * no normalized copy of the output is ever built. Detection is identical to the - * previous `strip -> slice(-4096) -> toLowerCase -> includes` implementation: - * escape sequences and carriage returns are removed with the same grammar, a - * sequence split across chunks is still recognized, an escape that never - * completes is still emitted as literal text (and so still breaks a match), and - * the signature must still fall inside the trailing 4096-character window. - */ +/** Scans the start of one Codex pane's output once for the unambiguous backfill timeout. */ export function createCodexBackfillErrorDetector(): CodexBackfillErrorDetector { + let tail = '' let armed = true - let state: number = STATE_TEXT - /** Escape prefix carried from earlier chunks; re-emitted verbatim if the sequence never completes. */ - let carriedEscape = '' - /** Index in the current chunk where the pending escape prefix starts (0 when it began earlier). */ - let pendingStart = 0 - /** Characters emitted into the normalized stream so far — the stream coordinate space. */ - let emitted = 0 - let matchedLength = 0 - let lastMatchStart = -1 - - const feed = (code: number): void => { - // ASCII fold only: the signature is ASCII, and no non-ASCII character can - // lowercase into a position that completes it (U+0130 always emits a - // combining mark right after its 'i', and the signature does not end in 'i'). - const folded = code >= 0x41 && code <= 0x5a ? code + 0x20 : code - while (matchedLength > 0 && SIGNATURE_CODES[matchedLength] !== folded) { - matchedLength = SIGNATURE_FAILURE[matchedLength - 1] - } - if (SIGNATURE_CODES[matchedLength] === folded) { - matchedLength += 1 - } - emitted += 1 - if (matchedLength === SIGNATURE_CODES.length) { - lastMatchStart = emitted - SIGNATURE_CODES.length - matchedLength = SIGNATURE_FAILURE[matchedLength - 1] - } - } - - const emitRange = (source: string, start: number, end: number): void => { - for (let index = start; index < end; index++) { - const code = source.charCodeAt(index) - if (code === CARRIAGE_RETURN) { - continue - } - feed(code) - } - } - - /** An escape prefix that never became a sequence stays in the output as literal text. */ - const flushPendingEscape = (chunk: string, end: number): void => { - if (carriedEscape.length > 0) { - emitRange(carriedEscape, 0, carriedEscape.length) - carriedEscape = '' - } - emitRange(chunk, pendingStart, end) - } - + let observedChars = 0 return { observe(chunk: string): string | null { if (!armed) { return null } - pendingStart = 0 - for (let index = 0; index < chunk.length; index++) { - const code = chunk.charCodeAt(index) - // Re-dispatches the same character once when a state falls back to text. - for (;;) { - if (state === STATE_TEXT) { - if (code === ESC) { - state = STATE_ESCAPE - pendingStart = index - } else if (code !== CARRIAGE_RETURN) { - feed(code) - } - break - } - if (state === STATE_ESCAPE) { - if (code === LEFT_BRACKET) { - state = STATE_CSI_PARAMS - break - } - if (code === RIGHT_BRACKET) { - // An OSC always matches (its terminator is optional), so nothing is re-emitted. - carriedEscape = '' - state = STATE_OSC - break - } - flushPendingEscape(chunk, index) - state = STATE_TEXT - continue - } - if (state === STATE_CSI_PARAMS || state === STATE_CSI_INTERMEDIATE) { - if (state === STATE_CSI_PARAMS && isCsiParameterByte(code)) { - break - } - if (isCsiIntermediateByte(code)) { - state = STATE_CSI_INTERMEDIATE - break - } - if (isCsiFinalByte(code)) { - carriedEscape = '' - state = STATE_TEXT - break - } - flushPendingEscape(chunk, index) - state = STATE_TEXT - continue - } - if (state === STATE_OSC) { - if (code === BEL) { - state = STATE_TEXT - } else if (code === ESC) { - state = STATE_OSC_ESCAPE - } - break - } - // STATE_OSC_ESCAPE - if (code === BACKSLASH) { - state = STATE_TEXT - break - } - // The OSC match ended before this ESC, which the scanner then re-reads. - state = STATE_ESCAPE - pendingStart = index - 1 - continue + const normalized = (tail + chunk).replace(ANSI_ESCAPE_PATTERN, '').replace(/\r/g, '') + tail = normalized.slice(-DETECTOR_BUFFER_MAX_CHARS) + observedChars += chunk.length + if (!tail.toLowerCase().includes(CODEX_BACKFILL_TIMEOUT_SIGNATURE)) { + if (observedChars >= CODEX_BACKFILL_SCAN_BUDGET_CHARS) { + armed = false + tail = '' } - } - let pendingLength = 0 - if (state === STATE_OSC) { - // An unterminated OSC is consumed to end of input, exactly as the regex did. - state = STATE_TEXT - } else if (state === STATE_OSC_ESCAPE) { - // The OSC match ends before the trailing ESC, which survives into the window. - state = STATE_ESCAPE - carriedEscape = '\x1b' - pendingLength = 1 - } else if (state !== STATE_TEXT) { - carriedEscape = `${carriedEscape}${chunk.slice(pendingStart)}` - if (carriedEscape.length > DETECTOR_BUFFER_MAX_CHARS) { - // A parameter run this long can never complete inside the retained - // window; treat it as the literal text it will be scanned as. - emitRange(carriedEscape, 0, carriedEscape.length) - carriedEscape = '' - state = STATE_TEXT - } else { - pendingLength = carriedEscape.length - } - } - const normalizedLength = emitted + pendingLength - if (lastMatchStart < 0 || lastMatchStart < normalizedLength - DETECTOR_BUFFER_MAX_CHARS) { return null } armed = false + tail = '' return CODEX_BACKFILL_RECOVERY_NOTICE } } diff --git a/src/renderer/src/lib/pane-manager/terminal-complex-script.test.ts b/src/renderer/src/lib/pane-manager/terminal-complex-script.test.ts index 46c23026ca2..044e9201d29 100644 --- a/src/renderer/src/lib/pane-manager/terminal-complex-script.test.ts +++ b/src/renderer/src/lib/pane-manager/terminal-complex-script.test.ts @@ -522,6 +522,11 @@ describe('terminalOutputPrefersRenderRefresh single-pass classification', () => '\x1b[38;5;3;44mfg then bg\x1b[0m', '\x1b[38;2;1;2;3;41mfg truecolor then bg\x1b[0m', '\x1b[38:5:9mcolon fg\x1b[0m', + // 38;2 consumes four parameters, so this 44 is the blue component, not a background. + '\x1b[38;2;1;2;44m', + // The colon form does not consume trailing parameters, so this 44 is a background. + '\x1b[38:5:1;44m', + '\x1b[38:2::1:2:3;44m', '\x1b[m', '\x1b[;m', '\x1b[38m', @@ -547,6 +552,8 @@ describe('terminalOutputPrefersRenderRefresh single-pass classification', () => 'replacement \ufffd', 'lone surrogate \ud83d', 'astral tail \u{2f81a}', + // An astral code point outside every risk range must advance past both halves. + 'math bold \u{1d400} x', '\x1b[41m\u{1f680}', '\u{1f680}\x1b[41m' ]