perf(terminal): bound the Codex backfill scan instead of rewriting it

The streaming detector was 40-70% slower in CPU than the regex it replaced
and had a pinned semantic difference for aborted escapes. Codex prints the
backfill timeout from its state-db startup gate, before the TUI draws, so the
signature can only appear in a pane's first few KB; the original detector is
restored verbatim and simply disarms after 256 KB of raw output. This is
faster than either version (66 ms -> 3.3 ms over the 23.5 MB corpus) and
removes 95% of the allocation, with no scanner rewrite on the hot path.

Also pins three classifier edge cases the corpus tests did not bite on:
38;2 parameter skip width, the 38:… colon guard, and astral advance.
This commit is contained in:
Neil
2026-09-02 14:31:00 -07:00
parent 84d73d9a82
commit 4ac6395caf
3 changed files with 46 additions and 346 deletions
@@ -1,64 +1,10 @@
import { describe, expect, it } from 'vitest'
import {
CODEX_BACKFILL_RECOVERY_NOTICE,
CODEX_BACKFILL_TIMEOUT_SIGNATURE,
CODEX_BACKFILL_SCAN_BUDGET_CHARS,
createCodexBackfillErrorDetector
} from './codex-backfill-error-detector'
const ANSI_ESCAPE_PATTERN =
// eslint-disable-next-line no-control-regex -- terminal escape sequences contain control bytes
/\u001b(?:\[[0-9;?]*[ -/]*[@-~]|\][^\u0007\u001b]*(?:\u0007|\u001b\\)?)/g
const DETECTOR_BUFFER_MAX_CHARS = 4096
/** The pre-optimization detector, kept verbatim as the differential oracle. */
function createReferenceDetector(): { observe(chunk: string): string | null } {
let tail = ''
let armed = true
return {
observe(chunk: string): string | null {
if (!armed) {
return null
}
const normalized = (tail + chunk).replace(ANSI_ESCAPE_PATTERN, '').replace(/\r/g, '')
tail = normalized.slice(-DETECTOR_BUFFER_MAX_CHARS)
if (!tail.toLowerCase().includes(CODEX_BACKFILL_TIMEOUT_SIGNATURE)) {
return null
}
armed = false
return CODEX_BACKFILL_RECOVERY_NOTICE
}
}
}
function splitAt(text: string, offsets: readonly number[]): string[] {
const chunks: string[] = []
let previous = 0
for (const offset of offsets) {
chunks.push(text.slice(previous, offset))
previous = offset
}
chunks.push(text.slice(previous))
return chunks
}
function observeAll(
detector: { observe(chunk: string): string | null },
chunks: readonly string[]
): (string | null)[] {
return chunks.map((chunk) => detector.observe(chunk))
}
function expectMatchesReference(chunks: readonly string[]): (string | null)[] {
const actual = observeAll(createCodexBackfillErrorDetector(), chunks)
const expected = observeAll(createReferenceDetector(), chunks)
expect(actual).toEqual(expected)
return actual
}
const ESC = String.fromCharCode(0x1b)
const SIGNATURE_LINE = '\u001b[31mError: timed out waiting for state DB backfill\u001b[0m\r\n'
describe('Codex backfill error detector', () => {
it('recognizes the timeout across ANSI-decorated chunks once', () => {
const detector = createCodexBackfillErrorDetector()
@@ -74,103 +20,28 @@ describe('Codex backfill error detector', () => {
expect(detector.observe('local database appears to be damaged')).toBeNull()
})
it('detects the signature at every chunk-boundary split, including inside escapes', () => {
for (let offset = 0; offset <= SIGNATURE_LINE.length; offset++) {
const results = expectMatchesReference(splitAt(SIGNATURE_LINE, [offset]))
expect(results.some((result) => result === CODEX_BACKFILL_RECOVERY_NOTICE)).toBe(true)
}
})
it('detects a signature interleaved with escape sequences at every split', () => {
const interleaved =
'timed \u001b[1mout \u001b[0;32mwaiting for state \u001b[Kdb \u001b[38;5;214mbackfill\u001b[0m\r\n'
for (let offset = 0; offset <= interleaved.length; offset++) {
const results = expectMatchesReference(splitAt(interleaved, [offset]))
expect(results.some((result) => result === CODEX_BACKFILL_RECOVERY_NOTICE)).toBe(true)
}
})
it('keeps a literal escape that never completes breaking the signature', () => {
// The unmatched ESC stays in the normalized stream, so the halves do not join.
expectMatchesReference(['timed out waiting for state db back\u001b', 'fill\n'])
it('still fires when the signature lands inside the scan budget', () => {
const detector = createCodexBackfillErrorDetector()
expect(detector.observe('timed out waiting for state db back\u001b')).toBeNull()
expect(detector.observe('fill\n')).toBeNull()
const filler = 'x'.repeat(1024)
const chunksBeforeSignature = Math.floor(CODEX_BACKFILL_SCAN_BUDGET_CHARS / filler.length) - 1
for (let index = 0; index < chunksBeforeSignature; index++) {
expect(detector.observe(filler)).toBeNull()
}
expect(detector.observe('Error: timed out waiting for state DB backfill\r\n')).toBe(
CODEX_BACKFILL_RECOVERY_NOTICE
)
})
it('ignores a signature pushed out of the trailing normalized window by one chunk', () => {
const chunk = `${CODEX_BACKFILL_TIMEOUT_SIGNATURE}${'x'.repeat(DETECTOR_BUFFER_MAX_CHARS)}`
expect(createCodexBackfillErrorDetector().observe(chunk)).toBeNull()
expect(createReferenceDetector().observe(chunk)).toBeNull()
})
it('still fires when the signature ends inside the trailing window', () => {
const padding = DETECTOR_BUFFER_MAX_CHARS - CODEX_BACKFILL_TIMEOUT_SIGNATURE.length
const chunk = `${CODEX_BACKFILL_TIMEOUT_SIGNATURE}${'x'.repeat(padding)}`
expectMatchesReference([chunk])
expect(createCodexBackfillErrorDetector().observe(chunk)).toBe(CODEX_BACKFILL_RECOVERY_NOTICE)
})
it('reads an aborted escape sequence once, the way a terminal parser does', () => {
// Documented difference from the pre-optimization detector: that one re-stripped
// its own retained output every chunk, so an aborted CSI prefix could fuse with a
// byte that arrived in a LATER chunk and swallow it. Reaching this needs an escape
// sequence aborted mid-sequence by another ESC, which no terminal renders as text.
const aborted = `${ESC}[31${ESC}[0mtimed out waiting for state d`
it('disarms once the budget is spent so a later signature is ignored', () => {
const detector = createCodexBackfillErrorDetector()
expect(detector.observe(aborted)).toBeNull()
expect(detector.observe('b backfill')).toBe(CODEX_BACKFILL_RECOVERY_NOTICE)
const reference = createReferenceDetector()
expect(reference.observe(aborted)).toBeNull()
// The reference fuses the retained `ESC [ 3 1` with the following 't' on its next
// pass and eats the signature's first character.
expect(reference.observe('b backfill')).toBeNull()
expect(detector.observe('x'.repeat(CODEX_BACKFILL_SCAN_BUDGET_CHARS))).toBeNull()
expect(detector.observe('Error: timed out waiting for state DB backfill\r\n')).toBeNull()
})
it('matches the reference detector over randomized agent-like output', () => {
const alphabet = [
'timed out waiting for state db backfill',
'Timed Out Waiting For State DB Backfill',
'timed out waiting for state db back',
'fill',
'\u001b[0m',
'\u001b[38;5;214m',
'\u001b[?25l',
'\u001b[2K',
'\u001b]0;codex build\u0007',
'\u001b]8;;https://example.com\u001b\\',
'\u001b]0;unterminated',
'\u001b',
'',
'\u001bM',
'\u001b[<0;1;2M',
'\r',
'\r\n',
'\n',
'working spinner ',
'local database appears to be damaged',
'x'.repeat(97),
'timed out\u001b[0m waiting for state db backfill'
]
let seed = 0x2f6e2b1
const random = (): number => {
seed = (seed * 1103515245 + 12345) & 0x7fffffff
return seed / 0x7fffffff
}
for (let iteration = 0; iteration < 400; iteration++) {
let text = ''
const pieces = 1 + Math.floor(random() * 14)
for (let piece = 0; piece < pieces; piece++) {
text += alphabet[Math.floor(random() * alphabet.length)] ?? ''
}
const offsets: number[] = []
const splits = Math.floor(random() * 5)
for (let split = 0; split < splits; split++) {
offsets.push(Math.floor(random() * (text.length + 1)))
}
offsets.sort((a, b) => a - b)
expectMatchesReference(splitAt(text, offsets))
}
it('charges raw chunk length, so escape-heavy output cannot extend the scan', () => {
const detector = createCodexBackfillErrorDetector()
const escapes = '\u001b[0m'.repeat(CODEX_BACKFILL_SCAN_BUDGET_CHARS / 4)
expect(detector.observe(escapes)).toBeNull()
expect(detector.observe('timed out waiting for state db backfill')).toBeNull()
})
})
@@ -5,220 +5,42 @@ export const CODEX_BACKFILL_RECOVERY_NOTICE = [
'Keep Orca open for a few minutes, then retry this pane. Orca attempts background recovery for managed local and WSL homes.'
].join('\n')
/**
* Width of the normalized-output window the signature must land inside. Kept
* from the original `normalized.slice(-4096)` detector so a signature followed
* by more than this many normalized characters in one chunk still does not fire.
*/
const ANSI_ESCAPE_PATTERN =
// eslint-disable-next-line no-control-regex -- terminal escape sequences contain control bytes
/\u001b(?:\[[0-9;?]*[ -/]*[@-~]|\][^\u0007\u001b]*(?:\u0007|\u001b\\)?)/g
const DETECTOR_BUFFER_MAX_CHARS = 4096
const SIGNATURE_CODES = Uint16Array.from(CODEX_BACKFILL_TIMEOUT_SIGNATURE, (character) =>
character.charCodeAt(0)
)
/** KMP prefix table so the streaming match needs no per-chunk buffer or backtracking. */
function buildSignatureFailureTable(codes: Uint16Array): Int32Array {
const failure = new Int32Array(codes.length)
let matched = 0
for (let index = 1; index < codes.length; index++) {
while (matched > 0 && codes[index] !== codes[matched]) {
matched = failure[matched - 1]
}
if (codes[index] === codes[matched]) {
matched += 1
}
failure[index] = matched
}
return failure
}
const SIGNATURE_FAILURE = buildSignatureFailureTable(SIGNATURE_CODES)
const STATE_TEXT = 0
const STATE_ESCAPE = 1
const STATE_CSI_PARAMS = 2
const STATE_CSI_INTERMEDIATE = 3
const STATE_OSC = 4
const STATE_OSC_ESCAPE = 5
const ESC = 0x1b
const CARRIAGE_RETURN = 0x0d
const BEL = 0x07
const LEFT_BRACKET = 0x5b
const RIGHT_BRACKET = 0x5d
const BACKSLASH = 0x5c
function isCsiParameterByte(code: number): boolean {
// Mirrors the old [0-9;?] class exactly — ':' and '<' '=' '>' are deliberately excluded.
return (code >= 0x30 && code <= 0x39) || code === 0x3b || code === 0x3f
}
function isCsiIntermediateByte(code: number): boolean {
return code >= 0x20 && code <= 0x2f
}
function isCsiFinalByte(code: number): boolean {
return code >= 0x40 && code <= 0x7e
}
/**
* Raw output after which the scan disarms. Codex prints the timeout from its
* state-db startup gate, before the TUI ever draws, so it can only appear in
* the first few KB of a pane's output; without a budget every later chunk is
* stripped and copied for the pane's whole life.
*/
export const CODEX_BACKFILL_SCAN_BUDGET_CHARS = 256 * 1024
export type CodexBackfillErrorDetector = { observe(chunk: string): string | null }
/**
* Scans one Codex pane's output once for the unambiguous backfill timeout.
*
* Streams the chunk through an ANSI/CR state machine feeding a KMP matcher, so
* no normalized copy of the output is ever built. Detection is identical to the
* previous `strip -> slice(-4096) -> toLowerCase -> includes` implementation:
* escape sequences and carriage returns are removed with the same grammar, a
* sequence split across chunks is still recognized, an escape that never
* completes is still emitted as literal text (and so still breaks a match), and
* the signature must still fall inside the trailing 4096-character window.
*/
/** Scans the start of one Codex pane's output once for the unambiguous backfill timeout. */
export function createCodexBackfillErrorDetector(): CodexBackfillErrorDetector {
let tail = ''
let armed = true
let state: number = STATE_TEXT
/** Escape prefix carried from earlier chunks; re-emitted verbatim if the sequence never completes. */
let carriedEscape = ''
/** Index in the current chunk where the pending escape prefix starts (0 when it began earlier). */
let pendingStart = 0
/** Characters emitted into the normalized stream so far — the stream coordinate space. */
let emitted = 0
let matchedLength = 0
let lastMatchStart = -1
const feed = (code: number): void => {
// ASCII fold only: the signature is ASCII, and no non-ASCII character can
// lowercase into a position that completes it (U+0130 always emits a
// combining mark right after its 'i', and the signature does not end in 'i').
const folded = code >= 0x41 && code <= 0x5a ? code + 0x20 : code
while (matchedLength > 0 && SIGNATURE_CODES[matchedLength] !== folded) {
matchedLength = SIGNATURE_FAILURE[matchedLength - 1]
}
if (SIGNATURE_CODES[matchedLength] === folded) {
matchedLength += 1
}
emitted += 1
if (matchedLength === SIGNATURE_CODES.length) {
lastMatchStart = emitted - SIGNATURE_CODES.length
matchedLength = SIGNATURE_FAILURE[matchedLength - 1]
}
}
const emitRange = (source: string, start: number, end: number): void => {
for (let index = start; index < end; index++) {
const code = source.charCodeAt(index)
if (code === CARRIAGE_RETURN) {
continue
}
feed(code)
}
}
/** An escape prefix that never became a sequence stays in the output as literal text. */
const flushPendingEscape = (chunk: string, end: number): void => {
if (carriedEscape.length > 0) {
emitRange(carriedEscape, 0, carriedEscape.length)
carriedEscape = ''
}
emitRange(chunk, pendingStart, end)
}
let observedChars = 0
return {
observe(chunk: string): string | null {
if (!armed) {
return null
}
pendingStart = 0
for (let index = 0; index < chunk.length; index++) {
const code = chunk.charCodeAt(index)
// Re-dispatches the same character once when a state falls back to text.
for (;;) {
if (state === STATE_TEXT) {
if (code === ESC) {
state = STATE_ESCAPE
pendingStart = index
} else if (code !== CARRIAGE_RETURN) {
feed(code)
}
break
}
if (state === STATE_ESCAPE) {
if (code === LEFT_BRACKET) {
state = STATE_CSI_PARAMS
break
}
if (code === RIGHT_BRACKET) {
// An OSC always matches (its terminator is optional), so nothing is re-emitted.
carriedEscape = ''
state = STATE_OSC
break
}
flushPendingEscape(chunk, index)
state = STATE_TEXT
continue
}
if (state === STATE_CSI_PARAMS || state === STATE_CSI_INTERMEDIATE) {
if (state === STATE_CSI_PARAMS && isCsiParameterByte(code)) {
break
}
if (isCsiIntermediateByte(code)) {
state = STATE_CSI_INTERMEDIATE
break
}
if (isCsiFinalByte(code)) {
carriedEscape = ''
state = STATE_TEXT
break
}
flushPendingEscape(chunk, index)
state = STATE_TEXT
continue
}
if (state === STATE_OSC) {
if (code === BEL) {
state = STATE_TEXT
} else if (code === ESC) {
state = STATE_OSC_ESCAPE
}
break
}
// STATE_OSC_ESCAPE
if (code === BACKSLASH) {
state = STATE_TEXT
break
}
// The OSC match ended before this ESC, which the scanner then re-reads.
state = STATE_ESCAPE
pendingStart = index - 1
continue
const normalized = (tail + chunk).replace(ANSI_ESCAPE_PATTERN, '').replace(/\r/g, '')
tail = normalized.slice(-DETECTOR_BUFFER_MAX_CHARS)
observedChars += chunk.length
if (!tail.toLowerCase().includes(CODEX_BACKFILL_TIMEOUT_SIGNATURE)) {
if (observedChars >= CODEX_BACKFILL_SCAN_BUDGET_CHARS) {
armed = false
tail = ''
}
}
let pendingLength = 0
if (state === STATE_OSC) {
// An unterminated OSC is consumed to end of input, exactly as the regex did.
state = STATE_TEXT
} else if (state === STATE_OSC_ESCAPE) {
// The OSC match ends before the trailing ESC, which survives into the window.
state = STATE_ESCAPE
carriedEscape = '\x1b'
pendingLength = 1
} else if (state !== STATE_TEXT) {
carriedEscape = `${carriedEscape}${chunk.slice(pendingStart)}`
if (carriedEscape.length > DETECTOR_BUFFER_MAX_CHARS) {
// A parameter run this long can never complete inside the retained
// window; treat it as the literal text it will be scanned as.
emitRange(carriedEscape, 0, carriedEscape.length)
carriedEscape = ''
state = STATE_TEXT
} else {
pendingLength = carriedEscape.length
}
}
const normalizedLength = emitted + pendingLength
if (lastMatchStart < 0 || lastMatchStart < normalizedLength - DETECTOR_BUFFER_MAX_CHARS) {
return null
}
armed = false
tail = ''
return CODEX_BACKFILL_RECOVERY_NOTICE
}
}
@@ -522,6 +522,11 @@ describe('terminalOutputPrefersRenderRefresh single-pass classification', () =>
'\x1b[38;5;3;44mfg then bg\x1b[0m',
'\x1b[38;2;1;2;3;41mfg truecolor then bg\x1b[0m',
'\x1b[38:5:9mcolon fg\x1b[0m',
// 38;2 consumes four parameters, so this 44 is the blue component, not a background.
'\x1b[38;2;1;2;44m',
// The colon form does not consume trailing parameters, so this 44 is a background.
'\x1b[38:5:1;44m',
'\x1b[38:2::1:2:3;44m',
'\x1b[m',
'\x1b[;m',
'\x1b[38m',
@@ -547,6 +552,8 @@ describe('terminalOutputPrefersRenderRefresh single-pass classification', () =>
'replacement \ufffd',
'lone surrogate \ud83d',
'astral tail \u{2f81a}',
// An astral code point outside every risk range must advance past both halves.
'math bold \u{1d400} x',
'\x1b[41m\u{1f680}',
'\u{1f680}\x1b[41m'
]