mirror of
https://github.com/stablyai/orca.git
synced 2026-10-02 00:02:05 +00:00
perf(terminal): bound the Codex backfill scan instead of rewriting it
The streaming detector was 40-70% slower in CPU than the regex it replaced and had a pinned semantic difference for aborted escapes. Codex prints the backfill timeout from its state-db startup gate, before the TUI draws, so the signature can only appear in a pane's first few KB; the original detector is restored verbatim and simply disarms after 256 KB of raw output. This is faster than either version (66 ms -> 3.3 ms over the 23.5 MB corpus) and removes 95% of the allocation, with no scanner rewrite on the hot path. Also pins three classifier edge cases the corpus tests did not bite on: 38;2 parameter skip width, the 38:… colon guard, and astral advance.
This commit is contained in:
@@ -1,64 +1,10 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import {
|
||||
CODEX_BACKFILL_RECOVERY_NOTICE,
|
||||
CODEX_BACKFILL_TIMEOUT_SIGNATURE,
|
||||
CODEX_BACKFILL_SCAN_BUDGET_CHARS,
|
||||
createCodexBackfillErrorDetector
|
||||
} from './codex-backfill-error-detector'
|
||||
|
||||
const ANSI_ESCAPE_PATTERN =
|
||||
// eslint-disable-next-line no-control-regex -- terminal escape sequences contain control bytes
|
||||
/\u001b(?:\[[0-9;?]*[ -/]*[@-~]|\][^\u0007\u001b]*(?:\u0007|\u001b\\)?)/g
|
||||
const DETECTOR_BUFFER_MAX_CHARS = 4096
|
||||
|
||||
/** The pre-optimization detector, kept verbatim as the differential oracle. */
|
||||
function createReferenceDetector(): { observe(chunk: string): string | null } {
|
||||
let tail = ''
|
||||
let armed = true
|
||||
return {
|
||||
observe(chunk: string): string | null {
|
||||
if (!armed) {
|
||||
return null
|
||||
}
|
||||
const normalized = (tail + chunk).replace(ANSI_ESCAPE_PATTERN, '').replace(/\r/g, '')
|
||||
tail = normalized.slice(-DETECTOR_BUFFER_MAX_CHARS)
|
||||
if (!tail.toLowerCase().includes(CODEX_BACKFILL_TIMEOUT_SIGNATURE)) {
|
||||
return null
|
||||
}
|
||||
armed = false
|
||||
return CODEX_BACKFILL_RECOVERY_NOTICE
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function splitAt(text: string, offsets: readonly number[]): string[] {
|
||||
const chunks: string[] = []
|
||||
let previous = 0
|
||||
for (const offset of offsets) {
|
||||
chunks.push(text.slice(previous, offset))
|
||||
previous = offset
|
||||
}
|
||||
chunks.push(text.slice(previous))
|
||||
return chunks
|
||||
}
|
||||
|
||||
function observeAll(
|
||||
detector: { observe(chunk: string): string | null },
|
||||
chunks: readonly string[]
|
||||
): (string | null)[] {
|
||||
return chunks.map((chunk) => detector.observe(chunk))
|
||||
}
|
||||
|
||||
function expectMatchesReference(chunks: readonly string[]): (string | null)[] {
|
||||
const actual = observeAll(createCodexBackfillErrorDetector(), chunks)
|
||||
const expected = observeAll(createReferenceDetector(), chunks)
|
||||
expect(actual).toEqual(expected)
|
||||
return actual
|
||||
}
|
||||
|
||||
const ESC = String.fromCharCode(0x1b)
|
||||
|
||||
const SIGNATURE_LINE = '\u001b[31mError: timed out waiting for state DB backfill\u001b[0m\r\n'
|
||||
|
||||
describe('Codex backfill error detector', () => {
|
||||
it('recognizes the timeout across ANSI-decorated chunks once', () => {
|
||||
const detector = createCodexBackfillErrorDetector()
|
||||
@@ -74,103 +20,28 @@ describe('Codex backfill error detector', () => {
|
||||
expect(detector.observe('local database appears to be damaged')).toBeNull()
|
||||
})
|
||||
|
||||
it('detects the signature at every chunk-boundary split, including inside escapes', () => {
|
||||
for (let offset = 0; offset <= SIGNATURE_LINE.length; offset++) {
|
||||
const results = expectMatchesReference(splitAt(SIGNATURE_LINE, [offset]))
|
||||
expect(results.some((result) => result === CODEX_BACKFILL_RECOVERY_NOTICE)).toBe(true)
|
||||
}
|
||||
})
|
||||
|
||||
it('detects a signature interleaved with escape sequences at every split', () => {
|
||||
const interleaved =
|
||||
'timed \u001b[1mout \u001b[0;32mwaiting for state \u001b[Kdb \u001b[38;5;214mbackfill\u001b[0m\r\n'
|
||||
for (let offset = 0; offset <= interleaved.length; offset++) {
|
||||
const results = expectMatchesReference(splitAt(interleaved, [offset]))
|
||||
expect(results.some((result) => result === CODEX_BACKFILL_RECOVERY_NOTICE)).toBe(true)
|
||||
}
|
||||
})
|
||||
|
||||
it('keeps a literal escape that never completes breaking the signature', () => {
|
||||
// The unmatched ESC stays in the normalized stream, so the halves do not join.
|
||||
expectMatchesReference(['timed out waiting for state db back\u001b', 'fill\n'])
|
||||
it('still fires when the signature lands inside the scan budget', () => {
|
||||
const detector = createCodexBackfillErrorDetector()
|
||||
expect(detector.observe('timed out waiting for state db back\u001b')).toBeNull()
|
||||
expect(detector.observe('fill\n')).toBeNull()
|
||||
const filler = 'x'.repeat(1024)
|
||||
const chunksBeforeSignature = Math.floor(CODEX_BACKFILL_SCAN_BUDGET_CHARS / filler.length) - 1
|
||||
for (let index = 0; index < chunksBeforeSignature; index++) {
|
||||
expect(detector.observe(filler)).toBeNull()
|
||||
}
|
||||
expect(detector.observe('Error: timed out waiting for state DB backfill\r\n')).toBe(
|
||||
CODEX_BACKFILL_RECOVERY_NOTICE
|
||||
)
|
||||
})
|
||||
|
||||
it('ignores a signature pushed out of the trailing normalized window by one chunk', () => {
|
||||
const chunk = `${CODEX_BACKFILL_TIMEOUT_SIGNATURE}${'x'.repeat(DETECTOR_BUFFER_MAX_CHARS)}`
|
||||
expect(createCodexBackfillErrorDetector().observe(chunk)).toBeNull()
|
||||
expect(createReferenceDetector().observe(chunk)).toBeNull()
|
||||
})
|
||||
|
||||
it('still fires when the signature ends inside the trailing window', () => {
|
||||
const padding = DETECTOR_BUFFER_MAX_CHARS - CODEX_BACKFILL_TIMEOUT_SIGNATURE.length
|
||||
const chunk = `${CODEX_BACKFILL_TIMEOUT_SIGNATURE}${'x'.repeat(padding)}`
|
||||
expectMatchesReference([chunk])
|
||||
expect(createCodexBackfillErrorDetector().observe(chunk)).toBe(CODEX_BACKFILL_RECOVERY_NOTICE)
|
||||
})
|
||||
|
||||
it('reads an aborted escape sequence once, the way a terminal parser does', () => {
|
||||
// Documented difference from the pre-optimization detector: that one re-stripped
|
||||
// its own retained output every chunk, so an aborted CSI prefix could fuse with a
|
||||
// byte that arrived in a LATER chunk and swallow it. Reaching this needs an escape
|
||||
// sequence aborted mid-sequence by another ESC, which no terminal renders as text.
|
||||
const aborted = `${ESC}[31${ESC}[0mtimed out waiting for state d`
|
||||
it('disarms once the budget is spent so a later signature is ignored', () => {
|
||||
const detector = createCodexBackfillErrorDetector()
|
||||
expect(detector.observe(aborted)).toBeNull()
|
||||
expect(detector.observe('b backfill')).toBe(CODEX_BACKFILL_RECOVERY_NOTICE)
|
||||
|
||||
const reference = createReferenceDetector()
|
||||
expect(reference.observe(aborted)).toBeNull()
|
||||
// The reference fuses the retained `ESC [ 3 1` with the following 't' on its next
|
||||
// pass and eats the signature's first character.
|
||||
expect(reference.observe('b backfill')).toBeNull()
|
||||
expect(detector.observe('x'.repeat(CODEX_BACKFILL_SCAN_BUDGET_CHARS))).toBeNull()
|
||||
expect(detector.observe('Error: timed out waiting for state DB backfill\r\n')).toBeNull()
|
||||
})
|
||||
|
||||
it('matches the reference detector over randomized agent-like output', () => {
|
||||
const alphabet = [
|
||||
'timed out waiting for state db backfill',
|
||||
'Timed Out Waiting For State DB Backfill',
|
||||
'timed out waiting for state db back',
|
||||
'fill',
|
||||
'\u001b[0m',
|
||||
'\u001b[38;5;214m',
|
||||
'\u001b[?25l',
|
||||
'\u001b[2K',
|
||||
'\u001b]0;codex build\u0007',
|
||||
'\u001b]8;;https://example.com\u001b\\',
|
||||
'\u001b]0;unterminated',
|
||||
'\u001b',
|
||||
'',
|
||||
'\u001bM',
|
||||
'\u001b[<0;1;2M',
|
||||
'\r',
|
||||
'\r\n',
|
||||
'\n',
|
||||
'working spinner ',
|
||||
'local database appears to be damaged',
|
||||
'x'.repeat(97),
|
||||
'timed out\u001b[0m waiting for state db backfill'
|
||||
]
|
||||
let seed = 0x2f6e2b1
|
||||
const random = (): number => {
|
||||
seed = (seed * 1103515245 + 12345) & 0x7fffffff
|
||||
return seed / 0x7fffffff
|
||||
}
|
||||
for (let iteration = 0; iteration < 400; iteration++) {
|
||||
let text = ''
|
||||
const pieces = 1 + Math.floor(random() * 14)
|
||||
for (let piece = 0; piece < pieces; piece++) {
|
||||
text += alphabet[Math.floor(random() * alphabet.length)] ?? ''
|
||||
}
|
||||
const offsets: number[] = []
|
||||
const splits = Math.floor(random() * 5)
|
||||
for (let split = 0; split < splits; split++) {
|
||||
offsets.push(Math.floor(random() * (text.length + 1)))
|
||||
}
|
||||
offsets.sort((a, b) => a - b)
|
||||
expectMatchesReference(splitAt(text, offsets))
|
||||
}
|
||||
it('charges raw chunk length, so escape-heavy output cannot extend the scan', () => {
|
||||
const detector = createCodexBackfillErrorDetector()
|
||||
const escapes = '\u001b[0m'.repeat(CODEX_BACKFILL_SCAN_BUDGET_CHARS / 4)
|
||||
expect(detector.observe(escapes)).toBeNull()
|
||||
expect(detector.observe('timed out waiting for state db backfill')).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
@@ -5,220 +5,42 @@ export const CODEX_BACKFILL_RECOVERY_NOTICE = [
|
||||
'Keep Orca open for a few minutes, then retry this pane. Orca attempts background recovery for managed local and WSL homes.'
|
||||
].join('\n')
|
||||
|
||||
/**
|
||||
* Width of the normalized-output window the signature must land inside. Kept
|
||||
* from the original `normalized.slice(-4096)` detector so a signature followed
|
||||
* by more than this many normalized characters in one chunk still does not fire.
|
||||
*/
|
||||
const ANSI_ESCAPE_PATTERN =
|
||||
// eslint-disable-next-line no-control-regex -- terminal escape sequences contain control bytes
|
||||
/\u001b(?:\[[0-9;?]*[ -/]*[@-~]|\][^\u0007\u001b]*(?:\u0007|\u001b\\)?)/g
|
||||
const DETECTOR_BUFFER_MAX_CHARS = 4096
|
||||
|
||||
const SIGNATURE_CODES = Uint16Array.from(CODEX_BACKFILL_TIMEOUT_SIGNATURE, (character) =>
|
||||
character.charCodeAt(0)
|
||||
)
|
||||
|
||||
/** KMP prefix table so the streaming match needs no per-chunk buffer or backtracking. */
|
||||
function buildSignatureFailureTable(codes: Uint16Array): Int32Array {
|
||||
const failure = new Int32Array(codes.length)
|
||||
let matched = 0
|
||||
for (let index = 1; index < codes.length; index++) {
|
||||
while (matched > 0 && codes[index] !== codes[matched]) {
|
||||
matched = failure[matched - 1]
|
||||
}
|
||||
if (codes[index] === codes[matched]) {
|
||||
matched += 1
|
||||
}
|
||||
failure[index] = matched
|
||||
}
|
||||
return failure
|
||||
}
|
||||
|
||||
const SIGNATURE_FAILURE = buildSignatureFailureTable(SIGNATURE_CODES)
|
||||
|
||||
const STATE_TEXT = 0
|
||||
const STATE_ESCAPE = 1
|
||||
const STATE_CSI_PARAMS = 2
|
||||
const STATE_CSI_INTERMEDIATE = 3
|
||||
const STATE_OSC = 4
|
||||
const STATE_OSC_ESCAPE = 5
|
||||
|
||||
const ESC = 0x1b
|
||||
const CARRIAGE_RETURN = 0x0d
|
||||
const BEL = 0x07
|
||||
const LEFT_BRACKET = 0x5b
|
||||
const RIGHT_BRACKET = 0x5d
|
||||
const BACKSLASH = 0x5c
|
||||
|
||||
function isCsiParameterByte(code: number): boolean {
|
||||
// Mirrors the old [0-9;?] class exactly — ':' and '<' '=' '>' are deliberately excluded.
|
||||
return (code >= 0x30 && code <= 0x39) || code === 0x3b || code === 0x3f
|
||||
}
|
||||
|
||||
function isCsiIntermediateByte(code: number): boolean {
|
||||
return code >= 0x20 && code <= 0x2f
|
||||
}
|
||||
|
||||
function isCsiFinalByte(code: number): boolean {
|
||||
return code >= 0x40 && code <= 0x7e
|
||||
}
|
||||
/**
|
||||
* Raw output after which the scan disarms. Codex prints the timeout from its
|
||||
* state-db startup gate, before the TUI ever draws, so it can only appear in
|
||||
* the first few KB of a pane's output; without a budget every later chunk is
|
||||
* stripped and copied for the pane's whole life.
|
||||
*/
|
||||
export const CODEX_BACKFILL_SCAN_BUDGET_CHARS = 256 * 1024
|
||||
|
||||
export type CodexBackfillErrorDetector = { observe(chunk: string): string | null }
|
||||
|
||||
/**
|
||||
* Scans one Codex pane's output once for the unambiguous backfill timeout.
|
||||
*
|
||||
* Streams the chunk through an ANSI/CR state machine feeding a KMP matcher, so
|
||||
* no normalized copy of the output is ever built. Detection is identical to the
|
||||
* previous `strip -> slice(-4096) -> toLowerCase -> includes` implementation:
|
||||
* escape sequences and carriage returns are removed with the same grammar, a
|
||||
* sequence split across chunks is still recognized, an escape that never
|
||||
* completes is still emitted as literal text (and so still breaks a match), and
|
||||
* the signature must still fall inside the trailing 4096-character window.
|
||||
*/
|
||||
/** Scans the start of one Codex pane's output once for the unambiguous backfill timeout. */
|
||||
export function createCodexBackfillErrorDetector(): CodexBackfillErrorDetector {
|
||||
let tail = ''
|
||||
let armed = true
|
||||
let state: number = STATE_TEXT
|
||||
/** Escape prefix carried from earlier chunks; re-emitted verbatim if the sequence never completes. */
|
||||
let carriedEscape = ''
|
||||
/** Index in the current chunk where the pending escape prefix starts (0 when it began earlier). */
|
||||
let pendingStart = 0
|
||||
/** Characters emitted into the normalized stream so far — the stream coordinate space. */
|
||||
let emitted = 0
|
||||
let matchedLength = 0
|
||||
let lastMatchStart = -1
|
||||
|
||||
const feed = (code: number): void => {
|
||||
// ASCII fold only: the signature is ASCII, and no non-ASCII character can
|
||||
// lowercase into a position that completes it (U+0130 always emits a
|
||||
// combining mark right after its 'i', and the signature does not end in 'i').
|
||||
const folded = code >= 0x41 && code <= 0x5a ? code + 0x20 : code
|
||||
while (matchedLength > 0 && SIGNATURE_CODES[matchedLength] !== folded) {
|
||||
matchedLength = SIGNATURE_FAILURE[matchedLength - 1]
|
||||
}
|
||||
if (SIGNATURE_CODES[matchedLength] === folded) {
|
||||
matchedLength += 1
|
||||
}
|
||||
emitted += 1
|
||||
if (matchedLength === SIGNATURE_CODES.length) {
|
||||
lastMatchStart = emitted - SIGNATURE_CODES.length
|
||||
matchedLength = SIGNATURE_FAILURE[matchedLength - 1]
|
||||
}
|
||||
}
|
||||
|
||||
const emitRange = (source: string, start: number, end: number): void => {
|
||||
for (let index = start; index < end; index++) {
|
||||
const code = source.charCodeAt(index)
|
||||
if (code === CARRIAGE_RETURN) {
|
||||
continue
|
||||
}
|
||||
feed(code)
|
||||
}
|
||||
}
|
||||
|
||||
/** An escape prefix that never became a sequence stays in the output as literal text. */
|
||||
const flushPendingEscape = (chunk: string, end: number): void => {
|
||||
if (carriedEscape.length > 0) {
|
||||
emitRange(carriedEscape, 0, carriedEscape.length)
|
||||
carriedEscape = ''
|
||||
}
|
||||
emitRange(chunk, pendingStart, end)
|
||||
}
|
||||
|
||||
let observedChars = 0
|
||||
return {
|
||||
observe(chunk: string): string | null {
|
||||
if (!armed) {
|
||||
return null
|
||||
}
|
||||
pendingStart = 0
|
||||
for (let index = 0; index < chunk.length; index++) {
|
||||
const code = chunk.charCodeAt(index)
|
||||
// Re-dispatches the same character once when a state falls back to text.
|
||||
for (;;) {
|
||||
if (state === STATE_TEXT) {
|
||||
if (code === ESC) {
|
||||
state = STATE_ESCAPE
|
||||
pendingStart = index
|
||||
} else if (code !== CARRIAGE_RETURN) {
|
||||
feed(code)
|
||||
}
|
||||
break
|
||||
}
|
||||
if (state === STATE_ESCAPE) {
|
||||
if (code === LEFT_BRACKET) {
|
||||
state = STATE_CSI_PARAMS
|
||||
break
|
||||
}
|
||||
if (code === RIGHT_BRACKET) {
|
||||
// An OSC always matches (its terminator is optional), so nothing is re-emitted.
|
||||
carriedEscape = ''
|
||||
state = STATE_OSC
|
||||
break
|
||||
}
|
||||
flushPendingEscape(chunk, index)
|
||||
state = STATE_TEXT
|
||||
continue
|
||||
}
|
||||
if (state === STATE_CSI_PARAMS || state === STATE_CSI_INTERMEDIATE) {
|
||||
if (state === STATE_CSI_PARAMS && isCsiParameterByte(code)) {
|
||||
break
|
||||
}
|
||||
if (isCsiIntermediateByte(code)) {
|
||||
state = STATE_CSI_INTERMEDIATE
|
||||
break
|
||||
}
|
||||
if (isCsiFinalByte(code)) {
|
||||
carriedEscape = ''
|
||||
state = STATE_TEXT
|
||||
break
|
||||
}
|
||||
flushPendingEscape(chunk, index)
|
||||
state = STATE_TEXT
|
||||
continue
|
||||
}
|
||||
if (state === STATE_OSC) {
|
||||
if (code === BEL) {
|
||||
state = STATE_TEXT
|
||||
} else if (code === ESC) {
|
||||
state = STATE_OSC_ESCAPE
|
||||
}
|
||||
break
|
||||
}
|
||||
// STATE_OSC_ESCAPE
|
||||
if (code === BACKSLASH) {
|
||||
state = STATE_TEXT
|
||||
break
|
||||
}
|
||||
// The OSC match ended before this ESC, which the scanner then re-reads.
|
||||
state = STATE_ESCAPE
|
||||
pendingStart = index - 1
|
||||
continue
|
||||
const normalized = (tail + chunk).replace(ANSI_ESCAPE_PATTERN, '').replace(/\r/g, '')
|
||||
tail = normalized.slice(-DETECTOR_BUFFER_MAX_CHARS)
|
||||
observedChars += chunk.length
|
||||
if (!tail.toLowerCase().includes(CODEX_BACKFILL_TIMEOUT_SIGNATURE)) {
|
||||
if (observedChars >= CODEX_BACKFILL_SCAN_BUDGET_CHARS) {
|
||||
armed = false
|
||||
tail = ''
|
||||
}
|
||||
}
|
||||
let pendingLength = 0
|
||||
if (state === STATE_OSC) {
|
||||
// An unterminated OSC is consumed to end of input, exactly as the regex did.
|
||||
state = STATE_TEXT
|
||||
} else if (state === STATE_OSC_ESCAPE) {
|
||||
// The OSC match ends before the trailing ESC, which survives into the window.
|
||||
state = STATE_ESCAPE
|
||||
carriedEscape = '\x1b'
|
||||
pendingLength = 1
|
||||
} else if (state !== STATE_TEXT) {
|
||||
carriedEscape = `${carriedEscape}${chunk.slice(pendingStart)}`
|
||||
if (carriedEscape.length > DETECTOR_BUFFER_MAX_CHARS) {
|
||||
// A parameter run this long can never complete inside the retained
|
||||
// window; treat it as the literal text it will be scanned as.
|
||||
emitRange(carriedEscape, 0, carriedEscape.length)
|
||||
carriedEscape = ''
|
||||
state = STATE_TEXT
|
||||
} else {
|
||||
pendingLength = carriedEscape.length
|
||||
}
|
||||
}
|
||||
const normalizedLength = emitted + pendingLength
|
||||
if (lastMatchStart < 0 || lastMatchStart < normalizedLength - DETECTOR_BUFFER_MAX_CHARS) {
|
||||
return null
|
||||
}
|
||||
armed = false
|
||||
tail = ''
|
||||
return CODEX_BACKFILL_RECOVERY_NOTICE
|
||||
}
|
||||
}
|
||||
|
||||
@@ -522,6 +522,11 @@ describe('terminalOutputPrefersRenderRefresh single-pass classification', () =>
|
||||
'\x1b[38;5;3;44mfg then bg\x1b[0m',
|
||||
'\x1b[38;2;1;2;3;41mfg truecolor then bg\x1b[0m',
|
||||
'\x1b[38:5:9mcolon fg\x1b[0m',
|
||||
// 38;2 consumes four parameters, so this 44 is the blue component, not a background.
|
||||
'\x1b[38;2;1;2;44m',
|
||||
// The colon form does not consume trailing parameters, so this 44 is a background.
|
||||
'\x1b[38:5:1;44m',
|
||||
'\x1b[38:2::1:2:3;44m',
|
||||
'\x1b[m',
|
||||
'\x1b[;m',
|
||||
'\x1b[38m',
|
||||
@@ -547,6 +552,8 @@ describe('terminalOutputPrefersRenderRefresh single-pass classification', () =>
|
||||
'replacement \ufffd',
|
||||
'lone surrogate \ud83d',
|
||||
'astral tail \u{2f81a}',
|
||||
// An astral code point outside every risk range must advance past both halves.
|
||||
'math bold \u{1d400} x',
|
||||
'\x1b[41m\u{1f680}',
|
||||
'\u{1f680}\x1b[41m'
|
||||
]
|
||||
|
||||
Reference in New Issue
Block a user