mirror of
https://github.com/stablyai/orca.git
synced 2026-09-28 16:02:45 +00:00
perf(terminal): take the ESC at hand before searching for one
The unconditional ground-state indexOf regressed dense back-to-back SGR/CSI streams, where the code unit at the cursor is already the ESC and the search pays call plus SIMD setup to find it in place. Check the current unit first and fall back to the native search otherwise. 0.9 MiB dense SGR/CSI medians: 2.04 ms before this PR, 2.56 ms with the unconditional search, 1.92 ms with the hybrid. Sparse colored logs and plain text keep the full search win (0.36 / 0.016 ms vs 1.54 / 1.41 ms baseline). Differential over the full VT alphabet with lone and split surrogates matched 388,416 cases against both prior implementations with zero mismatches; the ground-scan work budget now records 16 inspected code units and 2 native searches.
This commit is contained in:
@@ -471,11 +471,11 @@
|
||||
},
|
||||
"redGreenEvidence": {
|
||||
"status": "complete",
|
||||
"evidence": "New ASCII/UTF-16 work budgets failed baseline at 262,160/196,624 inspected code units; candidate inspects 13 in each with exact tails and completion. External differential matched 2,710,126 cases; existing fold fuzz covers 593,468 cases."
|
||||
"evidence": "New ASCII/UTF-16 work budgets failed baseline at 262,160/196,624 inspected code units; candidate inspects 16 code units and 2 native ESC searches in each with exact tails and completion. External differential matched 2,710,126 cases (plus 388,416 hybrid-vs-baseline cases over the full VT alphabet with lone/split surrogates, 0 mismatches); existing fold fuzz covers 593,468 cases."
|
||||
},
|
||||
"performanceBudget": {
|
||||
"required": true,
|
||||
"evidence": "Production HeadlessEmulator.writeSync CPU for 2 MiB colored logs improves 32.321 to 26.638 ms; sparse ANSI improves 26.757 to 21.982 ms. Agent-redraw CPU improves 402.308 to 394.633 ms. Rotated 100,000-write baseline/identical-control/candidate medians: plain echo 27.340/27.015/27.822 ms, colored echo 36.009/35.556/35.981 ms. Native ESC searches advance only in ground; existing plain-output fast path is unchanged. No new state, allocation, timer, provider call or transport behavior."
|
||||
"evidence": "Production HeadlessEmulator.writeSync CPU for 2 MiB colored logs improves 32.321 to 26.638 ms; sparse ANSI improves 26.757 to 21.982 ms. Agent-redraw CPU improves 402.308 to 394.633 ms. Rotated 100,000-write baseline/identical-control/candidate medians: plain echo 27.340/27.015/27.822 ms, colored echo 36.009/35.556/35.981 ms. Native ESC searches advance only in ground, and only when the current code unit is not already ESC, so dense back-to-back CSI streams do not pay a search per sequence: 0.9 MiB dense SGR/CSI medians are 2.04 ms baseline, 2.56 ms search-always, 1.92 ms shipped. Existing plain-output fast path is unchanged. No new state, allocation, timer, provider call or transport behavior."
|
||||
},
|
||||
"knownGaps": [
|
||||
"No launched Electron or live SSH/WSL/Windows/Linux process, native input or end-to-end UI-latency measurement.",
|
||||
|
||||
@@ -61,7 +61,9 @@ export function extractPartialEscapeTail(stream: string): string {
|
||||
for (let i = 0; i < stream.length; i++) {
|
||||
if (state === 'ground') {
|
||||
// Only ESC leaves ground; skip ordinary text without a per-code-unit walk.
|
||||
const escape = stream.indexOf('\x1b', i)
|
||||
// Check the current unit first: on dense escape streams it is usually the
|
||||
// ESC itself, and indexOf's call + SIMD setup costs more than the compare.
|
||||
const escape = stream.charCodeAt(i) === ESC ? i : stream.indexOf('\x1b', i)
|
||||
if (escape === -1) {
|
||||
return ''
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user