diff --git a/src/main/runtime/rpc/terminal-stream-byte-length.test.ts b/src/main/runtime/rpc/terminal-stream-byte-length.test.ts index bf07fe38fc8..60c4c86fe44 100644 --- a/src/main/runtime/rpc/terminal-stream-byte-length.test.ts +++ b/src/main/runtime/rpc/terminal-stream-byte-length.test.ts @@ -7,9 +7,24 @@ import { } from './terminal-stream-byte-length' import { TERMINAL_OUTPUT_BATCH_MAX_BYTES } from '../../../shared/terminal-multiplex-flow-control' -// Byte-for-byte copy of the pre-change implementation (shared/clipboard-text.ts +// Copy of the pre-change implementation (shared/clipboard-text.ts // measureClipboardTextByteLength), kept here so equivalence is checked against the -// ACTUAL old code path rather than a paraphrase of it. +// ACTUAL old code path rather than a paraphrase of it. The one deliberate deviation is +// `legacyCodePointAt`: raw `String.prototype.codePointAt` reads one code unit past the end +// of a sliced string once V8 optimizes its caller, so the naive copy is not a stable +// reference. See src/shared/utf8-byte-limits.ts (readUtf8CodePointAt). +function legacyCodePointAt(text: string, index: number): number { + const leadUnit = text.charCodeAt(index) + if (leadUnit < 0xd800 || leadUnit > 0xdbff || index + 1 >= text.length) { + return leadUnit + } + const trailUnit = text.charCodeAt(index + 1) + if (trailUnit < 0xdc00 || trailUnit > 0xdfff) { + return leadUnit + } + return (leadUnit - 0xd800) * 0x400 + (trailUnit - 0xdc00) + 0x10000 +} + function legacyUtf8ByteLengthForCodePoint(codePoint: number): number { if (codePoint <= 0x7f) { return 1 @@ -30,7 +45,7 @@ function legacyMeasure( const stopAfterBytes = options.stopAfterBytes let byteLength = 0 for (let index = 0; index < text.length; index += 1) { - const codePoint = text.codePointAt(index) ?? 0 + const codePoint = legacyCodePointAt(text, index) byteLength += legacyUtf8ByteLengthForCodePoint(codePoint) if (Number.isFinite(stopAfterBytes) && byteLength > (stopAfterBytes ?? 0)) { return { byteLength, exceededLimit: true } @@ -128,6 +143,39 @@ const EDGE_STRINGS = [ `${'é'.repeat(500)}\ud800` ] +// Regression for the intermittent "measurement diverged at 13 units" failure: the fuzzers +// build a rope and cut it at a fixed code-unit count, which can split a surrogate pair and +// leave the low half in the parent just past the slice. Optimized `codePointAt` pairs across +// that boundary, so the scan measured one byte too many, but only after the enclosing function +// tiered up, which made the failure look load-dependent. 13 code units is V8's minimum length +// for a sliced string, which is why the divergence started exactly there. +describe('measuring a prefix slice that cuts a surrogate pair in half', () => { + // Kept first in the file so the scan is still specializing on this shape when it tiers up. + it('measures the slice like the encoder does in every JIT tier', () => { + const sliced = 'abcdefghijkl\u{1f600}'.slice(0, 13) + expect(sliced.length).toBe(13) + expect(sliced.charCodeAt(12)).toBe(0xd83d) + // 12 ASCII bytes plus U+FFFD for the orphaned high surrogate. + expect(Buffer.byteLength(sliced, 'utf8')).toBe(15) + + const observedByteLengths = new Set() + const observedExceeded = new Set() + const observedMeasurements = new Set() + for (let iteration = 0; iteration < 200_000; iteration += 1) { + observedByteLengths.add(terminalStreamByteLength(sliced)) + observedExceeded.add(terminalStreamByteLengthExceeds(sliced, 15)) + observedMeasurements.add( + JSON.stringify(measureTerminalStreamByteLength(sliced, { stopAfterBytes: 15 })) + ) + } + expect([...observedByteLengths]).toEqual([15]) + expect([...observedExceeded]).toEqual([false]) + expect([...observedMeasurements]).toEqual([ + JSON.stringify({ byteLength: 15, exceededLimit: false }) + ]) + }) +}) + describe('terminal stream byte length equivalence with the legacy code-point scan', () => { it('matches the legacy total byte length on edge strings', () => { for (const text of EDGE_STRINGS) { diff --git a/src/renderer/src/components/editor/rich-markdown-source-owning-slice.ts b/src/renderer/src/components/editor/rich-markdown-source-owning-slice.ts index 51807b15557..622d5c0bc8b 100644 --- a/src/renderer/src/components/editor/rich-markdown-source-owning-slice.ts +++ b/src/renderer/src/components/editor/rich-markdown-source-owning-slice.ts @@ -1,5 +1,8 @@ import type { Slice } from '@tiptap/pm/model' -import { getUtf8ByteLengthForCodePoint } from '../../../../shared/utf8-byte-limits' +import { + getUtf8ByteLengthForCodePoint, + readUtf8CodePointAt +} from '../../../../shared/utf8-byte-limits' import { getRichMarkdownLeafVisibleText, isRichMarkdownVisibleBlockStart @@ -90,7 +93,7 @@ function addUtf8BytesWithinLimit( } let byteLength = current for (let index = 0; index < value.length; index += 1) { - const codePoint = value.codePointAt(index) ?? 0 + const codePoint = readUtf8CodePointAt(value, index) byteLength += getUtf8ByteLengthForCodePoint(codePoint) if (byteLength > limit) { return { byteLength, exceeded: true } diff --git a/src/renderer/src/components/editor/rich-markdown-source-reconcile.ts b/src/renderer/src/components/editor/rich-markdown-source-reconcile.ts index 2cff1ed560f..27a776ee163 100644 --- a/src/renderer/src/components/editor/rich-markdown-source-reconcile.ts +++ b/src/renderer/src/components/editor/rich-markdown-source-reconcile.ts @@ -5,7 +5,10 @@ import { makeDiff, makePatches } from '@sanity/diff-match-patch' -import { getUtf8ByteLengthForCodePoint } from '../../../../shared/utf8-byte-limits' +import { + getUtf8ByteLengthForCodePoint, + readUtf8CodePointAt +} from '../../../../shared/utf8-byte-limits' // Why: cap document size in UTF-16 code units (`.length`) since re-parse cost scales with length — the per-commit throwaway TipTap safety re-parse (~50-67ms here) must stay under the 300ms serialize debounce so it can't stall the main thread on slow/SSH hosts. const RECONCILE_SIZE_CAP_CODE_UNITS = 50_000 @@ -147,10 +150,7 @@ function getUtf8OffsetsAtCodeUnitIndices( for (const target of targets) { const boundedTarget = Math.max(0, Math.min(target, text.length)) while (codeUnitIndex < boundedTarget) { - const codePoint = text.codePointAt(codeUnitIndex) - if (codePoint === undefined) { - break - } + const codePoint = readUtf8CodePointAt(text, codeUnitIndex) byteOffset += getUtf8ByteLengthForCodePoint(codePoint) codeUnitIndex += codePoint > 0xffff ? 2 : 1 } diff --git a/src/renderer/src/components/terminal-pane/terminal-paste-chunks.ts b/src/renderer/src/components/terminal-pane/terminal-paste-chunks.ts index 7082f7aff9d..a67e2e6f4ca 100644 --- a/src/renderer/src/components/terminal-pane/terminal-paste-chunks.ts +++ b/src/renderer/src/components/terminal-pane/terminal-paste-chunks.ts @@ -1,7 +1,10 @@ import { BRACKETED_PASTE_END, BRACKETED_PASTE_START } from './terminal-bracketed-paste' import { TERMINAL_PASTE_CHUNK_MAX_BYTES } from './terminal-paste-limits' import type { TerminalPastePlan } from './terminal-paste-coordinator' -import { getUtf8ByteLengthForCodePoint } from '../../../../shared/utf8-byte-limits' +import { + getUtf8ByteLengthForCodePoint, + readUtf8CodePointAt +} from '../../../../shared/utf8-byte-limits' const TERMINAL_PASTE_ESCAPE_CODE_POINT = 0x1b const TERMINAL_PASTE_INERT_ESCAPE_CODE_POINT = 0x241b @@ -38,7 +41,7 @@ function* iterateTextByUtf8Bytes( let chunk = '' let chunkBytes = 0 for (let index = 0; index < text.length; index += 1) { - const codePoint = text.codePointAt(index) ?? 0 + const codePoint = readUtf8CodePointAt(text, index) const codeUnitLength = codePoint > 0xffff ? 2 : 1 // Why: iterator normalization avoids a full-size copy and keeps CRLF atomic across chunks. if ( diff --git a/src/renderer/src/lib/agent-draft-paste-content.ts b/src/renderer/src/lib/agent-draft-paste-content.ts index b6c381dc5a9..526e717de27 100644 --- a/src/renderer/src/lib/agent-draft-paste-content.ts +++ b/src/renderer/src/lib/agent-draft-paste-content.ts @@ -1,6 +1,9 @@ import { yieldToEventLoop } from '../../../shared/event-loop-yield' import type { GlobalSettings } from '../../../shared/types' -import { getUtf8ByteLengthForCodePoint } from '../../../shared/utf8-byte-limits' +import { + getUtf8ByteLengthForCodePoint, + readUtf8CodePointAt +} from '../../../shared/utf8-byte-limits' import { BRACKETED_PASTE_END, BRACKETED_PASTE_START, @@ -110,7 +113,7 @@ export function* iterateAgentDraftPasteContentChunks( let chunkBytes = 0 for (let index = 0; index < terminalContent.length; index += 1) { - const codePoint = terminalContent.codePointAt(index) ?? 0 + const codePoint = readUtf8CodePointAt(terminalContent, index) const codeUnitLength = codePoint > 0xffff ? 2 : 1 const sanitizedEscape = codePoint === AGENT_DRAFT_PASTE_ESCAPE_CODE_POINT const sanitized = sanitizedEscape @@ -150,7 +153,7 @@ function measureSanitizedUtf8ByteLength( let byteLength = 0 const stopAfterBytes = options.stopAfterBytes for (let index = 0; index < content.length; index += 1) { - const codePoint = content.codePointAt(index) ?? 0 + const codePoint = readUtf8CodePointAt(content, index) byteLength += getSanitizedUtf8ByteLengthForCodePoint(codePoint) if (Number.isFinite(stopAfterBytes) && byteLength > (stopAfterBytes ?? 0)) { return { byteLength, exceededLimit: true } @@ -166,7 +169,7 @@ async function isSanitizedDraftPasteOverLimit(content: string, maxBytes: number) let byteLength = 0 let nextYieldAt = AGENT_DRAFT_PASTE_PREFLIGHT_YIELD_CODE_UNITS for (let index = 0; index < content.length; index += 1) { - const codePoint = content.codePointAt(index) ?? 0 + const codePoint = readUtf8CodePointAt(content, index) byteLength += getSanitizedUtf8ByteLengthForCodePoint(codePoint) if (byteLength > maxBytes) { return true diff --git a/src/renderer/src/lib/comment-body-submit-state.ts b/src/renderer/src/lib/comment-body-submit-state.ts index 4a7066f6408..460200d3e10 100644 --- a/src/renderer/src/lib/comment-body-submit-state.ts +++ b/src/renderer/src/lib/comment-body-submit-state.ts @@ -1,4 +1,7 @@ -import { getUtf8ByteLengthForCodePoint } from '../../../shared/utf8-byte-limits' +import { + getUtf8ByteLengthForCodePoint, + readUtf8CodePointAt +} from '../../../shared/utf8-byte-limits' export const COMMENT_BODY_NONBLANK_SCAN_MAX_BYTES = 64 * 1024 @@ -16,7 +19,7 @@ function getCommentBodyPresence( let scannedBytes = 0 for (let index = 0; index < body.length; index += 1) { - const codePoint = body.codePointAt(index) ?? 0 + const codePoint = readUtf8CodePointAt(body, index) const codeUnitLength = codePoint > 0xffff ? 2 : 1 scannedBytes += getUtf8ByteLengthForCodePoint(codePoint) if (scannedBytes > maxScanBytes) { diff --git a/src/renderer/src/lib/paste-payload-metadata.ts b/src/renderer/src/lib/paste-payload-metadata.ts index f28e49b24f2..f29894446b2 100644 --- a/src/renderer/src/lib/paste-payload-metadata.ts +++ b/src/renderer/src/lib/paste-payload-metadata.ts @@ -1,5 +1,8 @@ import { yieldToEventLoop } from '../../../shared/event-loop-yield' -import { getUtf8ByteLengthForCodePoint } from '../../../shared/utf8-byte-limits' +import { + getUtf8ByteLengthForCodePoint, + readUtf8CodePointAt +} from '../../../shared/utf8-byte-limits' export type PastePayloadMetadata = { byteLength: number @@ -25,7 +28,7 @@ export function measurePastePayloadMetadata( let previousWasCarriageReturn = false for (let index = 0; index < text.length; index += 1) { - const codePoint = text.codePointAt(index) ?? 0 + const codePoint = readUtf8CodePointAt(text, index) byteLength += getUtf8ByteLengthForCodePoint(codePoint) hasControlSequences ||= isPasteControlSequenceCodePoint(codePoint) if (codePoint === 0x0d) { @@ -73,7 +76,7 @@ export async function measurePastePayloadMetadataWithYield( let previousWasCarriageReturn = false for (let index = 0; index < text.length; index += 1) { - const codePoint = text.codePointAt(index) ?? 0 + const codePoint = readUtf8CodePointAt(text, index) byteLength += getUtf8ByteLengthForCodePoint(codePoint) hasControlSequences ||= isPasteControlSequenceCodePoint(codePoint) if (codePoint === 0x0d) { @@ -110,7 +113,7 @@ export function countPastePayloadLines(text: string): number { export function hasPastePayloadControlSequence(text: string): boolean { for (let index = 0; index < text.length; index += 1) { - const codePoint = text.codePointAt(index) ?? 0 + const codePoint = readUtf8CodePointAt(text, index) if (isPasteControlSequenceCodePoint(codePoint)) { return true } diff --git a/src/shared/clipboard-text.ts b/src/shared/clipboard-text.ts index e8170af606b..c5d641e93d6 100644 --- a/src/shared/clipboard-text.ts +++ b/src/shared/clipboard-text.ts @@ -2,6 +2,7 @@ import { yieldToEventLoop } from './event-loop-yield' import { getUtf8ByteLengthForCodePoint, measureUtf8ByteLength, + readUtf8CodePointAt, type Utf8ByteLengthMeasurement } from './utf8-byte-limits' @@ -50,7 +51,7 @@ export async function measureClipboardTextByteLengthWithYield( let byteLength = 0 for (let index = 0; index < text.length; index += 1) { - const codePoint = text.codePointAt(index) ?? 0 + const codePoint = readUtf8CodePointAt(text, index) byteLength += getUtf8ByteLengthForCodePoint(codePoint) if (Number.isFinite(stopAfterBytes) && byteLength > (stopAfterBytes ?? 0)) { return { byteLength, exceededLimit: true } diff --git a/src/shared/utf8-byte-limits.test.ts b/src/shared/utf8-byte-limits.test.ts index 9b3c63c9cea..9556a5c39bc 100644 --- a/src/shared/utf8-byte-limits.test.ts +++ b/src/shared/utf8-byte-limits.test.ts @@ -1,5 +1,77 @@ import { describe, expect, it } from 'vitest' -import { getUtf8ChunkEndIndex, isUtf8ByteLengthWithinLimit } from './utf8-byte-limits' +import { + clampUtf8TextPrefix, + getUtf8ChunkEndIndex, + isUtf8ByteLengthWithinLimit, + measureUtf8ByteLength, + readUtf8CodePointAt +} from './utf8-byte-limits' + +// Once V8 optimizes the calling function, `String.prototype.codePointAt` on a sliced string +// pairs a trailing high surrogate with the code unit that follows the SLICE inside its parent +// (reproduced on Node 24 and 26). A prefix slice cut mid-pair then reads a code point the +// string does not contain, so a scan reports one byte too many — but only after tier-up, which +// is why it surfaced as an intermittent CI failure rather than a deterministic one. +// Build a rope, cut it mid-pair, and hammer the scan so the optimizing tier is under test. +const SLICE_BOUNDARY_PARENT_UNITS = [ + 0x72, 0x2e6, 0x7b, 0x54, 0xda9b, 0x568, 0x52, 0x26, 0x46, 0xc15b, 0x7, 0x768, 0xd9f6, 0xdcde +] +// V8 only creates a sliced string (rather than copying) at 13 code units or more. +const SLICE_BOUNDARY_UNITS = 13 +const TIER_UP_ITERATIONS = 200_000 + +function buildSliceEndingInLoneHighSurrogate(): string { + let text = '' + for (const unit of SLICE_BOUNDARY_PARENT_UNITS) { + text += String.fromCharCode(unit) + } + return text.slice(0, SLICE_BOUNDARY_UNITS) +} + +describe('scanning a prefix slice whose parent continues past the slice', () => { + const sliced = buildSliceEndingInLoneHighSurrogate() + + it('is a sliced string that ends in a lone high surrogate', () => { + expect(sliced.length).toBe(SLICE_BOUNDARY_UNITS) + expect(sliced.charCodeAt(SLICE_BOUNDARY_UNITS - 1)).toBe(0xd9f6) + expect(Buffer.byteLength(sliced, 'utf8')).toBe(22) + }) + + it('reads the trailing lone surrogate without pairing past the end in every JIT tier', () => { + const observed = new Set() + for (let iteration = 0; iteration < TIER_UP_ITERATIONS; iteration += 1) { + observed.add(readUtf8CodePointAt(sliced, SLICE_BOUNDARY_UNITS - 1)) + } + expect([...observed]).toEqual([0xd9f6]) + }) + + it('measures the same byte length as the encoder in every JIT tier', () => { + const expected = Buffer.byteLength(sliced, 'utf8') + const observed = new Set() + for (let iteration = 0; iteration < TIER_UP_ITERATIONS; iteration += 1) { + observed.add(measureUtf8ByteLength(sliced).byteLength) + } + expect([...observed]).toEqual([expected]) + }) + + it('does not report an exceeded limit at the true byte length in every JIT tier', () => { + const limit = Buffer.byteLength(sliced, 'utf8') + const observed = new Set() + for (let iteration = 0; iteration < TIER_UP_ITERATIONS; iteration += 1) { + observed.add(measureUtf8ByteLength(sliced, { stopAfterBytes: limit }).exceededLimit) + } + expect([...observed]).toEqual([false]) + }) + + it('keeps the whole slice when clamping to its true byte length in every JIT tier', () => { + const limit = Buffer.byteLength(sliced, 'utf8') + const observed = new Set() + for (let iteration = 0; iteration < TIER_UP_ITERATIONS; iteration += 1) { + observed.add(clampUtf8TextPrefix(sliced, limit).length) + } + expect([...observed]).toEqual([SLICE_BOUNDARY_UNITS]) + }) +}) describe('getUtf8ChunkEndIndex', () => { it.each([ diff --git a/src/shared/utf8-byte-limits.ts b/src/shared/utf8-byte-limits.ts index b32bd968d57..df8f99f68ad 100644 --- a/src/shared/utf8-byte-limits.ts +++ b/src/shared/utf8-byte-limits.ts @@ -8,6 +8,26 @@ export type Utf8TextTail = { bytes: number } +/** + * Bounds-safe replacement for `String.prototype.codePointAt`. + * + * Why: once V8 optimizes the calling function, `codePointAt` on a sliced string pairs a + * trailing high surrogate with the code unit that follows the SLICE inside its parent, so a + * prefix slice cut mid-pair reports a code point the string does not contain (and one byte + * too many). `charCodeAt` stays bounds-correct in every tier, so pair the units explicitly. + */ +export function readUtf8CodePointAt(text: string, index: number): number { + const leadUnit = text.charCodeAt(index) + if (leadUnit < 0xd800 || leadUnit > 0xdbff || index + 1 >= text.length) { + return leadUnit + } + const trailUnit = text.charCodeAt(index + 1) + if (trailUnit < 0xdc00 || trailUnit > 0xdfff) { + return leadUnit + } + return (leadUnit - 0xd800) * 0x400 + (trailUnit - 0xdc00) + 0x10000 +} + export function measureUtf8ByteLength( text: string, options: { stopAfterBytes?: number } = {} @@ -15,7 +35,7 @@ export function measureUtf8ByteLength( const stopAfterBytes = options.stopAfterBytes let byteLength = 0 for (let index = 0; index < text.length; index += 1) { - const codePoint = text.codePointAt(index) ?? 0 + const codePoint = readUtf8CodePointAt(text, index) byteLength += getUtf8ByteLengthForCodePoint(codePoint) if (Number.isFinite(stopAfterBytes) && byteLength > (stopAfterBytes ?? 0)) { return { byteLength, exceededLimit: true } @@ -69,7 +89,7 @@ export function clampUtf8TextPrefix(text: string, maxBytes: number): string { let bytes = 0 let end = 0 while (end < text.length) { - const codePoint = text.codePointAt(end) ?? 0 + const codePoint = readUtf8CodePointAt(text, end) const codePointBytes = getUtf8ByteLengthForCodePoint(codePoint) if (bytes + codePointBytes > maxBytes) { break @@ -84,7 +104,7 @@ export function getUtf8ChunkEndIndex(text: string, startIndex: number, maxBytes: let bytes = 0 let endIndex = startIndex while (endIndex < text.length) { - const codePoint = text.codePointAt(endIndex) ?? 0 + const codePoint = readUtf8CodePointAt(text, endIndex) const codePointBytes = getUtf8ByteLengthForCodePoint(codePoint) if (bytes > 0 && bytes + codePointBytes > maxBytes) { break @@ -123,6 +143,6 @@ function getPreviousUtf8CodePoint( } return { start, - bytes: getUtf8ByteLengthForCodePoint(text.codePointAt(start) ?? codeUnit) + bytes: getUtf8ByteLengthForCodePoint(readUtf8CodePointAt(text, start)) } }