diff --git a/src/shared/terminal-emoji-cell-width-agreement.test.ts b/src/shared/terminal-emoji-cell-width-agreement.test.ts index f0d32ddf660..a3e1505c626 100644 --- a/src/shared/terminal-emoji-cell-width-agreement.test.ts +++ b/src/shared/terminal-emoji-cell-width-agreement.test.ts @@ -34,8 +34,11 @@ const SEQUENCES: [string, string, number][] = [ ['desktop computer U+1F5A5 U+FE0F', '\u{1F5A5}️', 2], ['warning U+26A0 U+FE0F', '⚠️', 2], ['keycap U+0031 U+FE0F U+20E3', '1️⃣', 2], + // A variation selector inside a ZWJ cluster is the same promotion, one level in. + ['heart on fire U+2764 U+FE0F U+200D U+1F525', '❤️‍\u{1F525}', 2], // Same authority, same defect class: a modifier is part of its base's cluster. ['thumbs up + skin tone', '\u{1F44D}\u{1F3FD}', 2], + ['person + skin tone + ZWJ role', '\u{1F9D1}\u{1F3FD}‍\u{1F4BB}', 2], // Emoji that postdate xterm's frozen Unicode 11 width table. ['melting face U+1FAE0', '\u{1FAE0}', 2], ['bubble tea U+1F9CB', '\u{1F9CB}', 2], @@ -45,7 +48,11 @@ const SEQUENCES: [string, string, number][] = [ ['check mark U+2705', '✅', 2], ['regional-indicator flag', '\u{1F1E8}\u{1F1F3}', 2], ['ZWJ family', '\u{1F468}‍\u{1F469}‍\u{1F467}‍\u{1F466}', 2], - ['heart on fire U+2764 U+FE0F U+200D U+1F525', '❤️‍\u{1F525}', 2], + // A modifier only belongs to a base that takes one. After anything else it is + // its own cluster, exactly as it was before this provider learned the rule. + ['CJK + skin tone', '中\u{1F3FD}', 4], + ['non-modifier-base emoji + skin tone', '\u{1F355}\u{1F3FD}', 4], + ['ASCII + skin tone', 'a\u{1F3FD}', 3], // A variation selector after a non-emoji base is not a variation sequence. ['letter + U+FE0F', 'a️', 1], ['plain ASCII', 'ab', 2] @@ -80,9 +87,9 @@ describe('terminal emoji cell width agreement (STA-6740)', () => { }) it('reserves the second cell of a widened cluster instead of leaving it writable', async () => { - // Why: the reported symptom is lost characters, which needs the cell the - // cluster claims to actually be a wide-char placeholder — otherwise the next - // glyph lands inside the emoji and one of the two is overwritten. + // Why: advancing two cells is only half the contract. The second cell must + // also be a wide-char placeholder, or the next glyph lands inside the + // cluster and one of the two is overwritten. const parity = createTerminal() await writeToTerminal(parity.terminal, '\x1b[H\x1b[2J⚠️X') const line = parity.terminal.buffer.active.getLine(0) diff --git a/src/shared/terminal-emoji-width-ranges.derivation.test.ts b/src/shared/terminal-emoji-width-ranges.derivation.test.ts new file mode 100644 index 00000000000..7ab44f9866f --- /dev/null +++ b/src/shared/terminal-emoji-width-ranges.derivation.test.ts @@ -0,0 +1,162 @@ +/** + * Staleness gate for the frozen tables in terminal-emoji-width-ranges.ts. + * + * The bug those tables fix is a width table that stopped tracking Unicode. + * A frozen table can acquire that same bug silently, so this re-derives all + * three from the runtime's own Unicode data and xterm's Unicode 11 provider — + * the recipe recorded in the module header — and compares. Nothing here reads + * the committed ranges to build its expectation. + * + * Direction matters, because the tables are frozen at one Unicode version and + * this runs on whatever the test runner embeds: + * - "no missing entries" runs always. An older runtime simply has not assigned + * the newest code points, so it can never fail this direction spuriously, + * and a newer one fails it exactly when the table has gone stale. + * - exact equality runs only when the runtime's Unicode version matches the + * version the tables were generated against, where any difference is real. + */ +import { describe, expect, it } from 'vitest' +import { Terminal } from '@xterm/headless' +import { Unicode11Addon } from '@xterm/addon-unicode11' +import { + EMOJI_MODIFIER_BASE_RANGES, + EMOJI_PRESENTATION_WIDE_RANGES, + EMOJI_VARIATION_BASE_RANGES, + isEmojiModifierBase, + isEmojiPresentationWideCodepoint, + isEmojiVariationSequenceBase, + type CodepointRange +} from './terminal-emoji-width-ranges' + +/** The version the header records the tables as generated against. */ +const GENERATED_AGAINST_UNICODE = '17.0' + +const LAST_CODEPOINT = 0x10ffff +const SURROGATE_FIRST = 0xd800 +const SURROGATE_LAST = 0xdfff +const REGIONAL_INDICATOR_FIRST = 0x1f1e6 +const REGIONAL_INDICATOR_LAST = 0x1f1ff + +const EMOJI = /\p{Emoji}/u +const EMOJI_PRESENTATION = /\p{Emoji_Presentation}/u +const EMOJI_MODIFIER_BASE = /\p{Emoji_Modifier_Base}/u + +type UnicodeServiceInternals = { activeVersion: string; wcwidth(codepoint: number): number } + +function openUnicode11(): { terminal: Terminal; unicode: UnicodeServiceInternals } { + const terminal = new Terminal({ cols: 40, rows: 10, allowProposedApi: true }) + terminal.loadAddon(new Unicode11Addon()) + const unicode = (terminal as unknown as { _core: { unicodeService: UnicodeServiceInternals } }) + ._core.unicodeService + unicode.activeVersion = '11' + return { terminal, unicode } +} + +/** Re-derives the three sets straight from the recipe in the module header. */ +function deriveTables(unicode: UnicodeServiceInternals): { + wide: number[] + variationBase: number[] + modifierBase: number[] +} { + const wide: number[] = [] + const variationBase: number[] = [] + const modifierBase: number[] = [] + for (let codepoint = 0; codepoint <= LAST_CODEPOINT; codepoint += 1) { + if (codepoint >= SURROGATE_FIRST && codepoint <= SURROGATE_LAST) { + continue + } + const char = String.fromCodePoint(codepoint) + if (EMOJI_MODIFIER_BASE.test(char)) { + modifierBase.push(codepoint) + } + if (!EMOJI.test(char)) { + continue + } + if (EMOJI_PRESENTATION.test(char)) { + const isRegionalIndicator = + codepoint >= REGIONAL_INDICATOR_FIRST && codepoint <= REGIONAL_INDICATOR_LAST + if (unicode.wcwidth(codepoint) !== 2 && !isRegionalIndicator) { + wide.push(codepoint) + } + } else if (unicode.wcwidth(codepoint) === 1) { + variationBase.push(codepoint) + } + } + return { wide, variationBase, modifierBase } +} + +function expandRanges(ranges: readonly CodepointRange[]): number[] { + const codepoints: number[] = [] + for (const [first, last] of ranges) { + for (let codepoint = first; codepoint <= last; codepoint += 1) { + codepoints.push(codepoint) + } + } + return codepoints +} + +const hex = (value: number): string => `U+${value.toString(16).toUpperCase().padStart(4, '0')}` + +/** Contiguous code points collapse into one entry so a drift reads as ranges. */ +function summarize(codepoints: number[]): string[] { + const runs: [number, number][] = [] + for (const codepoint of codepoints) { + const last = runs.at(-1) + if (last && last[1] === codepoint - 1) { + last[1] = codepoint + continue + } + runs.push([codepoint, codepoint]) + } + return runs.map(([first, last]) => (first === last ? hex(first) : `${hex(first)}..${hex(last)}`)) +} + +const TABLES: [string, readonly CodepointRange[], (codepoint: number) => boolean][] = [ + ['emoji presentation wide', EMOJI_PRESENTATION_WIDE_RANGES, isEmojiPresentationWideCodepoint], + ['emoji variation base', EMOJI_VARIATION_BASE_RANGES, isEmojiVariationSequenceBase], + ['emoji modifier base', EMOJI_MODIFIER_BASE_RANGES, isEmojiModifierBase] +] + +describe('emoji width range derivation', () => { + const { terminal, unicode } = openUnicode11() + const derived = deriveTables(unicode) + const derivedByName: Record = { + 'emoji presentation wide': derived.wide, + 'emoji variation base': derived.variationBase, + 'emoji modifier base': derived.modifierBase + } + terminal.dispose() + + it.each(TABLES)('%s: the committed table holds every derived code point', (name, ranges) => { + const committed = new Set(expandRanges(ranges)) + const missing = derivedByName[name]!.filter((codepoint) => !committed.has(codepoint)) + expect(summarize(missing)).toEqual([]) + }) + + it.each(TABLES)('%s: the lookup agrees with the committed ranges', (name, ranges, lookup) => { + // Guards the binary search and the fast-path bound, not just the data. + const disagreeing = expandRanges(ranges).filter((codepoint) => !lookup(codepoint)) + expect(summarize(disagreeing)).toEqual([]) + expect(lookup(SURROGATE_FIRST)).toBe(false) + }) + + it.each(TABLES)('%s: matches the derivation exactly on the generated version', (name, ranges) => { + if (process.versions.unicode !== GENERATED_AGAINST_UNICODE) { + // A different embedded Unicode version legitimately derives a different + // set; the "no missing entries" case above still gates staleness there. + expect(process.versions.unicode).toBeTypeOf('string') + return + } + expect(summarize(expandRanges(ranges))).toEqual(summarize(derivedByName[name]!)) + }) + + it('excludes regional indicators so a flag stays two cells', () => { + for ( + let codepoint = REGIONAL_INDICATOR_FIRST; + codepoint <= REGIONAL_INDICATOR_LAST; + codepoint += 1 + ) { + expect(isEmojiPresentationWideCodepoint(codepoint)).toBe(false) + } + }) +}) diff --git a/src/shared/terminal-emoji-width-ranges.ts b/src/shared/terminal-emoji-width-ranges.ts index 8bce9c0d99a..0c954d139d2 100644 --- a/src/shared/terminal-emoji-width-ranges.ts +++ b/src/shared/terminal-emoji-width-ranges.ts @@ -3,22 +3,40 @@ // single width authority every Orca terminal (renderer pane, headless daemon // mirror, dashboard preview, restore-parity fixture) activates. // -// Both tables are DERIVED, not hand-written. Regenerate against the Unicode -// character database exposed by the runtime's ICU (Unicode 17.0 when generated) -// and xterm's Unicode 11 provider: +// WHY THE TABLES ARE FROZEN HERE rather than read from the runtime's Unicode +// data: this width is a cross-host contract. A pane's cells are laid out by +// whichever host runs the terminal and re-derived by whichever client renders +// it, and Electron's V8, plain Node and the mobile JSC ship different ICU +// versions. Deriving at runtime would make two Orca processes disagree about +// the same buffer purely because they embed different Unicode data — the exact +// failure this module exists to remove. Freezing the table moves the version +// skew to something a release controls. +// +// WHY NOT xterm's grapheme-clustering provider: it is upstream-experimental, +// leaves every post-Unicode-11 emoji below at one cell, and disagrees with +// other terminals on ZWJ-plus-variation clusters. It does not answer the +// measurement this module exists to fix. +// +// The tables are DERIVED, not hand-written, and +// terminal-emoji-width-ranges.derivation.test.ts re-derives all three from the +// runtime's Unicode data on every run so they cannot go stale the way xterm's +// Unicode 11 table did: // WIDE = \p{Emoji_Presentation} where xterm's v11 wcwidth is not 2, // minus regional indicators U+1F1E6..U+1F1FF (see below). // VSBASE = \p{Emoji} without \p{Emoji_Presentation} whose v11 wcwidth is 1, // i.e. the bases a following U+FE0F promotes to emoji presentation. +// MODBASE= \p{Emoji_Modifier_Base}, i.e. the bases a skin-tone modifier may +// attach to. +// Generated against Unicode 17.0. // // Regional indicators are excluded on purpose: xterm has no grapheme // segmentation, so a flag reaches two cells as 1 + 1. Widening each indicator // would make a flag four cells wide. /** Sorted, non-overlapping, ascending. */ -type CodepointRange = readonly [number, number] +export type CodepointRange = readonly [number, number] -const EMOJI_PRESENTATION_WIDE_RANGES: readonly CodepointRange[] = [ +export const EMOJI_PRESENTATION_WIDE_RANGES: readonly CodepointRange[] = [ [0x1f6d6, 0x1f6d8], [0x1f6dc, 0x1f6df], [0x1f6fb, 0x1f6fc], @@ -40,7 +58,7 @@ const EMOJI_PRESENTATION_WIDE_RANGES: readonly CodepointRange[] = [ [0x1faef, 0x1faf8] ] -const EMOJI_VARIATION_BASE_RANGES: readonly CodepointRange[] = [ +export const EMOJI_VARIATION_BASE_RANGES: readonly CodepointRange[] = [ [0x23, 0x23], [0x2a, 0x2a], [0x30, 0x39], @@ -156,6 +174,52 @@ const EMOJI_VARIATION_BASE_RANGES: readonly CodepointRange[] = [ [0x1f6f3, 0x1f6f3] ] +export const EMOJI_MODIFIER_BASE_RANGES: readonly CodepointRange[] = [ + [0x261d, 0x261d], + [0x26f9, 0x26f9], + [0x270a, 0x270d], + [0x1f385, 0x1f385], + [0x1f3c2, 0x1f3c4], + [0x1f3c7, 0x1f3c7], + [0x1f3ca, 0x1f3cc], + [0x1f442, 0x1f443], + [0x1f446, 0x1f450], + [0x1f466, 0x1f478], + [0x1f47c, 0x1f47c], + [0x1f481, 0x1f483], + [0x1f485, 0x1f487], + [0x1f48f, 0x1f48f], + [0x1f491, 0x1f491], + [0x1f4aa, 0x1f4aa], + [0x1f574, 0x1f575], + [0x1f57a, 0x1f57a], + [0x1f590, 0x1f590], + [0x1f595, 0x1f596], + [0x1f645, 0x1f647], + [0x1f64b, 0x1f64f], + [0x1f6a3, 0x1f6a3], + [0x1f6b4, 0x1f6b6], + [0x1f6c0, 0x1f6c0], + [0x1f6cc, 0x1f6cc], + [0x1f90c, 0x1f90c], + [0x1f90f, 0x1f90f], + [0x1f918, 0x1f91f], + [0x1f926, 0x1f926], + [0x1f930, 0x1f939], + [0x1f93c, 0x1f93e], + [0x1f977, 0x1f977], + [0x1f9b5, 0x1f9b6], + [0x1f9b8, 0x1f9b9], + [0x1f9bb, 0x1f9bb], + [0x1f9cd, 0x1f9cf], + [0x1f9d1, 0x1f9dd], + [0x1fac3, 0x1fac5], + [0x1faf0, 0x1faf8] +] + +/** Lowest code point in EMOJI_PRESENTATION_WIDE_RANGES; skips the table for ASCII and CJK. */ +const EMOJI_PRESENTATION_WIDE_FIRST = 0x1f6d6 + const EMOJI_MODIFIER_FIRST = 0x1f3fb const EMOJI_MODIFIER_LAST = 0x1f3ff @@ -178,7 +242,10 @@ function inRanges(ranges: readonly CodepointRange[], codepoint: number): boolean /** Emoji that default to emoji presentation but postdate xterm's frozen Unicode 11 width table. */ export function isEmojiPresentationWideCodepoint(codepoint: number): boolean { - return codepoint >= 0x1f6d6 && inRanges(EMOJI_PRESENTATION_WIDE_RANGES, codepoint) + return ( + codepoint >= EMOJI_PRESENTATION_WIDE_FIRST && + inRanges(EMOJI_PRESENTATION_WIDE_RANGES, codepoint) + ) } /** A code point U+FE0F may promote to emoji presentation, per emoji-variation-sequences. */ @@ -190,3 +257,8 @@ export function isEmojiVariationSequenceBase(codepoint: number): boolean { export function isEmojiModifier(codepoint: number): boolean { return codepoint >= EMOJI_MODIFIER_FIRST && codepoint <= EMOJI_MODIFIER_LAST } + +/** A code point a skin-tone modifier may attach to. */ +export function isEmojiModifierBase(codepoint: number): boolean { + return inRanges(EMOJI_MODIFIER_BASE_RANGES, codepoint) +} diff --git a/src/shared/terminal-unicode-provider.ts b/src/shared/terminal-unicode-provider.ts index 658594f8407..b1ba663aa7a 100644 --- a/src/shared/terminal-unicode-provider.ts +++ b/src/shared/terminal-unicode-provider.ts @@ -1,6 +1,7 @@ import type { IUnicodeHandling, IUnicodeVersionProvider } from '@xterm/xterm' import { isEmojiModifier, + isEmojiModifierBase, isEmojiPresentationWideCodepoint, isEmojiVariationSequenceBase } from './terminal-emoji-width-ranges' @@ -87,23 +88,24 @@ class OrcaUnicodeProvider implements IUnicodeVersionProvider { return createProperties(codepoint, 2, true) } - if (isEmojiModifier(codepoint) && precedingWidth === 2) { + if (isEmojiModifier(codepoint) && precedingWidth === 2 && isEmojiModifierBase(precedingKind)) { // Why: a skin-tone modifier belongs to its base's cluster. Left to the v11 // table it lands as a second wide cell and every later column shifts by two. + // Why gated on the base: a modifier after any other wide cell — a CJK + // ideograph, an emoji that takes no skin tone — is not a modifier sequence, + // and joining it there would narrow text this provider never measured. return createProperties(codepoint, 2, true) } const base = this.baseProvider.charProperties(codepoint, preceding) const shouldJoin = extractShouldJoin(base) + const baseWidth = extractWidth(base) // Why re-encode: the base provider reports char kind 0 and derives width // from its own wcwidth, so the rules above would lose both the preceding // code point and this provider's post-Unicode-11 widths. A joining code // point keeps the cluster width the base already resolved. - return createProperties( - codepoint, - shouldJoin ? extractWidth(base) : this.wcwidth(codepoint), - shouldJoin - ) + const width = shouldJoin || !isEmojiPresentationWideCodepoint(codepoint) ? baseWidth : 2 + return createProperties(codepoint, width, shouldJoin) } }