From e949f3d422fd22f7e5406f96b09a2b4aed6ff35b Mon Sep 17 00:00:00 2001 From: Merge Sim Date: Wed, 9 Sep 2026 01:58:09 -0700 Subject: [PATCH] fix(terminal): make the Orca width provider measure emoji clusters like other terminals MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A cursor-position probe (STA-6740) measured Orca advancing one cell for sequences Windows Terminal and Ghostty advance two for. The cause is that `OrcaUnicodeProvider` delegated every width decision to xterm's Unicode 11 tables, which measure single code points against a table frozen in 2018, while the TUIs writing into the pane measure whole grapheme clusters against a current one. Each cell of disagreement shifts the columns a TUI's next in-place redraw erases and reprints over. The provider is already the single width authority every Orca terminal activates — the live renderer pane, the headless daemon mirror that snapshot/restore serializes from, the dashboard preview, and the restore-parity fixture. This teaches that one authority three cluster rules it was missing: - U+FE0F promotes a valid emoji-variation base from one cell to two; - a skin-tone modifier joins its base's cluster instead of claiming a second wide cell (a `+2` column shift for every glyph after it); - emoji that default to emoji presentation but postdate Unicode 11 are two cells, not one. Regional indicators are deliberately left alone: xterm has no grapheme segmentation, so a flag already reaches two cells as 1 + 1. U+FE0E text presentation is not implemented — it would have to narrow an already-placed cluster, and xterm's printer only ever advances the cursor for a joined code point. Closes STA-6740. --- ...erminal-emoji-cell-width-agreement.test.ts | 94 +++++++++ src/shared/terminal-emoji-width-ranges.ts | 192 ++++++++++++++++++ src/shared/terminal-unicode-provider.ts | 57 +++++- 3 files changed, 342 insertions(+), 1 deletion(-) create mode 100644 src/shared/terminal-emoji-cell-width-agreement.test.ts create mode 100644 src/shared/terminal-emoji-width-ranges.ts diff --git a/src/shared/terminal-emoji-cell-width-agreement.test.ts b/src/shared/terminal-emoji-cell-width-agreement.test.ts new file mode 100644 index 00000000000..f0d32ddf660 --- /dev/null +++ b/src/shared/terminal-emoji-cell-width-agreement.test.ts @@ -0,0 +1,94 @@ +/** + * STA-6740: a cursor-position probe measured Orca advancing one cell for + * sequences other terminals advance two for. These assert the advance the way + * that probe does — by writing the bytes and reading where the cursor landed — + * because the advance, not the glyph, is what a TUI budgets its columns + * against, and a disagreement is what makes its next in-place redraw erase and + * reprint over the wrong cells. + * + * The restore case pins the same measurement on the other side of + * serialize/replay, so the snapshot path cannot grow a second width answer. + */ +import { describe, expect, it } from 'vitest' +import { + buildParityMainBufferSnapshot, + createRendererParityTerminal, + writeToTerminal, + SNAPSHOT_REPLAY_PREAMBLE_NORMAL, + type ParityTerminal +} from './terminal-restore-parity-fixture' + +/** Cells the cursor advanced for `text`, i.e. what CSI 6n would report. */ +async function cursorAdvance(parity: ParityTerminal, text: string): Promise { + await writeToTerminal(parity.terminal, `\x1b[H\x1b[2J${text}`) + return parity.terminal.buffer.active.cursorX +} + +function createTerminal(): ParityTerminal { + return createRendererParityTerminal({ cols: 80, rows: 24 }) +} + +// [label, sequence, expected cells] +const SEQUENCES: [string, string, number][] = [ + // Reported in STA-6740 as one cell in Orca, two elsewhere. + ['desktop computer U+1F5A5 U+FE0F', '\u{1F5A5}️', 2], + ['warning U+26A0 U+FE0F', '⚠️', 2], + ['keycap U+0031 U+FE0F U+20E3', '1️⃣', 2], + // Same authority, same defect class: a modifier is part of its base's cluster. + ['thumbs up + skin tone', '\u{1F44D}\u{1F3FD}', 2], + // Emoji that postdate xterm's frozen Unicode 11 width table. + ['melting face U+1FAE0', '\u{1FAE0}', 2], + ['bubble tea U+1F9CB', '\u{1F9CB}', 2], + ['playground slide U+1F6DD', '\u{1F6DD}', 2], + // Controls the report measured as already correct — these must not move. + ['CJK 中', '中', 2], + ['check mark U+2705', '✅', 2], + ['regional-indicator flag', '\u{1F1E8}\u{1F1F3}', 2], + ['ZWJ family', '\u{1F468}‍\u{1F469}‍\u{1F467}‍\u{1F466}', 2], + ['heart on fire U+2764 U+FE0F U+200D U+1F525', '❤️‍\u{1F525}', 2], + // A variation selector after a non-emoji base is not a variation sequence. + ['letter + U+FE0F', 'a️', 1], + ['plain ASCII', 'ab', 2] +] + +describe('terminal emoji cell width agreement (STA-6740)', () => { + it.each(SEQUENCES)('advances the cursor %s cells for %s', async (_label, text, expected) => { + const parity = createTerminal() + expect(await cursorAdvance(parity, text)).toBe(expected) + parity.terminal.dispose() + }) + + it('keeps the same advance after a snapshot is serialized and replayed', async () => { + const live = createTerminal() + const line = SEQUENCES.filter(([, , expected]) => expected > 0) + .map(([, text]) => text) + .join('') + await writeToTerminal(live.terminal, `\x1b[H\x1b[2J${line}`) + const liveCursor = live.terminal.buffer.active.cursorX + const snapshot = buildParityMainBufferSnapshot(live, 1) + + const restored = createTerminal() + await writeToTerminal(restored.terminal, SNAPSHOT_REPLAY_PREAMBLE_NORMAL) + await writeToTerminal(restored.terminal, snapshot.data) + + expect(restored.terminal.buffer.active.cursorX).toBe(liveCursor) + expect(restored.terminal.buffer.active.getLine(0)?.translateToString(true)).toBe( + live.terminal.buffer.active.getLine(0)?.translateToString(true) + ) + live.terminal.dispose() + restored.terminal.dispose() + }) + + it('reserves the second cell of a widened cluster instead of leaving it writable', async () => { + // Why: the reported symptom is lost characters, which needs the cell the + // cluster claims to actually be a wide-char placeholder — otherwise the next + // glyph lands inside the emoji and one of the two is overwritten. + const parity = createTerminal() + await writeToTerminal(parity.terminal, '\x1b[H\x1b[2J⚠️X') + const line = parity.terminal.buffer.active.getLine(0) + expect(line?.getCell(0)?.getWidth()).toBe(2) + expect(line?.getCell(1)?.getWidth()).toBe(0) + expect(line?.getCell(2)?.getChars()).toBe('X') + parity.terminal.dispose() + }) +}) diff --git a/src/shared/terminal-emoji-width-ranges.ts b/src/shared/terminal-emoji-width-ranges.ts new file mode 100644 index 00000000000..8bce9c0d99a --- /dev/null +++ b/src/shared/terminal-emoji-width-ranges.ts @@ -0,0 +1,192 @@ +// Emoji code-point ranges that decide how many terminal cells a grapheme +// cluster occupies. Consumed only by terminal-unicode-provider.ts, which is the +// single width authority every Orca terminal (renderer pane, headless daemon +// mirror, dashboard preview, restore-parity fixture) activates. +// +// Both tables are DERIVED, not hand-written. Regenerate against the Unicode +// character database exposed by the runtime's ICU (Unicode 17.0 when generated) +// and xterm's Unicode 11 provider: +// WIDE = \p{Emoji_Presentation} where xterm's v11 wcwidth is not 2, +// minus regional indicators U+1F1E6..U+1F1FF (see below). +// VSBASE = \p{Emoji} without \p{Emoji_Presentation} whose v11 wcwidth is 1, +// i.e. the bases a following U+FE0F promotes to emoji presentation. +// +// Regional indicators are excluded on purpose: xterm has no grapheme +// segmentation, so a flag reaches two cells as 1 + 1. Widening each indicator +// would make a flag four cells wide. + +/** Sorted, non-overlapping, ascending. */ +type CodepointRange = readonly [number, number] + +const EMOJI_PRESENTATION_WIDE_RANGES: readonly CodepointRange[] = [ + [0x1f6d6, 0x1f6d8], + [0x1f6dc, 0x1f6df], + [0x1f6fb, 0x1f6fc], + [0x1f7f0, 0x1f7f0], + [0x1f90c, 0x1f90c], + [0x1f972, 0x1f972], + [0x1f977, 0x1f979], + [0x1f9a3, 0x1f9a4], + [0x1f9ab, 0x1f9ad], + [0x1f9cb, 0x1f9cc], + [0x1fa74, 0x1fa77], + [0x1fa7b, 0x1fa7c], + [0x1fa83, 0x1fa8a], + [0x1fa8e, 0x1fa8f], + [0x1fa96, 0x1fac6], + [0x1fac8, 0x1fac8], + [0x1facd, 0x1fadc], + [0x1fadf, 0x1faea], + [0x1faef, 0x1faf8] +] + +const EMOJI_VARIATION_BASE_RANGES: readonly CodepointRange[] = [ + [0x23, 0x23], + [0x2a, 0x2a], + [0x30, 0x39], + [0xa9, 0xa9], + [0xae, 0xae], + [0x203c, 0x203c], + [0x2049, 0x2049], + [0x2122, 0x2122], + [0x2139, 0x2139], + [0x2194, 0x2199], + [0x21a9, 0x21aa], + [0x2328, 0x2328], + [0x23cf, 0x23cf], + [0x23ed, 0x23ef], + [0x23f1, 0x23f2], + [0x23f8, 0x23fa], + [0x24c2, 0x24c2], + [0x25aa, 0x25ab], + [0x25b6, 0x25b6], + [0x25c0, 0x25c0], + [0x25fb, 0x25fc], + [0x2600, 0x2604], + [0x260e, 0x260e], + [0x2611, 0x2611], + [0x2618, 0x2618], + [0x261d, 0x261d], + [0x2620, 0x2620], + [0x2622, 0x2623], + [0x2626, 0x2626], + [0x262a, 0x262a], + [0x262e, 0x262f], + [0x2638, 0x263a], + [0x2640, 0x2640], + [0x2642, 0x2642], + [0x265f, 0x2660], + [0x2663, 0x2663], + [0x2665, 0x2666], + [0x2668, 0x2668], + [0x267b, 0x267b], + [0x267e, 0x267e], + [0x2692, 0x2692], + [0x2694, 0x2697], + [0x2699, 0x2699], + [0x269b, 0x269c], + [0x26a0, 0x26a0], + [0x26a7, 0x26a7], + [0x26b0, 0x26b1], + [0x26c8, 0x26c8], + [0x26cf, 0x26cf], + [0x26d1, 0x26d1], + [0x26d3, 0x26d3], + [0x26e9, 0x26e9], + [0x26f0, 0x26f1], + [0x26f4, 0x26f4], + [0x26f7, 0x26f9], + [0x2702, 0x2702], + [0x2708, 0x2709], + [0x270c, 0x270d], + [0x270f, 0x270f], + [0x2712, 0x2712], + [0x2714, 0x2714], + [0x2716, 0x2716], + [0x271d, 0x271d], + [0x2721, 0x2721], + [0x2733, 0x2734], + [0x2744, 0x2744], + [0x2747, 0x2747], + [0x2763, 0x2764], + [0x27a1, 0x27a1], + [0x2934, 0x2935], + [0x2b05, 0x2b07], + [0x1f170, 0x1f171], + [0x1f17e, 0x1f17f], + [0x1f321, 0x1f321], + [0x1f324, 0x1f32c], + [0x1f336, 0x1f336], + [0x1f37d, 0x1f37d], + [0x1f396, 0x1f397], + [0x1f399, 0x1f39b], + [0x1f39e, 0x1f39f], + [0x1f3cb, 0x1f3ce], + [0x1f3d4, 0x1f3df], + [0x1f3f3, 0x1f3f3], + [0x1f3f5, 0x1f3f5], + [0x1f3f7, 0x1f3f7], + [0x1f43f, 0x1f43f], + [0x1f441, 0x1f441], + [0x1f4fd, 0x1f4fd], + [0x1f549, 0x1f54a], + [0x1f56f, 0x1f570], + [0x1f573, 0x1f579], + [0x1f587, 0x1f587], + [0x1f58a, 0x1f58d], + [0x1f590, 0x1f590], + [0x1f5a5, 0x1f5a5], + [0x1f5a8, 0x1f5a8], + [0x1f5b1, 0x1f5b2], + [0x1f5bc, 0x1f5bc], + [0x1f5c2, 0x1f5c4], + [0x1f5d1, 0x1f5d3], + [0x1f5dc, 0x1f5de], + [0x1f5e1, 0x1f5e1], + [0x1f5e3, 0x1f5e3], + [0x1f5e8, 0x1f5e8], + [0x1f5ef, 0x1f5ef], + [0x1f5f3, 0x1f5f3], + [0x1f5fa, 0x1f5fa], + [0x1f6cb, 0x1f6cb], + [0x1f6cd, 0x1f6cf], + [0x1f6e0, 0x1f6e5], + [0x1f6e9, 0x1f6e9], + [0x1f6f0, 0x1f6f0], + [0x1f6f3, 0x1f6f3] +] + +const EMOJI_MODIFIER_FIRST = 0x1f3fb +const EMOJI_MODIFIER_LAST = 0x1f3ff + +function inRanges(ranges: readonly CodepointRange[], codepoint: number): boolean { + let low = 0 + let high = ranges.length - 1 + while (low <= high) { + const mid = (low + high) >> 1 + const range = ranges[mid]! + if (codepoint < range[0]) { + high = mid - 1 + } else if (codepoint > range[1]) { + low = mid + 1 + } else { + return true + } + } + return false +} + +/** Emoji that default to emoji presentation but postdate xterm's frozen Unicode 11 width table. */ +export function isEmojiPresentationWideCodepoint(codepoint: number): boolean { + return codepoint >= 0x1f6d6 && inRanges(EMOJI_PRESENTATION_WIDE_RANGES, codepoint) +} + +/** A code point U+FE0F may promote to emoji presentation, per emoji-variation-sequences. */ +export function isEmojiVariationSequenceBase(codepoint: number): boolean { + return inRanges(EMOJI_VARIATION_BASE_RANGES, codepoint) +} + +/** Skin-tone modifiers U+1F3FB..U+1F3FF. */ +export function isEmojiModifier(codepoint: number): boolean { + return codepoint >= EMOJI_MODIFIER_FIRST && codepoint <= EMOJI_MODIFIER_LAST +} diff --git a/src/shared/terminal-unicode-provider.ts b/src/shared/terminal-unicode-provider.ts index ecd0fe7e782..658594f8407 100644 --- a/src/shared/terminal-unicode-provider.ts +++ b/src/shared/terminal-unicode-provider.ts @@ -1,4 +1,9 @@ import type { IUnicodeHandling, IUnicodeVersionProvider } from '@xterm/xterm' +import { + isEmojiModifier, + isEmojiPresentationWideCodepoint, + isEmojiVariationSequenceBase +} from './terminal-emoji-width-ranges' type XtermTerminalWithUnicodeCore = { unicode: IUnicodeHandling @@ -12,6 +17,7 @@ type XtermTerminalWithUnicodeCore = { const ORCA_UNICODE_VERSION = 'orca-11-zwj' const UNICODE11_VERSION = '11' const ZERO_WIDTH_JOINER = 0x200d +const EMOJI_PRESENTATION_SELECTOR = 0xfe0f function extractWidth(properties: number): 0 | 1 | 2 { return ((properties >> 1) & 3) as 0 | 1 | 2 @@ -21,16 +27,39 @@ function extractCharKind(properties: number): number { return properties >> 3 } +function extractShouldJoin(properties: number): boolean { + return (properties & 1) === 1 +} + function createProperties(charKind: number, width: 0 | 1 | 2, shouldJoin: boolean): number { return ((charKind & 0xffffff) << 3) | ((width & 3) << 1) | (shouldJoin ? 1 : 0) } +/** + * Orca's single authority for how many cells a grapheme cluster occupies. + * + * Every terminal Orca runs activates this provider, so the live pane, the + * headless daemon mirror the snapshot/restore path serializes from, and the + * dashboard preview budget columns identically. The rules below exist because + * xterm's Unicode 11 tables measure single code points against a table frozen + * in 2018, while the TUIs writing into the pane measure whole clusters against + * a current one — and every cell of disagreement shifts the columns an in-place + * redraw erases and reprints. + * + * Not implemented: U+FE0E text presentation, which would have to narrow an + * already-placed cluster. xterm's printer only ever advances the cursor for a + * joined code point, so a narrowing cluster would leave the cell width and the + * cursor disagreeing. + */ class OrcaUnicodeProvider implements IUnicodeVersionProvider { public readonly version = ORCA_UNICODE_VERSION public constructor(private readonly baseProvider: IUnicodeVersionProvider) {} public wcwidth(codepoint: number): 0 | 1 | 2 { + if (isEmojiPresentationWideCodepoint(codepoint)) { + return 2 + } return this.baseProvider.wcwidth(codepoint) } @@ -48,7 +77,33 @@ class OrcaUnicodeProvider implements IUnicodeVersionProvider { return createProperties(codepoint, precedingWidth, true) } - return this.baseProvider.charProperties(codepoint, preceding) + if ( + codepoint === EMOJI_PRESENTATION_SELECTOR && + precedingWidth === 1 && + isEmojiVariationSequenceBase(precedingKind) + ) { + // Why: U+FE0F switches its base to emoji presentation, which every other + // terminal advances two cells for; xterm keeps the text-presentation cell. + return createProperties(codepoint, 2, true) + } + + if (isEmojiModifier(codepoint) && precedingWidth === 2) { + // Why: a skin-tone modifier belongs to its base's cluster. Left to the v11 + // table it lands as a second wide cell and every later column shifts by two. + return createProperties(codepoint, 2, true) + } + + const base = this.baseProvider.charProperties(codepoint, preceding) + const shouldJoin = extractShouldJoin(base) + // Why re-encode: the base provider reports char kind 0 and derives width + // from its own wcwidth, so the rules above would lose both the preceding + // code point and this provider's post-Unicode-11 widths. A joining code + // point keeps the cluster width the base already resolved. + return createProperties( + codepoint, + shouldJoin ? extractWidth(base) : this.wcwidth(codepoint), + shouldJoin + ) } }