fix(terminal): make the Orca width provider measure emoji clusters like other terminals

A cursor-position probe (STA-6740) measured Orca advancing one cell for
sequences Windows Terminal and Ghostty advance two for. The cause is that
`OrcaUnicodeProvider` delegated every width decision to xterm's Unicode 11
tables, which measure single code points against a table frozen in 2018, while
the TUIs writing into the pane measure whole grapheme clusters against a
current one. Each cell of disagreement shifts the columns a TUI's next in-place
redraw erases and reprints over.

The provider is already the single width authority every Orca terminal
activates — the live renderer pane, the headless daemon mirror that
snapshot/restore serializes from, the dashboard preview, and the restore-parity
fixture. This teaches that one authority three cluster rules it was missing:

- U+FE0F promotes a valid emoji-variation base from one cell to two;
- a skin-tone modifier joins its base's cluster instead of claiming a second
  wide cell (a `+2` column shift for every glyph after it);
- emoji that default to emoji presentation but postdate Unicode 11 are two
  cells, not one.

Regional indicators are deliberately left alone: xterm has no grapheme
segmentation, so a flag already reaches two cells as 1 + 1.

U+FE0E text presentation is not implemented — it would have to narrow an
already-placed cluster, and xterm's printer only ever advances the cursor for a
joined code point.

Closes STA-6740.
This commit is contained in:
Merge Sim
2026-09-09 01:58:09 -07:00
parent 65631e449a
commit e949f3d422
3 changed files with 342 additions and 1 deletions
@@ -0,0 +1,94 @@
/**
* STA-6740: a cursor-position probe measured Orca advancing one cell for
* sequences other terminals advance two for. These assert the advance the way
* that probe does — by writing the bytes and reading where the cursor landed —
* because the advance, not the glyph, is what a TUI budgets its columns
* against, and a disagreement is what makes its next in-place redraw erase and
* reprint over the wrong cells.
*
* The restore case pins the same measurement on the other side of
* serialize/replay, so the snapshot path cannot grow a second width answer.
*/
import { describe, expect, it } from 'vitest'
import {
buildParityMainBufferSnapshot,
createRendererParityTerminal,
writeToTerminal,
SNAPSHOT_REPLAY_PREAMBLE_NORMAL,
type ParityTerminal
} from './terminal-restore-parity-fixture'
/** Cells the cursor advanced for `text`, i.e. what CSI 6n would report. */
async function cursorAdvance(parity: ParityTerminal, text: string): Promise<number> {
await writeToTerminal(parity.terminal, `\x1b[H\x1b[2J${text}`)
return parity.terminal.buffer.active.cursorX
}
function createTerminal(): ParityTerminal {
return createRendererParityTerminal({ cols: 80, rows: 24 })
}
// [label, sequence, expected cells]
const SEQUENCES: [string, string, number][] = [
// Reported in STA-6740 as one cell in Orca, two elsewhere.
['desktop computer U+1F5A5 U+FE0F', '\u{1F5A5}️', 2],
['warning U+26A0 U+FE0F', '⚠️', 2],
['keycap U+0031 U+FE0F U+20E3', '1️⃣', 2],
// Same authority, same defect class: a modifier is part of its base's cluster.
['thumbs up + skin tone', '\u{1F44D}\u{1F3FD}', 2],
// Emoji that postdate xterm's frozen Unicode 11 width table.
['melting face U+1FAE0', '\u{1FAE0}', 2],
['bubble tea U+1F9CB', '\u{1F9CB}', 2],
['playground slide U+1F6DD', '\u{1F6DD}', 2],
// Controls the report measured as already correct — these must not move.
['CJK 中', '中', 2],
['check mark U+2705', '✅', 2],
['regional-indicator flag', '\u{1F1E8}\u{1F1F3}', 2],
['ZWJ family', '\u{1F468}‍\u{1F469}‍\u{1F467}‍\u{1F466}', 2],
['heart on fire U+2764 U+FE0F U+200D U+1F525', '❤️‍\u{1F525}', 2],
// A variation selector after a non-emoji base is not a variation sequence.
['letter + U+FE0F', 'a️', 1],
['plain ASCII', 'ab', 2]
]
describe('terminal emoji cell width agreement (STA-6740)', () => {
it.each(SEQUENCES)('advances the cursor %s cells for %s', async (_label, text, expected) => {
const parity = createTerminal()
expect(await cursorAdvance(parity, text)).toBe(expected)
parity.terminal.dispose()
})
it('keeps the same advance after a snapshot is serialized and replayed', async () => {
const live = createTerminal()
const line = SEQUENCES.filter(([, , expected]) => expected > 0)
.map(([, text]) => text)
.join('')
await writeToTerminal(live.terminal, `\x1b[H\x1b[2J${line}`)
const liveCursor = live.terminal.buffer.active.cursorX
const snapshot = buildParityMainBufferSnapshot(live, 1)
const restored = createTerminal()
await writeToTerminal(restored.terminal, SNAPSHOT_REPLAY_PREAMBLE_NORMAL)
await writeToTerminal(restored.terminal, snapshot.data)
expect(restored.terminal.buffer.active.cursorX).toBe(liveCursor)
expect(restored.terminal.buffer.active.getLine(0)?.translateToString(true)).toBe(
live.terminal.buffer.active.getLine(0)?.translateToString(true)
)
live.terminal.dispose()
restored.terminal.dispose()
})
it('reserves the second cell of a widened cluster instead of leaving it writable', async () => {
// Why: the reported symptom is lost characters, which needs the cell the
// cluster claims to actually be a wide-char placeholder — otherwise the next
// glyph lands inside the emoji and one of the two is overwritten.
const parity = createTerminal()
await writeToTerminal(parity.terminal, '\x1b[H\x1b[2J⚠️X')
const line = parity.terminal.buffer.active.getLine(0)
expect(line?.getCell(0)?.getWidth()).toBe(2)
expect(line?.getCell(1)?.getWidth()).toBe(0)
expect(line?.getCell(2)?.getChars()).toBe('X')
parity.terminal.dispose()
})
})
+192
View File
@@ -0,0 +1,192 @@
// Emoji code-point ranges that decide how many terminal cells a grapheme
// cluster occupies. Consumed only by terminal-unicode-provider.ts, which is the
// single width authority every Orca terminal (renderer pane, headless daemon
// mirror, dashboard preview, restore-parity fixture) activates.
//
// Both tables are DERIVED, not hand-written. Regenerate against the Unicode
// character database exposed by the runtime's ICU (Unicode 17.0 when generated)
// and xterm's Unicode 11 provider:
// WIDE = \p{Emoji_Presentation} where xterm's v11 wcwidth is not 2,
// minus regional indicators U+1F1E6..U+1F1FF (see below).
// VSBASE = \p{Emoji} without \p{Emoji_Presentation} whose v11 wcwidth is 1,
// i.e. the bases a following U+FE0F promotes to emoji presentation.
//
// Regional indicators are excluded on purpose: xterm has no grapheme
// segmentation, so a flag reaches two cells as 1 + 1. Widening each indicator
// would make a flag four cells wide.
/** Sorted, non-overlapping, ascending. */
type CodepointRange = readonly [number, number]
const EMOJI_PRESENTATION_WIDE_RANGES: readonly CodepointRange[] = [
[0x1f6d6, 0x1f6d8],
[0x1f6dc, 0x1f6df],
[0x1f6fb, 0x1f6fc],
[0x1f7f0, 0x1f7f0],
[0x1f90c, 0x1f90c],
[0x1f972, 0x1f972],
[0x1f977, 0x1f979],
[0x1f9a3, 0x1f9a4],
[0x1f9ab, 0x1f9ad],
[0x1f9cb, 0x1f9cc],
[0x1fa74, 0x1fa77],
[0x1fa7b, 0x1fa7c],
[0x1fa83, 0x1fa8a],
[0x1fa8e, 0x1fa8f],
[0x1fa96, 0x1fac6],
[0x1fac8, 0x1fac8],
[0x1facd, 0x1fadc],
[0x1fadf, 0x1faea],
[0x1faef, 0x1faf8]
]
const EMOJI_VARIATION_BASE_RANGES: readonly CodepointRange[] = [
[0x23, 0x23],
[0x2a, 0x2a],
[0x30, 0x39],
[0xa9, 0xa9],
[0xae, 0xae],
[0x203c, 0x203c],
[0x2049, 0x2049],
[0x2122, 0x2122],
[0x2139, 0x2139],
[0x2194, 0x2199],
[0x21a9, 0x21aa],
[0x2328, 0x2328],
[0x23cf, 0x23cf],
[0x23ed, 0x23ef],
[0x23f1, 0x23f2],
[0x23f8, 0x23fa],
[0x24c2, 0x24c2],
[0x25aa, 0x25ab],
[0x25b6, 0x25b6],
[0x25c0, 0x25c0],
[0x25fb, 0x25fc],
[0x2600, 0x2604],
[0x260e, 0x260e],
[0x2611, 0x2611],
[0x2618, 0x2618],
[0x261d, 0x261d],
[0x2620, 0x2620],
[0x2622, 0x2623],
[0x2626, 0x2626],
[0x262a, 0x262a],
[0x262e, 0x262f],
[0x2638, 0x263a],
[0x2640, 0x2640],
[0x2642, 0x2642],
[0x265f, 0x2660],
[0x2663, 0x2663],
[0x2665, 0x2666],
[0x2668, 0x2668],
[0x267b, 0x267b],
[0x267e, 0x267e],
[0x2692, 0x2692],
[0x2694, 0x2697],
[0x2699, 0x2699],
[0x269b, 0x269c],
[0x26a0, 0x26a0],
[0x26a7, 0x26a7],
[0x26b0, 0x26b1],
[0x26c8, 0x26c8],
[0x26cf, 0x26cf],
[0x26d1, 0x26d1],
[0x26d3, 0x26d3],
[0x26e9, 0x26e9],
[0x26f0, 0x26f1],
[0x26f4, 0x26f4],
[0x26f7, 0x26f9],
[0x2702, 0x2702],
[0x2708, 0x2709],
[0x270c, 0x270d],
[0x270f, 0x270f],
[0x2712, 0x2712],
[0x2714, 0x2714],
[0x2716, 0x2716],
[0x271d, 0x271d],
[0x2721, 0x2721],
[0x2733, 0x2734],
[0x2744, 0x2744],
[0x2747, 0x2747],
[0x2763, 0x2764],
[0x27a1, 0x27a1],
[0x2934, 0x2935],
[0x2b05, 0x2b07],
[0x1f170, 0x1f171],
[0x1f17e, 0x1f17f],
[0x1f321, 0x1f321],
[0x1f324, 0x1f32c],
[0x1f336, 0x1f336],
[0x1f37d, 0x1f37d],
[0x1f396, 0x1f397],
[0x1f399, 0x1f39b],
[0x1f39e, 0x1f39f],
[0x1f3cb, 0x1f3ce],
[0x1f3d4, 0x1f3df],
[0x1f3f3, 0x1f3f3],
[0x1f3f5, 0x1f3f5],
[0x1f3f7, 0x1f3f7],
[0x1f43f, 0x1f43f],
[0x1f441, 0x1f441],
[0x1f4fd, 0x1f4fd],
[0x1f549, 0x1f54a],
[0x1f56f, 0x1f570],
[0x1f573, 0x1f579],
[0x1f587, 0x1f587],
[0x1f58a, 0x1f58d],
[0x1f590, 0x1f590],
[0x1f5a5, 0x1f5a5],
[0x1f5a8, 0x1f5a8],
[0x1f5b1, 0x1f5b2],
[0x1f5bc, 0x1f5bc],
[0x1f5c2, 0x1f5c4],
[0x1f5d1, 0x1f5d3],
[0x1f5dc, 0x1f5de],
[0x1f5e1, 0x1f5e1],
[0x1f5e3, 0x1f5e3],
[0x1f5e8, 0x1f5e8],
[0x1f5ef, 0x1f5ef],
[0x1f5f3, 0x1f5f3],
[0x1f5fa, 0x1f5fa],
[0x1f6cb, 0x1f6cb],
[0x1f6cd, 0x1f6cf],
[0x1f6e0, 0x1f6e5],
[0x1f6e9, 0x1f6e9],
[0x1f6f0, 0x1f6f0],
[0x1f6f3, 0x1f6f3]
]
const EMOJI_MODIFIER_FIRST = 0x1f3fb
const EMOJI_MODIFIER_LAST = 0x1f3ff
function inRanges(ranges: readonly CodepointRange[], codepoint: number): boolean {
let low = 0
let high = ranges.length - 1
while (low <= high) {
const mid = (low + high) >> 1
const range = ranges[mid]!
if (codepoint < range[0]) {
high = mid - 1
} else if (codepoint > range[1]) {
low = mid + 1
} else {
return true
}
}
return false
}
/** Emoji that default to emoji presentation but postdate xterm's frozen Unicode 11 width table. */
export function isEmojiPresentationWideCodepoint(codepoint: number): boolean {
return codepoint >= 0x1f6d6 && inRanges(EMOJI_PRESENTATION_WIDE_RANGES, codepoint)
}
/** A code point U+FE0F may promote to emoji presentation, per emoji-variation-sequences. */
export function isEmojiVariationSequenceBase(codepoint: number): boolean {
return inRanges(EMOJI_VARIATION_BASE_RANGES, codepoint)
}
/** Skin-tone modifiers U+1F3FB..U+1F3FF. */
export function isEmojiModifier(codepoint: number): boolean {
return codepoint >= EMOJI_MODIFIER_FIRST && codepoint <= EMOJI_MODIFIER_LAST
}
+56 -1
View File
@@ -1,4 +1,9 @@
import type { IUnicodeHandling, IUnicodeVersionProvider } from '@xterm/xterm'
import {
isEmojiModifier,
isEmojiPresentationWideCodepoint,
isEmojiVariationSequenceBase
} from './terminal-emoji-width-ranges'
type XtermTerminalWithUnicodeCore = {
unicode: IUnicodeHandling
@@ -12,6 +17,7 @@ type XtermTerminalWithUnicodeCore = {
const ORCA_UNICODE_VERSION = 'orca-11-zwj'
const UNICODE11_VERSION = '11'
const ZERO_WIDTH_JOINER = 0x200d
const EMOJI_PRESENTATION_SELECTOR = 0xfe0f
function extractWidth(properties: number): 0 | 1 | 2 {
return ((properties >> 1) & 3) as 0 | 1 | 2
@@ -21,16 +27,39 @@ function extractCharKind(properties: number): number {
return properties >> 3
}
function extractShouldJoin(properties: number): boolean {
return (properties & 1) === 1
}
function createProperties(charKind: number, width: 0 | 1 | 2, shouldJoin: boolean): number {
return ((charKind & 0xffffff) << 3) | ((width & 3) << 1) | (shouldJoin ? 1 : 0)
}
/**
* Orca's single authority for how many cells a grapheme cluster occupies.
*
* Every terminal Orca runs activates this provider, so the live pane, the
* headless daemon mirror the snapshot/restore path serializes from, and the
* dashboard preview budget columns identically. The rules below exist because
* xterm's Unicode 11 tables measure single code points against a table frozen
* in 2018, while the TUIs writing into the pane measure whole clusters against
* a current one — and every cell of disagreement shifts the columns an in-place
* redraw erases and reprints.
*
* Not implemented: U+FE0E text presentation, which would have to narrow an
* already-placed cluster. xterm's printer only ever advances the cursor for a
* joined code point, so a narrowing cluster would leave the cell width and the
* cursor disagreeing.
*/
class OrcaUnicodeProvider implements IUnicodeVersionProvider {
public readonly version = ORCA_UNICODE_VERSION
public constructor(private readonly baseProvider: IUnicodeVersionProvider) {}
public wcwidth(codepoint: number): 0 | 1 | 2 {
if (isEmojiPresentationWideCodepoint(codepoint)) {
return 2
}
return this.baseProvider.wcwidth(codepoint)
}
@@ -48,7 +77,33 @@ class OrcaUnicodeProvider implements IUnicodeVersionProvider {
return createProperties(codepoint, precedingWidth, true)
}
return this.baseProvider.charProperties(codepoint, preceding)
if (
codepoint === EMOJI_PRESENTATION_SELECTOR &&
precedingWidth === 1 &&
isEmojiVariationSequenceBase(precedingKind)
) {
// Why: U+FE0F switches its base to emoji presentation, which every other
// terminal advances two cells for; xterm keeps the text-presentation cell.
return createProperties(codepoint, 2, true)
}
if (isEmojiModifier(codepoint) && precedingWidth === 2) {
// Why: a skin-tone modifier belongs to its base's cluster. Left to the v11
// table it lands as a second wide cell and every later column shifts by two.
return createProperties(codepoint, 2, true)
}
const base = this.baseProvider.charProperties(codepoint, preceding)
const shouldJoin = extractShouldJoin(base)
// Why re-encode: the base provider reports char kind 0 and derives width
// from its own wcwidth, so the rules above would lose both the preceding
// code point and this provider's post-Unicode-11 widths. A joining code
// point keeps the cluster width the base already resolved.
return createProperties(
codepoint,
shouldJoin ? extractWidth(base) : this.wcwidth(codepoint),
shouldJoin
)
}
}