mirror of
https://github.com/stablyai/orca.git
synced 2026-10-01 08:01:56 +00:00
fix(terminal): scope the skin-tone rule to modifier bases, and gate the tables against staleness
Review findings on the width-authority change. The skin-tone rule gated only on the preceding cell being two wide, never on it being something a modifier can attach to. A modifier after a CJK ideograph joined it: `中🏽` measured two cells where it measured four before. The rule now also requires the preceding code point to be an emoji modifier base, which is the standard's own definition of what a modifier may modify — tighter than Extended_Pictographic and, for anything that is not a well-formed modifier sequence, it restores exactly the measurement this provider had before. The frozen tables could go stale the same way xterm's Unicode 11 table did, so the derivation is now committed as a test: it re-derives all three tables from the runtime's Unicode data and xterm's Unicode 11 provider and compares, and reads nothing from the committed ranges to build that expectation. "No missing entries" runs always — an older runtime cannot fail it spuriously — and exact equality runs on the Unicode version the tables were generated against. The module header now records why the tables are frozen at all: this width is a cross-host contract and Electron, Node and the mobile JSC embed different ICU versions, so deriving at runtime would make two Orca processes disagree about one buffer. Also: reuse the width the base provider already resolved instead of asking for it twice on the hot path, and fix two comments that described the ZWJ case as a control and claimed the reported symptom as established. Refs STA-6740, STA-7066.
This commit is contained in:
@@ -34,8 +34,11 @@ const SEQUENCES: [string, string, number][] = [
|
||||
['desktop computer U+1F5A5 U+FE0F', '\u{1F5A5}️', 2],
|
||||
['warning U+26A0 U+FE0F', '⚠️', 2],
|
||||
['keycap U+0031 U+FE0F U+20E3', '1️⃣', 2],
|
||||
// A variation selector inside a ZWJ cluster is the same promotion, one level in.
|
||||
['heart on fire U+2764 U+FE0F U+200D U+1F525', '❤️\u{1F525}', 2],
|
||||
// Same authority, same defect class: a modifier is part of its base's cluster.
|
||||
['thumbs up + skin tone', '\u{1F44D}\u{1F3FD}', 2],
|
||||
['person + skin tone + ZWJ role', '\u{1F9D1}\u{1F3FD}\u{1F4BB}', 2],
|
||||
// Emoji that postdate xterm's frozen Unicode 11 width table.
|
||||
['melting face U+1FAE0', '\u{1FAE0}', 2],
|
||||
['bubble tea U+1F9CB', '\u{1F9CB}', 2],
|
||||
@@ -45,7 +48,11 @@ const SEQUENCES: [string, string, number][] = [
|
||||
['check mark U+2705', '✅', 2],
|
||||
['regional-indicator flag', '\u{1F1E8}\u{1F1F3}', 2],
|
||||
['ZWJ family', '\u{1F468}\u{1F469}\u{1F467}\u{1F466}', 2],
|
||||
['heart on fire U+2764 U+FE0F U+200D U+1F525', '❤️\u{1F525}', 2],
|
||||
// A modifier only belongs to a base that takes one. After anything else it is
|
||||
// its own cluster, exactly as it was before this provider learned the rule.
|
||||
['CJK + skin tone', '中\u{1F3FD}', 4],
|
||||
['non-modifier-base emoji + skin tone', '\u{1F355}\u{1F3FD}', 4],
|
||||
['ASCII + skin tone', 'a\u{1F3FD}', 3],
|
||||
// A variation selector after a non-emoji base is not a variation sequence.
|
||||
['letter + U+FE0F', 'a️', 1],
|
||||
['plain ASCII', 'ab', 2]
|
||||
@@ -80,9 +87,9 @@ describe('terminal emoji cell width agreement (STA-6740)', () => {
|
||||
})
|
||||
|
||||
it('reserves the second cell of a widened cluster instead of leaving it writable', async () => {
|
||||
// Why: the reported symptom is lost characters, which needs the cell the
|
||||
// cluster claims to actually be a wide-char placeholder — otherwise the next
|
||||
// glyph lands inside the emoji and one of the two is overwritten.
|
||||
// Why: advancing two cells is only half the contract. The second cell must
|
||||
// also be a wide-char placeholder, or the next glyph lands inside the
|
||||
// cluster and one of the two is overwritten.
|
||||
const parity = createTerminal()
|
||||
await writeToTerminal(parity.terminal, '\x1b[H\x1b[2J⚠️X')
|
||||
const line = parity.terminal.buffer.active.getLine(0)
|
||||
|
||||
@@ -0,0 +1,162 @@
|
||||
/**
|
||||
* Staleness gate for the frozen tables in terminal-emoji-width-ranges.ts.
|
||||
*
|
||||
* The bug those tables fix is a width table that stopped tracking Unicode.
|
||||
* A frozen table can acquire that same bug silently, so this re-derives all
|
||||
* three from the runtime's own Unicode data and xterm's Unicode 11 provider —
|
||||
* the recipe recorded in the module header — and compares. Nothing here reads
|
||||
* the committed ranges to build its expectation.
|
||||
*
|
||||
* Direction matters, because the tables are frozen at one Unicode version and
|
||||
* this runs on whatever the test runner embeds:
|
||||
* - "no missing entries" runs always. An older runtime simply has not assigned
|
||||
* the newest code points, so it can never fail this direction spuriously,
|
||||
* and a newer one fails it exactly when the table has gone stale.
|
||||
* - exact equality runs only when the runtime's Unicode version matches the
|
||||
* version the tables were generated against, where any difference is real.
|
||||
*/
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { Terminal } from '@xterm/headless'
|
||||
import { Unicode11Addon } from '@xterm/addon-unicode11'
|
||||
import {
|
||||
EMOJI_MODIFIER_BASE_RANGES,
|
||||
EMOJI_PRESENTATION_WIDE_RANGES,
|
||||
EMOJI_VARIATION_BASE_RANGES,
|
||||
isEmojiModifierBase,
|
||||
isEmojiPresentationWideCodepoint,
|
||||
isEmojiVariationSequenceBase,
|
||||
type CodepointRange
|
||||
} from './terminal-emoji-width-ranges'
|
||||
|
||||
/** The version the header records the tables as generated against. */
|
||||
const GENERATED_AGAINST_UNICODE = '17.0'
|
||||
|
||||
const LAST_CODEPOINT = 0x10ffff
|
||||
const SURROGATE_FIRST = 0xd800
|
||||
const SURROGATE_LAST = 0xdfff
|
||||
const REGIONAL_INDICATOR_FIRST = 0x1f1e6
|
||||
const REGIONAL_INDICATOR_LAST = 0x1f1ff
|
||||
|
||||
const EMOJI = /\p{Emoji}/u
|
||||
const EMOJI_PRESENTATION = /\p{Emoji_Presentation}/u
|
||||
const EMOJI_MODIFIER_BASE = /\p{Emoji_Modifier_Base}/u
|
||||
|
||||
type UnicodeServiceInternals = { activeVersion: string; wcwidth(codepoint: number): number }
|
||||
|
||||
function openUnicode11(): { terminal: Terminal; unicode: UnicodeServiceInternals } {
|
||||
const terminal = new Terminal({ cols: 40, rows: 10, allowProposedApi: true })
|
||||
terminal.loadAddon(new Unicode11Addon())
|
||||
const unicode = (terminal as unknown as { _core: { unicodeService: UnicodeServiceInternals } })
|
||||
._core.unicodeService
|
||||
unicode.activeVersion = '11'
|
||||
return { terminal, unicode }
|
||||
}
|
||||
|
||||
/** Re-derives the three sets straight from the recipe in the module header. */
|
||||
function deriveTables(unicode: UnicodeServiceInternals): {
|
||||
wide: number[]
|
||||
variationBase: number[]
|
||||
modifierBase: number[]
|
||||
} {
|
||||
const wide: number[] = []
|
||||
const variationBase: number[] = []
|
||||
const modifierBase: number[] = []
|
||||
for (let codepoint = 0; codepoint <= LAST_CODEPOINT; codepoint += 1) {
|
||||
if (codepoint >= SURROGATE_FIRST && codepoint <= SURROGATE_LAST) {
|
||||
continue
|
||||
}
|
||||
const char = String.fromCodePoint(codepoint)
|
||||
if (EMOJI_MODIFIER_BASE.test(char)) {
|
||||
modifierBase.push(codepoint)
|
||||
}
|
||||
if (!EMOJI.test(char)) {
|
||||
continue
|
||||
}
|
||||
if (EMOJI_PRESENTATION.test(char)) {
|
||||
const isRegionalIndicator =
|
||||
codepoint >= REGIONAL_INDICATOR_FIRST && codepoint <= REGIONAL_INDICATOR_LAST
|
||||
if (unicode.wcwidth(codepoint) !== 2 && !isRegionalIndicator) {
|
||||
wide.push(codepoint)
|
||||
}
|
||||
} else if (unicode.wcwidth(codepoint) === 1) {
|
||||
variationBase.push(codepoint)
|
||||
}
|
||||
}
|
||||
return { wide, variationBase, modifierBase }
|
||||
}
|
||||
|
||||
function expandRanges(ranges: readonly CodepointRange[]): number[] {
|
||||
const codepoints: number[] = []
|
||||
for (const [first, last] of ranges) {
|
||||
for (let codepoint = first; codepoint <= last; codepoint += 1) {
|
||||
codepoints.push(codepoint)
|
||||
}
|
||||
}
|
||||
return codepoints
|
||||
}
|
||||
|
||||
const hex = (value: number): string => `U+${value.toString(16).toUpperCase().padStart(4, '0')}`
|
||||
|
||||
/** Contiguous code points collapse into one entry so a drift reads as ranges. */
|
||||
function summarize(codepoints: number[]): string[] {
|
||||
const runs: [number, number][] = []
|
||||
for (const codepoint of codepoints) {
|
||||
const last = runs.at(-1)
|
||||
if (last && last[1] === codepoint - 1) {
|
||||
last[1] = codepoint
|
||||
continue
|
||||
}
|
||||
runs.push([codepoint, codepoint])
|
||||
}
|
||||
return runs.map(([first, last]) => (first === last ? hex(first) : `${hex(first)}..${hex(last)}`))
|
||||
}
|
||||
|
||||
const TABLES: [string, readonly CodepointRange[], (codepoint: number) => boolean][] = [
|
||||
['emoji presentation wide', EMOJI_PRESENTATION_WIDE_RANGES, isEmojiPresentationWideCodepoint],
|
||||
['emoji variation base', EMOJI_VARIATION_BASE_RANGES, isEmojiVariationSequenceBase],
|
||||
['emoji modifier base', EMOJI_MODIFIER_BASE_RANGES, isEmojiModifierBase]
|
||||
]
|
||||
|
||||
describe('emoji width range derivation', () => {
|
||||
const { terminal, unicode } = openUnicode11()
|
||||
const derived = deriveTables(unicode)
|
||||
const derivedByName: Record<string, number[]> = {
|
||||
'emoji presentation wide': derived.wide,
|
||||
'emoji variation base': derived.variationBase,
|
||||
'emoji modifier base': derived.modifierBase
|
||||
}
|
||||
terminal.dispose()
|
||||
|
||||
it.each(TABLES)('%s: the committed table holds every derived code point', (name, ranges) => {
|
||||
const committed = new Set(expandRanges(ranges))
|
||||
const missing = derivedByName[name]!.filter((codepoint) => !committed.has(codepoint))
|
||||
expect(summarize(missing)).toEqual([])
|
||||
})
|
||||
|
||||
it.each(TABLES)('%s: the lookup agrees with the committed ranges', (name, ranges, lookup) => {
|
||||
// Guards the binary search and the fast-path bound, not just the data.
|
||||
const disagreeing = expandRanges(ranges).filter((codepoint) => !lookup(codepoint))
|
||||
expect(summarize(disagreeing)).toEqual([])
|
||||
expect(lookup(SURROGATE_FIRST)).toBe(false)
|
||||
})
|
||||
|
||||
it.each(TABLES)('%s: matches the derivation exactly on the generated version', (name, ranges) => {
|
||||
if (process.versions.unicode !== GENERATED_AGAINST_UNICODE) {
|
||||
// A different embedded Unicode version legitimately derives a different
|
||||
// set; the "no missing entries" case above still gates staleness there.
|
||||
expect(process.versions.unicode).toBeTypeOf('string')
|
||||
return
|
||||
}
|
||||
expect(summarize(expandRanges(ranges))).toEqual(summarize(derivedByName[name]!))
|
||||
})
|
||||
|
||||
it('excludes regional indicators so a flag stays two cells', () => {
|
||||
for (
|
||||
let codepoint = REGIONAL_INDICATOR_FIRST;
|
||||
codepoint <= REGIONAL_INDICATOR_LAST;
|
||||
codepoint += 1
|
||||
) {
|
||||
expect(isEmojiPresentationWideCodepoint(codepoint)).toBe(false)
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -3,22 +3,40 @@
|
||||
// single width authority every Orca terminal (renderer pane, headless daemon
|
||||
// mirror, dashboard preview, restore-parity fixture) activates.
|
||||
//
|
||||
// Both tables are DERIVED, not hand-written. Regenerate against the Unicode
|
||||
// character database exposed by the runtime's ICU (Unicode 17.0 when generated)
|
||||
// and xterm's Unicode 11 provider:
|
||||
// WHY THE TABLES ARE FROZEN HERE rather than read from the runtime's Unicode
|
||||
// data: this width is a cross-host contract. A pane's cells are laid out by
|
||||
// whichever host runs the terminal and re-derived by whichever client renders
|
||||
// it, and Electron's V8, plain Node and the mobile JSC ship different ICU
|
||||
// versions. Deriving at runtime would make two Orca processes disagree about
|
||||
// the same buffer purely because they embed different Unicode data — the exact
|
||||
// failure this module exists to remove. Freezing the table moves the version
|
||||
// skew to something a release controls.
|
||||
//
|
||||
// WHY NOT xterm's grapheme-clustering provider: it is upstream-experimental,
|
||||
// leaves every post-Unicode-11 emoji below at one cell, and disagrees with
|
||||
// other terminals on ZWJ-plus-variation clusters. It does not answer the
|
||||
// measurement this module exists to fix.
|
||||
//
|
||||
// The tables are DERIVED, not hand-written, and
|
||||
// terminal-emoji-width-ranges.derivation.test.ts re-derives all three from the
|
||||
// runtime's Unicode data on every run so they cannot go stale the way xterm's
|
||||
// Unicode 11 table did:
|
||||
// WIDE = \p{Emoji_Presentation} where xterm's v11 wcwidth is not 2,
|
||||
// minus regional indicators U+1F1E6..U+1F1FF (see below).
|
||||
// VSBASE = \p{Emoji} without \p{Emoji_Presentation} whose v11 wcwidth is 1,
|
||||
// i.e. the bases a following U+FE0F promotes to emoji presentation.
|
||||
// MODBASE= \p{Emoji_Modifier_Base}, i.e. the bases a skin-tone modifier may
|
||||
// attach to.
|
||||
// Generated against Unicode 17.0.
|
||||
//
|
||||
// Regional indicators are excluded on purpose: xterm has no grapheme
|
||||
// segmentation, so a flag reaches two cells as 1 + 1. Widening each indicator
|
||||
// would make a flag four cells wide.
|
||||
|
||||
/** Sorted, non-overlapping, ascending. */
|
||||
type CodepointRange = readonly [number, number]
|
||||
export type CodepointRange = readonly [number, number]
|
||||
|
||||
const EMOJI_PRESENTATION_WIDE_RANGES: readonly CodepointRange[] = [
|
||||
export const EMOJI_PRESENTATION_WIDE_RANGES: readonly CodepointRange[] = [
|
||||
[0x1f6d6, 0x1f6d8],
|
||||
[0x1f6dc, 0x1f6df],
|
||||
[0x1f6fb, 0x1f6fc],
|
||||
@@ -40,7 +58,7 @@ const EMOJI_PRESENTATION_WIDE_RANGES: readonly CodepointRange[] = [
|
||||
[0x1faef, 0x1faf8]
|
||||
]
|
||||
|
||||
const EMOJI_VARIATION_BASE_RANGES: readonly CodepointRange[] = [
|
||||
export const EMOJI_VARIATION_BASE_RANGES: readonly CodepointRange[] = [
|
||||
[0x23, 0x23],
|
||||
[0x2a, 0x2a],
|
||||
[0x30, 0x39],
|
||||
@@ -156,6 +174,52 @@ const EMOJI_VARIATION_BASE_RANGES: readonly CodepointRange[] = [
|
||||
[0x1f6f3, 0x1f6f3]
|
||||
]
|
||||
|
||||
export const EMOJI_MODIFIER_BASE_RANGES: readonly CodepointRange[] = [
|
||||
[0x261d, 0x261d],
|
||||
[0x26f9, 0x26f9],
|
||||
[0x270a, 0x270d],
|
||||
[0x1f385, 0x1f385],
|
||||
[0x1f3c2, 0x1f3c4],
|
||||
[0x1f3c7, 0x1f3c7],
|
||||
[0x1f3ca, 0x1f3cc],
|
||||
[0x1f442, 0x1f443],
|
||||
[0x1f446, 0x1f450],
|
||||
[0x1f466, 0x1f478],
|
||||
[0x1f47c, 0x1f47c],
|
||||
[0x1f481, 0x1f483],
|
||||
[0x1f485, 0x1f487],
|
||||
[0x1f48f, 0x1f48f],
|
||||
[0x1f491, 0x1f491],
|
||||
[0x1f4aa, 0x1f4aa],
|
||||
[0x1f574, 0x1f575],
|
||||
[0x1f57a, 0x1f57a],
|
||||
[0x1f590, 0x1f590],
|
||||
[0x1f595, 0x1f596],
|
||||
[0x1f645, 0x1f647],
|
||||
[0x1f64b, 0x1f64f],
|
||||
[0x1f6a3, 0x1f6a3],
|
||||
[0x1f6b4, 0x1f6b6],
|
||||
[0x1f6c0, 0x1f6c0],
|
||||
[0x1f6cc, 0x1f6cc],
|
||||
[0x1f90c, 0x1f90c],
|
||||
[0x1f90f, 0x1f90f],
|
||||
[0x1f918, 0x1f91f],
|
||||
[0x1f926, 0x1f926],
|
||||
[0x1f930, 0x1f939],
|
||||
[0x1f93c, 0x1f93e],
|
||||
[0x1f977, 0x1f977],
|
||||
[0x1f9b5, 0x1f9b6],
|
||||
[0x1f9b8, 0x1f9b9],
|
||||
[0x1f9bb, 0x1f9bb],
|
||||
[0x1f9cd, 0x1f9cf],
|
||||
[0x1f9d1, 0x1f9dd],
|
||||
[0x1fac3, 0x1fac5],
|
||||
[0x1faf0, 0x1faf8]
|
||||
]
|
||||
|
||||
/** Lowest code point in EMOJI_PRESENTATION_WIDE_RANGES; skips the table for ASCII and CJK. */
|
||||
const EMOJI_PRESENTATION_WIDE_FIRST = 0x1f6d6
|
||||
|
||||
const EMOJI_MODIFIER_FIRST = 0x1f3fb
|
||||
const EMOJI_MODIFIER_LAST = 0x1f3ff
|
||||
|
||||
@@ -178,7 +242,10 @@ function inRanges(ranges: readonly CodepointRange[], codepoint: number): boolean
|
||||
|
||||
/** Emoji that default to emoji presentation but postdate xterm's frozen Unicode 11 width table. */
|
||||
export function isEmojiPresentationWideCodepoint(codepoint: number): boolean {
|
||||
return codepoint >= 0x1f6d6 && inRanges(EMOJI_PRESENTATION_WIDE_RANGES, codepoint)
|
||||
return (
|
||||
codepoint >= EMOJI_PRESENTATION_WIDE_FIRST &&
|
||||
inRanges(EMOJI_PRESENTATION_WIDE_RANGES, codepoint)
|
||||
)
|
||||
}
|
||||
|
||||
/** A code point U+FE0F may promote to emoji presentation, per emoji-variation-sequences. */
|
||||
@@ -190,3 +257,8 @@ export function isEmojiVariationSequenceBase(codepoint: number): boolean {
|
||||
export function isEmojiModifier(codepoint: number): boolean {
|
||||
return codepoint >= EMOJI_MODIFIER_FIRST && codepoint <= EMOJI_MODIFIER_LAST
|
||||
}
|
||||
|
||||
/** A code point a skin-tone modifier may attach to. */
|
||||
export function isEmojiModifierBase(codepoint: number): boolean {
|
||||
return inRanges(EMOJI_MODIFIER_BASE_RANGES, codepoint)
|
||||
}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import type { IUnicodeHandling, IUnicodeVersionProvider } from '@xterm/xterm'
|
||||
import {
|
||||
isEmojiModifier,
|
||||
isEmojiModifierBase,
|
||||
isEmojiPresentationWideCodepoint,
|
||||
isEmojiVariationSequenceBase
|
||||
} from './terminal-emoji-width-ranges'
|
||||
@@ -87,23 +88,24 @@ class OrcaUnicodeProvider implements IUnicodeVersionProvider {
|
||||
return createProperties(codepoint, 2, true)
|
||||
}
|
||||
|
||||
if (isEmojiModifier(codepoint) && precedingWidth === 2) {
|
||||
if (isEmojiModifier(codepoint) && precedingWidth === 2 && isEmojiModifierBase(precedingKind)) {
|
||||
// Why: a skin-tone modifier belongs to its base's cluster. Left to the v11
|
||||
// table it lands as a second wide cell and every later column shifts by two.
|
||||
// Why gated on the base: a modifier after any other wide cell — a CJK
|
||||
// ideograph, an emoji that takes no skin tone — is not a modifier sequence,
|
||||
// and joining it there would narrow text this provider never measured.
|
||||
return createProperties(codepoint, 2, true)
|
||||
}
|
||||
|
||||
const base = this.baseProvider.charProperties(codepoint, preceding)
|
||||
const shouldJoin = extractShouldJoin(base)
|
||||
const baseWidth = extractWidth(base)
|
||||
// Why re-encode: the base provider reports char kind 0 and derives width
|
||||
// from its own wcwidth, so the rules above would lose both the preceding
|
||||
// code point and this provider's post-Unicode-11 widths. A joining code
|
||||
// point keeps the cluster width the base already resolved.
|
||||
return createProperties(
|
||||
codepoint,
|
||||
shouldJoin ? extractWidth(base) : this.wcwidth(codepoint),
|
||||
shouldJoin
|
||||
)
|
||||
const width = shouldJoin || !isEmojiPresentationWideCodepoint(codepoint) ? baseWidth : 2
|
||||
return createProperties(codepoint, width, shouldJoin)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user