${'{{a}}'.repeat(count)}
` @@ -252,11 +180,13 @@ describe('vue embedded-tokenizer recursion depth', () => { ['script', 'ts', 'typescript'], ['style', 'scss', 'scss'] ])('re-embeds a %s body after an over-budget opening line', (tag, lang, embeddedLanguageId) => { - const embeds = embeddedLanguagePerLine(createMonarchTokenizer('vue', vueMonarchLanguage), [ - `<${tag} lang="${lang}">a = "${'x'.repeat(EMBED_ENTRY_REST_OF_LINE_BUDGET)}"`, - ' b', - `${tag}>` - ]) + const embeds = endEmbeddedLanguages( + tokenizeLines(createMonarchTokenizer('vue', vueMonarchLanguage), [ + `<${tag} lang="${lang}">a = "${'x'.repeat(EMBED_ENTRY_REST_OF_LINE_BUDGET)}"`, + ' b', + `${tag}>` + ]) + ) expect(embeds).toEqual([null, embeddedLanguageId, null]) }) diff --git a/src/renderer/src/lib/monaco-languages/monarch-tokenizer-test-harness.ts b/src/renderer/src/lib/monaco-languages/monarch-tokenizer-test-harness.ts new file mode 100644 index 00000000000..08f8c14cd5a --- /dev/null +++ b/src/renderer/src/lib/monaco-languages/monarch-tokenizer-test-harness.ts @@ -0,0 +1,159 @@ +import type * as Monaco from 'monaco-editor' +import { compile } from 'monaco-editor/esm/vs/editor/standalone/common/monarch/monarchCompile.js' +import { MonarchTokenizer } from 'monaco-editor/esm/vs/editor/standalone/common/monarch/monarchLexer.js' +import { MAX_TOKENIZATION_LINE_LENGTH } from './monarch-embed-entry-budget' + +// Drives the real `MonarchTokenizer` shipped with monaco-editor rather than +// walking a grammar's rule table. A table walk cannot see the failures that +// actually reach the renderer — a grammar that throws on every `{expr}`, or +// that silently drops an embed, still has a well-formed rule table. + +/** One Monaco token. `language` is the (embedded) language the region belongs to. */ +export type MonarchToken = { offset: number; type: string; language: string } + +type MonarchEndState = { embeddedLanguageData?: { languageId: string } | null } + +export type MonarchTokenizerInstance = { + getInitialState: () => unknown + tokenize: ( + line: string, + hasEOL: boolean, + state: unknown + ) => { tokens: MonarchToken[]; endState: MonarchEndState } + _nestedTokenize: (...args: unknown[]) => unknown +} + +export function createMonarchTokenizer( + languageId: string, + language: Monaco.languages.IMonarchLanguage, + maxTokenizationLineLength = MAX_TOKENIZATION_LINE_LENGTH +): MonarchTokenizerInstance { + // Nested languages stay unregistered, so `nestedLanguageTokenize` emits one + // empty-typed token tagged with the embedded language id instead of running + // that language's tokenizer. That is what makes `token.language` a direct + // readout of which embed covers which region. + const languageService = { + languageIdCodec: { encodeLanguageId: () => 1, decodeLanguageId: () => '' }, + getLanguageIdByLanguageName: () => null, + getLanguageIdByMimeType: () => null, + isRegisteredLanguageId: () => false, + requestBasicLanguageFeatures: () => {} + } + const themeService = { getColorTheme: () => ({ tokenTheme: {} }) } + const configurationService = { + getValue: () => maxTokenizationLineLength, + onDidChangeConfiguration: () => ({ dispose: () => {} }) + } + + return new MonarchTokenizer( + languageService, + themeService, + languageId, + compile(languageId, language), + configurationService + ) as MonarchTokenizerInstance +} + +export type TokenizedLine = { + text: string + tokens: MonarchToken[] + /** Embedded language still active at end of line; `null` means that region renders unhighlighted. */ + endEmbeddedLanguageId: string | null +} + +/** Tokenizes `lines` as one document, threading tokenizer state line to line. */ +export function tokenizeLines( + tokenizer: MonarchTokenizerInstance, + lines: string[] +): TokenizedLine[] { + let state: unknown = tokenizer.getInitialState() + return lines.map((text) => { + const { tokens, endState } = tokenizer.tokenize(text, true, state) + state = endState + return { + text, + tokens, + endEmbeddedLanguageId: endState.embeddedLanguageData?.languageId ?? null + } + }) +} + +export function tokenizeMonarchDocument( + languageId: string, + language: Monaco.languages.IMonarchLanguage, + source: string +): TokenizedLine[] { + return tokenizeLines(createMonarchTokenizer(languageId, language), source.split('\n')) +} + +/** The embedded language each line *ends* in — `null` for no embed. */ +export function endEmbeddedLanguages(lines: TokenizedLine[]): (string | null)[] { + return lines.map((line) => line.endEmbeddedLanguageId) +} + +/** The distinct languages a line's tokens were attributed to, in order. */ +export function tokenLanguages(line: TokenizedLine): string[] { + return line.tokens + .map((token) => token.language) + .filter((language, index, all) => language !== all[index - 1]) +} + +/** + * Per line, which languages actually cover it. This is the readout that catches + * a silently dropped embed: the region falls back to the host grammar's own id + * instead of `html` / `typescript` / `scss`, and renders unhighlighted. + */ +export function tokenLanguagesPerLine(lines: TokenizedLine[]): string[][] { + return lines.map(tokenLanguages) +} + +/** Token type covering `index`, without the grammar's `tokenPostfix`. */ +export function tokenTypeAt(line: TokenizedLine, index: number): string { + const covering = line.tokens.findLast((token) => token.offset <= index) + return covering?.type.split('.').slice(0, -1).join('.') ?? '' +} + +/** One `text | offset:type@language … | embed=…` row per line, for snapshots. */ +export function formatTokenizedLines(lines: TokenizedLine[]): string[] { + return lines.map((line) => { + const tokens = line.tokens + .map((token) => `${token.offset}:${token.type || '-'}@${token.language}`) + .join(' ') + return `${line.text} | ${tokens} | embed=${line.endEmbeddedLanguageId ?? 'none'}` + }) +} + +export type TokenizeMeasurement = { maxNestedDepth: number; error: Error | undefined } + +/** + * Tokenizes `lines`, recording peak `_nestedTokenize` recursion — the real JS + * stack cost, since Monarch enters an embed by mutual recursion with no TCO. + * Embeds cannot nest, so this counts sequential embed enter/exit transitions on + * one line, each holding a frame until the line ends. Errors are captured rather + * than thrown so a caller can assert on frame count and failure together. + */ +export function measureNestedDepth( + tokenizer: MonarchTokenizerInstance, + lines: string[] +): TokenizeMeasurement { + const nestedTokenize = tokenizer._nestedTokenize.bind(tokenizer) + let depth = 0 + let maxNestedDepth = 0 + tokenizer._nestedTokenize = (...args: unknown[]) => { + depth += 1 + maxNestedDepth = Math.max(maxNestedDepth, depth) + try { + return nestedTokenize(...args) + } finally { + depth -= 1 + } + } + + let error: Error | undefined + try { + tokenizeLines(tokenizer, lines) + } catch (thrown) { + error = thrown as Error + } + return { maxNestedDepth, error } +} diff --git a/src/renderer/src/lib/monaco-languages/monarch-upstream-mdx-recursion.test.ts b/src/renderer/src/lib/monaco-languages/monarch-upstream-mdx-recursion.test.ts new file mode 100644 index 00000000000..b87a025bf1e --- /dev/null +++ b/src/renderer/src/lib/monaco-languages/monarch-upstream-mdx-recursion.test.ts @@ -0,0 +1,58 @@ +// @vitest-environment happy-dom +// Why happy-dom: monaco's `basic-languages` entry points import the full +// browser editor before they export the grammar. +import { language as mdxLanguage } from 'monaco-editor/esm/vs/basic-languages/mdx/mdx.js' +import { describe, expect, it } from 'vitest' +import { + EMBED_ENTRY_REST_OF_LINE_BUDGET, + MAX_TOKENIZATION_LINE_LENGTH +} from './monarch-embed-entry-budget' +import { createMonarchTokenizer, measureNestedDepth } from './monarch-tokenizer-test-harness' +import { svelteMonarchLanguage } from './register-svelte' + +// Why pin a third-party grammar: monaco's OWN shipped mdx grammar enters a `js` +// embed on every `{` and pops on `}` with no budget, so it reproduces the +// unbounded embed-entry recursion exactly. That makes it the proof this shape is +// monaco's, not something Orca's svelte/astro/vue grammars invented — and it is +// the tripwire for a monaco upgrade that changes the recursion shape. Do not +// delete as "not our code". + +/** One `js` embed enter/exit transition per repeat, in 3 characters. */ +const interpolations = (count: number): string => '{a}'.repeat(count) + +/** Longest run of them monaco will still tokenize at all. */ +const UNTOKENIZABLE_ABOVE = Math.floor(MAX_TOKENIZATION_LINE_LENGTH / 3) - 1 + +describe('upstream monaco mdx grammar', () => { + it('spends one stack frame per interpolation, unbounded', () => { + const frames = [50, 200, 500].map( + (count) => + measureNestedDepth(createMonarchTokenizer('mdx', mdxLanguage), [interpolations(count)]) + .maxNestedDepth + ) + + expect(frames).toEqual([50, 200, 500]) + }) + + it('exhausts the JS stack on a line monaco is still willing to tokenize', () => { + const line = interpolations(UNTOKENIZABLE_ABOVE) + expect(line.length).toBeLessThan(MAX_TOKENIZATION_LINE_LENGTH) + + const measurement = measureNestedDepth(createMonarchTokenizer('mdx', mdxLanguage), [line]) + + // The frame ceiling is runtime-dependent (~1145 measured here), so assert the + // failure rather than the number. + expect(measurement.error).toBeInstanceOf(RangeError) + expect(measurement.maxNestedDepth).toBeLessThan(UNTOKENIZABLE_ABOVE) + }) + + it('is what the embed-entry budget holds: the same shape stays bounded', () => { + const measurement = measureNestedDepth( + createMonarchTokenizer('svelte', svelteMonarchLanguage), + [interpolations(UNTOKENIZABLE_ABOVE)] + ) + + expect(measurement.error).toBeUndefined() + expect(measurement.maxNestedDepth).toBeLessThanOrEqual(EMBED_ENTRY_REST_OF_LINE_BUDGET) + }) +}) diff --git a/src/renderer/src/lib/monaco-languages/register-astro.test.ts b/src/renderer/src/lib/monaco-languages/register-astro.test.ts index 8d1458bdf52..694c4379ee8 100644 --- a/src/renderer/src/lib/monaco-languages/register-astro.test.ts +++ b/src/renderer/src/lib/monaco-languages/register-astro.test.ts @@ -1,112 +1,32 @@ import { describe, expect, it, vi } from 'vitest' +import { + endEmbeddedLanguages, + formatTokenizedLines, + tokenizeMonarchDocument, + tokenLanguages, + tokenLanguagesPerLine +} from './monarch-tokenizer-test-harness' import { astroLanguageConfiguration, astroMonarchLanguage, registerAstroLanguage } from './register-astro' -type MonarchAction = { - next?: string - nextEmbedded?: string - switchTo?: string -} -type MonarchRule = [RegExp, string | MonarchAction, string?] | { include: string } - -function normalizeState(nextState: string): string { - return nextState.startsWith('@') ? nextState.slice(1) : nextState +// Driven through the real `MonarchTokenizer`: a rule-table walk cannot tell a +// working grammar from one that throws on every `{expr}`, which is how broken +// Astro highlighting shipped green. +function tokenizeAstro(source: string) { + return tokenizeMonarchDocument('astro', astroMonarchLanguage, source) } -function isRuleEntry(rule: MonarchRule): rule is [RegExp, string | MonarchAction, string?] { - return Array.isArray(rule) -} - -function getRuleAction(rule: [RegExp, string | MonarchAction, string?]): MonarchAction | undefined { - const [, action, nextStateShortcut] = rule - return typeof action === 'object' - ? action - : nextStateShortcut - ? { next: nextStateShortcut } - : undefined -} - -function findRuleAction( - state: string, - source: string, - { embedPopOnly = false }: { embedPopOnly?: boolean } = {} -): MonarchAction | undefined { - const tokenizer = astroMonarchLanguage.tokenizer as Recorda {b}
` — +// ship with a green suite, because a broken grammar still has a valid table. +function tokenizeSvelte(source: string) { + return tokenizeMonarchDocument('svelte', svelteMonarchLanguage, source) } -function isRuleEntry(rule: MonarchRule): rule is [RegExp, string | MonarchAction, string?] { - return Array.isArray(rule) -} - -function getRuleAction(rule: [RegExp, string | MonarchAction, string?]): MonarchAction | undefined { - const [, action, nextStateShortcut] = rule - return typeof action === 'object' - ? action - : nextStateShortcut - ? { next: nextStateShortcut } - : undefined -} - -function findRuleAction( - state: string, - source: string, - { embedPopOnly = false }: { embedPopOnly?: boolean } = {} -): MonarchAction | undefined { - const tokenizer = svelteMonarchLanguage.tokenizer as Record{count} clicked
| 0:-@html 5:delimiter.curly.svelte@svelte 6:-@typescript 11:delimiter.curly.svelte@svelte 12:-@html | embed=html", + "{:else} | 0:keyword.control.svelte@svelte | embed=none", + "not yet
| 0:-@html | embed=html", + "{/if} | 0:-@html | embed=html", + " | 0:-@html | embed=html", + " | 0:-@html 17:delimiter.curly.svelte@svelte 18:-@typescript 27:delimiter.curly.svelte@svelte 28:-@html 29:delimiter.curly.svelte@svelte 30:-@typescript 35:delimiter.curly.svelte@svelte 36:-@html | embed=html", + "{@html 'raw'} | 0:keyword.control.svelte@svelte 6:-@typescript 21:delimiter.curly.svelte@svelte | embed=none", + " | | embed=html", + " | 0:tag.svelte@svelte | embed=none", ] `) }) -}) -describe('svelte tokenizer regressions', () => { - // Regression: when a Svelte file starts with `{#if}`, `{name}`, or `{@html}`, - // no html embed is active yet. Earlier drafts unconditionally emitted - // `nextEmbedded: '@pop'` from root, which Monaco rejects with - // "cannot pop embedded language if not inside one". The fix splits the - // entry-only `root` state from the html-embedded `markup` state. - it('does not pop a non-existent embed when a file starts with a Svelte block', () => { - const action = findRuleAction('root', '{#if foo}') - expect(action).toMatchObject({ switchTo: '@svelteBlockExpressionEnter' }) - expect(action?.nextEmbedded).toBeUndefined() + // Regression (the field failure): the first interpolation of a file threw + // "cannot pop embedded language if not inside one" — every Svelte file with a + // `{}` in it, which is essentially all of them. + it('highlights every interpolation of a markup line', () => { + const [line] = tokenizeSvelte('a {first} b {second} c
') + + expect(tokenLanguages(line)).toEqual([ + 'html', + 'svelte', + 'typescript', + 'svelte', + 'html', + 'svelte', + 'typescript', + 'svelte', + 'html' + ]) }) - it('starts the html embed and switches to markup when markup begins', () => { - expect(findRuleAction('root', '{{ message.toUpperCase() }}
@@ -150,121 +70,110 @@ const message = 'hello' p { color: rebeccapurple; } ` - const ruleActions = collectFixtureRuleActions(fixture) - - expect(ruleActions).toMatchInlineSnapshot(` + expect(formatTokenizedLines(tokenizeVue(fixture))).toMatchInlineSnapshot(` [ - { - "line": 1, - "matched": "", - "nextEmbedded": "html", - "nextState": undefined, - "state": "templateOpen", - "switchTo": "templateBody", - }, - { - "line": 2, - "matched": "{{", - "nextEmbedded": "@pop", - "nextState": undefined, - "state": "templateBody", - "switchTo": "templateExpressionEnter", - }, - { - "line": 2, - "matched": "}}", - "nextEmbedded": "@pop", - "nextState": undefined, - "state": "templateExpression", - "switchTo": "templateBodyReenter", - }, - { - "line": 3, - "matched": "", - "nextEmbedded": "@pop", - "nextState": "pop", - "state": "templateBody", - "switchTo": undefined, - }, - { - "line": 5, - "matched": "", - "nextEmbedded": "@pop", - "nextState": "pop", - "state": "scriptBody.typescript", - "switchTo": undefined, - }, - { - "line": 9, - "matched": "", - "nextEmbedded": "@pop", - "nextState": "pop", - "state": "styleBody.css", - "switchTo": undefined, - }, + " | 0:tag.vue@vue | embed=html", + "{{ message.toUpperCase() }}
| 0:-@html 5:delimiter.curly.vue@vue 7:-@typescript 30:delimiter.curly.vue@vue 32:-@html | embed=html", + " | 0:tag.vue@vue | embed=none", + " | | embed=none", + " | 0:tag.vue@vue | embed=none", + " | | embed=none", + " | 0:tag.vue@vue | embed=none", ] `) }) - it('tracks embedded languages from Vue block attributes', () => { - expect(findRuleAction('templateExpressionEnter', 'message }}')).toMatchObject({ - nextEmbedded: 'typescript', - switchTo: '@templateExpression' - }) - expect(findRuleAction('scriptLangValue.typescript', '"js"')).toMatchObject({ - switchTo: '@scriptOpen.javascript' - }) - expect(findRuleAction('scriptLangValue.javascript', '"ts"')).toMatchObject({ - switchTo: '@scriptOpen.typescript' - }) - expect(findRuleAction('scriptLangValue.typescript', 'js')).toMatchObject({ - switchTo: '@scriptOpen.javascript' - }) - expect(findRuleAction('styleLangValue.css', '"scss"')).toMatchObject({ - switchTo: '@styleOpen.scss' - }) - expect(findRuleAction('styleLangValue.css', 'less')).toMatchObject({ - switchTo: '@styleOpen.less' - }) + // Regression: every `{{ }}` threw "cannot pop embedded language if not inside + // one" once the template body lost its html embed. + it('highlights every interpolation in a template line', () => { + const [, line] = tokenizeVue('\n{{ a }} and {{ b }}
\n') + + expect(tokenLanguages(line)).toEqual([ + 'html', + 'vue', + 'typescript', + 'vue', + 'html', + 'vue', + 'typescript', + 'vue', + 'html' + ]) + }) + + it('embeds the template body as html', () => { + expect(endEmbeddedLanguages(tokenizeVue('\nx
\n'))).toEqual([ + 'html', + 'html', + null + ]) + }) + + it('keeps the template embedded across a comment before it', () => { + expect(languagesPerLine('\n\nx
\n')).toEqual([ + ['vue'], + ['vue'], + ['html'], + ['vue'] + ]) + }) + + it('does not enter typescript for an empty interpolation', () => { + // `{{}}` pops html on entry but never pushes typescript; the close must + // unwind only the state, or it pops an embed that is not there. + const [, line] = tokenizeVue('\n{{}}
\n') + + expect(tokenLanguages(line)).toEqual(['html', 'vue', 'html']) + }) +}) + +describe('vue embedded language attributes', () => { + it.each([ + ['`)).toEqual([ + ['vue'], + [embeddedLanguageId], + ['vue'] + ]) + }) + + it.each([ + ['`)).toEqual([ + ['vue'], + [embeddedLanguageId], + ['vue'] + ]) + }) +}) + +describe('vue root state invariant', () => { + // Structural on purpose: behaviour can only reach the root rules some fixture + // happens to exercise, and a root rule that pops an embed throws on the very + // first character of a file. Guard every root rule, exercised or not. + it('has no root rule that pops an embedded language', () => { + const rootRules = (vueMonarchLanguage.tokenizer as Record