fix(editor): share code-fence boundaries for Markdown links (#21574)

* fix(editor): share fence boundaries for Markdown links

* test(editor): assert highlighted lines across Monaco token spans

* perf(editor): scan fence lines without substring allocations

* fix(editor): match fence suffix whitespace rules
This commit is contained in:
Neil
2026-09-19 00:06:56 -07:00
committed by GitHub
parent 3de7273abb
commit 277c289bd4
7 changed files with 141 additions and 85 deletions
@@ -4,12 +4,38 @@ export type MarkdownFenceTracker = {
readonly insideFence: boolean
// Returns true when the line was consumed as a fence delimiter.
consume: (line: string) => boolean
consumeRange: (content: string, lineStart: number, lineEnd: number) => boolean
}
// Top-level fence delimiters may be indented by at most three spaces.
const FENCE_LINE = /^[ ]{0,3}(`{3,}|~{3,})/
// marked lets a closer trail a run of fence characters, e.g. ```~~~ closes a ``` block.
const CLOSING_FENCE_SUFFIX = /^[~`]*[ \t\r]*$/
const FENCE_LINE = /[ ]{0,3}(`{3,}|~{3,})/y
function hasClosingSuffix(content: string, start: number, end: number): boolean {
let sawWhitespace = false
for (let index = start; index < end; index += 1) {
const character = content.charCodeAt(index)
if (character === 32 || character === 9 || character === 13) {
sawWhitespace = true
continue
}
if ((character === 126 || character === 96) && !sawWhitespace) {
continue
}
if (character !== 126 && character !== 96) {
return false
}
return false
}
return true
}
function hasBacktick(content: string, start: number, end: number): boolean {
for (let index = start; index < end; index += 1) {
if (content.charCodeAt(index) === 96) {
return true
}
}
return false
}
/** Tracks fenced blocks using the editor parser's closing rules. */
export function createMarkdownFenceTracker(): MarkdownFenceTracker {
@@ -21,23 +47,31 @@ export function createMarkdownFenceTracker(): MarkdownFenceTracker {
return length > 0
},
consume(line: string): boolean {
const match = FENCE_LINE.exec(line)
if (!match) {
return this.consumeRange(line, 0, line.length)
},
consumeRange(content: string, lineStart: number, lineEnd: number): boolean {
FENCE_LINE.lastIndex = lineStart
const match = FENCE_LINE.exec(content)
if (!match || match.index !== lineStart || match[0].length > lineEnd - lineStart) {
return false
}
const lineMarker = match[1][0]
const lineLength = match[1].length
const suffix = line.slice(match[0].length)
const suffixStart = lineStart + match[0].length
if (length > 0) {
if (lineMarker === marker && lineLength >= length && CLOSING_FENCE_SUFFIX.test(suffix)) {
if (
lineMarker === marker &&
lineLength >= length &&
hasClosingSuffix(content, suffixStart, lineEnd)
) {
marker = ''
length = 0
}
return true
}
if (lineMarker === '`' && suffix.includes('`')) {
if (lineMarker === '`' && hasBacktick(content, suffixStart, lineEnd)) {
return false
}
marker = lineMarker
@@ -91,7 +125,7 @@ export function getMarkdownFenceRanges(content: string): MarkdownFenceRanges {
forEachMarkdownLine(content, (lineStart, lineEnd, nextLineStart) => {
const wasInside = tracker.insideFence
const isFenceLine = tracker.consume(content.slice(lineStart, lineEnd))
const isFenceLine = tracker.consumeRange(content, lineStart, lineEnd)
if (!wasInside && isFenceLine) {
openStart = lineStart
} else if (wasInside && !tracker.insideFence) {
@@ -1,6 +1,21 @@
import { afterEach, describe, expect, it, vi } from 'vitest'
import { marked } from 'marked'
import { normalizeMarkdownReferenceLinks } from './markdown-reference-link-normalization'
// Offsets marked itself treats as fenced code, so the sweep below compares the
// normalizer against the parser that decides what the file really means.
function markedCodeRanges(markdown: string): [number, number][] {
const ranges: [number, number][] = []
let offset = 0
for (const token of marked.lexer(markdown.replace(/\r\n|\r/g, '\n'))) {
if (token.type === 'code') {
ranges.push([offset, offset + token.raw.length])
}
offset += token.raw.length
}
return ranges
}
afterEach(() => {
vi.restoreAllMocks()
})
@@ -26,6 +41,25 @@ describe('normalizeMarkdownReferenceLinks', () => {
expect(normalizeMarkdownReferenceLinks(markdown)).toBe(markdown)
})
it('never removes a definition marked keeps inside a fenced code block', () => {
const definition = '[docs]: https://example.com/docs'
const survivors: string[] = []
for (const outer of ['```', '````', '~~~', '~~~~']) {
// Fence-shaped lines that do and do not close `outer`.
for (const inner of ['```', '````', '~~~', '````js', '``` ', '```` trailing', '~~~a`b']) {
const markdown = `${outer}\ncode\n${inner}\n${definition}\n[Docs]\n${outer}\n`
const offset = markdown.indexOf(definition)
const insideCode = markedCodeRanges(markdown).some(
([start, end]) => offset >= start && offset < end
)
if (insideCode && !normalizeMarkdownReferenceLinks(markdown).includes(definition)) {
survivors.push(markdown)
}
}
}
expect(survivors).toEqual([])
})
it('scans newline-heavy documents without splitting into line arrays', () => {
const split = vi.spyOn(String.prototype, 'split')
const body = Array.from({ length: 5000 }, (_, index) => `line ${index + 1}`).join('\n')
@@ -1,3 +1,9 @@
import {
createMarkdownFenceRangeCursor,
createMarkdownFenceTracker,
getMarkdownFenceRanges
} from './markdown-fence-scanner'
type ReferenceLinkDefinition = {
label: string
title: string | null
@@ -65,25 +71,12 @@ function splitReferenceDefinitions(content: string): {
markdown: string
} {
const definitions = new Map<string, ReferenceLinkDefinition>()
let activeFence: '`' | '~' | null = null
let activeFenceLength = 0
const fence = createMarkdownFenceTracker()
let markdown = ''
forEachReferenceDefinitionLine(content, (line, newline) => {
const fenceMatch = line.match(/^\s*(`{3,}|~{3,})/)
if (fenceMatch) {
const fenceChar = fenceMatch[1][0] as '`' | '~'
const fenceLength = fenceMatch[1].length
if (activeFence === null) {
activeFence = fenceChar
activeFenceLength = fenceLength
} else if (activeFence === fenceChar && fenceLength >= activeFenceLength) {
activeFence = null
activeFenceLength = 0
}
}
const definition = activeFence === null ? parseReferenceDefinition(line) : null
const isFenceLine = fence.consume(line)
const definition = isFenceLine || fence.insideFence ? null : parseReferenceDefinition(line)
if (definition) {
definitions.set(definition.label, definition)
return
@@ -148,40 +141,11 @@ function replaceReferenceLinks(
): string {
let result = ''
let index = 0
let activeFence: '`' | '~' | null = null
let activeFenceLength = 0
let isLineStart = true
const nonWhitespace = /\S/g
const fencePrefix = /(`{3,}|~{3,})/y
let fenceProbe = -1
let fenceMatch: RegExpExecArray | null = null
const isInsideFence = createMarkdownFenceRangeCursor(getMarkdownFenceRanges(markdown))
while (index < markdown.length) {
if (isLineStart) {
// Reuse the lookahead across blank lines, preserving cross-line fence semantics.
if (index > fenceProbe) {
nonWhitespace.lastIndex = index
fenceProbe = nonWhitespace.exec(markdown)?.index ?? markdown.length
fencePrefix.lastIndex = fenceProbe
fenceMatch = fencePrefix.exec(markdown)
}
if (fenceMatch) {
const fenceChar = fenceMatch[1][0] as '`' | '~'
const fenceLength = fenceMatch[1].length
if (activeFence === null) {
activeFence = fenceChar
activeFenceLength = fenceLength
} else if (activeFence === fenceChar && fenceLength >= activeFenceLength) {
activeFence = null
activeFenceLength = 0
}
}
}
if (activeFence || markdown[index] !== '[' || isEscaped(markdown, index)) {
const nextChar = markdown[index]
result += nextChar
isLineStart = nextChar === '\n'
if (isInsideFence(index) || markdown[index] !== '[' || isEscaped(markdown, index)) {
result += markdown[index]
index += 1
continue
}
@@ -189,7 +153,6 @@ function replaceReferenceLinks(
const closingTextIndex = findClosingBracket(markdown, index + 1)
if (closingTextIndex === -1) {
result += markdown[index]
isLineStart = false
index += 1
continue
}
@@ -198,7 +161,6 @@ function replaceReferenceLinks(
const afterText = markdown[closingTextIndex + 1]
if (afterText === '(') {
result += markdown[index]
isLineStart = false
index += 1
continue
}
@@ -211,7 +173,6 @@ function replaceReferenceLinks(
const definition = definitions.get(label)
if (definition) {
result += formatInlineReferenceLink(text, definition)
isLineStart = false
index = closingLabelIndex + 1
continue
}
@@ -220,14 +181,12 @@ function replaceReferenceLinks(
const definition = definitions.get(normalizeReferenceLabel(text))
if (definition) {
result += formatInlineReferenceLink(text, definition)
isLineStart = false
index = closingTextIndex + 1
continue
}
}
result += markdown[index]
isLineStart = false
index += 1
}
@@ -34,7 +34,7 @@ it.each(['\n', '\r\n', '\r'])('retains %j terminators while stripping code', (eo
)
})
it.each(['```~~~', ' ```'])('matches the parser on the closer %j', (closer) => {
it.each(['```~~~', '``` ~', ' ```'])('matches the parser on the closer %j', (closer) => {
const source = `\`\`\`\ncode\n${closer}\n<div>after</div>\n`
const code = marked.lexer(source)[0]
expect(code.type).toBe('code')
@@ -1,10 +1,12 @@
import type { IRange } from 'monaco-editor'
import { describe, expect, it } from 'vitest'
import { getMarkdownDocLinkTarget } from './markdown-doc-links'
import { createMarkdownFenceTracker } from './markdown-fence-scanner'
import { getMarkdownDocLinkDecorationRanges } from './monaco-markdown-doc-link-decorations'
// Why: the pre-offset implementation, kept verbatim as the equivalence oracle
// for the allocation-free scan that replaced it.
// Why: the pre-offset implementation, kept as the equivalence oracle for the
// allocation-free scan that replaced it. Fence detection is the shared
// scanner's contract, not this file's, so the oracle calls into it.
function referenceDecorationRanges(content: string): IRange[] {
const getInlineCodeSpans = (line: string): { start: number; end: number }[] => {
const spans: { start: number; end: number }[] = []
@@ -26,7 +28,7 @@ function referenceDecorationRanges(content: string): IRange[] {
spans.some((span) => index >= span.start && index < span.end)
const ranges: IRange[] = []
let insideFence = false
const fence = createMarkdownFenceTracker()
let lineStart = 0
let lineNumber = 1
for (let index = 0; index <= content.length; index += 1) {
@@ -39,11 +41,7 @@ function referenceDecorationRanges(content: string): IRange[] {
const currentLineNumber = lineNumber
lineNumber += 1
if (/^\s*(```|~~~)/.test(line)) {
insideFence = !insideFence
continue
}
if (insideFence) {
if (fence.consume(line) || fence.insideFence) {
continue
}
const inlineCodeSpans = getInlineCodeSpans(line)
@@ -1,5 +1,6 @@
import type { editor, IDisposable, IRange } from 'monaco-editor'
import { getMarkdownDocLinkTarget } from './markdown-doc-links'
import { createMarkdownFenceRangeCursor, getMarkdownFenceRanges } from './markdown-fence-scanner'
import { forEachLine } from './text-line-offsets'
const BACKTICK = 96
@@ -41,18 +42,10 @@ function isInsideSpan(index: number, spans: number[]): boolean {
return false
}
const FENCE_PREFIX_RE = /[^\S\n]*(?:```|~~~)/y
function startsCodeFence(content: string, lineStart: number, lineEnd: number): boolean {
FENCE_PREFIX_RE.lastIndex = lineStart
// Bound whitespace to this line so blank runs cannot trigger repeated suffix scans.
return FENCE_PREFIX_RE.test(content) && FENCE_PREFIX_RE.lastIndex <= lineEnd
}
export function getMarkdownDocLinkDecorationRanges(content: string): IRange[] {
const ranges: IRange[] = []
const inlineCodeSpans: number[] = []
let insideFence = false
const isInsideFence = createMarkdownFenceRangeCursor(getMarkdownFenceRanges(content))
// Why: `indexOf` on the whole document would rescan the tail once per line.
// Both cursors only ever move forward, and every probe position is
// monotonic, so the delimiter search stays linear in document length.
@@ -60,11 +53,7 @@ export function getMarkdownDocLinkDecorationRanges(content: string): IRange[] {
let nextClose = content.indexOf(']]')
forEachLine(content, (lineStart, lineEnd, lineNumber) => {
if (startsCodeFence(content, lineStart, lineEnd)) {
insideFence = !insideFence
return
}
if (insideFence) {
if (isInsideFence(lineStart)) {
return
}
+42
View File
@@ -0,0 +1,42 @@
import { test, expect } from './helpers/orca-app'
import { waitForActiveWorktree, waitForSessionReady } from './helpers/store'
import {
cleanupMarkdownFixture,
createMarkdownFixture,
getActiveWorktreeContext,
openMarkdownFixture
} from './helpers/markdown-editor-fixture'
test('source links stay outside mixed code fences', async ({ orcaPage }, testInfo) => {
await waitForSessionReady(orcaPage)
await waitForActiveWorktree(orcaPage)
const context = await getActiveWorktreeContext(orcaPage)
const file = await createMarkdownFixture(
context,
'markdown-fences',
'links',
testInfo.workerIndex,
'# Fence boundaries\n\n~~~text\n```\n[[inside-code]]\n~~~\n\n[[outside-code]]\n'
)
try {
await openMarkdownFixture(orcaPage, context, file)
await orcaPage.evaluate(() => {
const state = window.__store!.getState()
if (!state.activeFileId) {
throw new Error('missing active file')
}
state.setMarkdownViewMode(state.activeFileId, 'source')
})
const links = orcaPage.locator('.monaco-editor .view-line').filter({
has: orcaPage.locator('.monaco-markdown-doc-link')
})
await expect(links).toHaveCount(1)
await expect(links).toHaveText('[[outside-code]]')
await testInfo.attach('fence-links', {
body: await orcaPage.screenshot({ path: testInfo.outputPath('fence-links.png') }),
contentType: 'image/png'
})
} finally {
await cleanupMarkdownFixture(file)
}
})