fix(monaco): stop a truncated JSONL record poisoning every record after it

`@string` was pushed unconditionally, and Monarch state survives the line
break, so one truncated record left every later record inside the string state
— the whole rest of the file rendered as a single string. A truncated record is
a normal way for a .jsonl log to end.

Gate the push on a lookahead proving the closing quote is on this line, and
consume an unterminated remainder as `string.invalid` without pushing anything.

Measured against the real tokenizer over realistic JSONL. Two alternatives were
rejected: collapsing the string into one regex fixes the poisoning but loses
every `string.escape` token on well-formed records, and `includeLF` does not
fix the primary case at all (the `[^"\\]+` content rule swallows the newline
before an EOL rule can match). This candidate's token stream is byte-identical
to the previous grammar on every well-formed record — the existing inline
snapshot did not move — and is at or faster than it on 20k pathological lines.
This commit is contained in:
Neil
2026-09-10 23:47:22 -07:00
parent 7c26259d0c
commit 952e22a9bc
2 changed files with 42 additions and 9 deletions
@@ -84,16 +84,42 @@ describe('jsonl tokenization', () => {
expect(tokenTypeAt(line, 8)).toBe('string')
})
// KNOWN DEFECT, found by this suite once it started running the real
// tokenizer. Each JSONL line is an independent JSON value, but the `@string`
// state survives the line break, so one truncated record (common in logs)
// renders every record after it as a single string. `it.fails` pins it: this
// flips to a failure the moment the grammar is fixed, so the fix lands with
// this expectation inverted rather than silently.
it.fails('never carries string state across a record boundary', () => {
const [, second] = tokenizeJsonl('{"a": "unterminated\n{"b": 1}')
// Regression, found by this suite once it started running the real tokenizer:
// `@string` used to survive the line break, so one truncated record rendered
// every record after it as a single string.
it.each([
['mid-string', '{"a": "truncated here'],
['mid-escape', '{"a": "truncated\\'],
['on a trailing backslash', '{"a": "x\\']
])('does not let a record truncated %s poison the next one', (_name, truncated) => {
const [, second] = tokenizeJsonl(`${truncated}\n{"b": 1}`)
expect(tokenTypeAt(second, 1)).toBe('type.identifier')
expect(tokenTypeAt(second, 6)).toBe('number')
})
it('marks the unterminated remainder of a truncated record', () => {
const [first] = tokenizeJsonl('{"a": "truncated here')
expect(tokenTypeAt(first, 6)).toBe('string.invalid')
})
it.each([
['escaped quote', '{"m": "he said \\"hi\\""}', 15],
['escaped backslash', '{"m": "C:\\\\Users"}', 10],
['unicode escape', '{"m": "\\u00e9"}', 7],
['newline escape', '{"m": "a\\nb"}', 8]
])('still highlights an %s inside a well-formed record', (_name, record, escapeOffset) => {
// The fix must not cost escape fidelity on the common case: a candidate that
// collapsed the string into one regex lost every one of these.
const [line] = tokenizeJsonl(record)
expect(tokenTypeAt(line, escapeOffset)).toBe('string.escape')
})
it('flags an invalid escape inside a well-formed record', () => {
const [line] = tokenizeJsonl('{"m": "a\\qb"}')
expect(tokenTypeAt(line, 8)).toBe('string.escape.invalid')
})
})
@@ -34,7 +34,14 @@ export const jsonlMonarchLanguage: Monaco.languages.IMonarchLanguage = {
{ include: '@whitespace' },
// Property key vs string value are both quoted; color keys distinctly.
[/"(?:[^"\\]|\\.)*"(?=\s*:)/, 'type.identifier'],
[/"/, 'string', '@string'],
// Why the lookahead: each JSONL line is an independent value, but Monarch
// state survives the line break. Pushing `@string` unconditionally meant
// one truncated record (a normal way for a log to end) left every later
// record inside the string state, rendering the rest of the file as one
// string. Only enter the escape-aware state once a closing quote is known
// to be on this line; an unterminated remainder is consumed below instead.
[/"(?=(?:[^"\\]|\\.)*")/, 'string', '@string'],
[/"(?:[^"\\]|\\.)*\\?$/, 'string.invalid'],
[/[{}[\]]/, '@brackets'],
[/-?(?:0|[1-9]\d*)(?:\.\d+)?(?:[eE][+-]?\d+)?/, 'number'],
[/\b(?:true|false)\b/, 'keyword'],