Merge remote-tracking branch 'origin/main' into fix/chromium-samesite-enum-decode

This commit is contained in:
Merge Sim
2026-09-11 01:16:10 -07:00
37 changed files with 2399 additions and 780 deletions
+4
View File
@@ -31,6 +31,10 @@
# the reviewable change, and pin LF because they are compared byte-for-byte.
# Not -diff: the shell diff is the review surface when a wrapper does change.
/src/main/__fixtures__/shell-wrapper-snapshots/*.txt linguist-generated=true text eol=lf
# Captured agent PTY transcripts. -text, not `text eol=lf` like the wrapper snapshots above:
# these carry real CR and CRLF bytes as the terminal emitted them, and line-ending
# normalisation on a Windows checkout would rewrite the evidence the fixture exists to be.
/src/main/runtime/__fixtures__/*.txt -text
# Generated runtime English subset: compared byte-for-byte by
# verify:localization-runtime-catalog, so a CRLF checkout would fail the gate.
/src/renderer/src/i18n/en-runtime-required.json linguist-generated=true text eol=lf
+2
View File
@@ -103,7 +103,9 @@ docs/**
!docs/agent-skill-sharing-implementation-checklist.md
!docs/mobile-terminal-shortcut-bar.md
!docs/reference/
!docs/reference/agent-pty-transcript-capture.md
!docs/reference/agent-status-store.md
!docs/reference/antigravity-readiness-evidence.md
!docs/reference/git-compatibility.md
!docs/reference/headless-linux-server.md
!docs/reference/ime-regression-checklist.md
+4
View File
@@ -72,6 +72,10 @@ All changes must consider folder workspaces as well as git worktrees. Don't assu
The execution host owns agent status in one store, the hook server's, and every reader (sidebar, `worktree ps`, mobile, dashboard) subscribes to it. Before adding a producer, a cache, or a reader-side precedence rule, read [`docs/reference/agent-status-store.md`](./docs/reference/agent-status-store.md): new producers write into that store, and readers keep only presentation policy.
## Agent Terminal Screens
A rule that reads what an agent CLI paints on a terminal — readiness, blocked prompts, idle — must be written against a captured transcript, not a remembered screen. Record one with [`docs/reference/agent-pty-transcript-capture.md`](./docs/reference/agent-pty-transcript-capture.md), which keeps escapes and wrapping intact and scrubs account identifiers before they reach git. Antigravity readiness has no transcript yet and five failed attempts without one; before touching it, read [`docs/reference/antigravity-readiness-evidence.md`](./docs/reference/antigravity-readiness-evidence.md).
## Remote Wire Compatibility
Clients and remote Orca servers update independently, so mixed versions are the normal state. Before changing anything a paired client and host exchange — RPC params, stream frames, or the content either side publishes over them — follow [`docs/reference/remote-wire-compatibility.md`](./docs/reference/remote-wire-compatibility.md). A new optional field is safe; a new stream opcode must be capability-negotiated because decoders drop unknown opcodes silently; and changing what the host publishes reaches old clients even with no wire change.
@@ -0,0 +1,283 @@
/**
* Records a live agent CLI session through a real PTY into a test fixture, bytes intact.
*
* Why a PTY and not `agy | tee`: a pipe is not a terminal, so the CLI renders its
* non-interactive path — no alternate screen, no caret, no dialogs. The detector under
* test only ever sees the PTY shape, so that is the only shape worth capturing.
*
* Nothing here strips escapes, folds CRs, or rewraps lines: the transcript is written
* exactly as the terminal received it. See docs/reference/agent-pty-transcript-capture.md.
*/
import { createWriteStream, mkdirSync, readFileSync, writeFileSync } from 'node:fs'
import { dirname, join, resolve } from 'node:path'
import { pathToFileURL } from 'node:url'
import {
formatFindings,
redactTranscript,
scanTranscriptForSecrets
} from './pty-transcript-secret-scan.mjs'
const REPO_ROOT = resolve(import.meta.dirname, '..', '..')
const FIXTURE_DIR = join(REPO_ROOT, 'src', 'main', 'runtime', '__fixtures__')
const STOP_KEY = 0x1d // Ctrl-], consumed by the recorder and never forwarded to the agent.
const NAME_RE = /^[a-z0-9][a-z0-9-]*$/
const USAGE = `Capture a raw agent PTY transcript into src/main/runtime/__fixtures__/.
node config/scripts/capture-agent-pty-transcript.mjs --name <fixture-name> [options] -- <command> [args...]
node config/scripts/capture-agent-pty-transcript.mjs --scan <file...> [--redact]
Options
--name <fixture-name> Output fixture name, e.g. antigravity-ready-personal-non-gemini
--out <path> Write somewhere other than the fixture directory
--cols <n> --rows <n> Pin the PTY size (default: this terminal's size, else 120x40)
--duration <seconds> Stop unattended after N seconds
--send "<ms>:<text>" Type <text> into the PTY at <ms> (repeatable; \\r \\n \\t \\e escapes)
--note "<text>" Recorded in the <name>.meta.json sidecar
--scan <file...> Scan existing transcripts for identifiers/credentials and exit
--redact With --scan: rewrite each finding as a same-length placeholder
Press Ctrl-] to end a capture. That key is consumed here, so the agent keeps whatever
dialog it is showing — which is the only way to capture a dialog that owns the screen.`
function parseArgs(argv) {
const options = { cols: null, rows: null, duration: null, scan: [], sends: [], redact: false }
const command = []
let cursor = 0
let afterSeparator = false
while (cursor < argv.length) {
const arg = argv[cursor]
if (afterSeparator) {
command.push(arg)
cursor += 1
continue
}
if (arg === '--') {
afterSeparator = true
} else if (arg === '--redact') {
options.redact = true
} else if (arg === '--help' || arg === '-h') {
options.help = true
} else if (arg === '--scan') {
while (cursor + 1 < argv.length && !argv[cursor + 1].startsWith('--')) {
cursor += 1
options.scan.push(argv[cursor])
}
} else if (arg === '--send') {
cursor += 1
options.sends.push(parseSend(argv[cursor]))
} else if (arg.startsWith('--')) {
const key = arg.slice(2)
cursor += 1
options[key] = argv[cursor]
}
cursor += 1
}
for (const key of ['cols', 'rows', 'duration']) {
options[key] = options[key] == null ? null : Number(options[key])
}
return { options, command }
}
// String.fromCharCode, not a literal: the formatter rewrites an escape sequence into a raw
// control byte in source, which is unreadable and survives badly in diffs.
const ESC = String.fromCharCode(27)
const SEND_ESCAPES = { r: '\r', n: '\n', t: '\t', e: ESC, '\\': '\\' }
/** `"<ms>:<text>"` — a keystroke to deliver at a fixed offset, for an unattended dialog capture. */
function parseSend(value) {
const separator = String(value ?? '').indexOf(':')
if (separator === -1) {
throw new Error(`--send expects "<ms>:<text>", got ${String(value)}`)
}
const atMs = Number(value.slice(0, separator))
if (!Number.isFinite(atMs)) {
throw new Error(
`--send delay must be a number of milliseconds, got ${value.slice(0, separator)}`
)
}
const text = value
.slice(separator + 1)
.replace(/\\(.)/g, (whole, code) => SEND_ESCAPES[code] ?? whole)
return { atMs, text }
}
function runScan(files, redact) {
let failed = false
for (const file of files) {
const path = resolve(file)
const text = readFileSync(path, 'utf8')
if (redact) {
const { text: redacted, redacted: count } = redactTranscript(text)
writeFileSync(path, redacted)
console.log(`${file}: redacted ${count} span(s) in place, same length each.`)
continue
}
const findings = scanTranscriptForSecrets(text)
console.log(formatFindings(file, findings))
failed ||= findings.length > 0
}
return failed ? 1 : 0
}
function resolveSpawn(command) {
// node-pty cannot run a .cmd/.bat shim directly on Windows; those need cmd.exe.
if (process.platform === 'win32' && /\.(cmd|bat)$/i.test(command[0])) {
return { file: 'cmd.exe', args: ['/c', `"${command[0]}"`, ...command.slice(1)] }
}
return { file: command[0], args: command.slice(1) }
}
async function runCapture(options, command) {
const name = options.name
if (typeof name === 'string' && !NAME_RE.test(name)) {
console.error(`--name must be lowercase kebab-case; got ${name}`)
return 2
}
const outPath = options.out ? resolve(options.out) : join(FIXTURE_DIR, `${name}.txt`)
mkdirSync(dirname(outPath), { recursive: true })
const pty = await import('node-pty').catch((error) => {
console.error(
`node-pty failed to load. Build it for plain node first:
node config/scripts/ensure-native-runtime.mjs --runtime=node
${String(error)}`
)
return null
})
if (pty === null) {
return 2
}
const cols = options.cols ?? process.stdout.columns ?? 120
const rows = options.rows ?? process.stdout.rows ?? 40
const { file, args } = resolveSpawn(command)
const term = pty.spawn(file, args, {
name: 'xterm-256color',
cols,
rows,
cwd: process.cwd(),
env: { ...process.env, TERM: 'xterm-256color' },
encoding: null
})
const sink = createWriteStream(outPath)
let recording = true
term.onData((chunk) => {
const bytes = typeof chunk === 'string' ? Buffer.from(chunk, 'utf8') : chunk
// Why recording stops before the kill: an agent repaints an idle frame on its way out, so
// a transcript that keeps writing through shutdown ends on that frame instead of on the
// state you stopped to capture. A mid-turn or dialog capture cannot survive that.
if (recording) {
sink.write(bytes)
}
process.stdout.write(bytes)
})
const wasRaw = process.stdin.isTTY === true && process.stdin.isRaw === true
if (process.stdin.isTTY) {
process.stdin.setRawMode(true)
}
process.stdin.resume()
let stopping = false
const stop = () => {
if (stopping) {
return
}
stopping = true
recording = false
try {
term.kill()
} catch {
// The agent may have exited on its own; the transcript is already on disk.
}
}
process.stdin.on('data', (chunk) => {
if (chunk.includes(STOP_KEY)) {
stop()
return
}
term.write(chunk.toString('binary'))
})
// Why scripted input: a dialog capture has to be driven, and CI (or an agent) has no TTY to
// type into. The keystrokes ride the same PTY a human's would, so the capture is unchanged.
const sendTimers = options.sends.map((send) => setTimeout(() => term.write(send.text), send.atMs))
const durationTimer = options.duration === null ? null : setTimeout(stop, options.duration * 1000)
const exitCode = await new Promise((resolveExit) => {
term.onExit(({ exitCode: code }) => resolveExit(code ?? 0))
})
for (const timer of sendTimers) {
clearTimeout(timer)
}
if (durationTimer !== null) {
clearTimeout(durationTimer)
}
if (process.stdin.isTTY) {
process.stdin.setRawMode(wasRaw)
}
process.stdin.pause()
await new Promise((done) => sink.end(done))
writeMeta(outPath, { command, cols, rows, note: options.note ?? null, exitCode })
const findings = scanTranscriptForSecrets(readFileSync(outPath, 'utf8'))
console.log(`\nTranscript: ${outPath}`)
console.log(formatFindings('scrub check', findings))
if (findings.length > 0) {
console.log(
`Scrub with:
node config/scripts/capture-agent-pty-transcript.mjs --scan ${outPath} --redact`
)
}
return 0
}
function writeMeta(outPath, details) {
const metaPath = outPath.replace(/\.txt$/, '.meta.json')
writeFileSync(
metaPath,
`${JSON.stringify(
{
capturedAt: new Date().toISOString(),
platform: process.platform,
command: details.command,
cols: details.cols,
rows: details.rows,
note: details.note,
exitCode: details.exitCode
},
null,
2
)}\n`
)
}
async function main() {
const { options, command } = parseArgs(process.argv.slice(2))
if (options.help === true) {
console.log(USAGE)
return 0
}
if (options.scan.length > 0) {
return runScan(options.scan, options.redact)
}
if (command.length === 0 || (options.name === undefined && options.out === undefined)) {
console.error(USAGE)
return 2
}
return runCapture(options, command)
}
if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) {
main().then(
(code) => {
process.exitCode = code
},
(error) => {
console.error(error)
process.exitCode = 1
}
)
}
export { parseArgs, resolveSpawn }
@@ -0,0 +1,135 @@
// Finds account identifiers and credentials in a captured PTY transcript before it is committed.
import os from 'node:os'
// Why same-length replacements: a transcript's value is its exact wrapping and column
// alignment. Shortening a redacted span reflows the screen and destroys the evidence.
const EMAIL_DOMAIN = '@example.com'
const PLACEHOLDER_UUID = '00000000-0000-4000-8000-000000000000'
/** Ordered most-specific first; the first pattern to claim a span owns it. */
function buildPatterns() {
const username = os.userInfo().username
const hostname = os.hostname()
const patterns = [
{ kind: 'jwt', re: /\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{4,}/g },
{ kind: 'google-api-key', re: /\bAIza[0-9A-Za-z_-]{20,}/g },
{ kind: 'google-refresh-token', re: /\b1\/\/[0-9A-Za-z_-]{20,}/g },
{ kind: 'vendor-key', re: /\b(?:sk-|ghp_|gho_|github_pat_|xoxb-|xoxp-)[A-Za-z0-9_-]{16,}/g },
{ kind: 'bearer-token', re: /\bBearer\s+[A-Za-z0-9._~+/=-]{16,}/gi },
{ kind: 'email', re: /[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}/g },
// Why a UUID counts: agy prints a resumable conversation id on exit, and installation and
// project ids look the same. They identify the operator's session, not just its shape.
{ kind: 'uuid', re: /\b[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}\b/gi },
{ kind: 'opaque-token', re: /\b[A-Za-z0-9_-]{40,}\b/g }
]
if (username.length >= 3) {
patterns.splice(5, 0, { kind: 'local-username', re: literalPattern(username) })
}
if (hostname.length >= 3) {
patterns.splice(5, 0, { kind: 'local-hostname', re: literalPattern(hostname) })
}
return patterns
}
function literalPattern(value) {
return new RegExp(value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'), 'g')
}
/**
* @param {string} text raw transcript, escapes intact
* @returns {{kind: string, line: number, column: number, index: number, match: string}[]}
*/
export function scanTranscriptForSecrets(text) {
const claimed = []
const findings = []
for (const { kind, re } of buildPatterns()) {
re.lastIndex = 0
let match = re.exec(text)
while (match !== null) {
const start = match.index
const end = start + match[0].length
if (!claimed.some(([from, to]) => start < to && end > from)) {
claimed.push([start, end])
if (!isAlreadyScrubbed(kind, match[0])) {
findings.push({ kind, index: start, match: match[0], ...locate(text, start) })
}
}
match = re.exec(text)
}
}
return findings.sort((left, right) => left.index - right.index)
}
// Why: a scrubbed fixture must verify clean, so this scanner has to recognise its own
// placeholders — otherwise "prove it's gone" can never pass and the check gets ignored.
const PLACEHOLDER_DOMAIN_RE = /@(?:example\.(?:com|org|net)|localhost)$/i
function isAlreadyScrubbed(kind, match) {
if (kind === 'email') {
return PLACEHOLDER_DOMAIN_RE.test(match)
}
if (kind === 'uuid') {
return match.toLowerCase() === PLACEHOLDER_UUID
}
return /^(.)\1*$/.test(match)
}
function locate(text, index) {
let line = 1
let lineStart = 0
for (let cursor = 0; cursor < index; cursor += 1) {
if (text.charCodeAt(cursor) === 10) {
line += 1
lineStart = cursor + 1
}
}
return { line, column: index - lineStart + 1 }
}
/** Same-length stand-in so redaction cannot reflow the captured screen. */
export function placeholderFor(kind, length) {
if (kind === 'uuid' && length === PLACEHOLDER_UUID.length) {
return PLACEHOLDER_UUID
}
if (kind === 'email' && length > EMAIL_DOMAIN.length) {
return 'u'.repeat(length - EMAIL_DOMAIN.length) + EMAIL_DOMAIN
}
return kind === 'local-username' || kind === 'local-hostname'
? 'x'.repeat(length)
: 'X'.repeat(length)
}
/** @returns {{text: string, redacted: number}} */
export function redactTranscript(text) {
const findings = scanTranscriptForSecrets(text)
let out = ''
let cursor = 0
for (const finding of findings) {
out += text.slice(cursor, finding.index)
out += placeholderFor(finding.kind, finding.match.length)
cursor = finding.index + finding.match.length
}
return { text: out + text.slice(cursor), redacted: findings.length }
}
export function formatFindings(label, findings) {
if (findings.length === 0) {
return `${label}: clean — no account identifier or credential shapes found.`
}
const rows = findings.map(
(finding) => ` ${finding.line}:${finding.column} ${finding.kind} ${preview(finding.match)}`
)
return [`${label}: ${findings.length} finding(s) — scrub before committing.`, ...rows].join('\n')
}
// Why a codepoint test and not a character class: a control-byte range written as an escape is
// folded back into raw 0x00-0x1f bytes by the formatter, which makes this file binary to the VCS
// and leaves the one file gating real PTY data into history unreviewable in a diff.
function preview(value) {
const head = value.length <= 24 ? value : `${value.slice(0, 21)}...`
let printable = ''
for (const char of head) {
printable += (char.codePointAt(0) ?? 0) < 0x20 ? '?' : char
}
return printable
}
@@ -0,0 +1,133 @@
// The scrub gate is the only thing standing between a live agent transcript and a
// committed account identifier, so it is pinned on the shapes those transcripts carry.
import { readdirSync, readFileSync } from 'node:fs'
import os from 'node:os'
import { join } from 'node:path'
import { describe, expect, it } from 'vitest'
import {
formatFindings,
placeholderFor,
redactTranscript,
scanTranscriptForSecrets
} from './pty-transcript-secret-scan.mjs'
import { parseArgs, resolveSpawn } from './capture-agent-pty-transcript.mjs'
describe('pty transcript secret scan', () => {
it('finds the account row of a ready screen', () => {
const findings = scanTranscriptForSecrets('Antigravity CLI 1.1.17\njin.woo@acme.dev (Business)')
expect(findings).toHaveLength(1)
expect(findings[0]).toMatchObject({ kind: 'email', line: 2, column: 1 })
})
it('finds credentials an agent may echo while signing in', () => {
const kinds = scanTranscriptForSecrets(
[
'token: eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxMjM0NTY3ODkwIn0.dBjftJeZ4CVP',
'key: AIzaSyA1234567890abcdefghijklmnopqrstu',
'refresh: 1//0gLm34XyZabcdefghijklmnopqrstuvwx',
'Authorization: Bearer abcdefghijklmnopqrstuvwxyz012345'
].join('\n')
).map((finding) => finding.kind)
expect(kinds).toEqual(['jwt', 'google-api-key', 'google-refresh-token', 'bearer-token'])
})
it('flags this machine’s own username, which a prompt line leaks', () => {
const username = os.userInfo().username
const findings = scanTranscriptForSecrets(`~/Users/${username}/orca/repo\n> `)
expect(findings.some((finding) => finding.kind === 'local-username')).toBe(true)
})
it('finds the resumable conversation id agy prints on exit', () => {
const findings = scanTranscriptForSecrets(
'Resume with -c (or command below):\nagy --conversation=26dc1986-9eec-456a-a534-d93e5c1076c2'
)
expect(findings).toHaveLength(1)
expect(findings[0].kind).toBe('uuid')
expect(placeholderFor('uuid', findings[0].match.length)).toMatch(
/^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[0-9a-f]{4}-[0-9a-f]{12}$/
)
})
it('reports a clean transcript as clean', () => {
const findings = scanTranscriptForSecrets('Antigravity CLI 1.1.17\nSonnet 4.6 (High)\n> ')
expect(findings).toEqual([])
expect(formatFindings('fixture', findings)).toContain('clean')
})
it('claims a span once, so a token inside an email is not double-reported', () => {
const findings = scanTranscriptForSecrets('longlivedaccountname@corp.internal')
expect(findings).toHaveLength(1)
})
it('passes a fixture that is already scrubbed, so "prove it is gone" can succeed', () => {
const scrubbed = `uuuu@example.com\n${'X'.repeat(44)}`
expect(scanTranscriptForSecrets(scrubbed)).toEqual([])
})
})
describe('redaction', () => {
it('replaces every finding with the same number of characters', () => {
// Why length matters: the fixture's value is its exact wrapping. A shorter
// replacement reflows the screen and invalidates the capture.
const text = 'Antigravity CLI 1.1.17\njin.woo@acme.dev (Antigravity Business)\n> '
const { text: redacted, redacted: count } = redactTranscript(text)
expect(count).toBe(1)
expect(redacted).toHaveLength(text.length)
expect(redacted).not.toContain('jin.woo@acme.dev')
expect(scanTranscriptForSecrets(redacted)).toEqual([])
expect(redactTranscript(redacted).redacted).toBe(0)
})
it('keeps a redacted email shaped like an email', () => {
expect(placeholderFor('email', 'a@b.example.com'.length)).toMatch(/^u+@example\.com$/)
})
it('leaves the rest of the screen byte-for-byte untouched', () => {
const text = 'line one\nuser@corp.io\nline three'
expect(redactTranscript(text).text.split('\n')[2]).toBe('line three')
})
})
describe('committed transcripts', () => {
// Why in CI and not just in the recorder: a transcript is committed once and read forever.
// The capture-time warning is skippable; this is not.
const fixtureDir = join(import.meta.dirname, '..', '..', 'src', 'main', 'runtime', '__fixtures__')
const transcripts = readdirSync(fixtureDir).filter((entry) => entry.endsWith('.txt'))
it.each(transcripts)('%s carries no account identifier or credential', (name) => {
const findings = scanTranscriptForSecrets(readFileSync(join(fixtureDir, name), 'utf8'))
expect(formatFindings(name, findings)).toContain('clean')
})
})
describe('capture argv', () => {
it('splits recorder options from the agent command', () => {
const { options, command } = parseArgs([
'--name',
'antigravity-ready-personal-non-gemini',
'--cols',
'120',
'--',
'agy',
'--model',
'sonnet'
])
expect(options.name).toBe('antigravity-ready-personal-non-gemini')
expect(options.cols).toBe(120)
expect(command).toEqual(['agy', '--model', 'sonnet'])
})
it('collects a multi-file scan list', () => {
const { options } = parseArgs(['--scan', 'a.txt', 'b.txt', '--redact'])
expect(options.scan).toEqual(['a.txt', 'b.txt'])
expect(options.redact).toBe(true)
})
it('routes a Windows shim through cmd.exe, which node-pty cannot spawn directly', () => {
expect(resolveSpawn(['agy.cmd', '--model', 'sonnet'])).toEqual(
process.platform === 'win32'
? { file: 'cmd.exe', args: ['/c', '"agy.cmd"', '--model', 'sonnet'] }
: { file: 'agy.cmd', args: ['--model', 'sonnet'] }
)
})
})
@@ -46,6 +46,7 @@ const WINDOWS_SHIM_SPAWN_ALLOWLIST = [
'config/scripts/electron-builder-config.test.mjs',
'config/scripts/ensure-native-runtime.test.mjs',
'config/scripts/live-remote-freeze-rpc.mjs',
'config/scripts/pty-transcript-secret-scan.test.mjs',
'config/scripts/remote-agent-session-authority-repro.mjs',
// Platform-local build paths; the win32 branch is dead code on both.
'config/scripts/build-mac-local.mjs',
@@ -0,0 +1,129 @@
# Capturing an agent PTY transcript
Orca's readiness and blocked-prompt rules are text rules over what an agent CLI paints on a
terminal. They are only as good as the screens they were written against. This is how to record
one, byte for byte, so a rule can be pinned to evidence instead of to a remembered screen.
Related: [`antigravity-readiness-evidence.md`](./antigravity-readiness-evidence.md) names the
specific Antigravity transcripts that are still missing and what each one decides.
## The recorder
```
node config/scripts/capture-agent-pty-transcript.mjs --name <fixture-name> [options] -- <command> [args...]
```
It allocates a real PTY, spawns the agent inside it, mirrors the session to your terminal so you
can drive it by hand, and appends every byte it receives to
`src/main/runtime/__fixtures__/<fixture-name>.txt`. It does not strip escapes, fold `\r`, rewrap
lines, or normalise anything — the file is what the terminal received.
- **Ending a capture:** press <kbd>Ctrl</kbd>+<kbd>]</kbd>. The recorder consumes that key and
never forwards it, which is the only way to end a capture _while a dialog still owns the
screen_. Quitting the agent instead would first dismiss the dialog you came to record.
- `--cols N --rows M` pin the PTY size (default: your terminal's). Wrapping is part of the
evidence, so record the size — the sidecar does it for you.
- `--duration S` stops unattended after S seconds, for a screen that needs no interaction.
- `--send "<ms>:<text>"` types into the PTY at a fixed offset, repeatable, with `\r` `\n` `\t` `\e`
escapes. A dialog capture has to be driven, and an unattended run (CI, or an agent) has no TTY to
type into; the keystrokes ride the same PTY a human's would. For example, the committed
`antigravity-dialog-model-picker.txt` was recorded with
`--duration 24 --send "14000:/model" --send "16000:\r"`, which leaves the picker owning the
screen when the capture stops.
- `--note "<text>"` records the account type, plan, model and CLI version in the sidecar.
- `--out <path>` writes outside the fixture directory (use it for a first dry run).
Each capture also writes `<fixture-name>.meta.json` with the timestamp, platform, command,
PTY size, note and exit code. Commit it with the transcript; the version and account type behind
a screen are not recoverable from the bytes.
**Prerequisite:** `node-pty` must be built for plain Node:
```
node config/scripts/ensure-native-runtime.mjs --runtime=node
```
Orca itself does not need to be running, and the recorder never touches Orca state.
### Platform notes
- **macOS / Linux:** nothing special. `TERM=xterm-256color` is set for the child.
- **Windows:** run it from Windows Terminal / PowerShell, not a Git Bash (MSYS) pane — MSYS
rewrites arguments that start with `/`, which mangles the `cmd.exe /c` hand-off. A `.cmd` or
`.bat` agent shim cannot be spawned by node-pty directly, so the recorder routes those through
`cmd.exe` for you.
- **WSL:** capture _inside_ the distro (run the recorder from the distro's checkout). Recording
`wsl.exe` from the Windows side adds the login-shell banner to the transcript.
- **SSH:** record on the execution host. A transcript recorded locally is not evidence about what
a remote agent prints.
## Privacy: scrub before committing
A live agent screen routinely contains things that must not enter git history:
| Scrub | Why |
| ---------------------------------------------------------------------- | ---------------------------------------------------- |
| Account email / sign-in identifier | The account row on a ready screen prints it verbatim |
| Org, tenant or team name | Identifies a customer |
| Machine hostname and OS username | Appear in prompts, paths and the OSC title |
| Absolute home paths (`/Users/<you>`, `C:\Users\<you>`) | Contain the username |
| JWTs, `AIza…` keys, `1//…` refresh tokens, `Bearer …`, `sk-…`, `ghp_…` | Live credentials; a sign-in screen can echo one |
| Private repo, branch and ticket names | Leak roadmap detail |
| Anything you pasted into the agent during the capture | You typed it; it is in the transcript |
The recorder scans the file as soon as the capture ends and prints every hit with a line and
column. To scrub:
```
node config/scripts/capture-agent-pty-transcript.mjs --scan src/main/runtime/__fixtures__/<name>.txt --redact
```
Redaction replaces each finding with a **same-length** placeholder (`u…u@example.com`, `XXXX…`).
Length matters: a transcript's value is its exact wrapping and column alignment, and a shorter
replacement reflows the screen and destroys the evidence.
### Verify it is gone
1. `node config/scripts/capture-agent-pty-transcript.mjs --scan src/main/runtime/__fixtures__/<name>.txt`
must print `clean` and exit `0`. It recognises its own placeholders, so a scrubbed file passes.
2. Grep for the specifics the scanner cannot know:
`rg -n -i -- "$(whoami)|<your-email>|<your-org>|<your-hostname>" src/main/runtime/__fixtures__/<name>.txt`
3. Read it once with escapes visible: `LC_ALL=C cat -v src/main/runtime/__fixtures__/<name>.txt`.
The scanner matches shapes; only a human catches a project name.
4. Check the sidecar too — `--note` text is free-form and is committed.
`config/scripts/pty-transcript-secret-scan.test.mjs` re-scans every committed
`__fixtures__/*.txt`, so a transcript that skips step 1 fails the suite.
## Consuming a transcript in a test
Feed the raw bytes through the runtime rather than into a matcher directly: escape handling,
tail retention and title tracking all live in `onPtyData`, and a rule tested on pre-normalised
text is tested on something no pane ever sees.
`src/main/runtime/agent-transcript-pane-test-harness.ts` builds the pane;
`src/main/runtime/terminal-interactive-wait-visibility.test.ts` (cursor-agent) and
`src/main/runtime/antigravity-readiness-transcripts.test.ts` (Antigravity) are the two consumers.
## Worked example: the Antigravity captures
The six committed `antigravity-*.txt` fixtures were recorded this way on macOS against
`agy` 1.1.25. Two points generalise:
- **Reach a state without mutating the operator's config.** The ready-screen captures ran in a
directory the CLI already trusted, so no trust answer was written. Where a dialog could only be
reached by signing the operator out or deleting their settings, it was left uncaptured and
recorded as such rather than forced.
- **An environment variable is a legitimate capture knob** where a setting is not.
`AGY_CLI_HIDE_ACCOUNT_INFO=1` produced a second ready screen with no account row, which is
evidence no amount of reasoning about the first screen could have supplied. It changes nothing
on disk.
## Known gap in the existing captures
The three `cursor-agent-*.txt` fixtures contain **no escape bytes and no carriage returns**.
Whatever produced them went through a renderer and a clipboard, so they preserve wording and
box-drawing glyphs but not the caret, the cursor moves, the repaints, or whether the CLI uses the
alternate screen buffer. They are good enough for the wording-based rules built on them and are
not evidence for anything else. New captures made with this recorder keep those bytes; the
Antigravity scaffold asserts their presence so a pasted screen cannot pass as a capture.
@@ -0,0 +1,263 @@
# Antigravity readiness: what the transcripts show
`findAntigravityReadyPromptIndex` in `src/main/runtime/terminal-wait-detection.ts` decides whether
an Antigravity pane is ready for a prompt. It has been written five times, each version tuned
against a five-line screen typed from memory into a `.spec.ts` fixture. Three of the first four
were found worse than the bug they replaced, and the fifth was reverted.
Real transcripts now exist. They were recorded from a live `agy` on macOS with
[`agent-pty-transcript-capture.md`](./agent-pty-transcript-capture.md) and are committed under
`src/main/runtime/__fixtures__/`. `src/main/runtime/antigravity-readiness-transcripts.test.ts`
replays them through the runtime.
**Headline: on real output the current detector is inverted.** It refuses a genuinely ready screen
and accepts a live model picker. The five attempts argued about which extra condition to add; none
of them had noticed that the condition they all shared — a line beginning with the model name —
never matches a real Antigravity ready screen at all.
## Versions
| Thing | Value |
| ------------------------- | ----------------------------- |
| `agy --version` | `1.1.25` |
| Banner printed by the TUI | `Antigravity CLI 1.2.0` |
| Captured | 2026-09-10, macOS, 120x40 PTY |
The binary and its own banner disagree. Any rule keyed to a version string must read the banner,
not `--version`, and must tolerate the two disagreeing.
## What the captures are
| Fixture | What it is |
| -------------------------------------------- | --------------------------------------------------------- |
| `antigravity-ready-api-key-gemini-model.txt` | Ready screen, API-key identity, Gemini 3.7 Flash (Low) |
| `antigravity-ready-account-info-hidden.txt` | The same ready screen with `AGY_CLI_HIDE_ACCOUNT_INFO=1` |
| `antigravity-dialog-trust-workspace.txt` | Workspace trust dialog, live and unanswered |
| `antigravity-dialog-model-picker.txt` | `/model` picker, live and unanswered |
| `antigravity-dialog-command-palette.txt` | Slash-command palette, live and unanswered |
| `antigravity-dialog-dismissed.txt` | `/model` picker dismissed with esc, then settled |
| `antigravity-busy-mid-turn.txt` | A real turn, recording stopped while the spinner was live |
| `antigravity-busy-turn-ended.txt` | The same turn after it ended and the composer returned |
## What could not be captured, and why
Nothing below was faked. Each is a case the recorder could not reach without changing the
operator's account state or configuration, which is out of bounds.
| Missing | Why |
| ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
| `antigravity-ready-business-non-gemini.txt` | This machine has no OAuth session — the CLI prints _"You are currently not signed in"_ and authenticates from `GEMINI_API_KEY`. Reaching a Business ready screen means signing someone in. |
| A non-Gemini model on any ready screen | `agy models` offers 11 models, all Gemini, and `settings.json` pins `modelProvider: gemini`. A non-Gemini row is not reachable from this account. |
| `antigravity-dialog-sign-in.txt` | Unsetting `GEMINI_API_KEY` does not reach the sign-in dialog; the CLI refuses to start because `modelProvider` is pinned. Reaching it means editing the operator's `settings.json`. |
| `antigravity-dialog-theme-picker.txt` | There is no `/theme` command in 1.2.0 (`Unknown command: /theme`). The picker appears only in first-run onboarding, which means deleting the operator's config. |
| `antigravity-dialog-privacy-notice.txt` | First-run onboarding, as above. |
| `antigravity-dialog-update-banner.txt` | Cannot be forced; no update was pending during the session. |
Each remains as a named, skipping case in the suite so it is visible rather than forgotten.
## What the transcripts show
### 1. The ready screen's model row is not at the start of a line
The ready screen prints a block-glyph logo down the left, and the identity, model and path rows are
painted **on the same physical lines as the logo**. What Orca derives is:
```
▀▀▀▀▀▀ Gemini API key
▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low)
▄▀▀ ▀▀▄ ~
```
The detector requires `normalized.startsWith('gemini', trimmedStart)` on a trimmed line. The
trimmed line starts with `▀`. It never matches. Measured three ways on the real screen:
| Input | `isKnownReadyPromptPreview` |
| ------------------------------------------------------ | --------------------------- |
| Real ready screen | `false` |
| The same screen with the logo glyphs stripped | `true` |
| Real ready screen followed by the live `/model` picker | `true` |
So the logo — decoration, and suppressible with `AGY_CLI_HIDE_LOGO` — is what decides readiness
today, and the live dialog is what supplies the model line the ready screen could not.
### 2. The dialog is what satisfies the model rule
`/model` prints its options one per line:
```
Gemini 3.8 Flash
> Gemini 3.7 Flash (current)
Gemini 3.1 Pro
```
Those lines _do_ begin with `Gemini`, and a bare `>` composer line sits earlier in the same tail
from before the picker opened. Both halves of the rule are satisfied **while a dialog owns the
screen**, and the pane reads ready. This is the false-ready hazard the last three attempts were
each trying to close, reproduced from a real capture.
### 3. `>` is the dialog selection marker, not only the composer caret
Every dialog uses `>` to mark the highlighted row: `> Yes, I trust this folder`,
`> Gemini 3.7 Flash (current)`, `> /add-dir`. The idle composer is a line whose whole trimmed
content is `>`. That distinction is the only thing separating them, which means the relaxation
proposed in PRs #15840 and #15852 — accept any line _beginning_ with `>` — would make the trust
dialog and the model picker read as ready. On 1.2.0 the idle composer is a bare `>`; those PRs'
1.1.17 mode-banner claim could not be reproduced here and may be mode-specific.
### 4. There is no email account row, and the row can be switched off entirely
For an API-key user the identity row reads literally `Gemini API key`. There is no `@`, no
domain, nothing an account-row rule can key on. Separately, `AGY_CLI_HIDE_ACCOUNT_INFO=1` — a
supported environment variable in the binary — removes the row from a fully ready screen, which
`antigravity-ready-account-info-hidden.txt` captures.
### 5. Dialogs are drawn two different ways, and the banner is never reprinted
The trust dialog and the sign-in splash take the **alternate screen** (`ESC[?1049h` … `ESC[?1049l`).
The model picker and command palette are drawn **in place on the main screen** with erase-to-EOL.
After dismissal the CLI prints `⎿ Exited /model command` and redraws the composer — it does **not**
reprint the banner. The header stays where it was at startup.
### 6. Rows are positioned with cursor addressing, not newlines
The status row is written with absolute and relative moves (`ESC[13;99H`, `ESC[83X ESC[83C`), so
`? for shortcuts` and `Gemini 3.7 Flash · low` end up on one derived line. Any rule that assumes
one screen row equals one `\n`-delimited line is reading a different document than the user sees.
## 8. Busy frames park the caret exactly like idle frames — the spinner is what differs
The frame that ends a turn-in-progress and the frame that ends an idle screen park the cursor with
the **same bytes**. Only the hint row differs, and the park erases it:
```
idle: ? for shortcuts ESC[83X ESC[83C Gemini 3.7 Flash · low CR ESC[2A ESC[2C ESC[?25h
busy: esc to cancel ESC[85X ESC[85C Gemini 3.7 Flash · low CR ESC[2A ESC[2C ESC[?25h
```
So a rule that keys on "the caret is the last thing in the tail" cannot tell busy from idle **on the
frame alone**. What saves it is what comes next. Each spinner tick is its own repaint with its own
park, two rows higher than the frame's:
```
ESC[?25l CR ESC[2A ⣯ Generating ESC[11D ESC[?25h
ESC[?25l CR ESC[2A ⣟ Generating. ESC[12D ESC[?25h
```
That second `CR ESC[2A` splices the composer row away, so the retained tail during a live turn ends
on the spinner row, not on the caret. Measured on `antigravity-busy-mid-turn.txt`:
| Capture | last retained line | bare `>` line present |
| -------------------------------------------- | ------------------ | --------------------- |
| `antigravity-ready-api-key-gemini-model.txt` | `>` | **yes** |
| `antigravity-busy-mid-turn.txt` | `⣟ Generating...` | **no** |
**Consequence for a caret-based rule:** it already answers "not ready" for a real mid-turn capture,
because there is no bare caret in the tail to match. A constructed input that keeps the park bytes
and only edits the status text is not faithful to a live turn — a live turn has a spinner row
repainting _below_ the composer.
**The residual window, and the clause it implies.** Between a frame park and the next spinner tick
the tail does end on the bare caret and is indistinguishable from idle. The gap is one tick
interval. Any readiness path gated on sustained quiescence is safe, because ticks keep arriving and
the pane is never quiet; a path that only inspects retained text is not. For those paths the
evidence supports one clause, and only one:
> **A braille glyph (U+2800–U+28FF) on the last visible line of the retained tail means working.**
That predicate already exists in this file for cursor-agent (`CURSOR_BUSY_SPINNER_RE`) and should be
reused rather than reinvented. It must be scoped to the **last visible line**, not the whole tail:
a first-run transcript prints `⠾ Signing in...` during startup, which would otherwise pin a ready
screen as busy forever.
Nothing else in the capture distinguishes the two states. The hint row (`esc to cancel` versus
`? for shortcuts`) is erased by the park in both cases, the park offsets are identical, and
`ESC[?25l`/`ESC[?25h` fencing appears around every repaint, idle or busy.
## Confirmed / refuted, by attempt
Evidence column names the fixture; all quoted text is from the committed transcripts.
### Attempt 1 — the rule at HEAD
| # | Claim | Verdict | Evidence |
| ---- | -------------------------------------------------------- | --------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- |
| 1.1 | A ready screen prints the banner `Antigravity CLI` | **Confirmed** | `Antigravity CLI 1.2.0` in both ready fixtures |
| 1.1b | …and its last occurrence in the tail is the live one | **Refuted** | The trust dialog's own body says _"Antigravity CLI requires permission to read, edit, and execute files here"_, so `lastIndexOf` lands inside the dialog |
| 1.2 | The model row begins with the vendor word `Gemini` | **Refuted** | `▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low)` — the logo precedes it; never at line start |
| 1.3 | The caret line's whole trimmed content is `>` | **Confirmed** on 1.2.0 idle | bare `>` in both ready fixtures |
| 1.3b | …and only the composer prints `>` | **Refuted** | `> Yes, I trust this folder`, `> Gemini 3.7 Flash (current)`, `> /add-dir` |
| 1.4 | A ready screen prints the workspace path on its own line | **Refuted** | the path shares its line with logo glyphs (`▄▀▀ ▀▀▄ ~`) |
### Attempt 2 (loop 1) — blacklist the model line
| # | Claim | Verdict | Evidence |
| --- | ------------------------------------------ | ----------- | ---------------------------------------------------------------------------------------------------------------- |
| 2.1 | Dialog model-row wording is enumerable | **Refuted** | the palette lists 50+ commands with free-form descriptions; the picker prints whatever models the account offers |
| 2.2 | A dialog never reproduces a real model row | **Refuted** | the `/model` picker prints four real model rows, one per line, at line start |
### Attempt 3 (loop 2) — structural ordering on `headerIndex`
| # | Claim | Verdict | Evidence |
| --- | -------------------------------------------------- | ---------------------------------- | ---------------------------------------------------------------------------------------------------- |
| 3.1 | A live dialog is printed below the ready chrome | **Confirmed** for in-place dialogs | picker and palette append below the composer |
| 3.2 | The banner is reprinted when a dialog is dismissed | **Refuted** | `antigravity-dialog-dismissed.txt` shows `⎿ Exited /model command` and a redrawn composer, no banner |
| 3.3 | Antigravity does not use the alternate screen | **Refuted** | `ESC[?1049h` opens the trust dialog and the sign-in splash |
| 3.4 | No full repaint per keystroke | **Partly refuted** | typing `/mod` repaints the palette region on each keystroke with `ESC[K` |
Because of 3.2, `headerIndex` cannot be the anchor: it never advances. Ordering can only be
expressed against the model/caret positions, which is what 1.2 and 1.3b just invalidated.
### Attempt 4 (loop 3) — require a positive account row
| # | Claim | Verdict | Evidence |
| --- | ---------------------------------------------------- | ---------------------- | --------------------------------------------------------------------------------------------------------------------------- |
| 4.1 | Every ready screen prints an account row | **Refuted, twice** | API-key identity prints `Gemini API key` (no `@`); `AGY_CLI_HIDE_ACCOUNT_INFO=1` removes the row entirely |
| 4.2 | A startup dialog never contains an `@`-and-`.` token | **Not reachable here** | none of the captured dialogs contains one, but the palette shows free-form skill descriptions, which are user-authored text |
| 4.3 | The account row is distinguishable from prose | **Refuted** | the row is not a distinct line; it shares one with the logo |
### Attempt 5 (PR #19749, reverted) — ordering + account row
| # | Claim | Verdict | Evidence |
| --- | -------------------------------------------------------- | ----------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| 5.1 | Ordering plus an account row separates ready from dialog | **Refuted** | the account row is optional (4.1) and the ordering anchor never moves (3.2) |
| 5.2 | Executing both builds was sufficient verification | **Refuted** | the executed input was the hand-written fixture, so the check reproduced the fixture's assumptions. The real screen disagrees with that fixture on the model row, the path row and the account row |
| 5.3 | The wedge is a model-name problem | **Refuted** | it is a line-start problem. Even `Gemini 3.7 Flash (Low)` — a Gemini model — fails, because a logo glyph precedes it |
### Cross-cutting
| # | Question | Answer |
| --- | ---------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------- |
| X1 | Does `agy` set an OSC title distinguishing busy from idle? | **No.** Not one OSC title sequence appears in any capture. Title-based readiness is unavailable for this agent |
| X2 | Does it repaint with bare `\r`? | **Yes**, constantly, plus `ESC[K` and absolute cursor moves |
| X3 | Does the caret survive in the tail? | **Yes** — a bare `>` line is present in every ready capture |
| X4 | Banner-to-caret distance | ~8 derived lines on a 120x40 PTY; the banner falls outside the 6-line preview window, so only the full retained tail can see it |
| X5 | Pane title on the trust screen versus ready | Identical: none |
## Can attempt six be written?
Yes — but not as a variation on any of the five. Every one of them refined a predicate over
`\n`-delimited lines, and that is the layer where the evidence says the information is not.
What the captures support:
- **The one stable, dialog-free ready marker is a line whose entire trimmed content is `>`.** It is
present in every ready capture and absent from every dialog capture, because a dialog's `>` always
carries its selected row's label. This is a much narrower rule than any attempt used, and it is
the only one that survived contact with the transcripts.
- **Drop the model-row requirement.** It matches dialogs and not ready screens. Keeping it inverted
the detector.
- **Do not require an account row.** It is optional by environment variable and carries no email for
API-key users.
- **Do not anchor on `headerIndex`.** The banner is printed once and never reprinted.
- **The blocked-signal path already works** for the trust dialog: `antigravity-dialog-trust-workspace.txt`
is correctly refused today, by wording, not by structure.
What is still unknown and should be captured before shipping: the sign-in, theme, privacy and
update dialogs, and any ready screen where the composer is not idle (accept-edits and plan mode,
which PRs #15840 and #15852 describe from a screenshot). A bare-`>` rule is only as good as the
claim that those modes still end on a bare `>`; that claim is untested.
The honest summary is that this is a screen-shaped problem being solved with line-shaped tools. A
rule over the derived tail can be made much better than what ships today, but the durable fix is to
ask the terminal emulator what the bottom row of the screen actually is, rather than inferring it
from a byte stream that was written with cursor addressing.
+1
View File
@@ -29,6 +29,7 @@
"test": "node config/scripts/ensure-native-runtime.mjs --runtime=node && vitest run --config config/vitest.config.ts",
"test:skill-sharing:release": "vitest run --config config/vitest.config.ts src/main/skills src/main/runtime/rpc/methods/skills.test.ts src/relay/skill-install-handler.test.ts src/shared/skill-bundle-install-contract.test.ts src/shared/skill-install-contract.test.ts src/shared/skill-install-failure.test.ts src/shared/skill-package-manifest.test.ts",
"test:repro:remote-agent-session": "pnpm run build:cli && pnpm run build:electron-vite && node config/scripts/remote-agent-session-authority-repro.mjs",
"capture:agent-transcript": "node config/scripts/ensure-native-runtime.mjs --runtime=node && node config/scripts/capture-agent-pty-transcript.mjs",
"check:reliability-gates": "node config/scripts/check-reliability-gates.mjs",
"check:max-lines-ratchet": "node config/scripts/check-max-lines-ratchet.mjs",
"check:ts-nocheck-ratchet": "node config/scripts/check-ts-nocheck-ratchet.mjs",
@@ -0,0 +1,9 @@
{
"capturedAt": "2026-09-11T06:10:52.713Z",
"platform": "darwin",
"command": ["agy"],
"cols": 120,
"rows": 40,
"note": "agy TUI 1.2.0; recording stopped ~0.3s after submit, while the spinner was live; no shutdown repaint in the file",
"exitCode": 0
}
@@ -0,0 +1,38 @@
[?2026$p[?2027$p[>4m[=0;1u[?1049h[?25l[?5W[?2004h[>4;2m[=1;1u[?u
▄▀▀▄
▀▀▀▀▀▀
▀▀▀▀▀▀▀▀
▄▀▀ ▀▀▄
▄▀▀ ▀▀▄
Welcome to the Antigravity CLI. You are currently not signed in.
⣾ Signing in...␍ No authentication methods available.
Press ctrl+c or ctrl+d twice to exit.[>4m[=0;1u[?1049l[>4;2m[=1;1u[?u[0 q␍
▄▀▀▄ Antigravity CLI 1.2.0
▀▀▀▀▀▀ Gemini API key
▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low)
▄▀▀ ▀▀▄ ~
▄▀▀ ▀▀▄
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
>
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
? for shortcutsGemini 3.7 Flash · low␍[?25h[?25lI[?25h[?25ln ab
 G[?25h[?25lout 8[?25h[?25l0 wo[?25h[?25lrds,[?25h[?25lexpla[?25h[?25lin w[?25h[?25lhat a[?25h[?25l pse[?25h[?25lud[?25h[?25loter[?25h[?25lminal[?25h[?25l is.[?25h[?25l[?25h[?25l
? for shortcuts[?25h[?25lM
> In about 80 words, explain what a pseudoterminal is.
⣷ Generating...
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
>
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
esc to cancelGemini 3.7 Flash · low␍[?25h[?25lng
[?25h[?25l␍⣯ Generating
[?25h[?25l␍⣟ Generating.
[?25h
@@ -0,0 +1,9 @@
{
"capturedAt": "2026-09-11T06:13:00.364Z",
"platform": "darwin",
"command": ["agy"],
"cols": 120,
"rows": 40,
"note": "agy TUI 1.2.0; recording stopped after the turn ended and the composer returned, with the process still alive. This account's API key cannot complete a turn, so the turn ends in a backend error",
"exitCode": 0
}
@@ -0,0 +1,42 @@
[?2026$p[?2027$p[?5W[?2004h[>4;2m[=1;1u[?u[0 q␍
▄▀▀▄ Antigravity CLI 1.2.0
▀▀▀▀▀▀ Gemini API key
▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low)
▄▀▀ ▀▀▄ ~
▄▀▀ ▀▀▄
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
>
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
? for shortcutsGemini 3.7 Flash · low␍[?25h[?25lIn
 G[?25h[?25labo[?25h[?25lut 80[?25h[?25l wo[?25h[?25lrds[?25h[?25l, ex[?25h[?25lpla[?25h[?25lin wh[?25h[?25lat a[?25h[?25lpseudo[?25h[?25ltermi[?25h[?25lnal is[?25h[?25l.[?25h[?25l
? for shortcuts[?25h[?25lM
> In about 80 words, explain what a pseudoterminal is.
⣾ Generating...
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
>
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
esc to cancelGemini 3.7 Flash · low␍[?25h[?25l␍⣷ Generatin
[?25h[?25l␍⣯ Generating
[?25h[?25l␍⣟ Generating.
[?25h[?25l␍⡿ Generating...
[?25h[?25l␍⢿ Generatin
[?25h[?25l␍
⚠ Agent execution terminated due to error.
Error ID: 00000000-0000-4000-8000-000000000000-2
⢿ Generating...
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
>
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
esc to cancelGemini 3.7 Flash · low␍[?25h[?25l␍
? for shortcuts[?25h
@@ -0,0 +1,9 @@
{
"capturedAt": "2026-09-11T04:34:32.974Z",
"platform": "darwin",
"command": ["agy"],
"cols": 120,
"rows": 40,
"note": "agy TUI 1.2.0; slash-command palette live, unanswered",
"exitCode": 0
}
@@ -0,0 +1,41 @@
[?2026$p[?2027$p[?5W[?2004h[>4;2m[=1;1u[?u[0 q␍
▄▀▀▄ Antigravity CLI 1.2.0
▀▀▀▀▀▀ Gemini API key
▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low)
▄▀▀ ▀▀▄ ~
▄▀▀ ▀▀▄
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
>
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
? for shortcutsGemini 3.7 Flash · low␍[?25h[?25l/
> /add-dir  Add a directory to the workspace
/agents List available custom agents
/artifact View and review artifacts
/btw Ask a side question without interrupting the current task
/changelog Show release notes and changes
 ↓ 50 more

↑/↓ Navigate · enter Select · tab Complete
 Gemini 3.7 Flash · low␍[?25h[?25l
esc to cancel[?25h[>4m[=0;1u
[?2004l[0 q
@@ -0,0 +1,9 @@
{
"capturedAt": "2026-09-11T04:35:06.866Z",
"platform": "darwin",
"command": ["agy"],
"cols": 120,
"rows": 40,
"note": "agy TUI 1.2.0; /model picker opened then dismissed with esc, settled before stop",
"exitCode": 0
}
@@ -0,0 +1,54 @@
[?2026$p[?2027$p[?5W[?2004h[>4;2m[=1;1u[?u[0 q␍
▄▀▀▄ Antigravity CLI 1.2.0
▀▀▀▀▀▀ Gemini API key
▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low)
▄▀▀ ▀▀▄ ~
▄▀▀ ▀▀▄
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
>
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
? for shortcutsGemini 3.7 Flash · low␍[?25h[?25l/mod
> /model Set a model, or run a single prompt on another model
/permissioned-github Guidelines for interacting with GitHub and request permissions from the user when commands f...

↑/↓ Navigate · enter Select · tab Complete
esc to cancelGemini 3.7 Flash · low␍[?25h[?25l
/model


↑/↓ Navigate · enter Select · tab Complete
esc to cancelGemini 3.7 Flash · low␍[?25h[?25l[0 q
Switch Model
Gemini 3.8 Flash
> Gemini 3.7 Flash (current)
Gemini 3.6 Flash
Gemini 3.1 Pro

Effort ◂  ◉──────────────○──────────────○  ▸
  low  medium high 
 Faster responses, lighter reasoning — great for simpler tasks

Keyboard: ↑/↓ Navigate ←/→ Effort enter Select esc Go Back

 Gemini 3.7 Flash · low␍[0 q> /model
 ⎿ Exited /model command

────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
>
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
Gemini 3.7 Flash · low␍[?25h[?25l
? for shortcuts[?25h[>4m[=0;1u
[?2004l[0 q
Resume with -c (or command below):
agy --conversation=00000000-0000-4000-8000-000000000000
@@ -0,0 +1,9 @@
{
"capturedAt": "2026-09-11T04:34:10.855Z",
"platform": "darwin",
"command": ["agy"],
"cols": 120,
"rows": 40,
"note": "agy TUI 1.2.0; /model picker live, unanswered, killed while it owns the screen",
"exitCode": 0
}
@@ -0,0 +1,56 @@
[?2026$p[?2027$p[?5W[?2004h[>4;2m[=1;1u[?u[0 q␍
▄▀▀▄ Antigravity CLI 1.2.0
▀▀▀▀▀▀ Gemini API key
▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low)
▄▀▀ ▀▀▄ ~
▄▀▀ ▀▀▄
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
>
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
? for shortcutsGemini 3.7 Flash · low␍[?25h[?25l/mo
> /model Set a model, or run a single prompt on another model
/migrate-workflows Automatically migrate legacy workflows to modern skills across global and workspace configur...
/permissions Manage tool permissions
/agy-customizations Comprehensive guide and reference for the Antigravity Customization System. Use to explain h...
/permissioned-github Guidelines for interacting with GitHub and request permissions from the user when commands f...

↑/↓ Navigate · enter Select · tab Complete
? for shortcutsGemini 3.7 Flash · low␍[?25h[?25l
/model


↑/↓ Navigate · enter Select · tab Complete
esc to cancelGemini 3.7 Flash · low␍[?25h[?25l[0 q
Switch Model
> Gemini 3.8 Flash
Gemini 3.7 Flash (current)
Gemini 3.6 Flash
Gemini 3.1 Pro

Effort ◂  ●━━━━━━━━━━━━━━◉──────────────○  ▸
  low  medium  high 
 Balanced speed and reasoning quality for most tasks

Keyboard: ↑/↓ Navigate ←/→ Effort enter Select esc Go Back

? for shortcutsGemini 3.7 Flash · low␍ Gemini 3.8 Flash
> Gemini 3.7 Flash
◂  ◉──────────────○
 low  medium 
Faster responses, lighter reasoning — great for simpler tasks
 G[>4m[=0;1u␍[?25h[?2004l
@@ -0,0 +1,9 @@
{
"capturedAt": "2026-09-11T04:35:20.989Z",
"platform": "darwin",
"command": ["agy"],
"cols": 120,
"rows": 40,
"note": "agy TUI 1.2.0; workspace trust dialog live and unanswered in a throwaway untrusted directory",
"exitCode": 0
}
@@ -0,0 +1,12 @@
[?2026$p[?2027$p[>4m[=0;1u[?1049h[?25l[?5W[?2004h[>4;2m[=1;1u[?uAccessing workspace:
/private/tmp/agy-trust-scratch-77950
Do you trust the contents of this project?
Antigravity CLI requires permission to read, edit, and execute files here.
> Yes, I trust this folder
No, exit
↑/↓ Navigate · enter ConfirmGemini 3.7 Flash · low[>4m[=0;1u␍[?1049l[?25h[?2004l
@@ -0,0 +1,9 @@
{
"capturedAt": "2026-09-11T04:33:34.954Z",
"platform": "darwin",
"command": ["agy"],
"cols": 120,
"rows": 40,
"note": "same session as antigravity-ready-api-key-gemini-model but with AGY_CLI_HIDE_ACCOUNT_INFO=1",
"exitCode": 0
}
@@ -0,0 +1,13 @@
[?2026$p[?2027$p[?5W[?2004h[>4;2m[=1;1u[?u[0 q␍
▄▀▀▄ Antigravity CLI 1.2.0
▀▀▀▀▀▀ Gemini 3.7 Flash (Low)
▀▀▀▀▀▀▀▀ ~
▄▀▀ ▀▀▄
▄▀▀ ▀▀▄
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
>
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
? for shortcutsGemini 3.7 Flash · low␍[?25h[>4m[=0;1u
[?2004l[0 q
@@ -0,0 +1,9 @@
{
"capturedAt": "2026-09-11T04:33:14.819Z",
"platform": "darwin",
"command": ["agy"],
"cols": 120,
"rows": 40,
"note": "agy binary 1.1.25, TUI banner 1.2.0; Gemini API key identity (no OAuth sign-in); model Gemini 3.7 Flash (Low); workspace ~",
"exitCode": 0
}
@@ -0,0 +1,13 @@
[?2026$p[?2027$p[?5W[?2004h[>4;2m[=1;1u[?u[0 q␍
▄▀▀▄ Antigravity CLI 1.2.0
▀▀▀▀▀▀ Gemini API key
▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low)
▄▀▀ ▀▀▄ ~
▄▀▀ ▀▀▄
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
>
────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
? for shortcutsGemini 3.7 Flash · low␍[?25h[>4m[=0;1u
[?2004l[0 q
@@ -0,0 +1,79 @@
// One pane builder for every suite that replays a captured agent transcript through the runtime.
import { vi } from 'vitest'
import { OrcaRuntimeService } from './orca-runtime'
const TRANSCRIPT_PANE_LEAF_ID = '11111111-1111-4111-8111-111111111111'
const TRANSCRIPT_PANE_TAB_ID = 'tab-1'
const TRANSCRIPT_PANE_WORKTREE_ID = 'wt-1'
export const TRANSCRIPT_PANE_PTY_ID = 'pty-1'
export type TranscriptPaneOptions = {
paneTitle: string
foregroundProcess: string | null
data: string
/** Set for a pane whose PTY lives on an SSH host or WSL distro rather than locally. */
connectionId?: string
/** Simulates a PTY controller whose foreground probe never settles. */
foregroundProbeHangs?: boolean
onForegroundProbe?: () => void
}
export async function createTranscriptPane(
options: TranscriptPaneOptions
): Promise<{ runtime: OrcaRuntimeService; handle: string }> {
const runtime = new OrcaRuntimeService(null)
const internals = runtime as unknown as {
resolveTerminalWorkspaceLaunchScope: (selector: string) => Promise<unknown>
}
vi.spyOn(internals, 'resolveTerminalWorkspaceLaunchScope').mockResolvedValue({
id: TRANSCRIPT_PANE_WORKTREE_ID,
path: '/repo/app',
connectionId: options.connectionId ?? null,
repo: null,
folderWorkspace: null
})
runtime.setPtyController({
spawn: vi.fn().mockResolvedValue({ id: TRANSCRIPT_PANE_PTY_ID, incarnationId: 'inc-1' }),
write: () => true,
kill: () => true,
getForegroundProcess: (): Promise<string | null> => {
options.onForegroundProbe?.()
return options.foregroundProbeHangs === true
? new Promise<string | null>(() => {})
: Promise.resolve(options.foregroundProcess)
}
})
const terminal = await runtime.createTerminal(`id:${TRANSCRIPT_PANE_WORKTREE_ID}`, {
tabId: TRANSCRIPT_PANE_TAB_ID,
leafId: TRANSCRIPT_PANE_LEAF_ID,
title: 'Terminal'
})
runtime.attachWindow(1)
runtime.syncWindowGraph(1, {
tabs: [
{
tabId: TRANSCRIPT_PANE_TAB_ID,
worktreeId: TRANSCRIPT_PANE_WORKTREE_ID,
title: 'Terminal',
activeLeafId: TRANSCRIPT_PANE_LEAF_ID,
layout: null
}
],
leaves: [
{
tabId: TRANSCRIPT_PANE_TAB_ID,
worktreeId: TRANSCRIPT_PANE_WORKTREE_ID,
leafId: TRANSCRIPT_PANE_LEAF_ID,
paneRuntimeId: 1,
ptyId: TRANSCRIPT_PANE_PTY_ID,
paneTitle: options.paneTitle
}
]
})
// Why the guard: a restore seed is only applied to a never-written record, so the restore
// cases must not write an empty chunk first.
if (options.data.length > 0) {
runtime.onPtyData(TRANSCRIPT_PANE_PTY_ID, options.data, Date.now())
}
return { runtime, handle: terminal.handle }
}
@@ -0,0 +1,281 @@
/**
* Pins Antigravity readiness to captured transcripts instead of hand-written fixtures.
*
* Five detector attempts were tuned against a five-line screen someone typed from memory, and
* three of them shipped worse behaviour than the bug they replaced. Nothing here asserts what
* Antigravity prints: the transcripts do. Six are recorded from a live `agy`; the rest name
* themselves as skipped until someone can reach them.
*
* Four cases are pinned as KNOWN DEFECT: on real output the shipped detector refuses the ready
* screen and accepts the live model picker. Those assert what it does, not what it should.
*
* Capture protocol: docs/reference/agent-pty-transcript-capture.md
* What each transcript decides: docs/reference/antigravity-readiness-evidence.md
*/
import { existsSync, readFileSync } from 'node:fs'
import { join } from 'node:path'
import { describe, expect, it, vi } from 'vitest'
import { createTranscriptPane } from './agent-transcript-pane-test-harness'
import { extractLastOscTitle } from '../../shared/osc-title-extraction'
vi.mock('electron', () => ({
BrowserWindow: { fromId: vi.fn(() => null) },
webContents: { fromId: vi.fn(() => null) },
ipcMain: { on: vi.fn(), removeListener: vi.fn() },
app: { getPath: vi.fn(() => '/tmp') }
}))
const FIXTURE_DIR = join(__dirname, '__fixtures__')
const EVIDENCE_DOC = join(
__dirname,
'..',
'..',
'..',
'docs',
'reference',
'antigravity-readiness-evidence.md'
)
// Why asymmetric: a ready verdict has to survive the settle window, while a refusal only has to
// hold for one poll. Keeping the refusal short keeps seven transcripts off the suite's clock.
const READY_TIMEOUT_MS = 2_000
const REFUSAL_TIMEOUT_MS = 600
/** Antigravity's binary, as Orca launches and probes it (`tui-agent-config.ts` detectCmd). */
const ANTIGRAVITY_COMMAND = 'agy'
// String.fromCharCode, not a literal: the formatter rewrites an escape sequence into a raw
// control byte in source, which is unreadable and survives badly in diffs.
const ESC = String.fromCharCode(27)
type TranscriptCase = {
/** Fixture basename; `<name>.txt` under `__fixtures__/`. */
name: string
/** Capture in docs/reference/antigravity-readiness-evidence.md. */
capture: string
what: string
/** What a correct detector must answer. Not what the shipped one answers. */
expectReady: boolean
/**
* Set where the shipped detector contradicts the transcript. The case then runs inverted, so
* CI pins the defect instead of going permanently red — and flips to failing the moment
* someone fixes it, which is exactly when these expectations need re-reading.
*/
knownDefect?: string
}
const TRANSCRIPTS: readonly TranscriptCase[] = [
{
name: 'antigravity-ready-api-key-gemini-model',
capture: 'B',
what: 'ready screen, API-key identity — the account row reads "Gemini API key", not an email',
expectReady: true,
knownDefect: 'refused: the model row never starts a line, the logo shares it'
},
{
name: 'antigravity-ready-account-info-hidden',
capture: 'B',
what: 'ready screen with AGY_CLI_HIDE_ACCOUNT_INFO=1 — no account row at all',
expectReady: true,
knownDefect: 'refused: same line-start defect, and no account row exists to require'
},
{
name: 'antigravity-dialog-trust-workspace',
capture: 'C',
what: 'workspace trust dialog owning the screen',
expectReady: false
},
{
name: 'antigravity-dialog-model-picker',
capture: 'C',
what: 'model picker owning the screen',
expectReady: false,
knownDefect: "accepted: the picker's own `Gemini 3.x Flash` rows satisfy the model rule"
},
{
name: 'antigravity-dialog-command-palette',
capture: 'C',
what: 'slash-command palette owning the screen',
expectReady: false
},
{
name: 'antigravity-busy-mid-turn',
capture: 'E',
what: 'mid-turn, spinner live — the pane is working, not waiting for a prompt',
expectReady: false
},
{
// Expected ready because the turn is over and the composer is back on screen. The captured
// turn ends in a backend error, which is the only ending this account's key can produce.
name: 'antigravity-busy-turn-ended',
capture: 'E',
what: 'the turn has ended and the composer has returned, process still alive',
expectReady: true,
knownDefect: 'refused: the retained tail ends on the error block, with no composer row in it'
},
{
name: 'antigravity-dialog-dismissed',
capture: 'D',
what: 'the screen immediately after the model picker is dismissed',
expectReady: true,
knownDefect: 'refused: the banner is not reprinted and no model row starts a line'
},
// Not captured: this machine's agy has no OAuth session and offers only Gemini models, and
// reaching the rest would mean signing the operator out or deleting their config. See
// docs/reference/antigravity-readiness-evidence.md § What could not be captured.
{
name: 'antigravity-ready-business-non-gemini',
capture: 'A',
what: 'ready screen, Business account, non-Gemini model',
expectReady: true
},
{
name: 'antigravity-dialog-sign-in',
capture: 'C',
what: 'sign-in dialog owning the screen',
expectReady: false
},
{
name: 'antigravity-dialog-theme-picker',
capture: 'C',
what: 'theme picker owning the screen',
expectReady: false
},
{
name: 'antigravity-dialog-privacy-notice',
capture: 'C',
what: 'privacy notice owning the screen',
expectReady: false
},
{
name: 'antigravity-dialog-update-banner',
capture: 'C',
what: 'update banner owning the screen',
expectReady: false
}
]
function fixturePath(name: string): string {
return join(FIXTURE_DIR, `${name}.txt`)
}
/**
* A `tui-idle` wait ends three ways, and only one of them is readiness: it resolves satisfied, it
* resolves unsatisfied with a blocked reason, or it rejects with `timeout` because nothing ever
* looked ready. The orchestrator treats the last two identically — no prompt is delivered — so
* they are both `ready: false` here. This is the shape `worker-start` sees.
*/
async function readinessVerdict(
transcript: string,
timeoutMs: number
): Promise<{ ready: boolean; blockedReason: unknown; outcome: string }> {
const { runtime, handle } = await createTranscriptPane({
// Why the transcript's own title: every attempt guessed at Antigravity's title. A raw
// capture carries the OSC bytes, so the pane wears whatever the CLI actually set.
paneTitle: extractLastOscTitle(transcript) ?? ANTIGRAVITY_COMMAND,
foregroundProcess: ANTIGRAVITY_COMMAND,
data: transcript
})
try {
const result = (await runtime.waitForTerminal(handle, {
condition: 'tui-idle',
timeoutMs
})) as { satisfied?: boolean; blockedReason?: unknown }
return {
ready: result.satisfied === true,
blockedReason: result.blockedReason ?? null,
outcome: result.satisfied === true ? 'satisfied' : 'unsatisfied'
}
} catch (error) {
return { ready: false, blockedReason: null, outcome: `rejected: ${String(error)}` }
}
}
describe('Antigravity readiness, decided by captured transcripts', () => {
for (const transcript of TRANSCRIPTS) {
const path = fixturePath(transcript.name)
const captured = existsSync(path)
const label = `capture ${transcript.capture}: ${transcript.what}`
// A pinned defect asserts what the detector DOES, so CI is honest rather than permanently
// red; fixing the detector flips this case to failing, which is when these expectations
// need re-reading. The correct answer stays in `expectReady` and in the test's name.
const shipped =
transcript.knownDefect === undefined ? transcript.expectReady : !transcript.expectReady
const verdictName =
transcript.knownDefect === undefined
? `${label} → ${transcript.expectReady ? 'ready' : 'not ready'}`
: `${label} → must be ${transcript.expectReady ? 'ready' : 'not ready'}; KNOWN DEFECT, ${transcript.knownDefect}`
it.skipIf(!captured)(
verdictName,
async () => {
// A refusal only has to hold for one poll; a ready verdict has to survive the settle
// window. Keeping the refusal short keeps eleven transcripts off the suite's clock.
const verdict = await readinessVerdict(
readFileSync(path, 'utf8'),
transcript.expectReady ? READY_TIMEOUT_MS : REFUSAL_TIMEOUT_MS
)
// A silent dialog carries no blocked-signal wording, so the assertion is only that Orca
// does not call the pane ready and type a prompt into a dialog that owns the screen.
expect({ ready: verdict.ready, outcome: verdict.outcome }).toMatchObject({
ready: shipped
})
},
READY_TIMEOUT_MS + 10_000
)
it.skipIf(!captured)(`${label} was captured raw, not pasted from a rendered screen`, () => {
const text = readFileSync(path, 'utf8')
// Why: a transcript with no escape bytes went through a terminal's renderer and a
// human's clipboard. It cannot answer what the caret or chrome looked like.
expect(text).toContain(ESC)
})
}
it('documents every transcript the detector is allowed to depend on', () => {
// Why a test: the doc is the operator's checklist. A name that drifts out of it is a
// transcript nobody will capture, and a case that silently skips forever.
const doc = readFileSync(EVIDENCE_DOC, 'utf8')
for (const transcript of TRANSCRIPTS) {
expect(doc).toContain(`${transcript.name}.txt`)
}
})
it('reports how much evidence exists, so a fully skipped run is visible', () => {
const missing = TRANSCRIPTS.filter(
(transcript) => !existsSync(fixturePath(transcript.name))
).map((transcript) => `${transcript.name}.txt`)
if (missing.length > 0) {
console.info(
`Antigravity transcripts: ${TRANSCRIPTS.length - missing.length}/${TRANSCRIPTS.length} captured. Missing: ${missing.join(', ')}`
)
}
expect(missing.length).toBeLessThanOrEqual(TRANSCRIPTS.length)
})
})
describe('scaffold self-check', () => {
// Why these two live here: when a transcript lands and fails, the failure has to mean the
// capture disagreed with the detector — not that the harness or the timeouts are broken.
// Neither case is evidence about Antigravity; both are shapes the current detector already
// decides, used only to prove the plumbing reaches a verdict.
it('reaches a ready verdict through the harness', async () => {
const verdict = await readinessVerdict(
[
'Antigravity CLI 1.0.3',
'user@example.com (Antigravity Business)',
'Gemini 3.5 Flash (High)',
'~/orca/workspaces/orca/agy-dispatch-issue',
'>'
].join('\n'),
READY_TIMEOUT_MS
)
expect(verdict.ready).toBe(true)
})
it('reaches a not-ready verdict through the harness', async () => {
const verdict = await readinessVerdict(
'Do you trust this workspace directory?\nPress t to trust\n',
REFUSAL_TIMEOUT_MS
)
expect(verdict.ready).toBe(false)
})
})
@@ -3,7 +3,10 @@
import { readFileSync } from 'node:fs'
import { join } from 'node:path'
import { describe, expect, it, vi } from 'vitest'
import { OrcaRuntimeService } from './orca-runtime'
import {
createTranscriptPane as createPane,
TRANSCRIPT_PANE_PTY_ID as PTY_ID
} from './agent-transcript-pane-test-harness'
import { assertTerminalAgentSendable } from './rpc/terminal-agent-send-guard'
vi.mock('electron', () => ({
@@ -13,12 +16,11 @@ vi.mock('electron', () => ({
app: { getPath: vi.fn(() => '/tmp') }
}))
const LEAF_ID = '11111111-1111-4111-8111-111111111111'
const TAB_ID = 'tab-1'
const WORKTREE_ID = 'wt-1'
const PTY_ID = 'pty-1'
// Captured verbatim from cursor-agent 2026.08.11-e8db854 driven through Orca.
// cursor-agent 2026.08.11-e8db854's screens, but NOT raw PTY output: these files contain no
// escape bytes and no carriage returns, so they came through a terminal's renderer and a
// clipboard. They evidence wording, ordering and glyphs — which is all the rules below key on —
// and evidence nothing about the caret, cursor moves, repaints or the alternate screen buffer.
// Record new fixtures with config/scripts/capture-agent-pty-transcript.mjs, which keeps the bytes.
function fixture(name: string): string {
return readFileSync(join(__dirname, '__fixtures__', `${name}.txt`), 'utf8')
}
@@ -39,73 +41,6 @@ function agentStatusOsc(state: string): string {
return `]9999;${JSON.stringify({ state, prompt: 'ship it', agentType: 'claude' })}`
}
async function createPane(options: {
paneTitle: string
foregroundProcess: string | null
data: string
/** Set for a pane whose PTY lives on an SSH host or WSL distro rather than locally. */
connectionId?: string
/** Simulates a PTY controller whose foreground probe never settles. */
foregroundProbeHangs?: boolean
onForegroundProbe?: () => void
}): Promise<{ runtime: OrcaRuntimeService; handle: string }> {
const runtime = new OrcaRuntimeService(null)
const internals = runtime as unknown as {
resolveTerminalWorkspaceLaunchScope: (selector: string) => Promise<unknown>
}
vi.spyOn(internals, 'resolveTerminalWorkspaceLaunchScope').mockResolvedValue({
id: WORKTREE_ID,
path: '/repo/app',
connectionId: options.connectionId ?? null,
repo: null,
folderWorkspace: null
})
runtime.setPtyController({
spawn: vi.fn().mockResolvedValue({ id: PTY_ID, incarnationId: 'inc-1' }),
write: () => true,
kill: () => true,
getForegroundProcess: (): Promise<string | null> => {
options.onForegroundProbe?.()
return options.foregroundProbeHangs === true
? new Promise<string | null>(() => {})
: Promise.resolve(options.foregroundProcess)
}
})
const terminal = await runtime.createTerminal(`id:${WORKTREE_ID}`, {
tabId: TAB_ID,
leafId: LEAF_ID,
title: 'Terminal'
})
runtime.attachWindow(1)
runtime.syncWindowGraph(1, {
tabs: [
{
tabId: TAB_ID,
worktreeId: WORKTREE_ID,
title: 'Terminal',
activeLeafId: LEAF_ID,
layout: null
}
],
leaves: [
{
tabId: TAB_ID,
worktreeId: WORKTREE_ID,
leafId: LEAF_ID,
paneRuntimeId: 1,
ptyId: PTY_ID,
paneTitle: options.paneTitle
}
]
})
// Why the guard: a restore seed is only applied to a never-written record, so the restore
// cases must not write an empty chunk first.
if (options.data.length > 0) {
runtime.onPtyData(PTY_ID, options.data, Date.now())
}
return { runtime, handle: terminal.handle }
}
// cursor-agent renders a braille spinner in its OSC title while it works, and Orca reads
// that as `working`; the title is identical whether it is running a command or waiting.
const CURSOR_TITLE = '⠇ Cursor Agent'
@@ -1,11 +1,15 @@
import type * as Monaco from 'monaco-editor'
import { compile } from 'monaco-editor/esm/vs/editor/standalone/common/monarch/monarchCompile.js'
import { MonarchTokenizer } from 'monaco-editor/esm/vs/editor/standalone/common/monarch/monarchLexer.js'
import { describe, expect, it } from 'vitest'
import {
EMBED_ENTRY_REST_OF_LINE_BUDGET,
MAX_TOKENIZATION_LINE_LENGTH
} from './monarch-embed-entry-budget'
import {
createMonarchTokenizer,
endEmbeddedLanguages,
measureNestedDepth,
tokenizeLines
} from './monarch-tokenizer-test-harness'
import { astroMonarchLanguage } from './register-astro'
import { svelteMonarchLanguage } from './register-svelte'
import { vueMonarchLanguage } from './register-vue'
@@ -13,93 +17,15 @@ import { vueMonarchLanguage } from './register-vue'
// Monarch tokenizes embedded languages by mutual recursion: `_nestedTokenize`
// tail-calls `_myTokenize`, which tail-calls `_nestedTokenize` again for every
// embed entered mid-line. V8 has no TCO, so each mid-line embed entry costs
// real JS stack. Before the embed-entry budget, one 17_000-character line of
// `<script></script>` (under Monaco's own 20_000 line cap) reached ~1743 nested
// levels and died with `RangeError: Maximum call stack size exceeded` — the
// renderer-side STATUS_STACK_OVERFLOW this suite guards.
type MonarchEndState = { embeddedLanguageData?: { languageId: string } | null }
type MonarchTokenizerInstance = {
getInitialState: () => unknown
tokenize: (line: string, hasEOL: boolean, state: unknown) => { endState: MonarchEndState }
_nestedTokenize: (...args: unknown[]) => unknown
}
function createMonarchTokenizer(
languageId: string,
language: Monaco.languages.IMonarchLanguage,
maxTokenizationLineLength = MAX_TOKENIZATION_LINE_LENGTH
): MonarchTokenizerInstance {
// Nested languages stay unregistered: `_getNestedEmbeddedLanguageData` then
// hands back a null state, which changes what the embed *emits* but not
// whether monarch recurses into it — the depth measurement is unaffected.
const languageService = {
languageIdCodec: { encodeLanguageId: () => 1, decodeLanguageId: () => '' },
getLanguageIdByLanguageName: () => null,
getLanguageIdByMimeType: () => null,
isRegisteredLanguageId: () => false,
requestBasicLanguageFeatures: () => {}
}
const themeService = { getColorTheme: () => ({ tokenTheme: {} }) }
const configurationService = {
getValue: () => maxTokenizationLineLength,
onDidChangeConfiguration: () => ({ dispose: () => {} })
}
return new MonarchTokenizer(
languageService,
themeService,
languageId,
compile(languageId, language),
configurationService
) as MonarchTokenizerInstance
}
type TokenizeMeasurement = { maxNestedDepth: number; error: Error | undefined }
function measureNestedDepth(
tokenizer: MonarchTokenizerInstance,
lines: string[]
): TokenizeMeasurement {
const nestedTokenize = tokenizer._nestedTokenize.bind(tokenizer)
let depth = 0
let maxNestedDepth = 0
tokenizer._nestedTokenize = (...args: unknown[]) => {
depth += 1
maxNestedDepth = Math.max(maxNestedDepth, depth)
try {
return nestedTokenize(...args)
} finally {
depth -= 1
}
}
let error: Error | undefined
let state = tokenizer.getInitialState()
try {
for (const line of lines) {
state = tokenizer.tokenize(line, true, state).endState
}
} catch (thrown) {
error = thrown as Error
}
return { maxNestedDepth, error }
}
// The embedded language each line *ends* in — `null` means the line left the
// tokenizer with no embed, i.e. that region renders unhighlighted.
function embeddedLanguagePerLine(
tokenizer: MonarchTokenizerInstance,
lines: string[]
): (string | null)[] {
let state: unknown = tokenizer.getInitialState()
return lines.map((line) => {
const endState = tokenizer.tokenize(line, true, state).endState
state = endState
return endState.embeddedLanguageData?.languageId ?? null
})
}
// real JS stack. Embeds cannot nest (monarchLexer throws "cannot enter embedded
// language from within an embedded language"), so these are sequential
// enter/exit transitions on one line, each holding a frame until the line ends.
//
// Before the embed-entry budget, one 17_000-character line of `<script></script>`
// (under Monaco's own 20_000 line cap) reached ~1743 frames and threw
// `RangeError: Maximum call stack size exceeded`. Monaco's `safeTokenize` catches
// that per line, so the visible failure is a line that silently loses all
// highlighting; the frame count is what this suite bounds.
// 6600 is the largest `{a}` count under Monaco's line cap (19_800 chars); the
// filter below drops it for the longer chunk shapes, so the densest embed
@@ -127,7 +53,7 @@ const PATHOLOGICAL_LINES: [string, (count: number) => string][] = [
describe.each([
['svelte', svelteMonarchLanguage],
['astro', astroMonarchLanguage]
])('%s embedded-tokenizer recursion depth', (languageId, language) => {
])('%s embed-entry recursion', (languageId, language) => {
it.each(PATHOLOGICAL_LINES)(
'stays within the embed budget for a line of %s',
(_name, buildLine) => {
@@ -175,12 +101,14 @@ describe.each([
// budget, so the body starts unembedded. Every following short line must
// recover the embed (and the `lang=` language) instead of leaving the whole
// block unhighlighted until the closing tag.
const embeds = embeddedLanguagePerLine(createMonarchTokenizer(languageId, language), [
`<${tag} lang="${lang}">a = "${'x'.repeat(EMBED_ENTRY_REST_OF_LINE_BUDGET)}"`,
' b',
' c',
`</${tag}>`
])
const embeds = endEmbeddedLanguages(
tokenizeLines(createMonarchTokenizer(languageId, language), [
`<${tag} lang="${lang}">a = "${'x'.repeat(EMBED_ENTRY_REST_OF_LINE_BUDGET)}"`,
' b',
' c',
`</${tag}>`
])
)
expect(embeds).toEqual([null, embeddedLanguageId, embeddedLanguageId, null])
})
@@ -199,9 +127,9 @@ describe.each([
describe('unguarded embedded tokenizer', () => {
// Control: the same markup/expression shape with no budget on embed entry.
// Depth then tracks the interpolation count one-for-one, which is what took
// the renderer down; ~1700 levels is already a RangeError in this runtime,
// so the ramp stops short of the overflow to stay deterministic.
// The frame count then tracks the interpolation count one-for-one; ~1700
// frames is already a RangeError in this runtime, so the ramp stops short of
// the overflow to stay deterministic.
const perInterpolationEmbedLanguage: Monaco.languages.IMonarchLanguage = {
defaultToken: '',
tokenizer: {
@@ -225,7 +153,7 @@ describe('unguarded embedded tokenizer', () => {
})
})
describe('vue embedded-tokenizer recursion depth', () => {
describe('vue embed-entry recursion', () => {
const templateLine = (count: number): string =>
`<template><p>${'{{a}}'.repeat(count)}</p></template>`
@@ -252,11 +180,13 @@ describe('vue embedded-tokenizer recursion depth', () => {
['script', 'ts', 'typescript'],
['style', 'scss', 'scss']
])('re-embeds a %s body after an over-budget opening line', (tag, lang, embeddedLanguageId) => {
const embeds = embeddedLanguagePerLine(createMonarchTokenizer('vue', vueMonarchLanguage), [
`<${tag} lang="${lang}">a = "${'x'.repeat(EMBED_ENTRY_REST_OF_LINE_BUDGET)}"`,
' b',
`</${tag}>`
])
const embeds = endEmbeddedLanguages(
tokenizeLines(createMonarchTokenizer('vue', vueMonarchLanguage), [
`<${tag} lang="${lang}">a = "${'x'.repeat(EMBED_ENTRY_REST_OF_LINE_BUDGET)}"`,
' b',
`</${tag}>`
])
)
expect(embeds).toEqual([null, embeddedLanguageId, null])
})
@@ -0,0 +1,159 @@
import type * as Monaco from 'monaco-editor'
import { compile } from 'monaco-editor/esm/vs/editor/standalone/common/monarch/monarchCompile.js'
import { MonarchTokenizer } from 'monaco-editor/esm/vs/editor/standalone/common/monarch/monarchLexer.js'
import { MAX_TOKENIZATION_LINE_LENGTH } from './monarch-embed-entry-budget'
// Drives the real `MonarchTokenizer` shipped with monaco-editor rather than
// walking a grammar's rule table. A table walk cannot see the failures that
// actually reach the renderer — a grammar that throws on every `{expr}`, or
// that silently drops an embed, still has a well-formed rule table.
/** One Monaco token. `language` is the (embedded) language the region belongs to. */
export type MonarchToken = { offset: number; type: string; language: string }
type MonarchEndState = { embeddedLanguageData?: { languageId: string } | null }
export type MonarchTokenizerInstance = {
getInitialState: () => unknown
tokenize: (
line: string,
hasEOL: boolean,
state: unknown
) => { tokens: MonarchToken[]; endState: MonarchEndState }
_nestedTokenize: (...args: unknown[]) => unknown
}
export function createMonarchTokenizer(
languageId: string,
language: Monaco.languages.IMonarchLanguage,
maxTokenizationLineLength = MAX_TOKENIZATION_LINE_LENGTH
): MonarchTokenizerInstance {
// Nested languages stay unregistered, so `nestedLanguageTokenize` emits one
// empty-typed token tagged with the embedded language id instead of running
// that language's tokenizer. That is what makes `token.language` a direct
// readout of which embed covers which region.
const languageService = {
languageIdCodec: { encodeLanguageId: () => 1, decodeLanguageId: () => '' },
getLanguageIdByLanguageName: () => null,
getLanguageIdByMimeType: () => null,
isRegisteredLanguageId: () => false,
requestBasicLanguageFeatures: () => {}
}
const themeService = { getColorTheme: () => ({ tokenTheme: {} }) }
const configurationService = {
getValue: () => maxTokenizationLineLength,
onDidChangeConfiguration: () => ({ dispose: () => {} })
}
return new MonarchTokenizer(
languageService,
themeService,
languageId,
compile(languageId, language),
configurationService
) as MonarchTokenizerInstance
}
export type TokenizedLine = {
text: string
tokens: MonarchToken[]
/** Embedded language still active at end of line; `null` means that region renders unhighlighted. */
endEmbeddedLanguageId: string | null
}
/** Tokenizes `lines` as one document, threading tokenizer state line to line. */
export function tokenizeLines(
tokenizer: MonarchTokenizerInstance,
lines: string[]
): TokenizedLine[] {
let state: unknown = tokenizer.getInitialState()
return lines.map((text) => {
const { tokens, endState } = tokenizer.tokenize(text, true, state)
state = endState
return {
text,
tokens,
endEmbeddedLanguageId: endState.embeddedLanguageData?.languageId ?? null
}
})
}
export function tokenizeMonarchDocument(
languageId: string,
language: Monaco.languages.IMonarchLanguage,
source: string
): TokenizedLine[] {
return tokenizeLines(createMonarchTokenizer(languageId, language), source.split('\n'))
}
/** The embedded language each line *ends* in — `null` for no embed. */
export function endEmbeddedLanguages(lines: TokenizedLine[]): (string | null)[] {
return lines.map((line) => line.endEmbeddedLanguageId)
}
/** The distinct languages a line's tokens were attributed to, in order. */
export function tokenLanguages(line: TokenizedLine): string[] {
return line.tokens
.map((token) => token.language)
.filter((language, index, all) => language !== all[index - 1])
}
/**
* Per line, which languages actually cover it. This is the readout that catches
* a silently dropped embed: the region falls back to the host grammar's own id
* instead of `html` / `typescript` / `scss`, and renders unhighlighted.
*/
export function tokenLanguagesPerLine(lines: TokenizedLine[]): string[][] {
return lines.map(tokenLanguages)
}
/** Token type covering `index`, without the grammar's `tokenPostfix`. */
export function tokenTypeAt(line: TokenizedLine, index: number): string {
const covering = line.tokens.findLast((token) => token.offset <= index)
return covering?.type.split('.').slice(0, -1).join('.') ?? ''
}
/** One `text | offset:type@language … | embed=…` row per line, for snapshots. */
export function formatTokenizedLines(lines: TokenizedLine[]): string[] {
return lines.map((line) => {
const tokens = line.tokens
.map((token) => `${token.offset}:${token.type || '-'}@${token.language}`)
.join(' ')
return `${line.text} | ${tokens} | embed=${line.endEmbeddedLanguageId ?? 'none'}`
})
}
export type TokenizeMeasurement = { maxNestedDepth: number; error: Error | undefined }
/**
* Tokenizes `lines`, recording peak `_nestedTokenize` recursion — the real JS
* stack cost, since Monarch enters an embed by mutual recursion with no TCO.
* Embeds cannot nest, so this counts sequential embed enter/exit transitions on
* one line, each holding a frame until the line ends. Errors are captured rather
* than thrown so a caller can assert on frame count and failure together.
*/
export function measureNestedDepth(
tokenizer: MonarchTokenizerInstance,
lines: string[]
): TokenizeMeasurement {
const nestedTokenize = tokenizer._nestedTokenize.bind(tokenizer)
let depth = 0
let maxNestedDepth = 0
tokenizer._nestedTokenize = (...args: unknown[]) => {
depth += 1
maxNestedDepth = Math.max(maxNestedDepth, depth)
try {
return nestedTokenize(...args)
} finally {
depth -= 1
}
}
let error: Error | undefined
try {
tokenizeLines(tokenizer, lines)
} catch (thrown) {
error = thrown as Error
}
return { maxNestedDepth, error }
}
@@ -0,0 +1,58 @@
// @vitest-environment happy-dom
// Why happy-dom: monaco's `basic-languages` entry points import the full
// browser editor before they export the grammar.
import { language as mdxLanguage } from 'monaco-editor/esm/vs/basic-languages/mdx/mdx.js'
import { describe, expect, it } from 'vitest'
import {
EMBED_ENTRY_REST_OF_LINE_BUDGET,
MAX_TOKENIZATION_LINE_LENGTH
} from './monarch-embed-entry-budget'
import { createMonarchTokenizer, measureNestedDepth } from './monarch-tokenizer-test-harness'
import { svelteMonarchLanguage } from './register-svelte'
// Why pin a third-party grammar: monaco's OWN shipped mdx grammar enters a `js`
// embed on every `{` and pops on `}` with no budget, so it reproduces the
// unbounded embed-entry recursion exactly. That makes it the proof this shape is
// monaco's, not something Orca's svelte/astro/vue grammars invented — and it is
// the tripwire for a monaco upgrade that changes the recursion shape. Do not
// delete as "not our code".
/** One `js` embed enter/exit transition per repeat, in 3 characters. */
const interpolations = (count: number): string => '{a}'.repeat(count)
/** Longest run of them monaco will still tokenize at all. */
const UNTOKENIZABLE_ABOVE = Math.floor(MAX_TOKENIZATION_LINE_LENGTH / 3) - 1
describe('upstream monaco mdx grammar', () => {
it('spends one stack frame per interpolation, unbounded', () => {
const frames = [50, 200, 500].map(
(count) =>
measureNestedDepth(createMonarchTokenizer('mdx', mdxLanguage), [interpolations(count)])
.maxNestedDepth
)
expect(frames).toEqual([50, 200, 500])
})
it('exhausts the JS stack on a line monaco is still willing to tokenize', () => {
const line = interpolations(UNTOKENIZABLE_ABOVE)
expect(line.length).toBeLessThan(MAX_TOKENIZATION_LINE_LENGTH)
const measurement = measureNestedDepth(createMonarchTokenizer('mdx', mdxLanguage), [line])
// The frame ceiling is runtime-dependent (~1145 measured here), so assert the
// failure rather than the number.
expect(measurement.error).toBeInstanceOf(RangeError)
expect(measurement.maxNestedDepth).toBeLessThan(UNTOKENIZABLE_ABOVE)
})
it('is what the embed-entry budget holds: the same shape stays bounded', () => {
const measurement = measureNestedDepth(
createMonarchTokenizer('svelte', svelteMonarchLanguage),
[interpolations(UNTOKENIZABLE_ABOVE)]
)
expect(measurement.error).toBeUndefined()
expect(measurement.maxNestedDepth).toBeLessThanOrEqual(EMBED_ENTRY_REST_OF_LINE_BUDGET)
})
})
@@ -1,112 +1,32 @@
import { describe, expect, it, vi } from 'vitest'
import {
endEmbeddedLanguages,
formatTokenizedLines,
tokenizeMonarchDocument,
tokenLanguages,
tokenLanguagesPerLine
} from './monarch-tokenizer-test-harness'
import {
astroLanguageConfiguration,
astroMonarchLanguage,
registerAstroLanguage
} from './register-astro'
type MonarchAction = {
next?: string
nextEmbedded?: string
switchTo?: string
}
type MonarchRule = [RegExp, string | MonarchAction, string?] | { include: string }
function normalizeState(nextState: string): string {
return nextState.startsWith('@') ? nextState.slice(1) : nextState
// Driven through the real `MonarchTokenizer`: a rule-table walk cannot tell a
// working grammar from one that throws on every `{expr}`, which is how broken
// Astro highlighting shipped green.
function tokenizeAstro(source: string) {
return tokenizeMonarchDocument('astro', astroMonarchLanguage, source)
}
function isRuleEntry(rule: MonarchRule): rule is [RegExp, string | MonarchAction, string?] {
return Array.isArray(rule)
}
function getRuleAction(rule: [RegExp, string | MonarchAction, string?]): MonarchAction | undefined {
const [, action, nextStateShortcut] = rule
return typeof action === 'object'
? action
: nextStateShortcut
? { next: nextStateShortcut }
: undefined
}
function findRuleAction(
state: string,
source: string,
{ embedPopOnly = false }: { embedPopOnly?: boolean } = {}
): MonarchAction | undefined {
const tokenizer = astroMonarchLanguage.tokenizer as Record<string, MonarchRule[]>
const stateRules = tokenizer[state] ?? tokenizer[state.split('.')[0]]
const candidateRules = embedPopOnly
? stateRules.filter((rule) => {
if (!isRuleEntry(rule)) {
return false
}
return getRuleAction(rule)?.nextEmbedded === '@pop'
})
: stateRules
const matchedRule = candidateRules.find((rule) => {
if (!isRuleEntry(rule)) {
return false
}
const [regexp] = rule
regexp.lastIndex = 0
const match = regexp.exec(source)
return match !== null && match.index === 0
})
return matchedRule && isRuleEntry(matchedRule) ? getRuleAction(matchedRule) : undefined
}
function collectFixtureRuleActions(source: string): string[] {
const ruleActions: string[] = []
const tokenizer = astroMonarchLanguage.tokenizer as Record<string, MonarchRule[]>
const lines = source.split('\n')
const checks: { line: number; state: string; pattern: string }[] = [
{ line: 1, state: 'root', pattern: '---' },
{ line: 4, state: 'frontmatter', pattern: '---' },
// After the frontmatter closes we are back in `markupReenter`; the next
// non-structural character switches into `markup` with html active.
{ line: 6, state: 'markupReenter', pattern: '' },
{ line: 6, state: 'markup', pattern: '{' },
{ line: 6, state: 'astroExpression', pattern: '}' },
{ line: 8, state: 'markup', pattern: '<script' },
{ line: 8, state: 'scriptOpen.javascript', pattern: '>' },
{ line: 10, state: 'scriptBody.javascript', pattern: '</script>' },
{ line: 12, state: 'markup', pattern: '<style' },
{ line: 12, state: 'styleOpen.css', pattern: '>' },
{ line: 14, state: 'styleBody.css', pattern: '</style>' }
]
checks.forEach((check) => {
const line = lines.at(check.line - 1) ?? ''
const stateRules = tokenizer[check.state] ?? tokenizer[check.state.split('.')[0]]
const matchedRule = stateRules.find((rule) => {
if (!isRuleEntry(rule)) {
return false
}
const [regexp] = rule
regexp.lastIndex = 0
const match = regexp.exec(line)
return match !== null && match[0] === check.pattern
})
if (!matchedRule || !isRuleEntry(matchedRule)) {
return
}
const actionObject = getRuleAction(matchedRule)
const nextState = actionObject?.next ? normalizeState(actionObject.next) : '-'
const nextEmbedded = actionObject?.nextEmbedded ?? '-'
const switchTo = actionObject?.switchTo ? normalizeState(actionObject.switchTo) : '-'
ruleActions.push(
`${check.line}:${check.state}:${check.pattern || '<html>'} -> next=${nextState}, embedded=${nextEmbedded}, switch=${switchTo}`
)
})
return ruleActions
/** Which languages actually cover each line — a dropped embed shows up as `astro`. */
function languagesPerLine(source: string): string[][] {
return tokenLanguagesPerLine(tokenizeAstro(source))
}
describe('registerAstroLanguage registration', () => {
// Structural by necessity: covers the registration call itself (ids,
// extensions, idempotence), which tokenizing cannot observe.
it('registers the astro language, Monarch tokenizer, and configuration once', () => {
const languages: { id: string }[] = [{ id: 'typescript' }]
const register = vi.fn((entry: { id: string }) => {
@@ -140,8 +60,8 @@ describe('registerAstroLanguage registration', () => {
})
})
describe('astro tokenizer transitions', () => {
it('captures Astro tokenizer transitions for a representative component fixture', () => {
describe('astro tokenization', () => {
it('tokenizes a representative component', () => {
const fixture = `---
import Layout from '../layouts/Layout.astro'
const title = 'Home'
@@ -157,111 +77,120 @@ const title = 'Home'
h1 { color: rebeccapurple; }
</style>`
const ruleActions = collectFixtureRuleActions(fixture)
expect(ruleActions).toMatchInlineSnapshot(`
expect(formatTokenizedLines(tokenizeAstro(fixture))).toMatchInlineSnapshot(`
[
"1:root:--- -> next=-, embedded=typescript, switch=frontmatter",
"4:frontmatter:--- -> next=-, embedded=@pop, switch=markupReenter",
"6:markupReenter:<html> -> next=-, embedded=html, switch=markup",
"6:markup:{ -> next=-, embedded=@pop, switch=astroExpressionEnter",
"6:astroExpression:} -> next=-, embedded=@pop, switch=markupReenter",
"8:markup:<script -> next=-, embedded=@pop, switch=scriptOpen.javascript",
"8:scriptOpen.javascript:> -> next=-, embedded=$S2, switch=scriptBody.$S2",
"10:scriptBody.javascript:</script> -> next=-, embedded=@pop, switch=markupReenter",
"12:markup:<style -> next=-, embedded=@pop, switch=styleOpen.css",
"12:styleOpen.css:> -> next=-, embedded=$S2, switch=styleBody.$S2",
"14:styleBody.css:</style> -> next=-, embedded=@pop, switch=markupReenter",
"--- | 0:keyword.astro@astro | embed=typescript",
"import Layout from '../layouts/Layout.astro' | 0:-@typescript | embed=typescript",
"const title = 'Home' | 0:-@typescript | embed=typescript",
"--- | 0:keyword.astro@astro | embed=none",
" | | embed=html",
"<h1>{title}</h1> | 0:-@html 4:delimiter.curly.astro@astro 5:-@typescript 10:delimiter.curly.astro@astro 11:-@html | embed=html",
" | 0:-@html | embed=html",
"<script> | 0:tag.astro@astro | embed=javascript",
" console.log('hi') | 0:-@javascript | embed=javascript",
"</script> | 0:tag.astro@astro | embed=none",
" | | embed=html",
"<style lang="scss"> | 0:tag.astro@astro 6:white.astro@astro 7:attribute.name.astro@astro 11:delimiter.astro@astro 12:attribute.value.astro@astro 18:tag.astro@astro | embed=scss",
" h1 { color: rebeccapurple; } | 0:-@scss | embed=scss",
"</style> | 0:tag.astro@astro | embed=none",
]
`)
})
})
describe('astro tokenizer regressions', () => {
// Regression: a file that opens with a markup expression like `{title}` has
// no html embed active yet. If `root` itself ever emitted `nextEmbedded:
// '@pop'` Monaco would throw "cannot pop embedded language if not inside
// one" before any push had occurred. Enforce the invariant directly.
it('never pops an embedded language from the root state', () => {
const tokenizer = astroMonarchLanguage.tokenizer as Record<string, MonarchRule[]>
const popRules = tokenizer.root.filter((rule) => {
if (!isRuleEntry(rule)) {
return false
}
return getRuleAction(rule)?.nextEmbedded === '@pop'
})
expect(popRules).toHaveLength(0)
it('embeds the frontmatter fence as typescript', () => {
expect(
endEmbeddedLanguages(tokenizeAstro("---\nconst title = 'Home'\n---\n<h1>hi</h1>"))
).toEqual(['typescript', 'typescript', null, 'html'])
})
// Regression (verified live in the Electron app): when entry from root went
// straight to `@markup` with `nextEmbedded: 'html'`, while the embed-pop
// path also went via `@markupReenter`, Monarch's nested tokenizer reported
// "cannot pop embedded language if not inside one" on `{expr}` in markup.
// Routing every push of the html embed through `markupReenter` keeps the
// embed-stack invariant identical for every entry into `markup`.
it('routes all entries into markup through markupReenter', () => {
expect(findRuleAction('root', '<h1>Hello</h1>')).toMatchObject({
switchTo: '@markupReenter'
})
expect(findRuleAction('root', '{title}')).toMatchObject({
switchTo: '@markupReenter'
})
expect(findRuleAction('markupReenter', '<h1>Hello</h1>')).toMatchObject({
switchTo: '@markup',
nextEmbedded: 'html'
})
// Pins monaco-editor#1127: the pop rule's `^` survives Monaco's regex
// rebuild, so an indented or trailing `---` must not close the fence early.
it('keeps the frontmatter fence open past a --- that is not at column 0', () => {
expect(endEmbeddedLanguages(tokenizeAstro('---\n// ---\n ---\n---\n<h1>hi</h1>'))).toEqual([
'typescript',
'typescript',
'typescript',
null,
'html'
])
})
// Regression: while the html embed is active, only parent rules whose action
// pops the embed are consulted before delegating to html. The `markup`
// state must wire `nextEmbedded: '@pop'` on the structural rules so a
// trailing `<script>` or `<style>` after page markup can switch language.
it('pops the html embed when later script/style/comment markers appear', () => {
expect(findRuleAction('markup', '<script>', { embedPopOnly: true })).toMatchObject({
switchTo: '@scriptOpen.javascript',
nextEmbedded: '@pop'
})
expect(findRuleAction('markup', '<style lang="scss">', { embedPopOnly: true })).toMatchObject({
switchTo: '@styleOpen.css',
nextEmbedded: '@pop'
})
expect(findRuleAction('markup', '<!-- comment -->', { embedPopOnly: true })).toMatchObject({
switchTo: '@comment',
nextEmbedded: '@pop'
})
expect(findRuleAction('markup', '{title}', { embedPopOnly: true })).toMatchObject({
switchTo: '@astroExpressionEnter',
nextEmbedded: '@pop'
})
// Regression (verified live in the Electron app): an expression in the first
// markup line popped the html embed before any push, and Monarch threw
// "cannot pop embedded language if not inside one".
it('highlights an expression in the first markup line', () => {
const [line] = tokenizeAstro('<p>a {title} b</p>')
expect(tokenLanguages(line)).toEqual(['html', 'astro', 'typescript', 'astro', 'html'])
})
it('opens a file on an expression without popping a missing embed', () => {
// A file with no frontmatter that starts on `{expr}`: no html embed exists
// yet, so the entry rule must not pop one.
expect(tokenLanguages(tokenizeAstro('{title}')[0])).toEqual(['astro', 'typescript', 'astro'])
})
it('pops the html embed for a comment that follows markup', () => {
expect(languagesPerLine('<h1>hi</h1>\n<!-- a note -->\n<p>{x}</p>')).toEqual([
['html'],
['astro'],
['html', 'astro', 'typescript', 'astro', 'html']
])
})
it('does not enter typescript for an empty expression', () => {
// `{}` pops html on entry but never pushes typescript; the close must unwind
// only the state, or the tokenizer pops an embed that is not there.
expect(tokenLanguages(tokenizeAstro('<p>{}</p>')[0])).toEqual(['html', 'astro', 'html'])
})
})
describe('astro embedded language attributes', () => {
it('tracks embedded languages from Astro expressions and lang= attributes', () => {
expect(findRuleAction('astroExpressionEnter', 'title }')).toMatchObject({
nextEmbedded: 'typescript',
switchTo: '@astroExpression'
})
expect(findRuleAction('scriptLangValue.javascript', '"ts"')).toMatchObject({
switchTo: '@scriptOpen.typescript'
})
expect(findRuleAction('scriptLangValue.typescript', '"js"')).toMatchObject({
switchTo: '@scriptOpen.javascript'
})
expect(findRuleAction('scriptLangValue.javascript', 'ts')).toMatchObject({
switchTo: '@scriptOpen.typescript'
})
expect(findRuleAction('styleLangValue.css', '"scss"')).toMatchObject({
switchTo: '@styleOpen.scss'
})
expect(findRuleAction('styleLangValue.css', "'sass'")).toMatchObject({
switchTo: '@styleOpen.scss'
})
expect(findRuleAction('styleLangValue.css', 'less')).toMatchObject({
switchTo: '@styleOpen.less'
})
expect(findRuleAction('styleLangValue.scss', '"css"')).toMatchObject({
switchTo: '@styleOpen.css'
})
// Astro `<script>` defaults to JavaScript (unlike Svelte/Vue).
it.each([
['<script>', 'javascript'],
['<script lang="js">', 'javascript'],
['<script lang="ts">', 'typescript'],
["<script lang='typescript'>", 'typescript'],
['<script lang=ts>', 'typescript'],
['<script lang="unknown">', 'javascript']
])('embeds a %s body as %s', (openingTag, embeddedLanguageId) => {
expect(languagesPerLine(`<h1>hi</h1>\n${openingTag}\n a\n</script>`)).toEqual([
['html'],
['astro'],
[embeddedLanguageId],
['astro']
])
})
it.each([
['<style>', 'css'],
['<style lang="css">', 'css'],
['<style lang="scss">', 'scss'],
["<style lang='sass'>", 'scss'],
['<style lang=less>', 'less'],
['<style lang="unknown">', 'css']
])('embeds a %s body as %s', (openingTag, embeddedLanguageId) => {
expect(languagesPerLine(`<h1>hi</h1>\n${openingTag}\n h1 { color: red; }\n</style>`)).toEqual([
['html'],
['astro'],
[embeddedLanguageId],
['astro']
])
})
})
describe('astro root state invariant', () => {
// Structural on purpose: behaviour can only reach the root rules some fixture
// happens to exercise, and a root rule that pops an embed throws on the very
// first character of a file. Guard every root rule, exercised or not.
it('has no root rule that pops an embedded language', () => {
const rootRules = (astroMonarchLanguage.tokenizer as Record<string, unknown[]>).root
const popRules = rootRules.filter(
(rule) =>
Array.isArray(rule) && (rule[1] as { nextEmbedded?: string })?.nextEmbedded === '@pop'
)
expect(popRules).toEqual([])
})
})
@@ -1,4 +1,9 @@
import { describe, expect, it, vi } from 'vitest'
import {
formatTokenizedLines,
tokenizeMonarchDocument,
tokenTypeAt
} from './monarch-tokenizer-test-harness'
import {
JSONL_LANGUAGE_ID,
jsonlLanguageConfiguration,
@@ -6,6 +11,10 @@ import {
registerJsonlLanguage
} from './register-jsonl'
function tokenizeJsonl(source: string) {
return tokenizeMonarchDocument(JSONL_LANGUAGE_ID, jsonlMonarchLanguage, source)
}
function createMonacoMock(existingLanguageIds: string[] = []) {
return {
languages: {
@@ -52,3 +61,65 @@ describe('registerJsonlLanguage', () => {
expect(monaco.languages.setMonarchTokensProvider).not.toHaveBeenCalled()
})
})
describe('jsonl tokenization', () => {
it('tokenizes a representative pair of records', () => {
const fixture = `{"a": 1, "b": "x", "c": true, "d": null}
{"e": [1, -2.5e3], "f": "a\\"b"}`
expect(formatTokenizedLines(tokenizeJsonl(fixture))).toMatchInlineSnapshot(`
[
"{"a": 1, "b": "x", "c": true, "d": null} | 0:delimiter.curly.jsonl@jsonl 1:type.identifier.jsonl@jsonl 4:delimiter.jsonl@jsonl 5:white.jsonl@jsonl 6:number.jsonl@jsonl 7:delimiter.jsonl@jsonl 8:white.jsonl@jsonl 9:type.identifier.jsonl@jsonl 12:delimiter.jsonl@jsonl 13:white.jsonl@jsonl 14:string.jsonl@jsonl 17:delimiter.jsonl@jsonl 18:white.jsonl@jsonl 19:type.identifier.jsonl@jsonl 22:delimiter.jsonl@jsonl 23:white.jsonl@jsonl 24:keyword.jsonl@jsonl 28:delimiter.jsonl@jsonl 29:white.jsonl@jsonl 30:type.identifier.jsonl@jsonl 33:delimiter.jsonl@jsonl 34:white.jsonl@jsonl 35:keyword.jsonl@jsonl 39:delimiter.curly.jsonl@jsonl | embed=none",
"{"e": [1, -2.5e3], "f": "a\\"b"} | 0:delimiter.curly.jsonl@jsonl 1:type.identifier.jsonl@jsonl 4:delimiter.jsonl@jsonl 5:white.jsonl@jsonl 6:delimiter.square.jsonl@jsonl 7:number.jsonl@jsonl 8:delimiter.jsonl@jsonl 9:white.jsonl@jsonl 10:number.jsonl@jsonl 16:delimiter.square.jsonl@jsonl 17:delimiter.jsonl@jsonl 18:white.jsonl@jsonl 19:type.identifier.jsonl@jsonl 22:delimiter.jsonl@jsonl 23:white.jsonl@jsonl 24:string.jsonl@jsonl 26:string.escape.jsonl@jsonl 28:string.jsonl@jsonl 30:delimiter.curly.jsonl@jsonl | embed=none",
]
`)
})
it('colours a property key differently from a string value', () => {
// The `(?=\s*:)` lookahead is the only thing separating the two; a regression
// there makes every key look like a value.
const [line] = tokenizeJsonl('{"key": "value"}')
expect(tokenTypeAt(line, 1)).toBe('type.identifier')
expect(tokenTypeAt(line, 8)).toBe('string')
})
// Regression, found by this suite once it started running the real tokenizer:
// `@string` used to survive the line break, so one truncated record rendered
// every record after it as a single string.
it.each([
['mid-string', '{"a": "truncated here'],
['mid-escape', '{"a": "truncated\\'],
['on a trailing backslash', '{"a": "x\\']
])('does not let a record truncated %s poison the next one', (_name, truncated) => {
const [, second] = tokenizeJsonl(`${truncated}\n{"b": 1}`)
expect(tokenTypeAt(second, 1)).toBe('type.identifier')
expect(tokenTypeAt(second, 6)).toBe('number')
})
it('marks the unterminated remainder of a truncated record', () => {
const [first] = tokenizeJsonl('{"a": "truncated here')
expect(tokenTypeAt(first, 6)).toBe('string.invalid')
})
it.each([
['escaped quote', '{"m": "he said \\"hi\\""}', 15],
['escaped backslash', '{"m": "C:\\\\Users"}', 10],
['unicode escape', '{"m": "\\u00e9"}', 7],
['newline escape', '{"m": "a\\nb"}', 8]
])('still highlights an %s inside a well-formed record', (_name, record, escapeOffset) => {
// The fix must not cost escape fidelity on the common case: a candidate that
// collapsed the string into one regex lost every one of these.
const [line] = tokenizeJsonl(record)
expect(tokenTypeAt(line, escapeOffset)).toBe('string.escape')
})
it('flags an invalid escape inside a well-formed record', () => {
const [line] = tokenizeJsonl('{"m": "a\\qb"}')
expect(tokenTypeAt(line, 8)).toBe('string.escape.invalid')
})
})
@@ -34,7 +34,14 @@ export const jsonlMonarchLanguage: Monaco.languages.IMonarchLanguage = {
{ include: '@whitespace' },
// Property key vs string value are both quoted; color keys distinctly.
[/"(?:[^"\\]|\\.)*"(?=\s*:)/, 'type.identifier'],
[/"/, 'string', '@string'],
// Why the lookahead: each JSONL line is an independent value, but Monarch
// state survives the line break. Pushing `@string` unconditionally meant
// one truncated record (a normal way for a log to end) left every later
// record inside the string state, rendering the rest of the file as one
// string. Only enter the escape-aware state once a closing quote is known
// to be on this line; an unterminated remainder is consumed below instead.
[/"(?=(?:[^"\\]|\\.)*")/, 'string', '@string'],
[/"(?:[^"\\]|\\.)*\\?$/, 'string.invalid'],
[/[{}[\]]/, '@brackets'],
[/-?(?:0|[1-9]\d*)(?:\.\d+)?(?:[eE][+-]?\d+)?/, 'number'],
[/\b(?:true|false)\b/, 'keyword'],
@@ -1,124 +1,33 @@
import { describe, expect, it, vi } from 'vitest'
import {
endEmbeddedLanguages,
formatTokenizedLines,
tokenizeMonarchDocument,
tokenLanguages,
tokenLanguagesPerLine,
tokenTypeAt
} from './monarch-tokenizer-test-harness'
import {
registerSvelteLanguage,
svelteLanguageConfiguration,
svelteMonarchLanguage
} from './register-svelte'
type MonarchAction = {
next?: string
nextEmbedded?: string
switchTo?: string
}
type MonarchRule = [RegExp, string | MonarchAction, string?] | { include: string }
function normalizeState(nextState: string): string {
return nextState.startsWith('@') ? nextState.slice(1) : nextState
// These tests drive the real `MonarchTokenizer`. Walking the rule table instead
// let a grammar that threw on 100% of Svelte inputs — including `<p>a {b}</p>` —
// ship with a green suite, because a broken grammar still has a valid table.
function tokenizeSvelte(source: string) {
return tokenizeMonarchDocument('svelte', svelteMonarchLanguage, source)
}
function isRuleEntry(rule: MonarchRule): rule is [RegExp, string | MonarchAction, string?] {
return Array.isArray(rule)
}
function getRuleAction(rule: [RegExp, string | MonarchAction, string?]): MonarchAction | undefined {
const [, action, nextStateShortcut] = rule
return typeof action === 'object'
? action
: nextStateShortcut
? { next: nextStateShortcut }
: undefined
}
function findRuleAction(
state: string,
source: string,
{ embedPopOnly = false }: { embedPopOnly?: boolean } = {}
): MonarchAction | undefined {
const tokenizer = svelteMonarchLanguage.tokenizer as Record<string, MonarchRule[]>
const stateRules = tokenizer[state] ?? tokenizer[state.split('.')[0]]
// When the html embed is active inside `markup`, Monaco's
// `_findLeavingNestedLanguageOffset` only consults rules whose action has
// `nextEmbedded: '@pop'` — the zero-width `@rematch` catch-all is
// skipped. Mirror that when callers want to verify the "structural rule
// pops the embed" path.
const candidateRules = embedPopOnly
? stateRules.filter((rule) => {
if (!isRuleEntry(rule)) {
return false
}
return getRuleAction(rule)?.nextEmbedded === '@pop'
})
: stateRules
const matchedRule = candidateRules.find((rule) => {
if (!isRuleEntry(rule)) {
return false
}
const [regexp] = rule
regexp.lastIndex = 0
const match = regexp.exec(source)
return match !== null && match.index === 0
})
return matchedRule && isRuleEntry(matchedRule) ? getRuleAction(matchedRule) : undefined
}
function collectFixtureRuleActions(source: string): string[] {
const ruleActions: string[] = []
const tokenizer = svelteMonarchLanguage.tokenizer as Record<string, MonarchRule[]>
const lines = source.split('\n')
const checks: { line: number; state: string; pattern: string }[] = [
{ line: 1, state: 'root', pattern: '<script' },
{ line: 1, state: 'scriptOpen.typescript', pattern: '>' },
{ line: 4, state: 'scriptBody.typescript', pattern: '</script>' },
// After </script> pops back to root and the next non-structural character
// switches root -> markup with the html embed active.
{ line: 6, state: 'root', pattern: '' },
{ line: 7, state: 'markup', pattern: '{#if' },
{ line: 7, state: 'svelteBlockExpression', pattern: '}' },
{ line: 8, state: 'markup', pattern: '{' },
{ line: 8, state: 'svelteExpression', pattern: '}' },
{ line: 9, state: 'markup', pattern: '{:else' },
{ line: 9, state: 'svelteBlockExpressionEnter', pattern: '}' },
{ line: 11, state: 'markup', pattern: '{/if}' },
{ line: 13, state: 'markup', pattern: '{' },
{ line: 13, state: 'svelteExpression', pattern: '}' },
{ line: 14, state: 'markup', pattern: '{@html' },
{ line: 14, state: 'svelteExpression', pattern: '}' },
{ line: 16, state: 'markup', pattern: '<style' },
{ line: 16, state: 'styleOpen.css', pattern: '>' },
{ line: 18, state: 'styleBody.css', pattern: '</style>' }
]
checks.forEach((check) => {
const line = lines.at(check.line - 1) ?? ''
const stateRules = tokenizer[check.state] ?? tokenizer[check.state.split('.')[0]]
const matchedRule = stateRules.find((rule) => {
if (!isRuleEntry(rule)) {
return false
}
const [regexp] = rule
regexp.lastIndex = 0
const match = regexp.exec(line)
return match !== null && match[0] === check.pattern
})
if (!matchedRule || !isRuleEntry(matchedRule)) {
return
}
const actionObject = getRuleAction(matchedRule)
const nextState = actionObject?.next ? normalizeState(actionObject.next) : '-'
const nextEmbedded = actionObject?.nextEmbedded ?? '-'
const switchTo = actionObject?.switchTo ? normalizeState(actionObject.switchTo) : '-'
ruleActions.push(
`${check.line}:${check.state}:${check.pattern || '<html>'} -> next=${nextState}, embedded=${nextEmbedded}, switch=${switchTo}`
)
})
return ruleActions
/** Which languages actually cover each line — a dropped embed shows up as `svelte`. */
function languagesPerLine(source: string): string[][] {
return tokenLanguagesPerLine(tokenizeSvelte(source))
}
describe('registerSvelteLanguage registration', () => {
// Structural by necessity: this covers the registration call itself
// (ids, extensions, idempotence), which no amount of tokenizing can observe.
it('registers the svelte language, Monarch tokenizer, and configuration once', () => {
const languages: { id: string }[] = [{ id: 'typescript' }]
const register = vi.fn((entry: { id: string }) => {
@@ -152,8 +61,8 @@ describe('registerSvelteLanguage registration', () => {
})
})
describe('svelte tokenizer transitions', () => {
it('captures Svelte tokenizer transitions for a representative SFC fixture', () => {
describe('svelte tokenization', () => {
it('tokenizes a representative SFC', () => {
const fixture = `<script lang="ts">
let count = 0
$: doubled = count * 2
@@ -173,103 +82,159 @@ describe('svelte tokenizer transitions', () => {
h1 { color: rebeccapurple; }
</style>`
const ruleActions = collectFixtureRuleActions(fixture)
expect(ruleActions).toMatchInlineSnapshot(`
expect(formatTokenizedLines(tokenizeSvelte(fixture))).toMatchInlineSnapshot(`
[
"1:root:<script -> next=-, embedded=-, switch=scriptOpen.typescript",
"1:scriptOpen.typescript:> -> next=-, embedded=$S2, switch=scriptBody.$S2",
"4:scriptBody.typescript:</script> -> next=-, embedded=@pop, switch=markupReenter",
"6:root:<html> -> next=-, embedded=html, switch=markup",
"7:markup:{#if -> next=-, embedded=@pop, switch=svelteBlockExpressionEnter",
"7:svelteBlockExpression:} -> next=-, embedded=@pop, switch=markupReenter",
"8:markup:{ -> next=-, embedded=@pop, switch=svelteExpressionEnter",
"8:svelteExpression:} -> next=-, embedded=@pop, switch=markupReenter",
"9:markup:{:else -> next=-, embedded=@pop, switch=svelteBlockExpressionEnter",
"9:svelteBlockExpressionEnter:} -> next=-, embedded=-, switch=markupReenter",
"11:markup:{/if} -> next=-, embedded=-, switch=-",
"13:markup:{ -> next=-, embedded=@pop, switch=svelteExpressionEnter",
"13:svelteExpression:} -> next=-, embedded=@pop, switch=markupReenter",
"14:markup:{@html -> next=-, embedded=@pop, switch=svelteExpressionEnter",
"14:svelteExpression:} -> next=-, embedded=@pop, switch=markupReenter",
"16:markup:<style -> next=-, embedded=@pop, switch=styleOpen.css",
"16:styleOpen.css:> -> next=-, embedded=$S2, switch=styleBody.$S2",
"18:styleBody.css:</style> -> next=-, embedded=@pop, switch=markupReenter",
"<script lang="ts"> | 0:tag.svelte@svelte 7:white.svelte@svelte 8:attribute.name.svelte@svelte 12:delimiter.svelte@svelte 13:attribute.value.svelte@svelte 17:tag.svelte@svelte | embed=typescript",
" let count = 0 | 0:-@typescript | embed=typescript",
" $: doubled = count * 2 | 0:-@typescript | embed=typescript",
"</script> | 0:tag.svelte@svelte | embed=none",
" | | embed=html",
"<h1>Counter</h1> | 0:-@html | embed=html",
"{#if count > 0} | 0:keyword.control.svelte@svelte 4:-@typescript 14:keyword.control.svelte@svelte | embed=none",
" <p>{count} clicked</p> | 0:-@html 5:delimiter.curly.svelte@svelte 6:-@typescript 11:delimiter.curly.svelte@svelte 12:-@html | embed=html",
"{:else} | 0:keyword.control.svelte@svelte | embed=none",
" <p>not yet</p> | 0:-@html | embed=html",
"{/if} | 0:-@html | embed=html",
" | 0:-@html | embed=html",
"<button on:click={increment}>{count}</button> | 0:-@html 17:delimiter.curly.svelte@svelte 18:-@typescript 27:delimiter.curly.svelte@svelte 28:-@html 29:delimiter.curly.svelte@svelte 30:-@typescript 35:delimiter.curly.svelte@svelte 36:-@html | embed=html",
"{@html '<em>raw</em>'} | 0:keyword.control.svelte@svelte 6:-@typescript 21:delimiter.curly.svelte@svelte | embed=none",
" | | embed=html",
"<style> | 0:tag.svelte@svelte | embed=css",
" h1 { color: rebeccapurple; } | 0:-@css | embed=css",
"</style> | 0:tag.svelte@svelte | embed=none",
]
`)
})
})
describe('svelte tokenizer regressions', () => {
// Regression: when a Svelte file starts with `{#if}`, `{name}`, or `{@html}`,
// no html embed is active yet. Earlier drafts unconditionally emitted
// `nextEmbedded: '@pop'` from root, which Monaco rejects with
// "cannot pop embedded language if not inside one". The fix splits the
// entry-only `root` state from the html-embedded `markup` state.
it('does not pop a non-existent embed when a file starts with a Svelte block', () => {
const action = findRuleAction('root', '{#if foo}')
expect(action).toMatchObject({ switchTo: '@svelteBlockExpressionEnter' })
expect(action?.nextEmbedded).toBeUndefined()
// Regression (the field failure): the first interpolation of a file threw
// "cannot pop embedded language if not inside one" — every Svelte file with a
// `{}` in it, which is essentially all of them.
it('highlights every interpolation of a markup line', () => {
const [line] = tokenizeSvelte('<p>a {first} b {second} c</p>')
expect(tokenLanguages(line)).toEqual([
'html',
'svelte',
'typescript',
'svelte',
'html',
'svelte',
'typescript',
'svelte',
'html'
])
})
it('starts the html embed and switches to markup when markup begins', () => {
expect(findRuleAction('root', '<h1>Counter</h1>')).toMatchObject({
switchTo: '@markup',
nextEmbedded: 'html'
})
it('opens a file on a Svelte block without popping a missing embed', () => {
// No html embed exists yet at file start, so the block's entry rule must not
// pop one — Monarch throws outright if it does.
const [line] = tokenizeSvelte('{#if count > 0}')
expect(tokenTypeAt(line, 0)).toBe('keyword.control')
expect(tokenLanguages(line)).toEqual(['svelte', 'typescript', 'svelte'])
})
// Regression: while the html embed is active, only parent rules whose action
// pops the embed are consulted before delegating to html. The first draft
// omitted `nextEmbedded: '@pop'` from `<script>` / `<style>` / `<!--` rules,
// so a trailing `<style lang="scss">` after markup never reached the
// lang-switching code. The `markup` state wires the embed pop in.
it('pops the html embed when later script/style/comment markers appear', () => {
expect(findRuleAction('markup', '<script lang="ts">', { embedPopOnly: true })).toMatchObject({
switchTo: '@scriptOpen.typescript',
nextEmbedded: '@pop'
})
expect(findRuleAction('markup', '<style lang="scss">', { embedPopOnly: true })).toMatchObject({
switchTo: '@styleOpen.css',
nextEmbedded: '@pop'
})
expect(findRuleAction('markup', '<!-- comment -->', { embedPopOnly: true })).toMatchObject({
switchTo: '@comment',
nextEmbedded: '@pop'
})
it('enters the html embed when markup begins', () => {
expect(endEmbeddedLanguages(tokenizeSvelte('<h1>Counter</h1>'))).toEqual(['html'])
})
it('pops the html embed for a comment that follows markup', () => {
// Once html is active Monaco only consults parent rules that pop the embed,
// so `<!--` after markup is unreachable without one.
expect(languagesPerLine('<h1>hi</h1>\n<!-- a note -->\n<p>after</p>')).toEqual([
['html'],
['svelte'],
['html']
])
})
it('keeps markup embedded across a Svelte block', () => {
expect(
languagesPerLine(
'<h1>hi</h1>\n{#if ok}\n <p>yes</p>\n{:else}\n <p>no</p>\n{/if}\n<p>done</p>'
)
).toEqual([
['html'],
['svelte', 'typescript', 'svelte'],
['html'],
['svelte'],
['html'],
['html'],
['html']
])
})
it('keeps markup highlighted across a whole multi-line file', () => {
// A grammar that drops the embed leaves plain `svelte` on these rows, which
// is the silently-unhighlighted failure a rule-table walk cannot see.
expect(
languagesPerLine(
'<h1>hi</h1>\n<p>{a}</p>\n<!-- note -->\n<p>{b}</p>\n<button on:click={go}>x</button>'
)
).toEqual([
['html'],
['html', 'svelte', 'typescript', 'svelte', 'html'],
['svelte'],
['html', 'svelte', 'typescript', 'svelte', 'html'],
['html', 'svelte', 'typescript', 'svelte', 'html']
])
})
it('does not enter typescript for an empty expression', () => {
// `{}` pops html on entry but never pushes typescript; the close must unwind
// only the state, or the tokenizer pops an embed that is not there.
expect(tokenLanguages(tokenizeSvelte('<p>{}</p>')[0])).toEqual(['html', 'svelte', 'html'])
})
})
describe('svelte embedded language attributes', () => {
it('tracks embedded languages from Svelte block attributes and expressions', () => {
expect(findRuleAction('svelteExpressionEnter', 'count }')).toMatchObject({
nextEmbedded: 'typescript',
switchTo: '@svelteExpression'
})
expect(findRuleAction('svelteBlockExpressionEnter', 'count > 0}')).toMatchObject({
nextEmbedded: 'typescript',
switchTo: '@svelteBlockExpression'
})
expect(findRuleAction('scriptLangValue.typescript', '"js"')).toMatchObject({
switchTo: '@scriptOpen.javascript'
})
expect(findRuleAction('scriptLangValue.javascript', '"ts"')).toMatchObject({
switchTo: '@scriptOpen.typescript'
})
expect(findRuleAction('scriptLangValue.typescript', 'js')).toMatchObject({
switchTo: '@scriptOpen.javascript'
})
expect(findRuleAction('styleLangValue.css', '"scss"')).toMatchObject({
switchTo: '@styleOpen.scss'
})
expect(findRuleAction('styleLangValue.css', 'less')).toMatchObject({
switchTo: '@styleOpen.less'
})
expect(findRuleAction('styleLangValue.css', "'sass'")).toMatchObject({
switchTo: '@styleOpen.scss'
})
expect(findRuleAction('styleLangValue.scss', '"css"')).toMatchObject({
switchTo: '@styleOpen.css'
})
// `lang=` picks the embedded language for the block body. The assertion is on
// the body row, which is the region a reader sees highlighted (or not).
it.each([
['<script>', 'typescript'],
['<script lang="ts">', 'typescript'],
['<script lang="typescript">', 'typescript'],
['<script lang="js">', 'javascript'],
["<script lang='javascript'>", 'javascript'],
['<script lang=js>', 'javascript'],
['<script lang="unknown">', 'typescript']
])('embeds a %s body as %s', (openingTag, embeddedLanguageId) => {
expect(languagesPerLine(`<h1>hi</h1>\n${openingTag}\n a\n</script>`)).toEqual([
['html'],
['svelte'],
[embeddedLanguageId],
['svelte']
])
})
it.each([
['<style>', 'css'],
['<style lang="css">', 'css'],
['<style lang="scss">', 'scss'],
["<style lang='sass'>", 'scss'],
['<style lang=less>', 'less'],
['<style lang="unknown">', 'css']
])('embeds a %s body as %s', (openingTag, embeddedLanguageId) => {
expect(languagesPerLine(`<h1>hi</h1>\n${openingTag}\n h1 { color: red; }\n</style>`)).toEqual([
['html'],
['svelte'],
[embeddedLanguageId],
['svelte']
])
})
})
describe('svelte root state invariant', () => {
// Structural on purpose: behaviour can only reach the root rules some fixture
// happens to exercise, and a root rule that pops an embed throws on the very
// first character of a file. Guard every root rule, exercised or not.
it('has no root rule that pops an embedded language', () => {
const rootRules = (svelteMonarchLanguage.tokenizer as Record<string, unknown[]>).root
const popRules = rootRules.filter(
(rule) =>
Array.isArray(rule) && (rule[1] as { nextEmbedded?: string })?.nextEmbedded === '@pop'
)
expect(popRules).toEqual([])
})
})
@@ -1,110 +1,28 @@
import { describe, expect, it, vi } from 'vitest'
import {
endEmbeddedLanguages,
formatTokenizedLines,
tokenizeMonarchDocument,
tokenLanguages,
tokenLanguagesPerLine
} from './monarch-tokenizer-test-harness'
import { registerVueLanguage, vueLanguageConfiguration, vueMonarchLanguage } from './register-vue'
type MonarchAction = {
next?: string
nextEmbedded?: string
switchTo?: string
}
type MonarchRule = [RegExp, string | MonarchAction, string?] | { include: string }
function normalizeState(nextState: string): string {
return nextState.startsWith('@') ? nextState.slice(1) : nextState
// Driven through the real `MonarchTokenizer`: a rule-table walk cannot tell a
// working grammar from one that throws on every `{{ }}`, which is how broken
// Vue highlighting shipped green.
function tokenizeVue(source: string) {
return tokenizeMonarchDocument('vue', vueMonarchLanguage, source)
}
function isRuleEntry(rule: MonarchRule): rule is [RegExp, string | MonarchAction, string?] {
return Array.isArray(rule)
/** Which languages actually cover each line — a dropped embed shows up as `vue`. */
function languagesPerLine(source: string): string[][] {
return tokenLanguagesPerLine(tokenizeVue(source))
}
function getRuleAction(rule: [RegExp, string | MonarchAction, string?]): MonarchAction | undefined {
const [, action, nextStateShortcut] = rule
return typeof action === 'object'
? action
: nextStateShortcut
? { next: nextStateShortcut }
: undefined
}
function findRuleAction(state: string, source: string): MonarchAction | undefined {
const tokenizer = vueMonarchLanguage.tokenizer as Record<string, MonarchRule[]>
const stateRules = tokenizer[state] ?? tokenizer[state.split('.')[0]]
const matchedRule = stateRules.find((rule) => {
if (!isRuleEntry(rule)) {
return false
}
const [regexp] = rule
regexp.lastIndex = 0
const match = regexp.exec(source)
return match !== null && match.index === 0
})
return matchedRule && isRuleEntry(matchedRule) ? getRuleAction(matchedRule) : undefined
}
function collectFixtureRuleActions(source: string): {
line: number
state: string
matched: string
nextState?: string
nextEmbedded?: string
switchTo?: string
}[] {
const ruleActions: {
line: number
state: string
matched: string
nextState?: string
nextEmbedded?: string
switchTo?: string
}[] = []
const tokenizer = vueMonarchLanguage.tokenizer as Record<string, MonarchRule[]>
const lines = source.split('\n')
const checks: { line: number; state: string; pattern: string }[] = [
{ line: 1, state: 'root', pattern: '<template' },
{ line: 1, state: 'templateOpen', pattern: '>' },
{ line: 2, state: 'templateBody', pattern: '{{' },
{ line: 2, state: 'templateExpression', pattern: '}}' },
{ line: 3, state: 'templateBody', pattern: '</template>' },
{ line: 5, state: 'root', pattern: '<script' },
{ line: 5, state: 'scriptOpen.typescript', pattern: '>' },
{ line: 7, state: 'scriptBody.typescript', pattern: '</script>' },
{ line: 9, state: 'root', pattern: '<style' },
{ line: 9, state: 'styleOpen.css', pattern: '>' },
{ line: 11, state: 'styleBody.css', pattern: '</style>' }
]
checks.forEach((check) => {
const line = lines.at(check.line - 1) ?? ''
const stateRules = tokenizer[check.state] ?? tokenizer[check.state.split('.')[0]]
const matchedRule = stateRules.find((rule) => {
if (!isRuleEntry(rule)) {
return false
}
const [regexp] = rule
regexp.lastIndex = 0
const match = regexp.exec(line)
return match !== null && match[0] === check.pattern
})
if (!matchedRule || !isRuleEntry(matchedRule)) {
return
}
const actionObject = getRuleAction(matchedRule)
ruleActions.push({
line: check.line,
state: check.state,
matched: check.pattern,
nextState: actionObject?.next ? normalizeState(actionObject.next) : undefined,
nextEmbedded: actionObject?.nextEmbedded,
switchTo: actionObject?.switchTo ? normalizeState(actionObject.switchTo) : undefined
})
})
return ruleActions
}
describe('registerVueLanguage', () => {
describe('registerVueLanguage registration', () => {
// Structural by necessity: covers the registration call itself (ids,
// extensions, idempotence), which tokenizing cannot observe.
it('registers the vue language, Monarch tokenizer, and configuration once', () => {
const languages: { id: string }[] = [{ id: 'typescript' }]
const register = vi.fn((entry: { id: string }) => {
@@ -136,8 +54,10 @@ describe('registerVueLanguage', () => {
expect(setLanguageConfiguration).toHaveBeenCalledTimes(1)
expect(setLanguageConfiguration).toHaveBeenCalledWith('vue', vueLanguageConfiguration)
})
})
it('captures Vue tokenizer transitions for a representative SFC fixture', () => {
describe('vue tokenization', () => {
it('tokenizes a representative SFC', () => {
const fixture = `<template>
<p>{{ message.toUpperCase() }}</p>
</template>
@@ -150,121 +70,110 @@ const message = 'hello'
p { color: rebeccapurple; }
</style>`
const ruleActions = collectFixtureRuleActions(fixture)
expect(ruleActions).toMatchInlineSnapshot(`
expect(formatTokenizedLines(tokenizeVue(fixture))).toMatchInlineSnapshot(`
[
{
"line": 1,
"matched": "<template",
"nextEmbedded": undefined,
"nextState": "templateOpen",
"state": "root",
"switchTo": undefined,
},
{
"line": 1,
"matched": ">",
"nextEmbedded": "html",
"nextState": undefined,
"state": "templateOpen",
"switchTo": "templateBody",
},
{
"line": 2,
"matched": "{{",
"nextEmbedded": "@pop",
"nextState": undefined,
"state": "templateBody",
"switchTo": "templateExpressionEnter",
},
{
"line": 2,
"matched": "}}",
"nextEmbedded": "@pop",
"nextState": undefined,
"state": "templateExpression",
"switchTo": "templateBodyReenter",
},
{
"line": 3,
"matched": "</template>",
"nextEmbedded": "@pop",
"nextState": "pop",
"state": "templateBody",
"switchTo": undefined,
},
{
"line": 5,
"matched": "<script",
"nextEmbedded": undefined,
"nextState": "scriptOpen.typescript",
"state": "root",
"switchTo": undefined,
},
{
"line": 5,
"matched": ">",
"nextEmbedded": "$S2",
"nextState": undefined,
"state": "scriptOpen.typescript",
"switchTo": "scriptBody.$S2",
},
{
"line": 7,
"matched": "</script>",
"nextEmbedded": "@pop",
"nextState": "pop",
"state": "scriptBody.typescript",
"switchTo": undefined,
},
{
"line": 9,
"matched": "<style",
"nextEmbedded": undefined,
"nextState": "styleOpen.css",
"state": "root",
"switchTo": undefined,
},
{
"line": 9,
"matched": ">",
"nextEmbedded": "$S2",
"nextState": undefined,
"state": "styleOpen.css",
"switchTo": "styleBody.$S2",
},
{
"line": 11,
"matched": "</style>",
"nextEmbedded": "@pop",
"nextState": "pop",
"state": "styleBody.css",
"switchTo": undefined,
},
"<template> | 0:tag.vue@vue | embed=html",
" <p>{{ message.toUpperCase() }}</p> | 0:-@html 5:delimiter.curly.vue@vue 7:-@typescript 30:delimiter.curly.vue@vue 32:-@html | embed=html",
"</template> | 0:tag.vue@vue | embed=none",
" | | embed=none",
"<script setup lang="ts"> | 0:tag.vue@vue 7:white.vue@vue 8:attribute.name.vue@vue 13:white.vue@vue 14:attribute.name.vue@vue 18:delimiter.vue@vue 19:attribute.value.vue@vue 23:tag.vue@vue | embed=typescript",
"const message = 'hello' | 0:-@typescript | embed=typescript",
"</script> | 0:tag.vue@vue | embed=none",
" | | embed=none",
"<style scoped> | 0:tag.vue@vue 6:white.vue@vue 7:attribute.name.vue@vue 13:tag.vue@vue | embed=css",
"p { color: rebeccapurple; } | 0:-@css | embed=css",
"</style> | 0:tag.vue@vue | embed=none",
]
`)
})
it('tracks embedded languages from Vue block attributes', () => {
expect(findRuleAction('templateExpressionEnter', 'message }}')).toMatchObject({
nextEmbedded: 'typescript',
switchTo: '@templateExpression'
})
expect(findRuleAction('scriptLangValue.typescript', '"js"')).toMatchObject({
switchTo: '@scriptOpen.javascript'
})
expect(findRuleAction('scriptLangValue.javascript', '"ts"')).toMatchObject({
switchTo: '@scriptOpen.typescript'
})
expect(findRuleAction('scriptLangValue.typescript', 'js')).toMatchObject({
switchTo: '@scriptOpen.javascript'
})
expect(findRuleAction('styleLangValue.css', '"scss"')).toMatchObject({
switchTo: '@styleOpen.scss'
})
expect(findRuleAction('styleLangValue.css', 'less')).toMatchObject({
switchTo: '@styleOpen.less'
})
// Regression: every `{{ }}` threw "cannot pop embedded language if not inside
// one" once the template body lost its html embed.
it('highlights every interpolation in a template line', () => {
const [, line] = tokenizeVue('<template>\n <p>{{ a }} and {{ b }}</p>\n</template>')
expect(tokenLanguages(line)).toEqual([
'html',
'vue',
'typescript',
'vue',
'html',
'vue',
'typescript',
'vue',
'html'
])
})
it('embeds the template body as html', () => {
expect(endEmbeddedLanguages(tokenizeVue('<template>\n <p>x</p>\n</template>'))).toEqual([
'html',
'html',
null
])
})
it('keeps the template embedded across a comment before it', () => {
expect(languagesPerLine('<!-- a note -->\n<template>\n <p>x</p>\n</template>')).toEqual([
['vue'],
['vue'],
['html'],
['vue']
])
})
it('does not enter typescript for an empty interpolation', () => {
// `{{}}` pops html on entry but never pushes typescript; the close must
// unwind only the state, or it pops an embed that is not there.
const [, line] = tokenizeVue('<template>\n <p>{{}}</p>\n</template>')
expect(tokenLanguages(line)).toEqual(['html', 'vue', 'html'])
})
})
describe('vue embedded language attributes', () => {
it.each([
['<script>', 'typescript'],
['<script lang="ts">', 'typescript'],
['<script setup lang="typescript">', 'typescript'],
['<script lang="js">', 'javascript'],
['<script setup lang=js>', 'javascript'],
['<script lang="unknown">', 'typescript']
])('embeds a %s body as %s', (openingTag, embeddedLanguageId) => {
expect(languagesPerLine(`${openingTag}\n a\n</script>`)).toEqual([
['vue'],
[embeddedLanguageId],
['vue']
])
})
it.each([
['<style>', 'css'],
['<style scoped>', 'css'],
['<style lang="scss">', 'scss'],
["<style lang='sass'>", 'scss'],
['<style lang=less>', 'less'],
['<style lang="unknown">', 'css']
])('embeds a %s body as %s', (openingTag, embeddedLanguageId) => {
expect(languagesPerLine(`${openingTag}\n h1 { color: red; }\n</style>`)).toEqual([
['vue'],
[embeddedLanguageId],
['vue']
])
})
})
describe('vue root state invariant', () => {
// Structural on purpose: behaviour can only reach the root rules some fixture
// happens to exercise, and a root rule that pops an embed throws on the very
// first character of a file. Guard every root rule, exercised or not.
it('has no root rule that pops an embedded language', () => {
const rootRules = (vueMonarchLanguage.tokenizer as Record<string, unknown[]>).root
const popRules = rootRules.filter(
(rule) =>
Array.isArray(rule) && (rule[1] as { nextEmbedded?: string })?.nextEmbedded === '@pop'
)
expect(popRules).toEqual([])
})
})