Reduce redundant test coverage and unnecessary CI waits (#25806)

This commit is contained in:
Neil
2026-10-05 23:19:18 -07:00
committed by GitHub
parent 4e64fa9940
commit f9c8cd4fc3
14 changed files with 207 additions and 468 deletions
@@ -1,5 +1,5 @@
/**
* The mobile-view frame budget, held against Chromium's own JPEG encoder across the viewport range.
* The mobile-view frame budget, held against Chromium's own JPEG encoder at representative phone, tablet and wide viewports.
*
* `WORST_CASE_JPEG_BYTES_PER_PIXEL` is the one number the budget cannot derive, and every other
* check of it is circular: a case that encodes `noise(area * theConstant)` is measuring a byte
@@ -43,9 +43,7 @@ async function loadSweepModules() {
])
return {
budgetedMobileViewDeviceScaleFactor: request.budgetedMobileViewDeviceScaleFactor,
mobileBrowserFrameAreaBudget: request.mobileBrowserFrameAreaBudget,
WORST_CASE_JPEG_BYTES_PER_PIXEL: request.WORST_CASE_JPEG_BYTES_PER_PIXEL,
MOBILE_VIEW_DEVICE_SCALE_FACTOR: parameters.MOBILE_VIEW_DEVICE_SCALE_FACTOR,
BROWSER_FRAME_QUALITY: parameters.BROWSER_FRAME_QUALITY,
BRIDGE_MAX_MESSAGE_BYTES: caps.BRIDGE_MAX_MESSAGE_BYTES,
utf8ByteLength: caps.utf8ByteLength,
@@ -65,15 +63,23 @@ function sweep() {
return loaded
}
/** The viewport range the pane is mounted in, phone through tablet, in CSS pixels. */
const VIEWPORT_WIDTHS = [320, 360, 390, 393, 412, 430, 480, 600, 768, 834, 1024, 1280, 1400]
const VIEWPORT_HEIGHTS = [480, 640, 712, 720, 800, 896, 932, 1024, 1180, 1366, 1600]
type Viewport = { width: number; height: number }
const VIEWPORTS: Viewport[] = VIEWPORT_WIDTHS.flatMap((width) =>
VIEWPORT_HEIGHTS.map((height) => ({ width, height }))
)
// Sample aspect ratios and budget pressure without replaying one encoder contract 111 times.
const VIEWPORTS: Viewport[] = [
{ width: 320, height: 480 },
{ width: 360, height: 640 },
{ width: 390, height: 712 },
{ width: 393, height: 720 },
{ width: 430, height: 932 },
{ width: 600, height: 800 },
{ width: 768, height: 1024 },
{ width: 1024, height: 768 },
{ width: 320, height: 1600 },
{ width: 1400, height: 480 },
{ width: 834, height: 896 },
{ width: 1280, height: 640 }
]
let browser: Browser | null = null
let page: Page | null = null
@@ -363,22 +369,12 @@ function budgetedFrame(viewport: Viewport) {
}
}
/**
* The viewports the budget can actually fit, which are the ones it makes a promise about.
*
* Below a scale of one the module stops: asking for fewer device pixels than CSS pixels is a
* blurry frame rather than a working one, so a viewport too large for the cap keeps scale 1 and
* the frame that does not fit is C6 ruling 1's to drop. Split here so the promise and the
* exception are both asserted rather than averaged.
*/
const withinBudget = (viewport: Viewport) => budgetedFrame(viewport).scale > 1
describeSweep('the frame budget across the viewport range', () => {
it('keeps every viewport it budgets for inside one bridge message', async () => {
describeSweep('the frame budget at representative viewport sizes', () => {
it('keeps representative budgeted viewports inside one bridge message', async () => {
const overCap: string[] = []
let worstBytesPerPixel = 0
let bestBytesPerPixel = 1
for (const viewport of VIEWPORTS.filter(withinBudget)) {
for (const viewport of VIEWPORTS) {
const frame = budgetedFrame(viewport)
const imageBytes = await screencastNoiseJpegBytes(
frame,
@@ -411,15 +407,7 @@ describeSweep('the frame budget across the viewport range', () => {
}, 300_000)
it('does not budget below one device pixel per CSS pixel, and the shell drops what will not fit', async () => {
// The exception the split above names. These are real: a 1400x1180 viewport posts 1.2 MB.
const tooLarge = VIEWPORTS.filter((viewport) => !withinBudget(viewport))
// The 32 of the 143 the budget leaves at scale 1, a fixed number because the set is fixed.
expect(tooLarge.length).toBe(32)
const largest = tooLarge.reduce((left, right) =>
left.width * left.height > right.width * right.height ? left : right
)
const frame = budgetedFrame(largest)
const frame = budgetedFrame({ width: 1400, height: 1600 })
expect(frame.scale).toBe(1)
const imageBytes = await screencastNoiseJpegBytes(frame, 1)
expect(postThroughShell(new Uint8Array(imageBytes), frame)).toBeNull()
@@ -449,42 +437,4 @@ describeSweep('the frame budget across the viewport range', () => {
await context.close()
}
}, 120_000)
it('reads the frames this capture painted, never one left over from the last', () => {
// The four shapes measured on this rig at 20x CPU throttling, all arriving after the raster
// barrier: the black canvas the resize left, a full frame of the previous and larger viewport,
// and this capture's own two. Only the last two are this capture's, and the gap between the
// stale stamps and the paint was never under 86 ms.
const frames = [
{ bytes: 13_483, stamp: 914 },
{ bytes: 447_491, stamp: 939 },
{ bytes: 997_489, stamp: 1005 },
{ bytes: 997_489, stamp: 1024 }
]
expect(framesCarryingTheNoise(frames, 0, 1000).map((one) => one.bytes)).toEqual([
997_489, 997_489
])
// A frame the browser sent no capture time for is not admissible either: it cannot be told from
// the stale ones, and guessing it fresh is the understatement the gate exists to refuse.
expect(framesCarryingTheNoise([{ bytes: 997_489, stamp: null }], 0, 1000)).toEqual([])
// And the arrivals before the raster barrier stay out, which is the other half of the reading.
expect(framesCarryingTheNoise(frames, 3, 1000).map((one) => one.bytes)).toEqual([997_489])
})
it('never asks for more density than native, anywhere in the range', () => {
for (const viewport of VIEWPORTS) {
expect(budgetedFrame(viewport).scale).toBeLessThanOrEqual(
sweep().MOBILE_VIEW_DEVICE_SCALE_FACTOR
)
}
})
it('sweeps a range wide enough to contain the phones the pane runs on', () => {
// The set is fixed, so this is what says it still covers the case the old constant missed.
expect(VIEWPORTS).toContainEqual({ width: 390, height: 712 })
expect(VIEWPORTS).toContainEqual({ width: 393, height: 720 })
expect(VIEWPORTS).toContainEqual({ width: 360, height: 640 })
expect(VIEWPORTS.length).toBe(143)
expect(sweep().mobileBrowserFrameAreaBudget()).toBeGreaterThan(0)
})
})
@@ -11,8 +11,7 @@
* WebKit as well as Chromium, because the iOS shell is WKWebView and the two disagree: a `blob:`
* frame that Chromium admits under `frame-src blob:` is refused in WebKit by the
* `frame-ancestors 'none'` it inherits. `srcdoc` is what both admit under the policy that already
* ships, which is why this costs no CSP change and why a case below pins `frame-src 'none'` as still
* shipped.
* ships, so the preview needs no CSP change.
*
* The paint oracle is a pixel rather than a read inside the frame: the frame is an opaque origin, and
* WebKit refuses to evaluate in one, so reading its DOM would make the instrument engine-dependent.
@@ -243,8 +242,7 @@ for (const engine of ['chromium', 'webkit']) {
// frame's URL, which is `about:srcdoc` on one browser and empty on another.
expect(read.mountedSrcDoc).toContain('ARTIFACT_RENDERED')
expect(read.mountedSrc).toBeNull()
// The rendered frame carries the constant, so the token case below is about the frame the
// page mounts rather than about a string nothing reads.
// Check the sandbox on the mounted frame.
expect(read.mountedSandbox).toBe(read.declaredSandbox)
expect(read.mountedSandbox).toBe('allow-top-navigation-by-user-activation')
// The policy this document was served is the shell's own text plus the rig's report
@@ -298,8 +296,7 @@ for (const engine of ['chromium', 'webkit']) {
// The second fence, measured on its own: grant `allow-scripts` and keep the shipped policy,
// and the script still does not run, because a `srcdoc` frame inherits its embedder's
// `script-src 'self'` and the artifact's script is inline. So the seal does not rest on the
// sandbox attribute alone -- which is what makes the token list below a defence in depth
// rather than the only thing standing between the page and an agent's script.
// sandbox attribute alone.
const inherited = await open(browser(), {
signal: ctx.signal,
extra: { body: artifactScript(foreignOrigin) },
@@ -411,19 +408,6 @@ for (const engine of ['chromium', 'webkit']) {
expect(root.actError).toBeNull()
expect(root.ownOriginTopNavigations).toBe(1)
expect(root.topNavigations).toBe(0)
// `href=""` is the same navigation spelled as "this document", and it resolves the same way.
const empty = await open(browser(), {
signal: ctx.signal,
expectNavigation: 'main-frame',
act: async ({ frame }) => {
await frame?.click('#emptylink', { timeout: 2000 })
}
})
expect(empty.pixelBefore).toBe(ARTIFACT_RGB)
expect(empty.actError).toBeNull()
expect(empty.ownOriginTopNavigations).toBe(1)
expect(empty.topNavigations).toBe(0)
}, 180_000)
/**
@@ -834,33 +818,3 @@ for (const engine of ['chromium', 'webkit']) {
600_000
)
}
describe('the HTML preview needs no policy change', () => {
it('runs under a policy that still forbids every nested frame by URL', async () => {
const directives = (await readShellCsp()).split('; ')
// A `srcdoc` frame has no URL for `frame-src` to match, so the sealed box costs nothing here.
// Pinned so a future relaxation is a decision rather than a side effect of this component.
expect(directives).toContain("frame-src 'none'")
expect(directives).toContain("child-src 'none'")
expect(directives).toContain("script-src 'self'")
expect(directives).toContain("frame-ancestors 'none'")
})
it('grants exactly one sandbox token, and neither of the two that would unseal the frame', async () => {
const source = await readFileText('mobile/src/components/MobileHtmlPreview.web.tsx')
const match = /MOBILE_HTML_PREVIEW_SANDBOX = '([^']*)'/.exec(source)
expect(match).not.toBeNull()
const tokens = (match?.[1] ?? '').split(' ').filter((one) => one.length > 0)
expect(tokens).toEqual(['allow-top-navigation-by-user-activation'])
// Named rather than left to the list comparison: these two are the sealing invariant, and a
// reader of a failure should see which one was granted.
expect(tokens).not.toContain('allow-scripts')
expect(tokens).not.toContain('allow-same-origin')
})
})
/** One pixel of the frame's own fill, which is what says the artifact parsed and painted. */
async function readFileText(relativePath) {
const { readFile } = await import('node:fs/promises')
return await readFile(join(mobileDir, '..', relativePath), 'utf8')
}
@@ -30,6 +30,7 @@ const LIST_ROUTE = `/${MOBILE_WEB_APP_ROUTE_ROOT}/${HOST_ID}`
const SESSION_HREF = `${LIST_ROUTE}/session/wt-1`
const VIEWPORT = { width: 390, height: 844 }
const SAMPLE_MS = 1500
const SETTLED_GRACE_MS = 150
const bundles = mobileWebAppDependenciesPresent()
const describeRender = bundles ? describe : describe.skip
@@ -173,9 +174,13 @@ async function openList(browser, { animation = 'default', reducedMotion = 'no-pr
* Runs `action` on the probe, then reads both screens' left edge once per animation frame, and
* which screen a tap at the centre would land on. A hidden screen, or no screen hit, reads as null.
*/
function sampleFrames(page, action, { followUp = null, afterFrames = 0 } = {}) {
function sampleFrames(
page,
action,
{ followUp = null, afterFrames = 0, observeFullWindow = false } = {}
) {
return page.evaluate(
([name, sampleMs, nextAction, followAt]) =>
([name, sampleMs, nextAction, followAt, settledGraceMs, fullWindow]) =>
new Promise((resolve) => {
const leftOf = (id) => {
const node = document.querySelector(`[data-testid="${id}"]`)
@@ -190,6 +195,9 @@ function sampleFrames(page, action, { followUp = null, afterFrames = 0 } = {}) {
?.replace('stack-probe-', '') ?? null
const frames = []
const start = performance.now()
let settledAt = null
const destination = nextAction ?? name
const pushing = destination === 'push' || destination === 'pushOther'
globalThis.__orcaStackProbe[name]()
const tick = () => {
if (nextAction !== null && frames.length === followAt) {
@@ -202,7 +210,17 @@ function sampleFrames(page, action, { followUp = null, afterFrames = 0 } = {}) {
// Where the running slide starts, which is how a slide that restarts from 0 shows.
from: document.getAnimations()[0]?.effect?.getKeyframes()[0]?.transform ?? null
})
if (performance.now() - start < sampleMs) {
const last = frames.at(-1)
const arrived = pushing
? last.list === null && last.session === 0 && last.hit === 'session'
: last.list === 0 && last.session === null && last.hit === 'list'
const followUpDelivered = nextAction === null || frames.length > followAt
const settled = arrived && followUpDelivered && document.getAnimations().length === 0
const now = performance.now()
settledAt = settled ? (settledAt ?? now) : null
// Observe cleanup after actual arrival; retain the full window for interruption races.
const finished = !fullWindow && settledAt !== null && now - settledAt >= settledGraceMs
if (!finished && now - start < sampleMs) {
requestAnimationFrame(tick)
} else {
resolve(frames)
@@ -210,7 +228,7 @@ function sampleFrames(page, action, { followUp = null, afterFrames = 0 } = {}) {
}
requestAnimationFrame(tick)
}),
[action, SAMPLE_MS, followUp, afterFrames]
[action, SAMPLE_MS, followUp, afterFrames, SETTLED_GRACE_MS, observeFullWindow]
)
}
@@ -283,7 +301,11 @@ describeRender('the host stack transition on the page', () => {
const { errors, page } = await open()
await sampleFrames(page, 'push')
const pushed = await nodeCount(page)
const frames = await sampleFrames(page, 'back', { followUp: 'push', afterFrames: 3 })
const frames = await sampleFrames(page, 'back', {
followUp: 'push',
afterFrames: 3,
observeFullWindow: true
})
expect(frames.slice(0, 3).some((frame) => between(frame.session))).toBe(true)
expect(frames.at(-1)).toEqual(SESSION_SETTLED)
expect(await nodeCount(page)).toBe(pushed)
@@ -303,7 +325,11 @@ describeRender('the host stack transition on the page', () => {
await sampleFrames(page, 'back')
const baseline = await nodeCount(page)
const mounts = await sessionMounts(page)
const frames = await sampleFrames(page, 'push', { followUp: 'back', afterFrames: 6 })
const frames = await sampleFrames(page, 'push', {
followUp: 'back',
afterFrames: 6,
observeFullWindow: true
})
expect(between(frames[5].session)).toBe(true)
// Leaves from where the push stopped, not from 0: the exit's first keyframe is mid-screen.
const exitFrom = frames.slice(6).find((frame) => frame.from?.startsWith('matrix'))?.from
+43 -45
View File
@@ -90,7 +90,8 @@ describe('subscribeNativeChatTranscript', () => {
filePath,
onInitialSnapshot: (messages) => snapshots.push(messages),
onAppend: (messages) => appends.push(messages),
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
expect(sub.watching).toBe(true)
@@ -115,7 +116,8 @@ describe('subscribeNativeChatTranscript', () => {
filePath,
onInitialSnapshot: (messages) => snapshots.push(messages),
onAppend: () => {},
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await waitFor(() => snapshots.length === 1)
@@ -140,7 +142,8 @@ describe('subscribeNativeChatTranscript', () => {
lifecycles.push(lifecycle)
}
},
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await waitFor(() => lifecycles.length === 1)
@@ -170,7 +173,8 @@ describe('subscribeNativeChatTranscript', () => {
lifecycles.push(lifecycle)
}
},
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await waitFor(() => lifecycles.length === 1)
@@ -200,7 +204,8 @@ describe('subscribeNativeChatTranscript', () => {
snapshot = { messages, lifecycle }
},
onAppend: () => {},
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await waitFor(() => snapshot !== undefined)
@@ -269,7 +274,8 @@ describe('subscribeNativeChatTranscript', () => {
sessionId: 'ignored',
filePath,
onAppend: (messages) => batches.push(messages),
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await new Promise((resolve) => setTimeout(resolve, 20))
await appendFile(
@@ -295,7 +301,8 @@ describe('subscribeNativeChatTranscript', () => {
sessionId: 'ignored',
filePath,
onAppend: (messages) => seen.push(...messages),
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await new Promise((resolve) => setTimeout(resolve, 20))
await appendFile(
@@ -323,7 +330,8 @@ describe('subscribeNativeChatTranscript', () => {
onInitialSnapshot: (messages, hasMore) =>
snapshots.push({ ids: messages.map((message) => message.id), hasMore }),
onAppend: () => {},
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await waitFor(() => snapshots.length === 1)
@@ -342,7 +350,8 @@ describe('subscribeNativeChatTranscript', () => {
onInitialSnapshot: (messages, hasMore) =>
snapshots.push({ ids: messages.map((message) => message.id), hasMore }),
onAppend: () => {},
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await waitFor(() => snapshots.length === 1)
@@ -395,7 +404,8 @@ describe('subscribeNativeChatTranscript', () => {
onInitialSnapshot: (messages, hasMore) =>
snapshots.push({ ids: messages.map((message) => message.id), hasMore }),
onAppend: () => {},
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await waitFor(() => snapshots.length === 1)
@@ -412,7 +422,8 @@ describe('subscribeNativeChatTranscript', () => {
sessionId: 'ignored',
filePath,
onAppend: (messages) => batches.push(messages),
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await appendFile(filePath, claudeLine('a-1', 'assistant', 'reply'))
@@ -427,31 +438,6 @@ describe('subscribeNativeChatTranscript', () => {
expect(ids).toContain('a-1')
})
it('appends a turn in the gap between initial read and first watcher drain exactly once', async () => {
// Simulate the read/subscribe race: a turn lands after the caller's
// readSession EOF but before the watcher's first drain. Seeding at 0 means
// the first drain reads it; the assembler later dedups by deterministic id.
const filePath = await tempFile(claudeLine('u-1', 'user', 'first'))
const seen: NativeChatMessage[] = []
// The gap turn is written BEFORE subscribe completes its first drain.
await appendFile(filePath, claudeLine('a-gap', 'assistant', 'raced reply'))
const sub = await subscribeNativeChatTranscript({
agent: 'claude',
sessionId: 'ignored',
filePath,
onAppend: (messages) => seen.push(...messages),
debounceMs: 5
})
await waitFor(() => seen.some((m) => m.id === 'a-gap'))
sub.unsubscribe()
// The raced turn is present, and not duplicated within a single drain pass.
expect(seen.filter((m) => m.id === 'a-gap')).toHaveLength(1)
})
it('recovers cleanly when a read throws (subscription not left deaf)', async () => {
const filePath = await tempFile(claudeLine('u-1', 'user', 'hi'))
const seen: NativeChatMessage[] = []
@@ -461,7 +447,8 @@ describe('subscribeNativeChatTranscript', () => {
sessionId: 'ignored',
filePath,
onAppend: (messages) => seen.push(...messages),
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
// Make the file unreadable mid-flight (EACCES on the read path). The drain's
@@ -490,7 +477,8 @@ describe('subscribeNativeChatTranscript', () => {
sessionId: 'ignored',
filePath,
onAppend: () => {},
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
expect(getActiveNativeChatWatcherCount()).toBe(before + 1)
@@ -511,7 +499,8 @@ describe('subscribeNativeChatTranscript', () => {
sessionId: 'ignored',
filePath,
onAppend: (messages) => seen.push(...messages),
debounceMs: 10
debounceMs: 10,
reconciliationIntervalMs: 20
})
// Fire several appends back-to-back within the debounce window.
@@ -537,7 +526,8 @@ describe('subscribeNativeChatTranscript', () => {
sessionId: 'ignored',
filePath,
onAppend: (messages) => seen.push(...messages),
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await waitFor(() => seen.some((m) => m.id === 'u-1'))
@@ -566,7 +556,8 @@ describe('subscribeNativeChatTranscript', () => {
sessionId: 'ignored',
filePath,
onAppend: (messages) => seen.push(...messages),
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
// Replace the file with shorter content (simulates rotation to a new,
@@ -592,7 +583,8 @@ describe('subscribeNativeChatTranscript', () => {
sessionId: 'ignored',
filePath,
onAppend: (messages) => seen.push(...messages),
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await waitFor(() => seen.some((message) => message.id === 'u-old'))
@@ -622,7 +614,8 @@ describe('subscribeNativeChatTranscript', () => {
seen.splice(0, seen.length, ...messages)
},
onAppend: (messages) => seen.push(...messages),
debounceMs: 5
debounceMs: 5,
reconciliationIntervalMs: 20
})
await waitFor(() => seen.some((message) => message.id === 'atomic-old'))
@@ -648,7 +641,8 @@ describe('subscribeNativeChatTranscript', () => {
sessionId: 'ignored',
filePath,
onAppend: (messages) => seen.push(...messages),
debounceMs: 0
debounceMs: 0,
reconciliationIntervalMs: 20
})
await waitFor(() => seen.some((message) => message.id === 'race-old'))
@@ -672,7 +666,8 @@ describe('subscribeNativeChatTranscript', () => {
sessionId: 'ignored',
filePath,
onAppend: (messages) => seen.push(...messages),
debounceMs: 0
debounceMs: 0,
reconciliationIntervalMs: 20
})
await waitFor(() => seen.some((message) => message.id === 'unlink-old'))
@@ -717,6 +712,7 @@ describe('subscribeNativeChatTranscript (resolve-poll for a not-yet-created file
filePath,
onAppend: (messages) => seen.push(...messages),
debounceMs: 5,
reconciliationIntervalMs: 20,
resolvePollIntervalMs: 20
})
@@ -741,6 +737,7 @@ describe('subscribeNativeChatTranscript (resolve-poll for a not-yet-created file
sessionId: ' ',
onAppend: () => {},
debounceMs: 5,
reconciliationIntervalMs: 20,
resolvePollIntervalMs: 10
})
@@ -762,6 +759,7 @@ describe('subscribeNativeChatTranscript (resolve-poll for a not-yet-created file
filePath,
onAppend: () => {},
debounceMs: 5,
reconciliationIntervalMs: 20,
resolvePollIntervalMs: 20
})
@@ -387,23 +387,17 @@ describe('OrcaRuntimeService', () => {
'Browser automation is unavailable on this host, and the cause could not be determined.'
}
])
const browserCalls = Object.entries(runtime).filter(
([name, value]) => /^browser[A-Z]/.test(name) && typeof value === 'function'
)
expect(browserCalls.length).toBeGreaterThan(50)
for (const [name, call] of browserCalls) {
const invoke =
name === 'browserScreencast'
? () =>
(call as CallableFunction)(
{ format: 'jpeg' },
{ sendBinary: () => true, emit: () => undefined }
)
: () => (call as CallableFunction)({})
await expect(Promise.resolve().then(invoke)).rejects.toMatchObject({
code: 'browser_unavailable'
})
}
await expect(
Promise.resolve().then(() => runtime.browserGoto({ url: 'https://example.com' }))
).rejects.toMatchObject({ code: 'browser_unavailable' })
await expect(
Promise.resolve().then(() =>
runtime.browserScreencast(
{ format: 'jpeg' },
{ sendBinary: () => true, emit: () => undefined }
)
)
).rejects.toMatchObject({ code: 'browser_unavailable' })
})
it('reports the driver as missing instead of telling a configured operator to configure it', () => {
@@ -30,31 +30,6 @@ describe('OrcaRuntimeService', () => {
expect(runtime.getRuntimeId()).toBeTruthy()
})
it('reports runtime protocol, capabilities, and mobile aliases on status', () => {
const runtime = createRuntime()
const status = runtime.getStatus()
expect(typeof status.runtimeProtocolVersion).toBe('number')
expect(typeof status.minCompatibleRuntimeClientVersion).toBe('number')
expect(status.runtimeProtocolVersion).toBe(status.protocolVersion)
expect(status.minCompatibleRuntimeClientVersion).toBe(status.minCompatibleMobileVersion)
expect(status.capabilities).toContain('terminal.binary-stream.v1')
expect(status.capabilities).toContain('workspace-ports.v1')
expect(status.capabilities).toContain('mobile.tasks.v1')
expect(status.capabilities).toContain('terminal.quick-commands.v1')
expect(status.capabilities).toContain('session-tabs.split-group-placement.v1')
expect(status.capabilities).toContain('worktree.create-idempotency.v1')
expect(status.worktreeCreateIdempotency).toEqual({ dedupeTtlMs: 60_000 })
expect(status.capabilities).toContain('files.mutation-ownership.v1')
expect(status.capabilities).toContain('project-host-setup.v1')
expect(status.capabilities).toContain('linear.issue-attribute-filter.v1')
expect(status.capabilities).not.toContain('browser.screencast.v1')
expect(typeof status.protocolVersion).toBe('number')
expect(typeof status.minCompatibleMobileVersion).toBe('number')
expect(status.protocolVersion).toBeGreaterThanOrEqual(1)
expect(status.minCompatibleMobileVersion).toBeGreaterThanOrEqual(0)
})
it('reports the configured Windows terminal shell on status', () => {
const runtime = new OrcaRuntimeService({
...store,
@@ -265,20 +240,6 @@ describe('OrcaRuntimeService', () => {
expect(runtime.getStatus().capabilities).toContain('browser.screencast.v1')
})
// Paired desktops open a chat on this host only when it says it admits them by the client's
// chosen launch mode; without it, every paired launch quietly becomes a terminal.
it('advertises that it admits structured sessions by the client-chosen launch mode', () => {
expect(createRuntime().getStatus().capabilities).toContain(
'agent-session.structured.client-launch-mode.v1'
)
})
it('advertises safe Codex reset-credit RPC support as a static capability', () => {
const runtime = createRuntime()
expect(runtime.getStatus().capabilities).toContain('accounts.codex-reset-credit.v1')
})
it('routes mobile Codex reset consumption through the account mutation coordinator', async () => {
const runtime = createRuntime()
const expectedScope = {
@@ -2,12 +2,9 @@ import { describe, expect, it, vi } from 'vitest'
import {
AGENT_PROMPT_BRACKETED_PASTE_END,
AGENT_PROMPT_BRACKETED_PASTE_START,
buildAgentPromptPasteBytes,
resolveAgentPromptSubmitDelayForAgent
buildAgentPromptPasteBytes
} from '../../../shared/agent-prompt-injection'
import { TUI_AGENT_CONFIG } from '../../../shared/tui-agent-config'
import { ORCA_DISPATCH_PROMPT_LEAD_LINE } from '../../../shared/orca-dispatch-status-prompt'
import type { TuiAgent } from '../../../shared/tui-agent'
import { OrcaRuntimeService } from '../orca-runtime'
import { acknowledgeAgentPromptSubmit } from '../orca-runtime-test-mocks.spec'
import {
@@ -626,58 +623,52 @@ describe('OrcaRuntimeService', () => {
}
)
it.each(
(Object.keys(TUI_AGENT_CONFIG) as TuiAgent[]).filter(
(agent) => agent !== 'claude' && agent !== 'codex'
)
)('submits through the agent-specific PTY timing policy for %s', async (agent) => {
vi.useFakeTimers()
try {
const writes: string[] = []
const runtime = new OrcaRuntimeService(store)
runtime.setPtyController({
spawn: vi.fn().mockResolvedValue({ id: 'pty-bg' }),
write: (_ptyId, data) => {
writes.push(data)
if (agent === 'omp' && data.endsWith('\r')) {
runtime.onPtyData('pty-bg', '\x1b]0;Codex working\x07', Date.now())
} else {
acknowledgeAgentPromptSubmit(runtime, 'pty-bg', data)
}
return true
},
kill: () => true,
getForegroundProcess: async () => null
})
const { handle } = await runtime.createTerminal(`path:${TEST_WORKTREE_PATH}`, {
launchAgent: agent
})
it.each(['aider', 'antigravity', 'omp'] as const)(
'submits through the agent-specific PTY timing policy for %s',
async (agent) => {
vi.useFakeTimers()
try {
const writes: string[] = []
const runtime = new OrcaRuntimeService(store)
runtime.setPtyController({
spawn: vi.fn().mockResolvedValue({ id: 'pty-bg' }),
write: (_ptyId, data) => {
writes.push(data)
if (agent === 'omp' && data.endsWith('\r')) {
runtime.onPtyData('pty-bg', '\x1b]0;Codex working\x07', Date.now())
} else {
acknowledgeAgentPromptSubmit(runtime, 'pty-bg', data)
}
return true
},
kill: () => true,
getForegroundProcess: async () => null
})
const { handle } = await runtime.createTerminal(`path:${TEST_WORKTREE_PATH}`, {
launchAgent: agent
})
// The agent's own policy, not the byte-only delay: antigravity adds a per-line settle
// (#21665), and advancing fake timers by less than the policy waits leaves the submit
// pending until the real 30 s timeout.
const submitDelayMs = resolveAgentPromptSubmitDelayForAgent(
process.platform,
'review this change',
agent
)
const sendPromise = runtime.sendTerminalAgentPrompt(handle, 'review this change', {
inputKind: 'driving'
})
if (agent === 'omp') {
const prompt =
agent === 'antigravity' ? 'review this change\nfollow up' : 'review this change'
const submitDelayMs = agent === 'antigravity' ? 591 : 501
const sendPromise = runtime.sendTerminalAgentPrompt(handle, prompt, {
inputKind: 'driving'
})
if (agent === 'omp') {
await sendPromise
expect(writes).toEqual([`${buildAgentPromptPasteBytes('review this change')}\r`])
return
}
await vi.advanceTimersByTimeAsync(submitDelayMs - 1)
expect(writes).not.toContain('\r')
await vi.advanceTimersByTimeAsync(1)
await sendPromise
expect(writes).toEqual([`${buildAgentPromptPasteBytes('review this change')}\r`])
return
expect(writes.filter((data) => data === '\r')).toHaveLength(1)
} finally {
vi.useRealTimers()
}
await vi.advanceTimersByTimeAsync(submitDelayMs - 1)
expect(writes).not.toContain('\r')
await vi.advanceTimersByTimeAsync(1)
await sendPromise
expect(writes.filter((data) => data === '\r')).toHaveLength(1)
} finally {
vi.useRealTimers()
}
})
)
})
-9
View File
@@ -181,15 +181,6 @@ describe('Subprocess: Relay entry point', () => {
}
})
it('prints sentinel on startup', async () => {
relay = spawn()
await relay.sentinelReceived
}, 10_000)
it('keeps the Node-18 relay bundle free of unsupported array copy methods', () => {
expect(readFileSync(relayEntry, 'utf8')).not.toContain('.toReversed(')
})
it('loads node-pty after an in-place dependency repair without restarting', async () => {
tmpDir = mkdtempSync(path.join(tmpdir(), 'relay-native-repair-'))
const repairedRelayEntry = path.join(tmpDir, 'relay.js')
@@ -41,20 +41,6 @@ const session: NativeChatLiveSession = {
}
describe('NativeChatMessageList assistant messages', () => {
it('keeps prose selectable and places non-selectable controls after it', () => {
render(<NativeChatMessageList session={session} isWorking={false} expandSignal={false} />)
const prose = screen.getByText('Selectable agent response.')
const row = prose.closest('.group')
const copyButton = screen.getByRole('button', { name: 'Copy message' })
const controls = copyButton.parentElement
expect(row).toHaveClass('select-text')
expect(controls).toHaveClass('select-none', 'can-hover:pointer-events-none', 'mt-1')
expect(controls).not.toHaveClass('absolute')
expect(prose.compareDocumentPosition(controls!)).toBe(Node.DOCUMENT_POSITION_FOLLOWING)
})
it('keeps a running tool live when transcript lifecycle metadata is absent', () => {
render(
<NativeChatMessageList
@@ -50,45 +50,6 @@ describe('MessageRow control visibility', () => {
})
})
it('appends time to the existing agent controls and inherits their reveal', () => {
renderMessage('assistant')
const copy = screen.getByRole('button', { name: 'Copy message' })
const scroll = screen.getByRole('button', { name: 'Scroll this message to top' })
const time = screen.getByRole('time')
expect(Array.from(copy.parentElement!.children)).toEqual([copy, scroll, time])
expect(copy.parentElement).toHaveClass(
'can-hover:opacity-0',
'can-hover:pointer-events-none',
'group-hover:opacity-100',
'[.group:has(:focus-visible)_&]:opacity-100',
'group-hover:pointer-events-auto',
'[.group:has(:focus-visible)_&]:pointer-events-auto'
)
expect(copy.parentElement).not.toHaveClass('opacity-0', 'pointer-events-none')
expect(time).not.toHaveAttribute('tabindex')
copy.focus()
expect(copy).toHaveFocus()
})
it('gives user bubbles a copy button and timestamp that only hide on hover-capable devices', () => {
renderMessage('user')
const copy = screen.getByRole('button', { name: 'Copy message' })
const time = screen.getByRole('time')
expect(Array.from(copy.parentElement!.children)).toEqual([copy, time])
expect(copy.parentElement).toHaveClass(
'can-hover:opacity-0',
'can-hover:pointer-events-none',
'group-hover:opacity-100',
'[.group:has(:focus-visible)_&]:opacity-100',
'group-hover:pointer-events-auto',
'[.group:has(:focus-visible)_&]:pointer-events-auto'
)
expect(copy.parentElement).not.toHaveClass('opacity-0', 'pointer-events-none')
expect(copy.parentElement!.parentElement).toHaveClass('group')
time.focus()
expect(time).toHaveFocus()
})
it('copies the sent message text from a user bubble', async () => {
const writeClipboardText = vi.fn().mockResolvedValue(undefined)
Object.assign(window, { api: { ui: { writeClipboardText } } })
@@ -131,20 +92,6 @@ describe('MessageRow control visibility', () => {
expect(screen.queryByRole('time')).toBeNull()
expect(screen.queryByRole('button')).toBeNull()
})
it.each(['reasoning', 'system'] as const)(
'keeps %s rows upright and scopes their faint text',
(role) => {
const { container } = renderMessage(role)
const row = container.querySelector('[data-native-chat-message-tone="faint"]')
expect(row).toHaveClass('text-chat-foreground-faint')
expect(row).not.toHaveClass('italic')
expect(row).toContainElement(screen.getByText('Message text'))
if (role === 'reasoning') {
expect(row).toHaveClass('border-l-2')
}
}
)
})
describe('MessageRow send mode', () => {
@@ -27,14 +27,6 @@ describe('notice rows', () => {
screen.getByText('Context compacted').parentElement?.querySelectorAll('.bg-border')
).toHaveLength(2)
})
it.each([
['warning', 'text-[color:var(--warning,#f59e0b)]'],
['error', 'text-destructive'],
['notice', 'text-muted-foreground']
])('renders %s using its existing color treatment', (tone, className) => {
renderStatus({ kind: 'status', text: 'Readable notice', tone })
expect(screen.getByText('Readable notice').parentElement?.parentElement).toHaveClass(className)
})
it('renders a plan as readable markdown in the card primitive', () => {
renderStatus({
kind: 'status',
@@ -100,16 +92,4 @@ describe('notice rows', () => {
expect(screen.getByText(words)).toHaveClass('text-muted-foreground', 'text-sm')
expect(screen.queryByText('Words an older host wrote')).toBeNull()
})
it('renders future presentation and tone values as untinted text', () => {
renderStatus({
kind: 'status',
text: 'Future readable text',
tone: 'future-tone',
presentation: 'future-presentation'
})
expect(screen.getByText('Future readable text').parentElement?.parentElement).toHaveClass(
'text-foreground'
)
expect(screen.getByText('Future readable text').parentElement?.querySelector('svg')).toBeNull()
})
})
@@ -187,7 +187,12 @@ vi.mock('./NativeChatQuestionCard', () => ({
}))
import { NativeChatStructuredSession } from './NativeChatStructuredSession'
import { seededEntry, seedOutbox } from './NativeChatStructuredSession.test-harness'
import {
advanceProbeClock,
seededEntry,
seedOutbox,
useProbeClock
} from './NativeChatStructuredSession.test-harness'
import {
appendStructuredAgentSessionOutboxMessage,
getStructuredAgentSessionOutbox
@@ -218,6 +223,7 @@ describe('NativeChatStructuredSession delivery', () => {
// Resent under its own id until the host answers, so its row says only that it is still sending.
it('says a send whose answer was lost is sending until it confirms on its own, with no Retry', async () => {
useProbeClock()
mocks.mode = 'outbox'
mocks.call.mockRejectedValueOnce(new Error('socket closed')).mockResolvedValueOnce({
ok: true,
@@ -238,19 +244,19 @@ describe('NativeChatStructuredSession delivery', () => {
const send = mocks.composerProps?.structuredTransport?.send as
| ((text: string, attachments: readonly { id: string; path: string }[]) => boolean)
| undefined
expect(send?.('hello', [])).toBe(true)
await act(async () => {
expect(send?.('hello', [])).toBe(true)
})
// From the moment it is sent, through the lost answer, until the host confirms it.
await waitFor(() => expect(screen.getByText('Sending…')).toBeTruthy())
await waitFor(() =>
expect(getStructuredAgentSessionOutbox('session-1')).toMatchObject([{ state: 'unconfirmed' }])
)
expect(getStructuredAgentSessionOutbox('session-1')).toMatchObject([{ state: 'unconfirmed' }])
expect(screen.getByText('Sending…')).toBeTruthy()
expect(screen.queryByText('Message delivery is unconfirmed.')).toBeNull()
expect(screen.queryByRole('button', { name: /Retry/ })).toBeNull()
await waitFor(() => expect(mocks.call).toHaveBeenCalledTimes(2), { timeout: 5000 })
await advanceProbeClock(1000)
expect(mocks.call).toHaveBeenCalledTimes(2)
expect(mocks.call.mock.calls[1]?.[2]).toEqual(mocks.call.mock.calls[0]?.[2])
await waitFor(() => expect(screen.queryByText('Sending…')).toBeNull())
expect(screen.queryByText('Sending…')).toBeNull()
expect(getStructuredAgentSessionOutbox('session-1')).toEqual([])
expect(screen.queryByText('Message delivery is unconfirmed.')).toBeNull()
}, 10000)
@@ -282,6 +288,7 @@ describe('NativeChatStructuredSession delivery', () => {
}
it('says a send reopened mid-send is sending while it is resent, until it settles', async () => {
useProbeClock()
mocks.mode = 'outbox'
mocks.submissions = []
mocks.call.mockResolvedValue({
@@ -298,53 +305,45 @@ describe('NativeChatStructuredSession delivery', () => {
expect(screen.getByText('Sending…')).toBeTruthy()
expect(screen.queryByText('Message delivery is unconfirmed.')).toBeNull()
expect(screen.queryByRole('button', { name: /Retry/ })).toBeNull()
await waitFor(() => expect(mocks.call).toHaveBeenCalledOnce(), { timeout: 3000 })
await advanceProbeClock(1000)
expect(mocks.call).toHaveBeenCalledOnce()
expect(mocks.call.mock.calls[0]?.[2]).toMatchObject({
envelope: { clientOperationId: 'op-sent' }
})
await waitFor(() => expect(getStructuredAgentSessionOutbox('session-reopened')).toEqual([]))
expect(getStructuredAgentSessionOutbox('session-reopened')).toEqual([])
expect(screen.queryByText('Sending…')).toBeNull()
expect(screen.queryByText('Message delivery is unconfirmed.')).toBeNull()
}, 10000)
it.each([
['a live unknown', {}],
['a recovered unknown', { recovered: true }],
["an older host's recovered unknown", { reason: 'host_restarted_before_acknowledgement' }]
])(
'says a send reopened mid-send is unconfirmed, with its Retry, once the journal holds %s',
async (label, patch) => {
mocks.mode = 'outbox'
const sessionId = `session-reopened-${label.replace(/\W+/g, '-')}`
mocks.submissions = [
{
clientMessageId: 'op-sent',
fence: 1,
payloadFingerprint: 'fp',
dispatchState: 'unknown',
providerItemId: null,
reason: null,
submittedAt: 1,
resolvedAt: null,
...patch
}
]
seedMidSend(sessionId)
it('says a send reopened mid-send is unconfirmed, with its Retry, when the journal holds unknown', async () => {
useProbeClock()
mocks.mode = 'outbox'
const sessionId = 'session-reopened-unknown'
mocks.submissions = [
{
clientMessageId: 'op-sent',
fence: 1,
payloadFingerprint: 'fp',
dispatchState: 'unknown',
providerItemId: null,
reason: null,
submittedAt: 1,
resolvedAt: null
}
]
seedMidSend(sessionId)
renderSession(sessionId)
renderSession(sessionId)
await waitFor(() => expect(screen.getByText('Message delivery is unconfirmed.')).toBeTruthy())
expect(screen.getByRole('button', { name: /Retry/ })).toBeTruthy()
expect(screen.queryByText('Sending…')).toBeNull()
await act(async () => {
await new Promise((resolve) => setTimeout(resolve, 1500))
})
expect(mocks.call).not.toHaveBeenCalled()
},
10000
)
expect(screen.getByText('Message delivery is unconfirmed.')).toBeTruthy()
expect(screen.getByRole('button', { name: /Retry/ })).toBeTruthy()
expect(screen.queryByText('Sending…')).toBeNull()
await advanceProbeClock(1500)
expect(mocks.call).not.toHaveBeenCalled()
})
it('says a send a Stop outlived is unconfirmed when reopened, as nothing resends it', async () => {
useProbeClock()
mocks.mode = 'outbox'
mocks.submissions = []
seedMidSend('session-reopened-stopped', { outlivedStop: true })
@@ -354,9 +353,7 @@ describe('NativeChatStructuredSession delivery', () => {
expect(screen.getByText('Message delivery is unconfirmed.')).toBeTruthy()
expect(screen.getByRole('button', { name: /Retry/ })).toBeTruthy()
expect(screen.queryByText('Sending…')).toBeNull()
await act(async () => {
await new Promise((resolve) => setTimeout(resolve, 1500))
})
await advanceProbeClock(1500)
expect(mocks.call).not.toHaveBeenCalled()
}, 10000)
@@ -1,40 +0,0 @@
// @vitest-environment happy-dom
import '@testing-library/jest-dom/vitest'
import { cleanup, render, screen } from '@testing-library/react'
import { afterEach, describe, expect, it } from 'vitest'
import { NativeChatToolRun } from './NativeChatToolRun'
afterEach(cleanup)
describe('chat code typography opt-in', () => {
it('keeps live tool names and file paths outside code-size styling', () => {
render(
<NativeChatToolRun
blocks={[{ type: 'tool-call', name: 'Read', input: { file_path: 'src/app.ts' } }]}
expandSignal
activeTurnIsWorking
/>
)
expect(screen.getByText('Read src/app.ts')).not.toHaveAttribute('data-native-chat-code-content')
expect(screen.getByTitle('src/app.ts')).not.toHaveAttribute('data-native-chat-code-content')
})
it('opts command previews and expanded output into code size', () => {
const { container } = render(
<NativeChatToolRun
blocks={[
{ type: 'tool-call', callId: 'shell', name: 'Bash', input: { command: 'git status' } },
{ type: 'tool-result', callId: 'shell', output: 'working tree clean' }
]}
expandSignal
activeTurnIsWorking
/>
)
expect(screen.getByTitle('git status')).toHaveAttribute('data-native-chat-code-content')
expect(container.querySelector('pre')).toHaveAttribute('data-native-chat-code-content')
expect(screen.getByText('Ran')).not.toHaveAttribute('data-native-chat-code-content')
const header = container.querySelector('button')
expect(header).not.toBeNull()
expect(header?.querySelector('[data-native-chat-code-content]')).toBeNull()
})
})
@@ -19,6 +19,7 @@ vi.mock('@/runtime/structured-agent-session-client', () => ({
import { setLocalRuntimeCapabilitiesForTests } from '@/runtime/local-runtime-capabilities'
import { RuntimeRpcCallError } from '@/runtime/runtime-rpc-result'
import { useStructuredAgentSessionOutbox } from './use-structured-agent-session-outbox'
import { advanceProbeClock, useProbeClock } from './NativeChatStructuredSession.test-harness'
import { structuredAgentSessionDeliveryNotices } from './structured-agent-session-delivery-notices'
import { agentJournalSubmissionKey } from '../../../../shared/agent-session-journal-item-key'
import { agentSessionWriteNoticeEnglish } from '../../../../shared/agent-session-refusal-notice'
@@ -43,7 +44,10 @@ function outboxProps(submissions: AgentJournalSubmission[]): OutboxProps {
const NO_JOURNAL_ITEMS: readonly AgentJournalRenderItem[] = []
// Why: every hook here shares the session outbox store; one left mounted would drain the next test's.
afterEach(cleanup)
afterEach(() => {
cleanup()
vi.useRealTimers()
})
function shownFailure(entry: StructuredAgentSessionOutboxEntry | undefined): string | undefined {
return (
@@ -297,6 +301,7 @@ describe('a send the host rejected because the agent never started', () => {
// One stored host fact, one rendering: a message the host recorded and rejected, whose record this
// chat does not hold, keeps showing why on every mount, with no control, and is never sent again.
it('keeps a message the host rejected, with no control, on every mount, and never resends it', async () => {
useProbeClock()
writeOutbox('session-1', [
{
...rejectedBeforeRestart(),
@@ -333,9 +338,8 @@ describe('a send the host rejected because the agent never started', () => {
)
const first = mount()
await waitFor(() => expect(first.result.current.outbox[0]?.state).toBe('rejected'), {
timeout: 4000
})
await advanceProbeClock(1000)
expect(first.result.current.outbox[0]?.state).toBe('rejected')
const onFirst = notice(first.result.current.outbox, first.result.current.failedHere)
first.unmount()
const second = mount()
@@ -344,7 +348,7 @@ describe('a send the host rejected because the agent never started', () => {
for (const shown of [onFirst, onSecond]) {
expect(shown).toEqual({ text: REASON })
}
await act(() => new Promise((resolve) => setTimeout(resolve, 1500)))
await advanceProbeClock(1500)
// Only the first mount's question about the send it left in doubt; nothing resends it.
expect(mocks.call).toHaveBeenCalledOnce()
expect(second.result.current.outbox.map((entry) => entry.state)).toEqual(['rejected'])