From e7dc9b60995d73a0206c34891188e45cd9b708f2 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:41:24 -0700 Subject: [PATCH 01/69] test: honor background launch in paired client window helpers (#18978) --- AGENTS.md | 6 +++ tests/AGENTS.md | 5 +- .../helpers/paired-client-window-reveal.ts | 15 ++++-- .../paired-client-window-reveal.unit.test.ts | 47 ++++++++++++++++++- 4 files changed, 63 insertions(+), 10 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 8b0156ba6b1..306c9c8d5ed 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -4,6 +4,12 @@ All UI work — layout, color, typography, spacing, component selection, UX beha ## Electron UI Validation +Always run tests and agent-launched apps in the background with `ORCA_BACKGROUND_LAUNCH=1`. +Never steal monitor focus or reveal test windows: no `show()`, `showInactive()`, `bringToFront()`, +`app.focus()`, or OS activation. Use CDP screenshots of hidden renderers. Keep native-focus and +visible-window tests paused on the user's desktop; run them on an isolated display or CI. +Rebuild modified launch-policy code before running an app; stale build wrappers are not safe. + Use the `$electron` skill and Playwright CDP for rendered Orca UI checks. Do not use computer-use for Orca UI validation. # Style diff --git a/tests/AGENTS.md b/tests/AGENTS.md index 26987a87c25..30f522652fe 100644 --- a/tests/AGENTS.md +++ b/tests/AGENTS.md @@ -18,6 +18,5 @@ Rules when adding tests or scripts: - Do not reveal windows in explicit background or headless runs. Only an explicitly headful run may call `showInactive()`; never call `show()` or `bringToFront()` in automated background checks. - Tag a spec `@headful` only when it needs real pixels; it still runs in the background. -- `ORCA_E2E_FOREGROUND=1` is the only opt-out, for runs whose subject _is_ native focus (IME and - other OS-level key injection). Clear `ORCA_BACKGROUND_LAUNCH` for that isolated run and add a - comment saying why; an explicit background request takes precedence. +- Native-focus tests belong on an isolated display or CI. Do not set `ORCA_E2E_FOREGROUND=1` + on the user’s desktop; it cannot override explicit background mode. diff --git a/tests/e2e/helpers/paired-client-window-reveal.ts b/tests/e2e/helpers/paired-client-window-reveal.ts index 302d573c3d1..1ec5b635d77 100644 --- a/tests/e2e/helpers/paired-client-window-reveal.ts +++ b/tests/e2e/helpers/paired-client-window-reveal.ts @@ -29,22 +29,24 @@ export function assertPairedClientWindowRevealed(report: PairedClientWindowRevea export type PairedClientWindowFocusReport = PairedClientWindowRevealReport & { isFocused: boolean } /** - * Brings a paired client to the front, which a launched-but-background window never is. Main-side - * policies that ask whether the reader is looking at a WebContents read the OS focus state, so a - * spec driving real presses through such a policy has to put the window there first. + * Native-focus coverage must run on an isolated display or CI, never in background mode. */ export async function focusPairedClientWindow( client: RevealablePairedClient, { timeoutMs = 15_000 }: { timeoutMs?: number } = {} ): Promise { + await client.app.evaluate(() => { + if (process.env.ORCA_BACKGROUND_LAUNCH === '1') { + throw new Error('Native focus is forbidden by ORCA_BACKGROUND_LAUNCH') + } + }) const revealed = await revealPairedClientWindow(client) const deadline = Date.now() + timeoutMs let isFocused = false while (!isFocused) { isFocused = await client.app.evaluate(({ app, BrowserWindow }) => { const window = BrowserWindow.getAllWindows()[0] - // Why steal: nothing else in the run is asking for the front, and the window manager keeps - // the launching terminal there otherwise. + // Native-focus coverage requires a dedicated foreground session. app.focus({ steal: true }) window?.focus() return window?.isFocused() ?? false @@ -61,6 +63,9 @@ export async function revealPairedClientWindow( client: RevealablePairedClient ): Promise { const report = await client.app.evaluate(({ BrowserWindow }) => { + if (process.env.ORCA_BACKGROUND_LAUNCH === '1') { + throw new Error('Window reveal is forbidden by ORCA_BACKGROUND_LAUNCH') + } const windows = BrowserWindow.getAllWindows() const window = windows[0] const wasVisible = window?.isVisible() ?? false diff --git a/tests/e2e/helpers/paired-client-window-reveal.unit.test.ts b/tests/e2e/helpers/paired-client-window-reveal.unit.test.ts index dfb83e4c441..706088e6762 100644 --- a/tests/e2e/helpers/paired-client-window-reveal.unit.test.ts +++ b/tests/e2e/helpers/paired-client-window-reveal.unit.test.ts @@ -1,5 +1,10 @@ -import { describe, expect, it } from 'vitest' -import { assertPairedClientWindowRevealed } from './paired-client-window-reveal' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + assertPairedClientWindowRevealed, + focusPairedClientWindow, + revealPairedClientWindow, + type RevealablePairedClient +} from './paired-client-window-reveal' describe('assertPairedClientWindowRevealed', () => { it('accepts a window that the reveal made visible', () => { @@ -42,3 +47,41 @@ describe('assertPairedClientWindowRevealed', () => { ).toThrow(/stayed hidden after showInactive\(\)/) }) }) + +describe('paired client background safety', () => { + afterEach(() => vi.unstubAllEnvs()) + + function makeClient() { + const showInactive = vi.fn() + const focus = vi.fn() + const getAllWindows = vi.fn(() => [{ isVisible: () => false, showInactive, focus }]) + const evaluate = vi.fn(async (callback) => + callback({ + app: { focus }, + BrowserWindow: { getAllWindows } + }) + ) + const client = { + app: { evaluate }, + page: { waitForFunction: vi.fn() } + } as unknown as RevealablePairedClient + return { client, showInactive, focus, getAllWindows } + } + + it('rejects an explicit reveal before touching native windows', async () => { + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', '1') + const { client, getAllWindows, showInactive } = makeClient() + await expect(revealPairedClientWindow(client)).rejects.toThrow('Window reveal is forbidden') + expect(getAllWindows).not.toHaveBeenCalled() + expect(showInactive).not.toHaveBeenCalled() + }) + + it.each(['0', '1'])('rejects focus in background mode with foreground=%s', async (foreground) => { + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', '1') + vi.stubEnv('ORCA_E2E_FOREGROUND', foreground) + const { client, focus, getAllWindows } = makeClient() + await expect(focusPairedClientWindow(client)).rejects.toThrow('Native focus is forbidden') + expect(getAllWindows).not.toHaveBeenCalled() + expect(focus).not.toHaveBeenCalled() + }) +}) From 6d691a4c04c40fb3734f7066aecd002fe126e0cc Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:43:36 -0700 Subject: [PATCH 02/69] test: bound release checkout lock fixtures and gate delayed imports (#18981) --- .../release-checkout.unit.test.ts | 40 +++++++++++++++---- 1 file changed, 33 insertions(+), 7 deletions(-) diff --git a/tests/e2e/cross-version-wire/release-checkout.unit.test.ts b/tests/e2e/cross-version-wire/release-checkout.unit.test.ts index 7057a38babd..106e2ea778e 100644 --- a/tests/e2e/cross-version-wire/release-checkout.unit.test.ts +++ b/tests/e2e/cross-version-wire/release-checkout.unit.test.ts @@ -263,12 +263,24 @@ afterEach(() => { describe('release checkout materialization', () => { it('single-flights concurrent consumers of one release identity', async () => { const cacheRoot = temporaryCacheRoot() + let publications = 0 + const options = { + cacheRoot, + testHooks: { + populateStaging: async (context: CheckoutStagingContext) => { + publications++ + await populateMinimalStaging(context) + } + } + } const checkouts = await Promise.all([ - materializeReleaseCheckout('v1.4.190', { cacheRoot }), - materializeReleaseCheckout('v1.4.190', { cacheRoot }), - materializeReleaseCheckout('v1.4.190', { cacheRoot }) + materializeReleaseCheckout('v1.4.190', options), + materializeReleaseCheckout('v1.4.190', options), + materializeReleaseCheckout('v1.4.190', options) ]) + expect(publications).toBe(1) + expect(new Set(checkouts.map(({ root }) => root))).toHaveLength(1) expect(relative(cacheRoot, checkouts[0]!.root)).not.toMatch(/^\.\./) }) @@ -291,18 +303,29 @@ describe('release checkout materialization', () => { ) const cacheRoot = temporaryCacheRoot() - const first = await materializeReleaseCheckout(firstRef, { cacheRoot }) + const options = { cacheRoot, testHooks: { populateStaging: populateMinimalStaging } } + const first = await materializeReleaseCheckout(firstRef, options) const dependency = join(first.root, 'delayed-dependency.mjs') const entry = join(first.root, 'delayed-entry.mjs') + const importStarted = join(cacheRoot, 'import-started') + const continueImport = join(cacheRoot, 'continue-import') writeFileSync(dependency, "export const loaded = 'first-release'\n") writeFileSync( entry, - 'await new Promise((resolve) => setTimeout(resolve, 100))\n' + + "import { existsSync, writeFileSync } from 'node:fs'\n" + + `writeFileSync(${JSON.stringify(importStarted)}, '')\n` + + `while (!existsSync(${JSON.stringify(continueImport)})) await new Promise((resolve) => setTimeout(resolve, 10))\n` + "export const loaded = (await import('./delayed-dependency.mjs')).loaded\n" ) const loading = importReleaseCheckoutModule(first, '/delayed-entry.mjs') - const second = await materializeReleaseCheckout(secondRef, { cacheRoot }) + let second: ReleaseCheckout + try { + await waitForFile(importStarted, 5_000) + second = await materializeReleaseCheckout(secondRef, options) + } finally { + writeFileSync(continueImport, '') + } await expect(loading).resolves.toMatchObject({ loaded: 'first-release' }) expect(first.root).not.toBe(second.root) @@ -311,7 +334,10 @@ describe('release checkout materialization', () => { it('causally single-flights a rival process before publishing an in-use checkout', async () => { const cacheRoot = temporaryCacheRoot() const scratch = temporaryCacheRoot() - const published = await materializeReleaseCheckout('v1.4.190', { cacheRoot }) + const published = await materializeReleaseCheckout('v1.4.190', { + cacheRoot, + testHooks: { populateStaging: populateMinimalStaging } + }) await expect(runContentionPhase(published, scratch, 'locked', false)).resolves.toBe(true) // In the same causally acknowledged interleaving, a no-lock materializer From 84432d3aa1584c135283fb4be22ee25e1bb5258f Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:45:34 -0700 Subject: [PATCH 03/69] fix(native-chat): repair a structured chat tab permanently fenced by an inherited publication epoch (#18906) * fix(native-chat): repair a structured tab fenced out by a returning publisher A publication epoch is retired whenever another publisher takes over a worktree, and a retired epoch is then rejected forever. But a live publisher can return after transient interlopers - a `removed:` retraction, then a headless rebuild whose version restarts at 1 - and the structured tab publish inherits the worktree's existing epoch rather than minting one, so it arrives under the blacklisted epoch and is dropped. The chat tab never reaches the tab bar. The fence is right to reject the frame: it cannot tell a returning publisher apart from a delayed frame queued by a dead generation, whose version can outrank the live cursor. So the drop is no longer final - it schedules one bounded, debounced authoritative `session.tabs.listAll`, and only that census may revive an epoch, and only the one it names current. Subscription frames stay fenced exactly as before. * fix(native-chat): decay the structured tab repair cap and prune its state The attempt cap latched: three transient RPC failures left `exhausted` set for the renderer's lifetime, permanently hiding a chat tab behind a single console warning. It now decays, so a worktree that has been quiet for a minute gets its full budget back. The repair map was also missing from the sweep that drops publisher cursors for vanished worktrees, leaking an entry per deleted worktree. Pruning it there required inverting the repair lane's dependency on the inventory refresh, which is now injected. --------- Co-authored-by: Merge Sim --- ...tured-session-retired-epoch-repair.test.ts | 198 ++++++++++++++++++ .../inventory-refresh.ts | 10 +- .../retired-epoch-repair.test.ts | 121 +++++++++++ .../retired-epoch-repair.ts | 131 ++++++++++++ .../snapshot-apply.ts | 37 +++- .../subscription.ts | 19 +- .../publisher-identity-fences.ts | 13 ++ 7 files changed, 518 insertions(+), 11 deletions(-) create mode 100644 src/renderer/src/runtime/local-structured-session-retired-epoch-repair.test.ts create mode 100644 src/renderer/src/runtime/local-structured-session-tabs-sync/retired-epoch-repair.test.ts create mode 100644 src/renderer/src/runtime/local-structured-session-tabs-sync/retired-epoch-repair.ts diff --git a/src/renderer/src/runtime/local-structured-session-retired-epoch-repair.test.ts b/src/renderer/src/runtime/local-structured-session-retired-epoch-repair.test.ts new file mode 100644 index 00000000000..95edbadbf58 --- /dev/null +++ b/src/renderer/src/runtime/local-structured-session-retired-epoch-repair.test.ts @@ -0,0 +1,198 @@ +/** + * A live publisher can return to a worktree after another one briefly owned it, and the epoch it + * returns under is already in `retired`. The fence rejects that frame — correctly, because it + * cannot tell it apart from a delayed frame queued by a dead generation — so the drop has to be + * repaired from authority instead of being final. + * + * The sequence below is the measured one: a renderer publication, a `removed:` retraction, a + * headless rebuild whose version restarts at 1, then the same renderer epoch returning at a higher + * version carrying a newly published chat tab. + */ + +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { RuntimeMobileSessionTabsResult } from '../../../shared/runtime-types' +import type { Tab } from '../../../shared/tab-types' +import { + applyLocalStructuredSessionTabSnapshots, + resetLocalStructuredSessionVersionForTests +} from './local-structured-session-tabs-sync' +import { localStructuredSessionEpochHistoryByWorktree } from './local-structured-session-tabs-sync/inventory-generation-fence' +import type { WebSessionTabsSyncState } from './web-session-tabs-sync' +import { resetWebSessionFocusIntentForTests } from './web-session-focus-intent' + +const WORKTREE = 'folder:ws-1' +const ROOT_GROUP = 'local-root-group' +const RENDERER_EPOCH = 'renderer:53c8f87d' +const HEADLESS_EPOCH = 'headless:pty-backed:mtovsn3x' +const REMOVED_EPOCH = 'removed:mtovryl4' + +afterEach(() => { + resetWebSessionFocusIntentForTests() + resetLocalStructuredSessionVersionForTests() +}) + +/** + * The worktree must stay "known" or the trailing cursor sweep deletes its epoch history every + * round and nothing ever accumulates in `retired` — which makes this whole scenario vacuous. + */ +function stateWithCoordinatorTerminal(): WebSessionTabsSyncState { + const terminalTab: Tab = { + id: 'u-term-1', + entityId: 'term-1', + groupId: ROOT_GROUP, + worktreeId: WORKTREE, + contentType: 'terminal', + label: 'Terminal 1', + customLabel: null, + color: null, + sortOrder: 0, + createdAt: 1 + } + return { + activeBrowserTabId: null, + activeBrowserTabIdByWorktree: {}, + activeFileId: null, + activeFileIdByWorktree: {}, + activeGroupIdByWorktree: { [WORKTREE]: ROOT_GROUP }, + activeTabId: 'u-term-1', + activeTabIdByWorktree: { [WORKTREE]: 'u-term-1' }, + activeTabType: 'terminal', + activeTabTypeByWorktree: { [WORKTREE]: 'terminal' }, + activeWorktreeId: WORKTREE, + agentStatusByPaneKey: {}, + agentStatusEpoch: 0, + browserCertificateFailuresByPageId: {}, + browserPagesByWorkspace: {}, + browserTabsByWorktree: {}, + folderWorkspaces: [{ id: 'ws-1', name: 'ws', folderPath: '/tmp/ws' }], + groupsByWorktree: { + [WORKTREE]: [ + { id: ROOT_GROUP, worktreeId: WORKTREE, activeTabId: 'u-term-1', tabOrder: ['u-term-1'] } + ] + }, + layoutByWorktree: { [WORKTREE]: { type: 'leaf', groupId: ROOT_GROUP } }, + openFiles: [], + ptyIdsByTabId: { 'term-1': ['pty-1'] }, + remoteBrowserPageHandlesByPageId: {}, + tabBarOrderByWorktree: {}, + tabsByWorktree: {}, + terminalLayoutsByTabId: {}, + unifiedTabsByWorktree: { [WORKTREE]: [terminalTab] }, + unreadTerminalTabs: {}, + sortEpoch: 0 + } as unknown as WebSessionTabsSyncState +} + +function frame( + publicationEpoch: string, + snapshotVersion: number, + sessionId: string | null +): RuntimeMobileSessionTabsResult { + const id = sessionId ? `agent-session:${sessionId}` : null + return { + worktree: WORKTREE, + publicationEpoch, + snapshotVersion, + activeGroupId: ROOT_GROUP, + activeTabId: null, + activeTabType: null, + tabGroups: [{ id: ROOT_GROUP, activeTabId: null, tabOrder: id ? [id] : [] }], + tabs: id + ? [ + { + type: 'agent-session', + id, + title: 'Claude Chat', + sessionId, + agent: 'claude', + isActive: false + } + ] + : [] + } as RuntimeMobileSessionTabsResult +} + +function chatTabs(state: WebSessionTabsSyncState): string[] { + return (state.unifiedTabsByWorktree[WORKTREE] ?? []) + .filter((tab) => tab.contentType === 'agent-session') + .map((tab) => tab.label) +} + +/** Everything up to and including the drop; returns the state the repair has to fix. */ +function replayUntilDrop( + onRetiredEpochDrop?: (worktreeId: string, publicationEpoch: string) => void +): WebSessionTabsSyncState { + let state = stateWithCoordinatorTerminal() + state = applyLocalStructuredSessionTabSnapshots(state, [frame(RENDERER_EPOCH, 6, null)]) + state = applyLocalStructuredSessionTabSnapshots(state, [frame(REMOVED_EPOCH, 0, null)]) + state = applyLocalStructuredSessionTabSnapshots(state, [frame(HEADLESS_EPOCH, 1, null)]) + return applyLocalStructuredSessionTabSnapshots( + state, + [frame(RENDERER_EPOCH, 7, 'claude-1')], + undefined, + undefined, + onRetiredEpochDrop ? { onRetiredEpochDrop } : {} + ) +} + +describe('retired-epoch repair for a returning publisher', () => { + it('POSITIVE CONTROL: the same frame lands when no epoch has been retired', () => { + const applied = applyLocalStructuredSessionTabSnapshots(stateWithCoordinatorTerminal(), [ + frame(RENDERER_EPOCH, 7, 'claude-1') + ]) + + expect(chatTabs(applied)).toEqual(['Claude Chat']) + }) + + it('drops the returning publisher and reports it to the repair lane', () => { + const onRetiredEpochDrop = vi.fn() + + const dropped = replayUntilDrop(onRetiredEpochDrop) + + expect(chatTabs(dropped)).toEqual([]) + expect(onRetiredEpochDrop).toHaveBeenCalledWith(WORKTREE, RENDERER_EPOCH) + }) + + it('lands the tab when the authoritative census re-delivers the same frame', () => { + const dropped = replayUntilDrop() + expect(chatTabs(dropped)).toEqual([]) + + const repaired = applyLocalStructuredSessionTabSnapshots( + dropped, + [frame(RENDERER_EPOCH, 7, 'claude-1')], + undefined, + undefined, + { authoritative: true } + ) + + expect(chatTabs(repaired)).toEqual(['Claude Chat']) + }) + + it('a non-authoritative redelivery stays dropped, so only authority repairs it', () => { + const dropped = replayUntilDrop() + + const redelivered = applyLocalStructuredSessionTabSnapshots(dropped, [ + frame(RENDERER_EPOCH, 7, 'claude-1') + ]) + + expect(chatTabs(redelivered)).toEqual([]) + }) + + it('revives only the epoch authority names, leaving other generations fenced', () => { + const dropped = replayUntilDrop() + applyLocalStructuredSessionTabSnapshots( + dropped, + [frame(RENDERER_EPOCH, 7, 'claude-1')], + undefined, + undefined, + { authoritative: true } + ) + + // The census named the renderer epoch current, so the headless generation it displaced is now + // the retired one — and a delayed frame from it must still be rejected. + const history = localStructuredSessionEpochHistoryByWorktree.get(WORKTREE) + expect(history?.current).toBe(RENDERER_EPOCH) + expect(history?.retired).toContain(HEADLESS_EPOCH) + expect(history?.retired).not.toContain(RENDERER_EPOCH) + }) +}) diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-refresh.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-refresh.ts index af756458707..136e4fa2e30 100644 --- a/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-refresh.ts +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-refresh.ts @@ -21,9 +21,13 @@ export function restoreLocalStructuredSessionTabsOnce( ) } -/** Fetch the current host inventory even after the startup restore has settled. */ +/** Fetch the current host inventory even after the startup restore has settled. + * + * `authoritative` is opt-in and belongs to the repair lane alone: the startup restore stays + * fenced exactly as before, so nothing about first paint changes. */ export function refreshLocalStructuredSessionTabs( - expectedGeneration = localStructuredSessionGeneration() + expectedGeneration = localStructuredSessionGeneration(), + options: { authoritative?: boolean } = {} ): Promise { return window.api.runtime .call({ method: 'session.tabs.listAll', params: {} }) @@ -34,7 +38,7 @@ export function refreshLocalStructuredSessionTabs( const result = response.result as { snapshots?: RuntimeMobileSessionTabsResult[] } const snapshots = result.snapshots ?? [] if (isCurrentLocalStructuredSessionGeneration(expectedGeneration)) { - applyStructuredSessionTabSnapshots(snapshots) + applyStructuredSessionTabSnapshots(snapshots, undefined, options) } return snapshots }) diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync/retired-epoch-repair.test.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync/retired-epoch-repair.test.ts new file mode 100644 index 00000000000..563a7c3598b --- /dev/null +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync/retired-epoch-repair.test.ts @@ -0,0 +1,121 @@ +/** + * The repair lane must not become worse than the bug it fixes: a publisher that keeps re-sending a + * retired epoch would otherwise drive an unbounded refetch loop, and a cap that never decays would + * hide a chat tab for the renderer's lifetime after a run of transient RPC failures. + */ + +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { localStructuredSessionEpochHistoryByWorktree } from './inventory-generation-fence' +import { + forgetRetiredEpochRepairsOutside, + resetRetiredEpochRepairsForTests, + scheduleRetiredEpochRepair +} from './retired-epoch-repair' + +const WORKTREE = 'folder:ws-1' +const EPOCH = 'renderer:53c8f87d' + +const runRepair = vi.fn(async (_generation: number) => undefined) + +function markRetired(): void { + localStructuredSessionEpochHistoryByWorktree.set(WORKTREE, { + current: 'headless:pty-backed:x', + retired: [EPOCH] + }) +} + +beforeEach(() => { + vi.useFakeTimers() + runRepair.mockReset() + runRepair.mockImplementation(async () => undefined) + resetRetiredEpochRepairsForTests() + localStructuredSessionEpochHistoryByWorktree.clear() +}) + +afterEach(() => { + resetRetiredEpochRepairsForTests() + localStructuredSessionEpochHistoryByWorktree.clear() + vi.useRealTimers() +}) + +describe('retired-epoch repair scheduling', () => { + it('asks the host once for a burst of drops on one worktree', async () => { + markRetired() + + scheduleRetiredEpochRepair(WORKTREE, EPOCH, runRepair) + scheduleRetiredEpochRepair(WORKTREE, EPOCH, runRepair) + scheduleRetiredEpochRepair(WORKTREE, EPOCH, runRepair) + expect(runRepair).not.toHaveBeenCalled() + + await vi.advanceTimersByTimeAsync(300) + + expect(runRepair).toHaveBeenCalledTimes(1) + }) + + it('stops after a bounded number of attempts and says so', async () => { + markRetired() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + + // The epoch stays retired, so every refresh counts as a failed repair. + for (let attempt = 0; attempt < 6; attempt += 1) { + scheduleRetiredEpochRepair(WORKTREE, EPOCH, runRepair) + await vi.advanceTimersByTimeAsync(5000) + } + + expect(runRepair).toHaveBeenCalledTimes(3) + expect(warn).toHaveBeenCalledWith( + '[structured-session-tabs] retired publication epoch still unrepaired', + expect.objectContaining({ worktree: WORKTREE, publicationEpoch: EPOCH }) + ) + warn.mockRestore() + }) + + it('decays the cap instead of latching, so a later drop is still repairable', async () => { + markRetired() + vi.spyOn(console, 'warn').mockImplementation(() => {}) + + for (let attempt = 0; attempt < 4; attempt += 1) { + scheduleRetiredEpochRepair(WORKTREE, EPOCH, runRepair) + await vi.advanceTimersByTimeAsync(5000) + } + expect(runRepair).toHaveBeenCalledTimes(3) + + // A quiet minute later the worktree gets its budget back rather than staying hidden forever. + await vi.advanceTimersByTimeAsync(60_000) + scheduleRetiredEpochRepair(WORKTREE, EPOCH, runRepair) + await vi.advanceTimersByTimeAsync(300) + + expect(runRepair).toHaveBeenCalledTimes(4) + }) + + it('rearms once a repair actually revives the epoch', async () => { + markRetired() + runRepair.mockImplementation(async () => { + localStructuredSessionEpochHistoryByWorktree.set(WORKTREE, { + current: EPOCH, + retired: ['headless:pty-backed:x'] + }) + }) + + scheduleRetiredEpochRepair(WORKTREE, EPOCH, runRepair) + await vi.advanceTimersByTimeAsync(300) + expect(runRepair).toHaveBeenCalledTimes(1) + + markRetired() + scheduleRetiredEpochRepair(WORKTREE, EPOCH, runRepair) + await vi.advanceTimersByTimeAsync(300) + + expect(runRepair).toHaveBeenCalledTimes(2) + }) + + it('forgets repair state for worktrees that no longer exist', async () => { + markRetired() + scheduleRetiredEpochRepair(WORKTREE, EPOCH, runRepair) + + forgetRetiredEpochRepairsOutside(new Set(['folder:other'])) + await vi.advanceTimersByTimeAsync(5000) + + // The pending refetch for the vanished worktree is cancelled, not merely orphaned. + expect(runRepair).not.toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync/retired-epoch-repair.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync/retired-epoch-repair.ts new file mode 100644 index 00000000000..1916fb21546 --- /dev/null +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync/retired-epoch-repair.ts @@ -0,0 +1,131 @@ +/** + * Repairing a structured session snapshot the retired-epoch fence rejected. + * + * The fence is right to reject a subscription frame carrying a retired epoch — it cannot tell that + * frame apart from a delayed one queued by a dead publisher generation, whose version can be + * higher than the live cursor. What it cannot do is notice when the epoch's publisher is actually + * still alive and has simply returned after another publisher briefly owned the worktree. + * + * So the drop is not treated as final: it schedules one authoritative `session.tabs.listAll`, whose + * answer settles which epoch is current. If the epoch really is dead the census changes nothing; if + * it is live, the census carries it and the tab lands. The fence itself is never relaxed for + * subscription frames. + * + * The refresh is injected rather than imported so this module depends on nothing that in turn + * depends on the snapshot apply — which is what lets the apply prune this module's state. + */ + +import { + isCurrentLocalStructuredSessionGeneration, + localStructuredSessionEpochHistoryByWorktree, + localStructuredSessionGeneration +} from './inventory-generation-fence' + +/** Bounded so a publisher that keeps re-sending a retired epoch cannot drive an endless refetch. */ +const MAX_REPAIR_ATTEMPTS = 3 +const BASE_REPAIR_DELAY_MS = 250 +const MAX_REPAIR_DELAY_MS = 5000 +/** + * The cap decays rather than latching. A run of transient RPC failures must not hide a chat tab for + * the renderer's lifetime; once a worktree has been quiet this long, a fresh drop is a fresh + * problem and gets its full budget back. + */ +const REPAIR_ATTEMPT_DECAY_MS = 60_000 + +type RepairState = { + attempts: number + lastAttemptAt: number + timer: ReturnType | null +} + +export type RetiredEpochRepairRunner = (expectedGeneration: number) => Promise + +const repairsByWorktree = new Map() + +function repairState(worktreeId: string, now: number): RepairState { + const existing = repairsByWorktree.get(worktreeId) + if (!existing) { + const created: RepairState = { attempts: 0, lastAttemptAt: now, timer: null } + repairsByWorktree.set(worktreeId, created) + return created + } + if (now - existing.lastAttemptAt >= REPAIR_ATTEMPT_DECAY_MS) { + existing.attempts = 0 + } + return existing +} + +/** + * Schedules the authoritative refetch for a dropped snapshot, coalescing repeat drops for the same + * worktree into the one already pending. + */ +export function scheduleRetiredEpochRepair( + worktreeId: string, + publicationEpoch: string, + runRepair: RetiredEpochRepairRunner +): void { + const now = Date.now() + const state = repairState(worktreeId, now) + if (state.timer !== null) { + return + } + if (state.attempts >= MAX_REPAIR_ATTEMPTS) { + console.warn('[structured-session-tabs] retired publication epoch still unrepaired', { + worktree: worktreeId, + publicationEpoch, + attempts: state.attempts, + retryAfterMs: Math.max(0, REPAIR_ATTEMPT_DECAY_MS - (now - state.lastAttemptAt)) + }) + return + } + const generation = localStructuredSessionGeneration() + const delay = Math.min(BASE_REPAIR_DELAY_MS * 2 ** state.attempts, MAX_REPAIR_DELAY_MS) + state.attempts += 1 + state.lastAttemptAt = now + state.timer = setTimeout(() => { + state.timer = null + if (!isCurrentLocalStructuredSessionGeneration(generation)) { + repairsByWorktree.delete(worktreeId) + return + } + void runRepair(generation) + .then(() => { + // Why re-check rather than trust the call: a refresh that succeeds without reviving the + // epoch has not repaired anything, and counting it as success would loop forever. + const stillRetired = + localStructuredSessionEpochHistoryByWorktree + .get(worktreeId) + ?.retired.includes(publicationEpoch) ?? false + if (!stillRetired) { + repairsByWorktree.delete(worktreeId) + } + }) + .catch((error) => { + console.warn('[structured-session-tabs] retired-epoch repair refresh failed', error) + }) + }, delay) +} + +/** + * Drops repair state for worktrees that no longer exist, alongside the publisher cursors it + * shadows — without this every deleted worktree leaks an entry for the renderer's lifetime. + */ +export function forgetRetiredEpochRepairsOutside(knownWorktreeIds: ReadonlySet): void { + for (const [worktreeId, state] of repairsByWorktree) { + if (!knownWorktreeIds.has(worktreeId)) { + if (state.timer !== null) { + clearTimeout(state.timer) + } + repairsByWorktree.delete(worktreeId) + } + } +} + +export function resetRetiredEpochRepairsForTests(): void { + for (const state of repairsByWorktree.values()) { + if (state.timer !== null) { + clearTimeout(state.timer) + } + } + repairsByWorktree.clear() +} diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync/snapshot-apply.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync/snapshot-apply.ts index fc254de62dc..ad0d99bfeac 100644 --- a/src/renderer/src/runtime/local-structured-session-tabs-sync/snapshot-apply.ts +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync/snapshot-apply.ts @@ -7,7 +7,9 @@ import { } from '../web-session-tabs-sync' import type { WebSessionTabsSyncState } from '../web-session-tabs-sync' import { + hasRetiredValue, noteRetiredValue, + reviveRetiredValue, sameSessionTabsPublicationLineage } from '../web-session-tabs-sync/publisher-identity-fences' import { @@ -21,16 +23,30 @@ import { localStructuredSessionVersionByWorktree, supersedeLocalStructuredSessionGeneration } from './inventory-generation-fence' +import { forgetRetiredEpochRepairsOutside } from './retired-epoch-repair' import { projectLocalStructuredSessionTabs } from './snapshot-projection' export const LOCAL_STRUCTURED_SESSION_OWNER = 'local-structured-session' +export type StructuredSessionSnapshotApplyOptions = { + /** + * Marks these snapshots as an authoritative `session.tabs.listAll` response, which exempts them + * from the retired-epoch fence. A census is the synchronous answer to a request we just issued, + * so it cannot be the delayed frame from a dead generation that the fence exists to reject — + * whereas a subscription frame can be, and stays fenced. + */ + authoritative?: boolean + /** Called for each snapshot the retired-epoch fence rejects; the repair lane listens here. */ + onRetiredEpochDrop?: (worktreeId: string, publicationEpoch: string) => void +} + export function applyStructuredSessionTabSnapshots( snapshots: readonly RuntimeMobileSessionTabsResult[], - owner = LOCAL_STRUCTURED_SESSION_OWNER + owner = LOCAL_STRUCTURED_SESSION_OWNER, + options: StructuredSessionSnapshotApplyOptions = {} ): void { const settleStructuredSessionMirror = applyWebSessionTabsStorePatch( - (state) => applyLocalStructuredSessionTabSnapshots(state, snapshots, owner), + (state) => applyLocalStructuredSessionTabSnapshots(state, snapshots, owner, undefined, options), { frames: [] } ) settleStructuredSessionMirror() @@ -65,7 +81,8 @@ export function applyLocalStructuredSessionTabSnapshots< state: State, snapshots: readonly RuntimeMobileSessionTabsResult[], owner = LOCAL_STRUCTURED_SESSION_OWNER, - now = Date.now() + now = Date.now(), + options: StructuredSessionSnapshotApplyOptions = {} ): State { let next = state for (const snapshot of snapshots) { @@ -78,8 +95,17 @@ export function applyLocalStructuredSessionTabSnapshots< prior && sameSessionTabsPublicationLineage(prior.publicationEpoch, snapshot.publicationEpoch) ) const epochHistory = localStructuredSessionEpochHistoryByWorktree.get(snapshot.worktree) - if (epochHistory?.retired.includes(snapshot.publicationEpoch) && !sharesLineage) { - continue + // Why not just drop: an epoch is retired whenever another publisher takes over the worktree, + // but a live publisher can return after transient interlopers (a `removed:` retraction, then a + // headless rebuild), and the structured publish inherits the worktree's existing epoch rather + // than minting its own. So a retired epoch is not proof of a dead generation — only authority + // can settle it, and the repair lane goes and asks. + if (hasRetiredValue(epochHistory, snapshot.publicationEpoch) && !sharesLineage) { + if (!options.authoritative) { + options.onRetiredEpochDrop?.(snapshot.worktree, snapshot.publicationEpoch) + continue + } + reviveRetiredValue(epochHistory, snapshot.publicationEpoch) } if (prior && sharesLineage && snapshot.snapshotVersion <= prior.snapshotVersion) { continue @@ -114,5 +140,6 @@ export function applyLocalStructuredSessionTabSnapshots< localStructuredSessionEpochHistoryByWorktree.delete(worktreeId) } } + forgetRetiredEpochRepairsOutside(knownWorktreeIds) return next } diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync/subscription.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync/subscription.ts index b074cfb1c41..fef55f07a16 100644 --- a/src/renderer/src/runtime/local-structured-session-tabs-sync/subscription.ts +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync/subscription.ts @@ -9,7 +9,20 @@ import { refreshLocalStructuredSessionTabs, restoreLocalStructuredSessionTabsOnce } from './inventory-refresh' -import { applyStructuredSessionTabSnapshots } from './snapshot-apply' +import { scheduleRetiredEpochRepair } from './retired-epoch-repair' +import { + applyStructuredSessionTabSnapshots, + type StructuredSessionSnapshotApplyOptions +} from './snapshot-apply' + +// The refresh is supplied here rather than imported by the repair lane, so nothing the snapshot +// apply depends on depends back on it. +const REPAIR_DROPPED_EPOCHS: StructuredSessionSnapshotApplyOptions = { + onRetiredEpochDrop: (worktreeId, publicationEpoch) => + scheduleRetiredEpochRepair(worktreeId, publicationEpoch, (generation) => + refreshLocalStructuredSessionTabs(generation, { authoritative: true }) + ) +} type SessionTabsEvent = | (RuntimeMobileSessionTabsResult & { type: 'snapshot' | 'updated' }) @@ -84,9 +97,9 @@ export async function startLocalStructuredSessionTabsSync(args: { } const event = response.result as SessionTabsEvent if (event.type === 'snapshots') { - applyStructuredSessionTabSnapshots(event.snapshots) + applyStructuredSessionTabSnapshots(event.snapshots, undefined, REPAIR_DROPPED_EPOCHS) } else if (event.type === 'snapshot' || event.type === 'updated') { - applyStructuredSessionTabSnapshots([event]) + applyStructuredSessionTabSnapshots([event], undefined, REPAIR_DROPPED_EPOCHS) } else if (event.type === 'end' && generation === subscriptionGeneration) { // Reattach with one refresh so a runtime-restart boundary cannot strand stale tabs. subscriptionGeneration += 1 diff --git a/src/renderer/src/runtime/web-session-tabs-sync/publisher-identity-fences.ts b/src/renderer/src/runtime/web-session-tabs-sync/publisher-identity-fences.ts index 96ebe2293c6..fbab01b591b 100644 --- a/src/renderer/src/runtime/web-session-tabs-sync/publisher-identity-fences.ts +++ b/src/renderer/src/runtime/web-session-tabs-sync/publisher-identity-fences.ts @@ -36,6 +36,19 @@ export function noteRetiredValue( return history } +/** + * Un-retires one value, leaving every other retired generation fenced. + * + * Only an authority that names the value current may call this; reviving on a delayed frame's own + * say-so is exactly the resurrection `retired` exists to prevent. + */ +export function reviveRetiredValue(history: RetiredValueHistory | undefined, value: string): void { + const index = history?.retired.indexOf(value) ?? -1 + if (history && index >= 0) { + history.retired.splice(index, 1) + } +} + function normalizeSessionTabsRuntimeId(runtimeId: unknown): string | undefined { if (typeof runtimeId !== 'string') { return undefined From fd10758eae985564b9f3258074fe6f175a47364e Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 19:13:55 -0700 Subject: [PATCH 04/69] ci: expose existing E2E spec selection for manual dispatch (#18987) --- .github/workflows/e2e.yml | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index 3560d302a79..5f80c2090ad 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -27,6 +27,10 @@ on: description: Ref to check out (defaults to the workflow ref) required: false type: string + test_files: + description: JSON array of specs to run; empty runs the full suite + required: false + type: string schedule: # Why: GitHub cron uses UTC; these slots map to 10am and 3pm # America/Phoenix for the default-branch E2E run. From 681119dc05bab24469037ab50ce0be6c3cb1faa2 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 19:33:47 -0700 Subject: [PATCH 05/69] test: isolate window mocks from inherited launch flags (#18989) * test: isolate mocked window activation from inherited launch flags * Preserve background window regressions added on main --- src/main/ipc/dashboard-popout.test.ts | 8 ++++++++ src/main/ipc/notifications-retention-lifecycle.test.ts | 10 +++++++++- .../window/createMainWindow-startup-reveal.test.ts | 10 +++++++++- src/main/window/dashboard-popout-window.test.ts | 8 ++++++++ src/main/window/focus-existing-window.test.ts | 8 +++++++- 5 files changed, 41 insertions(+), 3 deletions(-) diff --git a/src/main/ipc/dashboard-popout.test.ts b/src/main/ipc/dashboard-popout.test.ts index ce10b3556fc..9bead362816 100644 --- a/src/main/ipc/dashboard-popout.test.ts +++ b/src/main/ipc/dashboard-popout.test.ts @@ -97,6 +97,14 @@ function makeStore(enabled = true) { } } +// These cases exercise foreground behavior against Electron mocks. +beforeEach(() => { + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', undefined) + vi.stubEnv('ORCA_E2E_HEADLESS', undefined) + vi.stubEnv('ORCA_E2E_HEADFUL', undefined) +}) +afterEach(() => vi.unstubAllEnvs()) + describe('registerDashboardPopoutHandlers', () => { let store: ReturnType diff --git a/src/main/ipc/notifications-retention-lifecycle.test.ts b/src/main/ipc/notifications-retention-lifecycle.test.ts index 278ca8f13ae..9705cf85a58 100644 --- a/src/main/ipc/notifications-retention-lifecycle.test.ts +++ b/src/main/ipc/notifications-retention-lifecycle.test.ts @@ -1,4 +1,4 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { getAllWindowsMock, @@ -30,6 +30,14 @@ vi.mock('../tray/system-tray', async () => import { registerNotificationHandlers } from './notifications' +// These cases exercise foreground behavior against Electron mocks. +beforeEach(() => { + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', undefined) + vi.stubEnv('ORCA_E2E_HEADLESS', undefined) + vi.stubEnv('ORCA_E2E_HEADFUL', undefined) +}) +afterEach(() => vi.unstubAllEnvs()) + describe('registerNotificationHandlers', () => { beforeEach(() => { vi.useFakeTimers() diff --git a/src/main/window/createMainWindow-startup-reveal.test.ts b/src/main/window/createMainWindow-startup-reveal.test.ts index f103881ea83..bd7eeadc6b2 100644 --- a/src/main/window/createMainWindow-startup-reveal.test.ts +++ b/src/main/window/createMainWindow-startup-reveal.test.ts @@ -1,4 +1,4 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' vi.mock('electron', async () => (await import('./createMainWindow-test-harness')).electronModuleMock() @@ -22,6 +22,14 @@ import { withPlatform } from './createMainWindow-test-harness' +// These cases exercise foreground behavior against Electron mocks. +beforeEach(() => { + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', undefined) + vi.stubEnv('ORCA_E2E_HEADLESS', undefined) + vi.stubEnv('ORCA_E2E_HEADFUL', undefined) +}) +afterEach(() => vi.unstubAllEnvs()) + describe('createMainWindow', () => { beforeEach(() => { resetMainWindowMocks() diff --git a/src/main/window/dashboard-popout-window.test.ts b/src/main/window/dashboard-popout-window.test.ts index 86735811bb8..0e64d573f76 100644 --- a/src/main/window/dashboard-popout-window.test.ts +++ b/src/main/window/dashboard-popout-window.test.ts @@ -171,6 +171,14 @@ function makeStore(ui: Record = {}): { const RENDERER_URL = 'http://localhost:5173' +// These cases exercise foreground behavior against Electron mocks. +beforeEach(() => { + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', undefined) + vi.stubEnv('ORCA_E2E_HEADLESS', undefined) + vi.stubEnv('ORCA_E2E_HEADFUL', undefined) +}) +afterEach(() => vi.unstubAllEnvs()) + describe('createOrFocusDashboardPopout', () => { beforeEach(() => { instances.length = 0 diff --git a/src/main/window/focus-existing-window.test.ts b/src/main/window/focus-existing-window.test.ts index 9f5dc522150..697b423ab37 100644 --- a/src/main/window/focus-existing-window.test.ts +++ b/src/main/window/focus-existing-window.test.ts @@ -1,5 +1,5 @@ import type { App, BrowserWindow } from 'electron' -import { afterEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { focusExistingMainWindow } from './focus-existing-window' type FakeWindowOptions = { @@ -78,6 +78,12 @@ function makeTimer(): { } } +// These cases exercise foreground behavior against Electron mocks. +beforeEach(() => { + vi.stubEnv('ORCA_BACKGROUND_LAUNCH', undefined) + vi.stubEnv('ORCA_E2E_HEADLESS', undefined) + vi.stubEnv('ORCA_E2E_HEADFUL', undefined) +}) afterEach(() => vi.unstubAllEnvs()) describe('focusExistingMainWindow', () => { From bedbe5997ba86cc43d8142fa21d92f512a559567 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 19:45:42 -0700 Subject: [PATCH 06/69] test: match explorer filenames independently of git badges (#18997) --- tests/e2e/file-explorer-watch-refresh.spec.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/e2e/file-explorer-watch-refresh.spec.ts b/tests/e2e/file-explorer-watch-refresh.spec.ts index d8a1bc1e4e7..85fbd66409e 100644 --- a/tests/e2e/file-explorer-watch-refresh.spec.ts +++ b/tests/e2e/file-explorer-watch-refresh.spec.ts @@ -35,7 +35,7 @@ test('refreshes the visible tree after external Windows file changes', async ({ const row = (name: string) => orcaPage .locator('[data-file-explorer-row]') - .filter({ hasText: new RegExp(`^${name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}$`) }) + .filter({ has: orcaPage.getByText(name, { exact: true }) }) rmSync(originalPath, { force: true }) rmSync(renamedPath, { force: true }) From 712cf1facb124149c7076bc1f80912e0196b043d Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 19:57:29 -0700 Subject: [PATCH 07/69] test: synchronize large repository recovery with Retry request (#18999) --- tests/e2e/helpers/git-status-retry-barrier.ts | 61 +++++++++++++++++++ .../git-status-retry-barrier.unit.test.ts | 39 ++++++++++++ .../source-control-large-file-count.spec.ts | 19 ++++-- 3 files changed, 114 insertions(+), 5 deletions(-) create mode 100644 tests/e2e/helpers/git-status-retry-barrier.ts create mode 100644 tests/e2e/helpers/git-status-retry-barrier.unit.test.ts diff --git a/tests/e2e/helpers/git-status-retry-barrier.ts b/tests/e2e/helpers/git-status-retry-barrier.ts new file mode 100644 index 00000000000..e166313d003 --- /dev/null +++ b/tests/e2e/helpers/git-status-retry-barrier.ts @@ -0,0 +1,61 @@ +import type { ElectronApplication } from '@stablyai/playwright-test' + +type StatusArgs = { worktreePath?: string; admissionTier?: string } +type StatusHandler = (event: unknown, args?: StatusArgs) => unknown +type RetryBarrier = { + captured: boolean + release: () => void + original: StatusHandler +} +type BarrierScope = typeof globalThis & { __gitStatusRetryBarrier?: RetryBarrier } + +export async function installGitStatusRetryBarrier( + app: ElectronApplication, + repoPath: string +): Promise { + await app.evaluate(({ ipcMain }, repoPath) => { + const scope = globalThis as BarrierScope + const handlers = (ipcMain as unknown as { _invokeHandlers: Map }) + ._invokeHandlers + const original = handlers.get('git:status') + if (!original || scope.__gitStatusRetryBarrier) { + throw new Error('Git status handler unavailable or retry barrier already installed') + } + let release!: () => void + const pending = new Promise((resolve) => { + release = resolve + }) + const state: RetryBarrier = { captured: false, release, original } + scope.__gitStatusRetryBarrier = state + handlers.set('git:status', async (event, args) => { + if ( + !state.captured && + args?.worktreePath === repoPath && + args.admissionTier === 'interactive' + ) { + state.captured = true + await pending + } + return original(event, args) + }) + }, repoPath) +} + +export async function hasCapturedGitStatusRetry(app: ElectronApplication): Promise { + return app.evaluate(() => (globalThis as BarrierScope).__gitStatusRetryBarrier?.captured ?? false) +} + +export async function restoreGitStatusRetryHandler(app: ElectronApplication): Promise { + await app.evaluate(({ ipcMain }) => { + const scope = globalThis as BarrierScope + const state = scope.__gitStatusRetryBarrier + if (!state) { + return + } + const handlers = (ipcMain as unknown as { _invokeHandlers: Map }) + ._invokeHandlers + handlers.set('git:status', state.original) + state.release() + delete scope.__gitStatusRetryBarrier + }) +} diff --git a/tests/e2e/helpers/git-status-retry-barrier.unit.test.ts b/tests/e2e/helpers/git-status-retry-barrier.unit.test.ts new file mode 100644 index 00000000000..74bdd8d8159 --- /dev/null +++ b/tests/e2e/helpers/git-status-retry-barrier.unit.test.ts @@ -0,0 +1,39 @@ +import type { ElectronApplication } from '@stablyai/playwright-test' +import { describe, expect, it, vi } from 'vitest' +import { + hasCapturedGitStatusRetry, + installGitStatusRetryBarrier, + restoreGitStatusRetryHandler +} from './git-status-retry-barrier' + +describe('Git status retry barrier', () => { + it('holds the target interactive request and restores the real handler on cleanup', async () => { + const original = vi.fn(async (_event: unknown, args: unknown) => args) + const handlers = new Map([['git:status', original]]) + const app = { + evaluate: (callback: (electron: unknown, arg?: unknown) => unknown, arg?: unknown) => + Promise.resolve(callback({ ipcMain: { _invokeHandlers: handlers } }, arg)) + } as unknown as ElectronApplication + await installGitStatusRetryBarrier(app, 'target-repo') + try { + const handler = handlers.get('git:status')! + const background = { worktreePath: 'target-repo', admissionTier: 'background' } + const otherRepo = { worktreePath: 'another-repo', admissionTier: 'interactive' } + await expect(handler({}, background)).resolves.toEqual(background) + await expect(handler({}, otherRepo)).resolves.toEqual(otherRepo) + expect(await hasCapturedGitStatusRetry(app)).toBe(false) + + const retry = { worktreePath: 'target-repo', admissionTier: 'interactive' } + const event = {} + const pending = handler(event, retry) + expect(await hasCapturedGitStatusRetry(app)).toBe(true) + expect(original).toHaveBeenCalledTimes(2) + await restoreGitStatusRetryHandler(app) + await expect(pending).resolves.toEqual(retry) + expect(original).toHaveBeenLastCalledWith(event, retry) + expect(handlers.get('git:status')).toBe(original) + } finally { + await restoreGitStatusRetryHandler(app) + } + }) +}) diff --git a/tests/e2e/source-control-large-file-count.spec.ts b/tests/e2e/source-control-large-file-count.spec.ts index 28a5e7758f2..85f3710b308 100644 --- a/tests/e2e/source-control-large-file-count.spec.ts +++ b/tests/e2e/source-control-large-file-count.spec.ts @@ -20,6 +20,11 @@ import type { ElectronApplication, Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' import { waitForSessionReady } from './helpers/store' +import { + hasCapturedGitStatusRetry, + installGitStatusRetryBarrier, + restoreGitStatusRetryHandler +} from './helpers/git-status-retry-barrier' import { createLargeFileCountRepo, removeLargeFileCountRepo, @@ -434,13 +439,17 @@ test.describe('Source Control large file count (#8013)', () => { ) expect(hugeState).not.toBeNull() - // Why: watcher refreshes stay parked while huge; the visible Retry is the - // explicit recovery path after the underlying change count drops. - removeLargeFileCountUntrackedTree(fixture.repoPath) - await expect(tooManyChangesBanner).toBeVisible() const retryButton = tooManyChangesBanner.locator('..').getByRole('button', { name: 'Retry' }) await expect(retryButton).toBeVisible() - await retryButton.click() + // Keep automatic refreshes from removing Retry before its real request starts. + await installGitStatusRetryBarrier(electronApp, fixture.repoPath) + try { + await retryButton.click() + await expect.poll(() => hasCapturedGitStatusRetry(electronApp)).toBe(true) + removeLargeFileCountUntrackedTree(fixture.repoPath) + } finally { + await restoreGitStatusRetryHandler(electronApp) + } await expect(tooManyChangesBanner).not.toBeVisible() await expect .poll(() => From 1c41d59203d1d68c30b85d3e5f8a86478e45aa40 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:02:20 -0700 Subject: [PATCH 08/69] perf(relay): drain fragmented frame buffers in linear time (#18891) * perf(relay): drain fragmented frame buffers in linear time * style: follow block-body lint in relay buffer checks --- .../scripts/relay-frame-buffer-benchmark.mjs | 62 +++++++++++++++++ src/shared/relay-frame-buffer.test.ts | 69 +++++++++++++++++++ src/shared/relay-frame-buffer.ts | 45 ++++++++---- 3 files changed, 163 insertions(+), 13 deletions(-) create mode 100644 config/scripts/relay-frame-buffer-benchmark.mjs create mode 100644 src/shared/relay-frame-buffer.test.ts diff --git a/config/scripts/relay-frame-buffer-benchmark.mjs b/config/scripts/relay-frame-buffer-benchmark.mjs new file mode 100644 index 00000000000..24d7b565400 --- /dev/null +++ b/config/scripts/relay-frame-buffer-benchmark.mjs @@ -0,0 +1,62 @@ +#!/usr/bin/env node +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { stripTypeScriptTypes } from 'node:module' +import { performance } from 'node:perf_hooks' + +// Pass the pre-change source saved with git show :src/shared/relay-frame-buffer.ts. +const baselinePath = process.argv[2] +if (!baselinePath) { + throw new Error('Usage: node config/scripts/relay-frame-buffer-benchmark.mjs ') +} +async function load(source) { + return ( + await import( + `data:text/javascript;base64,${Buffer.from(stripTypeScriptTypes(source)).toString('base64')}` + ) + ).RelayFrameBuffer +} +const Before = await load(readFileSync(baselinePath, 'utf8')) +const After = await load( + readFileSync(new URL('../../src/shared/relay-frame-buffer.ts', import.meta.url), 'utf8') +) +function median(values) { + return values.sort((a, b) => a - b)[Math.floor(values.length / 2)] +} +for (const count of [1, 256, 16384, 65536]) { + const chunks = Array.from({ length: count }, (_, index) => Buffer.alloc(64, index % 256)) + const expected = Buffer.concat(chunks) + for (const mode of ['take', 'discard']) { + const times = [[], []] + for (let round = 0; round < 9; round += 1) { + for (const arm of round % 2 === 0 ? [0, 1] : [1, 0]) { + const FrameBuffer = arm === 0 ? Before : After + const buffer = new FrameBuffer() + for (const chunk of chunks) { + buffer.append(chunk) + } + const start = performance.now() + const output = buffer[mode](expected.length) + times[arm].push(performance.now() - start) + if (mode === 'take') { + assert.deepEqual(output, expected) + } + assert.equal(buffer.length, 0) + buffer.append(Buffer.from('tail')) + assert.equal(buffer.drain().toString(), 'tail') + } + } + const beforeMs = median(times[0]), + afterMs = median(times[1]) + console.log( + JSON.stringify({ + mode, + chunks: count, + bytes: expected.length, + beforeMs, + afterMs, + speedup: beforeMs / afterMs + }) + ) + } +} diff --git a/src/shared/relay-frame-buffer.test.ts b/src/shared/relay-frame-buffer.test.ts new file mode 100644 index 00000000000..55dd744e57a --- /dev/null +++ b/src/shared/relay-frame-buffer.test.ts @@ -0,0 +1,69 @@ +import { describe, expect, it, vi } from 'vitest' +import { RelayFrameBuffer } from './relay-frame-buffer' + +describe('RelayFrameBuffer', () => { + it('preserves a byte stream across fragmented peeks, takes, discards and drains', () => { + const buffer = new RelayFrameBuffer() + let expected = Buffer.alloc(0) + for (let step = 0; step < 5000; step += 1) { + const chunk = Buffer.from([step % 256, (step + 1) % 256, (step + 2) % 256]) + buffer.append(chunk) + expected = Buffer.concat([expected, chunk]) + if (step % 3 === 0) { + const count = Math.min(expected.length, 5) + expect(buffer.peek(count).subarray(0, count)).toEqual(expected.subarray(0, count)) + expect(buffer.take(count)).toEqual(expected.subarray(0, count)) + expected = expected.subarray(count) + } + if (step % 7 === 0) { + const count = Math.min(expected.length, 4) + buffer.discard(count) + expected = expected.subarray(count) + } + if (step % 101 === 0) { + expect(buffer.drain()).toEqual(expected) + expected = Buffer.alloc(0) + } + expect(buffer.length).toBe(expected.length) + } + expect(buffer.drain()).toEqual(expected) + expect(buffer.drain()).toEqual(Buffer.alloc(0)) + }) + + it('releases consumed references and amortizes storage compaction in a large backlog', () => { + const buffer = new RelayFrameBuffer() + const chunks = Array.from({ length: 32768 }, (_, index) => Buffer.from([index % 256])) + for (const chunk of chunks) { + buffer.append(chunk) + } + const shifted = vi.spyOn(Array.prototype, 'shift') + let shiftCount: number + try { + buffer.discard(16000) + shiftCount = shifted.mock.calls.length + } finally { + shifted.mockRestore() + } + expect(shiftCount).toBe(0) + const storage = buffer as unknown as { chunks: (Buffer | undefined)[]; head: number } + expect(storage.chunks.slice(0, storage.head).every((chunk) => chunk === undefined)).toBe(true) + expect(buffer.take(1000)).toEqual(Buffer.concat(chunks.slice(16000, 17000))) + expect(storage.chunks.length).toBeLessThan(chunks.length) + expect(buffer.drain()).toEqual(Buffer.concat(chunks.slice(17000))) + expect(storage.chunks).toHaveLength(0) + expect(buffer.length).toBe(0) + }) + + it('keeps single-chunk views and clears partial data before reuse', () => { + const buffer = new RelayFrameBuffer() + const chunk = Buffer.from('abcdef') + buffer.append(chunk) + expect(buffer.peek(2)).toBe(chunk) + const taken = buffer.take(2) + expect(taken.buffer).toBe(chunk.buffer) + expect(taken.toString()).toBe('ab') + buffer.clear() + buffer.append(Buffer.from('fresh')) + expect(buffer.drain().toString()).toBe('fresh') + }) +}) diff --git a/src/shared/relay-frame-buffer.ts b/src/shared/relay-frame-buffer.ts index 5083804f1e1..a851426c6d8 100644 --- a/src/shared/relay-frame-buffer.ts +++ b/src/shared/relay-frame-buffer.ts @@ -1,5 +1,6 @@ export class RelayFrameBuffer { - private chunks: Buffer[] = [] + private chunks: (Buffer | undefined)[] = [] + private head = 0 private bytes = 0 get length(): number { @@ -13,23 +14,28 @@ export class RelayFrameBuffer { clear(): void { this.chunks = [] + this.head = 0 this.bytes = 0 } drain(): Buffer { - const out = this.chunks.length === 1 ? this.chunks[0] : Buffer.concat(this.chunks, this.bytes) + const out = + this.chunks.length - this.head === 1 + ? this.chunks[this.head]! + : Buffer.concat(this.chunks.slice(this.head) as Buffer[], this.bytes) this.clear() return out } peek(count: number): Buffer { - const first = this.chunks[0] + const first = this.chunks[this.head]! if (first.length >= count) { return first } const out = Buffer.allocUnsafe(count) let copied = 0 - for (const part of this.chunks) { + for (let index = this.head; index < this.chunks.length; index += 1) { + const part = this.chunks[index]! copied += part.copy(out, copied, 0, Math.min(part.length, count - copied)) if (copied >= count) { break @@ -39,43 +45,56 @@ export class RelayFrameBuffer { } take(count: number): Buffer { - const first = this.chunks[0] + const first = this.chunks[this.head]! if (first.length === count) { - this.chunks.shift() + this.removeHead() this.bytes -= count return first } if (first.length > count) { - this.chunks[0] = first.subarray(count) + this.chunks[this.head] = first.subarray(count) this.bytes -= count return first.subarray(0, count) } const out = Buffer.allocUnsafe(count) let copied = 0 while (copied < count) { - const part = this.chunks[0] + const part = this.chunks[this.head]! const take = Math.min(part.length, count - copied) part.copy(out, copied, 0, take) copied += take if (take === part.length) { - this.chunks.shift() + this.removeHead() } else { - this.chunks[0] = part.subarray(take) + this.chunks[this.head] = part.subarray(take) } } this.bytes -= count return out } + private removeHead(): void { + this.chunks[this.head] = undefined + this.head += 1 + // Amortize compaction without retaining consumed buffers. + if ( + this.head === this.chunks.length || + (this.head >= 1024 && this.head * 2 >= this.chunks.length) + ) { + this.chunks = this.chunks.slice(this.head) + this.head = 0 + } + } + discard(count: number): void { let remaining = count while (remaining > 0) { - const part = this.chunks[0] + const part = this.chunks[this.head]! if (part.length <= remaining) { - this.chunks.shift() + this.removeHead() remaining -= part.length } else { - this.chunks[0] = part.subarray(remaining) + this.chunks[this.head] = part.subarray(remaining) remaining = 0 } } From bf87b1290f612fed83fb5e4bb514a0d2d0446b0b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:02:25 -0700 Subject: [PATCH 09/69] perf(repos): avoid quadratic icon source scans (#18892) * perf(repos): avoid quadratic icon source scans * perf: avoid repeated malformed HTML icon scans * bench: balance icon parser timing samples --- .../repo-icon-source-href-benchmark.mjs | 55 ++++++++++++++ src/main/repo-icon-file-detection.test.ts | 56 +++++++++++++- src/main/repo-icon-file-detection.ts | 10 +-- src/main/repo-icon-source-href.test.ts | 76 +++++++++++++++++++ src/main/repo-icon-source-href.ts | 46 +++++++++++ 5 files changed, 233 insertions(+), 10 deletions(-) create mode 100644 config/scripts/repo-icon-source-href-benchmark.mjs create mode 100644 src/main/repo-icon-source-href.test.ts create mode 100644 src/main/repo-icon-source-href.ts diff --git a/config/scripts/repo-icon-source-href-benchmark.mjs b/config/scripts/repo-icon-source-href-benchmark.mjs new file mode 100644 index 00000000000..76c42d261b4 --- /dev/null +++ b/config/scripts/repo-icon-source-href-benchmark.mjs @@ -0,0 +1,55 @@ +import assert from 'node:assert/strict' +import { performance } from 'node:perf_hooks' +import { extractIconHref } from '../../src/main/repo-icon-source-href.ts' + +// Original production expressions, preserved for the before/after measurement. +const html = + /]*\brel=["'](?:icon|shortcut icon)["'])(?=[^>]*\bhref=["']([^"'?]+))[^>]*>/i +const object = + /(?=[^}]*\brel\s*:\s*["'](?:icon|shortcut icon)["'])(?=[^}]*\bhref\s*:\s*["']([^"'?]+))[^}]*/i +const original = (source) => source.match(html)?.[1] ?? source.match(object)?.[1] ?? null + +function measurePair(source) { + original(source) + extractIconHref(source) + const beforeSamples = [] + const afterSamples = [] + for (let run = 0; run < 5; run++) { + const measurements = [ + [original, beforeSamples], + [extractIconHref, afterSamples] + ] + if (run % 2 === 1) { + measurements.reverse() + } + for (const [fn, samples] of measurements) { + const started = performance.now() + fn(source) + samples.push(performance.now() - started) + } + } + return { + beforeMs: beforeSamples.sort((a, b) => a - b)[2], + afterMs: afterSamples.sort((a, b) => a - b)[2] + } +} + +const results = [] +for (const size of [8192, 16384, 32768]) { + for (const shape of ['no icon', 'rel without href', 'unterminated link starts']) { + const source = + shape === 'unterminated link starts' + ? ' { expect(stat).toHaveBeenCalled() }) }) + +describe('declared repo icons through production filesystem routes', () => { + it.each([ + ['local', false], + ['ssh', false], + ['local', true], + ['ssh', true] + ] as const)('preserves declared icon detection on %s (no icon: %s)', async (kind, noIcon) => { + const directory = await mkdtemp(join(tmpdir(), 'orca-icon-href-')) + const source = noIcon + ? 'a'.repeat(256 * 1024) + : `${'a'.repeat(32768)}{ rel: "icon", href: "/first.png", href: "/chosen.png" }` + try { + await mkdir(join(directory, 'public')) + await writeFile(join(directory, 'index.html'), source) + await writeFile(join(directory, 'public', 'chosen.png'), Buffer.from(PNG_BASE64, 'base64')) + const provider = remoteFilesystemProvider({ + stat: async (path) => { + const info = await stat(path) + return { + type: info.isFile() ? 'file' : 'directory', + size: info.size, + mtime: info.mtimeMs + } + }, + readFile: async (path) => { + const buffer = await readFile(path) + const isBinary = path.endsWith('.png') + return { + content: buffer.toString(isBinary ? 'base64' : 'utf8'), + isBinary, + mimeType: isBinary ? 'image/png' : 'text/html' + } + } + }) + const route: ExecutionHostFilesystemRoute = + kind === 'local' ? { kind: 'local', hostId: 'local' } : sshRoute('icon-oracle', provider) + await expect(detectRepoFileIcon(directory, route)).resolves.toEqual( + noIcon + ? null + : { + type: 'image', + src: `data:image/png;base64,${PNG_BASE64}`, + source: 'file', + label: 'public/chosen.png' + } + ) + } finally { + await rm(directory, { recursive: true, force: true }) + } + }) +}) diff --git a/src/main/repo-icon-file-detection.ts b/src/main/repo-icon-file-detection.ts index f4289924b08..831248b94b3 100644 --- a/src/main/repo-icon-file-detection.ts +++ b/src/main/repo-icon-file-detection.ts @@ -3,6 +3,7 @@ import { buildImageDataUri } from '../shared/image-data-uri' import { MAX_REPO_ICON_UPLOAD_BYTES, type RepoIcon } from '../shared/repo-icon' import type { ExecutionHostFilesystemRoute } from './providers/execution-host-provider-dispatch' import type { IFilesystemProvider } from './providers/types' +import { extractIconHref } from './repo-icon-source-href' import { iconHrefCandidates } from './repo-icon-href-candidates' import { joinWorktreeRelativePath } from './runtime/runtime-relative-paths' @@ -49,11 +50,6 @@ const REPO_ICON_SOURCE_FILE_CANDIDATES = [ // not read large app entrypoints just to find a small favicon href. const MAX_REPO_ICON_SOURCE_BYTES = 256 * 1024 -const LINK_ICON_HTML_RE = - /]*\brel=["'](?:icon|shortcut icon)["'])(?=[^>]*\bhref=["']([^"'?]+))[^>]*>/i -const LINK_ICON_OBJECT_RE = - /(?=[^}]*\brel\s*:\s*["'](?:icon|shortcut icon)["'])(?=[^}]*\bhref\s*:\s*["']([^"'?]+))[^}]*/i - type DetectedImageFormat = { mimeType: 'image/png' | 'image/webp' } @@ -98,10 +94,6 @@ function detectImageFormat(buffer: Buffer): DetectedImageFormat | null { return null } -function extractIconHref(source: string): string | null { - return source.match(LINK_ICON_HTML_RE)?.[1] ?? source.match(LINK_ICON_OBJECT_RE)?.[1] ?? null -} - function repoIconFromImageBuffer(buffer: Buffer, relativePath: string): RepoIcon | null { const format = detectImageFormat(buffer) if (!format) { diff --git a/src/main/repo-icon-source-href.test.ts b/src/main/repo-icon-source-href.test.ts new file mode 100644 index 00000000000..7c74511f3d6 --- /dev/null +++ b/src/main/repo-icon-source-href.test.ts @@ -0,0 +1,76 @@ +import { describe, expect, it } from 'vitest' +import { extractIconHref } from './repo-icon-source-href' + +// Original production expressions are the compatibility oracle. +const HTML_RE = + /]*\brel=["'](?:icon|shortcut icon)["'])(?=[^>]*\bhref=["']([^"'?]+))[^>]*>/i +const OBJECT_RE = + /(?=[^}]*\brel\s*:\s*["'](?:icon|shortcut icon)["'])(?=[^}]*\bhref\s*:\s*["']([^"'?]+))[^}]*/i + +export function originalIconHref(source: string): string | null { + return source.match(HTML_RE)?.[1] ?? source.match(OBJECT_RE)?.[1] ?? null +} + +describe('repo icon source href compatibility', () => { + it.each([ + '', + '', + '', + '', + '', + '', + 'plain source without icon properties', + '{ rel: "icon", href: "/first.png", href: "/last.png" }', + '{ href: "/first.png", rel: "icon", rel: "stylesheet" }', + '{ rel: "icon" } { href: "/unrelated.png" }', + '{ rel: "icon", href: "/first.png" } { rel: "icon", href: "/last.png" }', + '{ rel: "icon", href: "/object.png" } ', + '{ rel: "ICON", href: "/UPPER.png" }', + '}\n\n{href: "/line.png",\nrel : "shortcut icon"}', + '{ rel: "icon", href: "/unterminated}after brace', + '{ rel: "icon", href: "?query" }', + '{ rel: "icon", href: "/before?query" }', + '', + ' { + expect(extractIconHref(source)).toBe(originalIconHref(source)) + }) + + it('matches the original across generated malformed property sequences', () => { + const tokens = [ + '}', + '{', + ' ', + 'rel:"icon"', + 'href:"a"', + 'href:"b"', + 'rel:"other"', + 'x', + '\n', + '', + 'rel="icon"', + 'href="a"', + 'href="b"' + ] + let seed = 97 + for (let sample = 0; sample < 3000; sample++) { + let source = '' + for (let token = 0; token < 12; token++) { + seed = (Math.imul(seed, 1664525) + 1013904223) >>> 0 + source += tokens[seed % tokens.length] + } + expect(extractIconHref(source), source).toBe(originalIconHref(source)) + } + }) + + it('handles a maximum-size unterminated HTML tag region', () => { + expect(extractIconHref(' { + expect(extractIconHref('a'.repeat(256 * 1024))).toBeNull() + }) +}) diff --git a/src/main/repo-icon-source-href.ts b/src/main/repo-icon-source-href.ts new file mode 100644 index 00000000000..b8363da3711 --- /dev/null +++ b/src/main/repo-icon-source-href.ts @@ -0,0 +1,46 @@ +const LINK_START_RE = /]*\brel=["'](?:icon|shortcut icon)["'])(?=[^>]*\bhref=["']([^"'?]+))[^>]*>/iy +const LINK_ICON_OBJECT_RE = + /(?=[^}]*\brel\s*:\s*["'](?:icon|shortcut icon)["'])(?=[^}]*\bhref\s*:\s*["']([^"'?]+))[^}]*/iy + +function extractHtmlIconHref(source: string): string | null { + LINK_START_RE.lastIndex = 0 + let match: RegExpExecArray | null + while ((match = LINK_START_RE.exec(source))) { + LINK_ICON_HTML_RE.lastIndex = match.index + const href = LINK_ICON_HTML_RE.exec(source)?.[1] + if (href !== undefined) { + return href + } + // Later link starts before the same closing angle see only a subset of these attributes. + const closingAngle = source.indexOf('>', LINK_START_RE.lastIndex) + if (closingAngle === -1) { + return null + } + LINK_START_RE.lastIndex = closingAngle + 1 + } + return null +} + +export function extractIconHref(source: string): string | null { + const htmlHref = extractHtmlIconHref(source) + if (htmlHref !== null) { + return htmlHref + } + let start = 0 + while (start <= source.length) { + // Every suffix before the next closing brace sees the same candidate properties. + LINK_ICON_OBJECT_RE.lastIndex = start + const href = LINK_ICON_OBJECT_RE.exec(source)?.[1] + if (href !== undefined) { + return href + } + const closingBrace = source.indexOf('}', start) + if (closingBrace === -1) { + return null + } + start = closingBrace + 1 + } + return null +} From 926f3ff5859170f3bfafb54dc0386d20f39a4c7c Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:02:30 -0700 Subject: [PATCH 10/69] perf(browser): assemble fragmented tunnel frames once (#18893) --- .../benchmark-browser-tunnel-framing.mjs | 110 ++++++++++++++++++ ...wser-network-tunnel-stream-framing.test.ts | 92 +++++++++++++++ .../browser-network-tunnel-stream-framing.ts | 65 +++++++---- 3 files changed, 246 insertions(+), 21 deletions(-) create mode 100644 config/scripts/benchmark-browser-tunnel-framing.mjs diff --git a/config/scripts/benchmark-browser-tunnel-framing.mjs b/config/scripts/benchmark-browser-tunnel-framing.mjs new file mode 100644 index 00000000000..e91fd0887f6 --- /dev/null +++ b/config/scripts/benchmark-browser-tunnel-framing.mjs @@ -0,0 +1,110 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { readFileSync } from 'node:fs' +import { stripTypeScriptTypes } from 'node:module' +import { performance } from 'node:perf_hooks' + +// Run from the worktree root: node config/scripts/benchmark-browser-tunnel-framing.mjs [base-ref] +const path = 'src/shared/browser-network-tunnel-stream-framing.ts' +const baselineRef = process.argv[2] ?? 'HEAD' +const beforeSource = execFileSync('git', ['show', `${baselineRef}:${path}`], { + encoding: 'utf8' +}) +const afterSource = readFileSync(path, 'utf8') +const load = (source) => + import( + `data:text/javascript;base64,${Buffer.from( + stripTypeScriptTypes(source, { mode: 'transform' }) + ).toString('base64')}` + ) +const before = await load(beforeSource) +const after = await load(afterSource) + +function measure(module, chunks, payload, repetitions) { + let frameCount = 0 + let lastFrame + const onFrame = (frame) => { + frameCount++ + lastFrame = frame + } + const onError = (error) => { + throw error + } + const run = () => { + const decoder = new module.BrowserNetworkTunnelStreamFrameDecoder(onFrame, onError) + for (const chunk of chunks) { + decoder.feed(chunk) + } + } + run() + assert.deepEqual(lastFrame, payload) + const samples = [] + for (let sample = 0; sample < 5; sample++) { + const start = performance.now() + for (let iteration = 0; iteration < repetitions; iteration++) { + run() + } + samples.push((performance.now() - start) / repetitions) + } + assert.equal(frameCount, 1 + 5 * repetitions) + return samples.sort((a, b) => a - b)[2] +} + +function countCopies(module, chunks) { + const originalSet = Uint8Array.prototype.set + const originalSlice = Uint8Array.prototype.slice + let copied = 0 + Uint8Array.prototype.set = function (source, offset) { + copied += source.length + return originalSet.call(this, source, offset) + } + Uint8Array.prototype.slice = function (...args) { + const result = originalSlice.apply(this, args) + copied += result.length + return result + } + try { + const decoder = new module.BrowserNetworkTunnelStreamFrameDecoder( + () => {}, + (error) => { + throw error + } + ) + for (const chunk of chunks) { + decoder.feed(chunk) + } + } finally { + Uint8Array.prototype.set = originalSet + Uint8Array.prototype.slice = originalSlice + } + return copied +} + +const rows = [] +for (const [payloadBytes, chunkBytes, repetitions] of [ + [1, 5, 10000], + [64 * 1024, 65540, 1000], + [64 * 1024, 4096, 100], + [64 * 1024, 256, 25], + [64 * 1024, 16, 5], + [64 * 1024, 1, 1] +]) { + const payload = Uint8Array.from({ length: payloadBytes }, (_, index) => index % 251) + const encoded = before.encodeBrowserNetworkTunnelStreamFrame(payload) + const chunks = [] + for (let offset = 0; offset < encoded.length; offset += chunkBytes) { + chunks.push(encoded.subarray(offset, offset + chunkBytes)) + } + const beforeMs = measure(before, chunks, payload, repetitions) + const afterMs = measure(after, chunks, payload, repetitions) + rows.push({ + payloadBytes, + chunkBytes, + beforeMs: +beforeMs.toFixed(6), + afterMs: +afterMs.toFixed(6), + speedup: +(beforeMs / afterMs).toFixed(2), + beforeCopiedBytes: countCopies(before, chunks), + afterCopiedBytes: countCopies(after, chunks) + }) +} +console.log(JSON.stringify({ node: process.version, baselineRef, rows }, null, 2)) diff --git a/src/shared/browser-network-tunnel-stream-framing.test.ts b/src/shared/browser-network-tunnel-stream-framing.test.ts index 75eb07d0996..621f297a525 100644 --- a/src/shared/browser-network-tunnel-stream-framing.test.ts +++ b/src/shared/browser-network-tunnel-stream-framing.test.ts @@ -55,6 +55,98 @@ describe('browser network tunnel stream framing', () => { ) }) + it.each([1, 2, 3, 16, 256, 4096, 65556])( + 'preserves maximum-size frames split into %i-byte chunks', + (chunkSize) => { + const payload = Uint8Array.from({ length: 65552 }, (_, index) => index % 251) + const encoded = encodeBrowserNetworkTunnelStreamFrame(payload) + const frames: Uint8Array[] = [] + const onError = vi.fn() + const decoder = new BrowserNetworkTunnelStreamFrameDecoder( + (frame) => frames.push(frame), + onError + ) + for (let offset = 0; offset < encoded.length; offset += chunkSize) { + decoder.feed(encoded.subarray(offset, offset + chunkSize)) + } + expect(frames).toEqual([payload]) + expect(onError).not.toHaveBeenCalled() + } + ) + + it('copies fragmented bytes once instead of recopying the growing carry', () => { + const encoded = encodeBrowserNetworkTunnelStreamFrame(new Uint8Array(65536)) + const decoder = new BrowserNetworkTunnelStreamFrameDecoder( + () => {}, + () => {} + ) + const originalSet = Uint8Array.prototype.set + let copiedBytes = 0 + const set = vi + .spyOn(Uint8Array.prototype, 'set') + .mockImplementation(function (this: Uint8Array, source, offset) { + copiedBytes += source.length + originalSet.call(this, source, offset) + }) + try { + for (const byte of encoded) { + decoder.feed(new Uint8Array([byte])) + } + expect(copiedBytes).toBe(encoded.length) + } finally { + set.mockRestore() + } + }) + + it('owns partial input and emitted frames independently of caller buffers', () => { + const frames: Uint8Array[] = [] + const decoder = new BrowserNetworkTunnelStreamFrameDecoder( + (frame) => frames.push(frame), + () => {} + ) + const first = new Uint8Array([0, 0, 0, 3, 1]) + decoder.feed(first) + first.fill(255) + const rest = new Uint8Array([2, 3]) + decoder.feed(rest) + rest.fill(255) + decoder.feed(encodeBrowserNetworkTunnelStreamFrame(new Uint8Array([4]))) + expect(frames).toEqual([new Uint8Array([1, 2, 3]), new Uint8Array([4])]) + }) + + it('enforces the retained cap before decoding complete frames in a feed', () => { + const onFrame = vi.fn() + const onError = vi.fn() + const decoder = new BrowserNetworkTunnelStreamFrameDecoder(onFrame, onError, 16, 8) + decoder.feed(new Uint8Array([0, 0])) + decoder.feed(new Uint8Array([0, 1, 7, 0, 0, 0, 1])) + decoder.feed(new Uint8Array([8])) + expect(onFrame).not.toHaveBeenCalled() + expect(onError).toHaveBeenCalledExactlyOnceWith( + expect.objectContaining({ message: 'browser_tunnel_stream_buffer_overflow' }) + ) + }) + + it('counts the retained header and payload against the exact cap', () => { + const onFrame = vi.fn() + const onError = vi.fn() + const decoder = new BrowserNetworkTunnelStreamFrameDecoder(onFrame, onError, 16, 7) + decoder.feed(new Uint8Array([0, 0, 0, 3, 1])) + decoder.feed(new Uint8Array([2, 3])) + expect(onFrame).toHaveBeenCalledExactlyOnceWith(new Uint8Array([1, 2, 3])) + expect(onError).not.toHaveBeenCalled() + }) + + it('stops decoding coalesced frames when the callback closes the decoder', () => { + const onFrame = vi.fn(() => decoder.close()) + const onError = vi.fn() + const decoder = new BrowserNetworkTunnelStreamFrameDecoder(onFrame, onError) + decoder.feed(new Uint8Array([0, 0, 0, 1, 7, 0, 0, 0, 1, 8])) + decoder.feed(new Uint8Array([0, 0, 0, 1, 9])) + expect(onFrame).toHaveBeenCalledExactlyOnceWith(new Uint8Array([7])) + expect(onError).not.toHaveBeenCalled() + }) + it('serializes writes and rejects bounded queue overflow', () => { const callbacks: ((error?: Error | null) => void)[] = [] const writes: Uint8Array[] = [] diff --git a/src/shared/browser-network-tunnel-stream-framing.ts b/src/shared/browser-network-tunnel-stream-framing.ts index 734a304520c..042224253ee 100644 --- a/src/shared/browser-network-tunnel-stream-framing.ts +++ b/src/shared/browser-network-tunnel-stream-framing.ts @@ -15,7 +15,10 @@ export function encodeBrowserNetworkTunnelStreamFrame(frame: Uint8Array): Uint8A } export class BrowserNetworkTunnelStreamFrameDecoder { - private retained = new Uint8Array() + private readonly header = new Uint8Array(LENGTH_BYTES) + private headerBytes = 0 + private frame: Uint8Array | null = null + private frameBytes = 0 private closed = false constructor( @@ -29,30 +32,50 @@ export class BrowserNetworkTunnelStreamFrameDecoder { if (this.closed || chunk.byteLength === 0) { return } - if (this.retained.byteLength + chunk.byteLength > this.maxRetainedBytes) { + if (this.headerBytes + this.frameBytes + chunk.byteLength > this.maxRetainedBytes) { this.fail(new Error('browser_tunnel_stream_buffer_overflow')) return } - const combined = new Uint8Array(this.retained.byteLength + chunk.byteLength) - combined.set(this.retained) - combined.set(chunk, this.retained.byteLength) let offset = 0 - while (combined.byteLength - offset >= LENGTH_BYTES) { - const length = new DataView( - combined.buffer, - combined.byteOffset + offset, - LENGTH_BYTES - ).getUint32(0, false) - if (length === 0 || length > this.maxFrameBytes) { - this.fail(new Error('browser_tunnel_stream_frame_invalid')) + while (offset < chunk.byteLength) { + if (this.headerBytes < LENGTH_BYTES) { + let length: number + if (this.headerBytes === 0 && chunk.byteLength - offset >= LENGTH_BYTES) { + length = new DataView(chunk.buffer, chunk.byteOffset + offset, LENGTH_BYTES).getUint32( + 0, + false + ) + this.headerBytes = LENGTH_BYTES + offset += LENGTH_BYTES + } else { + const count = Math.min(LENGTH_BYTES - this.headerBytes, chunk.byteLength - offset) + this.header.set(chunk.subarray(offset, offset + count), this.headerBytes) + this.headerBytes += count + offset += count + if (this.headerBytes < LENGTH_BYTES) { + return + } + length = new DataView(this.header.buffer).getUint32(0, false) + } + if (length === 0 || length > this.maxFrameBytes) { + this.fail(new Error('browser_tunnel_stream_frame_invalid')) + return + } + this.frame = new Uint8Array(length) + } + const frame = this.frame! + const count = Math.min(frame.byteLength - this.frameBytes, chunk.byteLength - offset) + frame.set(chunk.subarray(offset, offset + count), this.frameBytes) + this.frameBytes += count + offset += count + if (this.frameBytes < frame.byteLength) { return } - const end = offset + LENGTH_BYTES + length - if (end > combined.byteLength) { - break - } + this.frame = null + this.frameBytes = 0 + this.headerBytes = 0 try { - this.onFrame(combined.slice(offset + LENGTH_BYTES, end)) + this.onFrame(frame) } catch (error) { this.fail(error instanceof Error ? error : new Error(String(error))) return @@ -60,14 +83,14 @@ export class BrowserNetworkTunnelStreamFrameDecoder { if (this.closed) { return } - offset = end } - this.retained = combined.slice(offset) } close(): void { this.closed = true - this.retained = new Uint8Array() + this.frame = null + this.frameBytes = 0 + this.headerBytes = 0 } private fail(error: Error): void { From 5a33e2acb05286d4aa03382e8bcfeeb023078f59 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:02:34 -0700 Subject: [PATCH 11/69] perf(editor): reuse Markdown source blocks while positioning review notes (#18895) --- .../rich-markdown-review-annotations.ts | 8 +- .../rich-markdown-review-note-positioning.ts | 11 +- ...ich-markdown-review-rail-benchmark.test.ts | 154 ++++++++++++++++++ .../rich-markdown-review-rail-blocks.test.ts | 139 ++++++++++++++++ .../rich-markdown-review-rail-blocks.ts | 31 ++++ 5 files changed, 334 insertions(+), 9 deletions(-) create mode 100644 src/renderer/src/components/editor/rich-markdown-review-rail-benchmark.test.ts create mode 100644 src/renderer/src/components/editor/rich-markdown-review-rail-blocks.test.ts create mode 100644 src/renderer/src/components/editor/rich-markdown-review-rail-blocks.ts diff --git a/src/renderer/src/components/editor/rich-markdown-review-annotations.ts b/src/renderer/src/components/editor/rich-markdown-review-annotations.ts index 619d128bb65..90bd170a5c1 100644 --- a/src/renderer/src/components/editor/rich-markdown-review-annotations.ts +++ b/src/renderer/src/components/editor/rich-markdown-review-annotations.ts @@ -139,7 +139,7 @@ export function getRichMarkdownAnnotationHighlightRangesForComment( comment: DiffComment, markdownSourceLineOffset: number, // Why optional: callers looping over comments pass one shared build. - prebuiltBlocks?: RichMarkdownCommentBlock[] + prebuiltBlocks?: readonly RichMarkdownCommentBlock[] ): RichMarkdownAnnotationHighlightRange[] { const blocks = prebuiltBlocks ?? buildRichMarkdownCommentBlocks(editor) const selectedText = comment.selectedText?.trim() @@ -192,13 +192,15 @@ export function getRichMarkdownCommentAnchorTop( block: RichMarkdownCommentBlock, containerRect: DOMRect, containerScrollTop: number, - markdownSourceLineOffset: number + markdownSourceLineOffset: number, + prebuiltBlocks?: readonly RichMarkdownCommentBlock[] ): number | null { try { const ranges = getRichMarkdownAnnotationHighlightRangesForComment( editor, comment, - markdownSourceLineOffset + markdownSourceLineOffset, + prebuiltBlocks ) // Why: range notes should sort by the start of the selected text. Anchoring // to the end puts overlapping ranges with the same final line in creation diff --git a/src/renderer/src/components/editor/rich-markdown-review-note-positioning.ts b/src/renderer/src/components/editor/rich-markdown-review-note-positioning.ts index c54996d0ffa..4fdc12a51b9 100644 --- a/src/renderer/src/components/editor/rich-markdown-review-note-positioning.ts +++ b/src/renderer/src/components/editor/rich-markdown-review-note-positioning.ts @@ -1,9 +1,7 @@ import type { Editor } from '@tiptap/react' import type { DiffComment } from '../../../../shared/diff-comment-types' -import { - buildRichMarkdownCommentBlocks, - getRichMarkdownCommentAnchorTop -} from './rich-markdown-review-annotations' +import { getRichMarkdownCommentAnchorTop } from './rich-markdown-review-annotations' +import { getRichMarkdownReviewRailBlocks } from './rich-markdown-review-rail-blocks' import { stackRichMarkdownReviewNotePositions, type RichMarkdownReviewNotePosition @@ -23,7 +21,7 @@ export function measureRichMarkdownReviewNotePositions({ markdownSourceLineOffset }: MeasureRichMarkdownReviewNotePositionsOptions): RichMarkdownReviewNotePosition[] { const containerRect = container.getBoundingClientRect() - const blocks = buildRichMarkdownCommentBlocks(editor) + const blocks = getRichMarkdownReviewRailBlocks(editor) const nextPositions = markdownComments .map((comment): RichMarkdownReviewNotePosition | null => { const bodyLineNumber = Math.max(1, comment.lineNumber - markdownSourceLineOffset) @@ -39,7 +37,8 @@ export function measureRichMarkdownReviewNotePositions({ block, containerRect, container.scrollTop, - markdownSourceLineOffset + markdownSourceLineOffset, + blocks ) return top === null ? null : { comment, top } }) diff --git a/src/renderer/src/components/editor/rich-markdown-review-rail-benchmark.test.ts b/src/renderer/src/components/editor/rich-markdown-review-rail-benchmark.test.ts new file mode 100644 index 00000000000..7090769f71f --- /dev/null +++ b/src/renderer/src/components/editor/rich-markdown-review-rail-benchmark.test.ts @@ -0,0 +1,154 @@ +// @vitest-environment happy-dom +// Run: ORCA_REVIEW_RAIL_BENCH=1 pnpm test src/renderer/src/components/editor/rich-markdown-review-rail-benchmark.test.ts +import { expect, it, vi } from 'vitest' +import { Editor as TiptapEditor } from '@tiptap/core' +import { createRichMarkdownExtensions } from './rich-markdown-extensions' +import { createRichMarkdownEditorCodec } from './rich-markdown-source-transport' +import { measureRichMarkdownReviewNotePositions } from './rich-markdown-review-note-positioning' + +// Baseline measurement body from the parent revision, before sharing source blocks. +import type { Editor } from '@tiptap/react' +import type { DiffComment } from '../../../../shared/diff-comment-types' +import { + buildRichMarkdownCommentBlocks, + getRichMarkdownCommentAnchorTop +} from './rich-markdown-review-annotations' +import { + stackRichMarkdownReviewNotePositions, + type RichMarkdownReviewNotePosition +} from './rich-markdown-review-note-layout' + +type MeasureRichMarkdownReviewNotePositionsOptions = { + container: HTMLDivElement + editor: Editor + markdownComments: DiffComment[] + markdownSourceLineOffset: number +} + +function measureBaseline({ + container, + editor, + markdownComments, + markdownSourceLineOffset +}: MeasureRichMarkdownReviewNotePositionsOptions): RichMarkdownReviewNotePosition[] { + const containerRect = container.getBoundingClientRect() + const blocks = buildRichMarkdownCommentBlocks(editor) + const nextPositions = markdownComments + .map((comment): RichMarkdownReviewNotePosition | null => { + const bodyLineNumber = Math.max(1, comment.lineNumber - markdownSourceLineOffset) + const block = blocks.find( + (candidate) => candidate.startLine <= bodyLineNumber && bodyLineNumber <= candidate.endLine + ) + if (!block) { + return null + } + const top = getRichMarkdownCommentAnchorTop( + editor, + comment, + block, + containerRect, + container.scrollTop, + markdownSourceLineOffset + ) + return top === null ? null : { comment, top } + }) + .filter((position): position is RichMarkdownReviewNotePosition => position !== null) + return stackRichMarkdownReviewNotePositions( + nextPositions, + measureReviewNoteHeights(container, nextPositions) + ) +} + +function measureReviewNoteHeights( + container: HTMLDivElement, + positions: RichMarkdownReviewNotePosition[] +): Map { + const measuredHeights = new Map() + for (const pos of positions) { + const el = container.querySelector(`[data-rich-markdown-review-note-id="${pos.comment.id}"]`) + if (el) { + measuredHeights.set(pos.comment.id, el.getBoundingClientRect().height) + } + } + return measuredHeights +} + +it.skipIf(process.env.ORCA_REVIEW_RAIL_BENCH !== '1')( + 'benchmarks full review rail measurements', + () => { + for (const blockCount of [250, 1000]) { + const editor = new TiptapEditor({ + element: document.createElement('div'), + extensions: createRichMarkdownExtensions({ codec: createRichMarkdownEditorCodec() }), + content: { + type: 'doc', + content: Array.from({ length: blockCount }, (_, index) => ({ + type: 'paragraph', + content: [{ type: 'text', text: `Paragraph ${index} with reviewable content.` }] + })) + } + }) + try { + vi.spyOn(editor.view, 'coordsAtPos').mockReturnValue({ + top: 100, + bottom: 120, + left: 0, + right: 10 + }) + const container = document.createElement('div') + const markdownComments: DiffComment[] = Array.from({ length: 5 }, (_, index) => ({ + id: `note-${index}`, + worktreeId: 'workspace', + filePath: 'notes.md', + source: 'markdown', + lineNumber: 1 + index * 40, + body: 'Review', + createdAt: index, + side: 'modified' + })) + const args = { editor, container, markdownComments, markdownSourceLineOffset: 0 } + const serialize = vi.spyOn(editor.markdown!, 'serialize') + const baseline = measureBaseline(args) + const baselineCalls = serialize.mock.calls.length + serialize.mockClear() + expect(measureRichMarkdownReviewNotePositions(args)).toEqual(baseline) + const coldCalls = serialize.mock.calls.length + serialize.mockClear() + expect(measureRichMarkdownReviewNotePositions(args)).toEqual(baseline) + const warmCalls = serialize.mock.calls.length + serialize.mockRestore() + const time = (run: () => unknown) => { + const start = performance.now() + for (let index = 0; index < 20; index++) { + run() + } + return (performance.now() - start) / 20 + } + const before: number[] = [] + const after: number[] = [] + for (let round = 0; round < 5; round++) { + before.push(time(() => measureBaseline(args))) + after.push(time(() => measureRichMarkdownReviewNotePositions(args))) + } + const median = (values: number[]) => values.sort((a, b) => a - b)[2]! + const result = { + blockCount, + comments: markdownComments.length, + beforeMs: median(before), + afterMs: median(after), + baselineCalls, + coldCalls, + warmCalls + } + process.stdout.write(`${JSON.stringify(result)}\n`) + expect(baselineCalls).toBe((2 * blockCount - 1) * 6) + expect(coldCalls).toBe(2 * blockCount - 1) + expect(warmCalls).toBe(0) + } finally { + editor.destroy() + vi.restoreAllMocks() + } + } + }, + 120_000 +) diff --git a/src/renderer/src/components/editor/rich-markdown-review-rail-blocks.test.ts b/src/renderer/src/components/editor/rich-markdown-review-rail-blocks.test.ts new file mode 100644 index 00000000000..18c00a47fb5 --- /dev/null +++ b/src/renderer/src/components/editor/rich-markdown-review-rail-blocks.test.ts @@ -0,0 +1,139 @@ +// @vitest-environment happy-dom +import { afterEach, describe, expect, it, vi } from 'vitest' +import { Editor, type JSONContent } from '@tiptap/core' +import type { DiffComment } from '../../../../shared/diff-comment-types' +import { createRichMarkdownExtensions } from './rich-markdown-extensions' +import { createRichMarkdownEditorCodec } from './rich-markdown-source-transport' +import { buildRichMarkdownCommentBlocks } from './rich-markdown-review-annotations' +import { getRichMarkdownReviewRailBlocks } from './rich-markdown-review-rail-blocks' +import { measureRichMarkdownReviewNotePositions } from './rich-markdown-review-note-positioning' + +const editors: Editor[] = [] + +function createEditor( + content: string | JSONContent = '# Heading\n\nFirst paragraph\n\n- One\n- Two' +) { + const editor = new Editor({ + element: document.createElement('div'), + extensions: createRichMarkdownExtensions({ codec: createRichMarkdownEditorCodec() }), + content, + ...(typeof content === 'string' ? { contentType: 'markdown' as const } : {}) + }) + editors.push(editor) + // Settle the trailing-node plugin before measuring selection-only transactions. + editor.view.dispatch(editor.state.tr.setMeta('addToHistory', false)) + return editor +} + +afterEach(() => { + for (const editor of editors.splice(0)) { + editor.destroy() + } + vi.restoreAllMocks() +}) + +describe('review rail source block reuse', () => { + it.each([ + '```ts\nconst value = 1\n\nvalue++\n```\n\nAfter code', + '| First | Second |\n| --- | --- |\n| one | two |\n\nAfter table', + '
Toggle\n\nInside\n\n
\n\nAfter toggle' + ])('preserves multiline block boundaries for %s', (source) => { + const editor = createEditor(source) + expect(getRichMarkdownReviewRailBlocks(editor)).toEqual(buildRichMarkdownCommentBlocks(editor)) + }) + + it('preserves block lines and reuses them after selection-only transactions', () => { + const editor = createEditor() + const expected = buildRichMarkdownCommentBlocks(editor) + const serialize = vi.spyOn(editor.markdown!, 'serialize') + const blocks = getRichMarkdownReviewRailBlocks(editor) + expect(blocks).toEqual(expected) + expect(serialize).toHaveBeenCalledTimes(2 * editor.state.doc.childCount - 1) + serialize.mockClear() + editor.commands.setTextSelection(3) + expect(getRichMarkdownReviewRailBlocks(editor)).toBe(blocks) + expect(serialize).not.toHaveBeenCalled() + }) + + it('rebuilds after edits and undo while keeping editors isolated', () => { + const editor = createEditor() + const original = getRichMarkdownReviewRailBlocks(editor) + editor.commands.insertContentAt(1, 'changed\ntext') + const changed = getRichMarkdownReviewRailBlocks(editor) + expect(changed).not.toBe(original) + expect(changed).toEqual(buildRichMarkdownCommentBlocks(editor)) + editor.commands.undo() + expect(getRichMarkdownReviewRailBlocks(editor)).toEqual(original) + expect(getRichMarkdownReviewRailBlocks(createEditor())).not.toBe(original) + }) + + it('invalidates an in-place serializer replacement and a manager replacement', () => { + const editor = createEditor() + const original = getRichMarkdownReviewRailBlocks(editor) + const serialize = editor.markdown!.serialize.bind(editor.markdown) + vi.spyOn(editor.markdown!, 'serialize').mockImplementation( + (content) => `${serialize(content)}\nextra line` + ) + const changed = getRichMarkdownReviewRailBlocks(editor) + expect(changed).not.toEqual(original) + expect(changed).toEqual(buildRichMarkdownCommentBlocks(editor)) + editor.markdown = createEditor().markdown + expect(getRichMarkdownReviewRailBlocks(editor)).toEqual(original) + }) + + it('does not reuse fallback lines after a missing serializer becomes available', () => { + const editor = createEditor() + const markdown = editor.markdown + editor.markdown = undefined + const fallback = getRichMarkdownReviewRailBlocks(editor) + expect(fallback).toEqual(buildRichMarkdownCommentBlocks(editor)) + editor.markdown = markdown + expect(getRichMarkdownReviewRailBlocks(editor)).not.toEqual(fallback) + expect(getRichMarkdownReviewRailBlocks(editor)).toEqual(buildRichMarkdownCommentBlocks(editor)) + }) + + it('avoids serialization for every comment and repeated scroll while refreshing geometry', () => { + const editor = createEditor() + let sourceTop = 100 + const coords = vi.spyOn(editor.view, 'coordsAtPos').mockImplementation(() => ({ + top: sourceTop, + bottom: sourceTop + 20, + left: 0, + right: 10 + })) + const container = document.createElement('div') + const comment: DiffComment = { + id: 'note', + worktreeId: 'workspace', + filePath: 'notes.md', + source: 'markdown', + lineNumber: 1, + body: 'Review', + createdAt: 1, + side: 'modified' + } + const markdownComments = Array.from({ length: 5 }, (_, index) => ({ + ...comment, + id: `note-${index}`, + selectedText: index === 0 ? 'Heading' : undefined + })) + const serialize = vi.spyOn(editor.markdown!, 'serialize') + const measure = () => + measureRichMarkdownReviewNotePositions({ + editor, + container, + markdownComments, + markdownSourceLineOffset: 0 + }) + expect(measure()[0]?.top).toBe(100) + expect(serialize).toHaveBeenCalledTimes(2 * editor.state.doc.childCount - 1) + serialize.mockClear() + sourceTop = 200 + container.scrollTop = 30 + for (let index = 0; index < 60; index++) { + expect(measure()[0]?.top).toBe(230) + } + expect(serialize).not.toHaveBeenCalled() + expect(coords).toHaveBeenCalledTimes(61 * markdownComments.length) + }) +}) diff --git a/src/renderer/src/components/editor/rich-markdown-review-rail-blocks.ts b/src/renderer/src/components/editor/rich-markdown-review-rail-blocks.ts new file mode 100644 index 00000000000..35dd1bb9eb5 --- /dev/null +++ b/src/renderer/src/components/editor/rich-markdown-review-rail-blocks.ts @@ -0,0 +1,31 @@ +import type { Editor } from '@tiptap/core' +import { + buildRichMarkdownCommentBlocks, + type RichMarkdownCommentBlock +} from './rich-markdown-review-annotations' + +type ReviewRailBlocks = { + doc: Editor['state']['doc'] + markdown: Editor['markdown'] + serialize: NonNullable['serialize'] | undefined + blocks: readonly RichMarkdownCommentBlock[] +} + +// Keep only the current document per editor; scrolling changes geometry, not source lines. +const blocksByEditor = new WeakMap() + +export function getRichMarkdownReviewRailBlocks( + editor: Editor +): readonly RichMarkdownCommentBlock[] { + const doc = editor.state.doc + const markdown = editor.markdown + const serialize = markdown?.serialize + const cached = blocksByEditor.get(editor) + if (cached?.doc === doc && cached.markdown === markdown && cached.serialize === serialize) { + return cached.blocks + } + + const blocks = buildRichMarkdownCommentBlocks(editor) + blocksByEditor.set(editor, { doc, markdown, serialize, blocks }) + return blocks +} From b55e0cffca2d7ae3435775563cf5e60dbbdc57d9 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:02:39 -0700 Subject: [PATCH 12/69] perf(editor): reuse live Markdown search matches across unrelated renders (#18903) --- ...rich-markdown-search-matches-cache.test.ts | 93 ++++++++++++ .../rich-markdown-search-matches-cache.ts | 36 +++++ .../useRichMarkdownSearch.reuse.test.tsx | 135 ++++++++++++++++++ .../editor/useRichMarkdownSearch.ts | 19 ++- 4 files changed, 277 insertions(+), 6 deletions(-) create mode 100644 src/renderer/src/components/editor/rich-markdown-search-matches-cache.test.ts create mode 100644 src/renderer/src/components/editor/rich-markdown-search-matches-cache.ts create mode 100644 src/renderer/src/components/editor/useRichMarkdownSearch.reuse.test.tsx diff --git a/src/renderer/src/components/editor/rich-markdown-search-matches-cache.test.ts b/src/renderer/src/components/editor/rich-markdown-search-matches-cache.test.ts new file mode 100644 index 00000000000..8859d8c4b2e --- /dev/null +++ b/src/renderer/src/components/editor/rich-markdown-search-matches-cache.test.ts @@ -0,0 +1,93 @@ +import { Schema } from '@tiptap/pm/model' +import { describe, expect, it, vi } from 'vitest' +import { findRichMarkdownSearchMatches } from './rich-markdown-search' +import { createRichMarkdownSearchMatchesCache } from './rich-markdown-search-matches-cache' + +const schema = new Schema({ + nodes: { + doc: { content: 'paragraph+' }, + paragraph: { content: 'text*' }, + text: {} + } +}) + +function createDoc(text = 'Beta beta betas', count = 1) { + return schema.node( + 'doc', + null, + Array.from({ length: count }, () => schema.node('paragraph', null, schema.text(text))) + ) +} + +describe('rich markdown search match reuse', () => { + it('keeps separate live and debounced results without rewalking the document', () => { + const doc = createDoc() + const find = createRichMarkdownSearchMatchesCache() + const walk = vi.spyOn(doc, 'nodesBetween') + const highlighted = find(doc, 'beta') + const live = find(doc, 'betas') + for (let index = 0; index < 100; index++) { + expect(find(doc, 'beta')).toBe(highlighted) + expect(find(doc, 'betas')).toBe(live) + } + expect(walk).toHaveBeenCalledTimes(2) + }) + + it('keys by document and both matching options', () => { + const find = createRichMarkdownSearchMatchesCache() + const doc = createDoc() + const anyCase = find(doc, 'beta') + expect(find(doc, 'beta', { matchCase: false, wholeWord: false })).toBe(anyCase) + for (const options of [ + { matchCase: true }, + { wholeWord: true }, + { matchCase: true, wholeWord: true } + ]) { + expect(find(doc, 'beta', options)).toEqual( + findRichMarkdownSearchMatches(doc, 'beta', options) + ) + } + const changedDoc = createDoc('A different beta') + expect(find(changedDoc, 'beta')).toEqual(findRichMarkdownSearchMatches(changedDoc, 'beta')) + expect(find(doc, 'beta')).not.toBe(anyCase) + }) + + it('bounds retained results to two queries for one document', () => { + const find = createRichMarkdownSearchMatchesCache() + const doc = createDoc() + const first = find(doc, 'beta') + find(doc, 'Beta') + find(doc, 'betas') + expect(find(doc, 'beta')).not.toBe(first) + }) +}) + +it.skipIf(process.env.ORCA_SEARCH_CACHE_BENCH !== '1')( + 'benchmarks repeated live-match checks', + () => { + for (const blockCount of [250, 1000]) { + const doc = createDoc('Beta beta betas with searchable content', blockCount) + const find = createRichMarkdownSearchMatchesCache() + const expected = findRichMarkdownSearchMatches(doc, 'beta') + expect(find(doc, 'beta')).toEqual(expected) + const measure = (run: () => unknown) => { + const samples: number[] = [] + for (let round = 0; round < 5; round++) { + const start = performance.now() + for (let index = 0; index < 100; index++) { + run() + } + samples.push((performance.now() - start) / 100) + } + return samples.sort((a, b) => a - b)[2]! + } + const beforeMs = measure(() => + findRichMarkdownSearchMatches(doc, 'beta').some((match) => match.touchesReadOnlyAtom) + ) + const afterMs = measure(() => find(doc, 'beta').some((match) => match.touchesReadOnlyAtom)) + process.stdout.write( + `${JSON.stringify({ blockCount, matchCount: expected.length, beforeMs, afterMs })}\n` + ) + } + } +) diff --git a/src/renderer/src/components/editor/rich-markdown-search-matches-cache.ts b/src/renderer/src/components/editor/rich-markdown-search-matches-cache.ts new file mode 100644 index 00000000000..1b057881356 --- /dev/null +++ b/src/renderer/src/components/editor/rich-markdown-search-matches-cache.ts @@ -0,0 +1,36 @@ +import type { Node as ProseMirrorNode } from '@tiptap/pm/model' +import type { TextMatchOptions } from './markdown-preview-search' +import { findRichMarkdownSearchMatches, type RichMarkdownSearchMatch } from './rich-markdown-search' + +type CachedMatches = { + query: string + matchCase: boolean + wholeWord: boolean + matches: RichMarkdownSearchMatch[] +} + +export function createRichMarkdownSearchMatchesCache(): typeof findRichMarkdownSearchMatches { + let previousDoc: ProseMirrorNode | undefined + let entries: CachedMatches[] = [] + + return (doc, query, options: TextMatchOptions = {}, stats) => { + if (doc !== previousDoc) { + previousDoc = doc + entries = [] + } + const matchCase = options.matchCase ?? false + const wholeWord = options.wholeWord ?? false + const existing = entries.find( + (entry) => + entry.query === query && entry.matchCase === matchCase && entry.wholeWord === wholeWord + ) + if (existing) { + return existing.matches + } + + const matches = findRichMarkdownSearchMatches(doc, query, options, stats) + // The live replacement guard and debounced highlights can have different queries. + entries = [...entries.slice(-1), { query, matchCase, wholeWord, matches }] + return matches + } +} diff --git a/src/renderer/src/components/editor/useRichMarkdownSearch.reuse.test.tsx b/src/renderer/src/components/editor/useRichMarkdownSearch.reuse.test.tsx new file mode 100644 index 00000000000..3aa0cea32dd --- /dev/null +++ b/src/renderer/src/components/editor/useRichMarkdownSearch.reuse.test.tsx @@ -0,0 +1,135 @@ +// @vitest-environment happy-dom +import { act, renderHook } from '@testing-library/react' +import { Schema, type Node as ProseMirrorNode } from '@tiptap/pm/model' +import { EditorState } from '@tiptap/pm/state' +import type { Editor } from '@tiptap/react' +import { afterEach, describe, expect, it, vi } from 'vitest' +import * as visibleText from './rich-markdown-visible-text-map' +import { useRichMarkdownSearch } from './useRichMarkdownSearch' + +vi.mock('@/store', () => ({ + useAppStore: (select: (state: unknown) => unknown) => select({ keybindings: {} }) +})) + +const schema = new Schema({ + nodes: { + doc: { content: 'paragraph+' }, + paragraph: { content: 'inline*' }, + text: { group: 'inline' }, + atom: { + group: 'inline', + inline: true, + atom: true, + attrs: { label: {} }, + leafText: (node) => node.attrs.label + } + } +}) + +function createDoc(atomLabel = 'locked') { + return schema.node( + 'doc', + null, + schema.node('paragraph', null, [ + schema.text('beta beta '), + schema.node('atom', { label: atomLabel }) + ]) + ) +} + +function mountSearch() { + let state = EditorState.create({ doc: createDoc() }) + const listeners = new Set<() => void>() + const editor = { + get state() { + return state + }, + commands: { focus: vi.fn() }, + on: (_event: string, listener: () => void) => listeners.add(listener), + off: (_event: string, listener: () => void) => listeners.delete(listener), + registerPlugin: vi.fn(), + unregisterPlugin: vi.fn(), + view: { + dispatch: (tr: EditorState['tr']) => { + state = state.apply(tr) + if (tr.docChanged) { + listeners.forEach((listener) => listener()) + } + } + } + } as unknown as Editor + const hook = renderHook(() => + useRichMarkdownSearch({ + editor, + rootRef: { current: null }, + scrollContainerRef: { current: null } + }) + ) + return { + hook, + editor, + replaceDocSilently: (doc: ProseMirrorNode) => { + state = EditorState.create({ doc }) + } + } +} + +afterEach(() => { + vi.useRealTimers() + vi.restoreAllMocks() +}) + +describe('live rich markdown search reuse', () => { + it('does not rescan while typing replacement text, navigating, or publishing debounced highlights', () => { + vi.useFakeTimers() + const walk = vi.spyOn(visibleText, 'createRichMarkdownVisibleTextMap') + const { hook } = mountSearch() + act(() => hook.result.current.openSearch()) + act(() => hook.result.current.searchActions.setSearchQuery('beta')) + expect(walk).toHaveBeenCalledTimes(1) + act(() => vi.advanceTimersByTime(150)) + expect(hook.result.current.searchState.matchCount).toBe(2) + for (let index = 0; index < 20; index++) { + act(() => hook.result.current.searchActions.setReplaceQuery(`replacement ${index}`)) + act(() => hook.result.current.searchActions.moveToMatch(1)) + } + expect(walk).toHaveBeenCalledTimes(1) + act(() => hook.result.current.searchActions.closeSearch()) + act(() => hook.result.current.openSearch()) + act(() => hook.result.current.searchActions.setSearchQuery('beta')) + expect(walk).toHaveBeenCalledTimes(2) + hook.unmount() + }) + + it('uses the immediate query for atom guards and replacements during the debounce window', () => { + vi.useFakeTimers() + const { hook, editor } = mountSearch() + act(() => hook.result.current.openSearch()) + act(() => hook.result.current.searchActions.setSearchQuery('beta')) + act(() => vi.advanceTimersByTime(150)) + act(() => hook.result.current.searchActions.setSearchQuery('locked')) + expect(hook.result.current.searchState.matchCount).toBe(2) + expect(hook.result.current.searchState.replaceDisabled).toBe(true) + const doc = editor.state.doc + act(() => hook.result.current.searchActions.replaceAllMatches()) + expect(editor.state.doc).toBe(doc) + act(() => hook.result.current.searchActions.setSearchQuery('beta')) + act(() => hook.result.current.searchActions.setReplaceQuery('changed')) + act(() => hook.result.current.searchActions.replaceAllMatches()) + expect(editor.state.doc.textContent).toBe('changed changed locked') + hook.unmount() + }) + + it('checks the latest document even before an editor update reaches React', () => { + vi.useFakeTimers() + const { hook, editor, replaceDocSilently } = mountSearch() + act(() => hook.result.current.openSearch()) + act(() => hook.result.current.searchActions.setSearchQuery('beta')) + act(() => vi.advanceTimersByTime(150)) + replaceDocSilently(createDoc('beta')) + const doc = editor.state.doc + act(() => hook.result.current.searchActions.replaceCurrentMatch()) + expect(editor.state.doc).toBe(doc) + hook.unmount() + }) +}) diff --git a/src/renderer/src/components/editor/useRichMarkdownSearch.ts b/src/renderer/src/components/editor/useRichMarkdownSearch.ts index 2508505b85b..1a0f132e67c 100644 --- a/src/renderer/src/components/editor/useRichMarkdownSearch.ts +++ b/src/renderer/src/components/editor/useRichMarkdownSearch.ts @@ -1,5 +1,4 @@ -import { useCallback, useEffect, useMemo, useRef, useState } from 'react' -import type { RefObject } from 'react' +import { useCallback, useEffect, useMemo, useRef, useState, type RefObject } from 'react' import type { Editor } from '@tiptap/react' import { TextSelection } from '@tiptap/pm/state' import { getShortcutPlatform } from '@/lib/shortcut-platform' @@ -14,6 +13,7 @@ import { findRichMarkdownSearchMatches, richMarkdownSearchPluginKey } from './rich-markdown-search' +import { createRichMarkdownSearchMatchesCache } from './rich-markdown-search-matches-cache' export function useRichMarkdownSearch({ editor, @@ -27,6 +27,13 @@ export function useRichMarkdownSearch({ const searchInputRef = useRef(null) const keybindings = useAppStore((state) => state.keybindings) const [isSearchOpen, setIsSearchOpen] = useState(false) + const findMatches = useMemo( + () => + editor && isSearchOpen + ? createRichMarkdownSearchMatchesCache() + : findRichMarkdownSearchMatches, + [editor, isSearchOpen] + ) const [isReplaceMode, setIsReplaceMode] = useState(false) const [searchQuery, setSearchQuery] = useState('') const [replaceQuery, setReplaceQuery] = useState('') @@ -57,13 +64,13 @@ export function useRichMarkdownSearch({ if (!editor || !isSearchOpen || !searchRequestQuery) { return [] } - return findRichMarkdownSearchMatches(editor.state.doc, searchRequestQuery, { + return findMatches(editor.state.doc, searchRequestQuery, { matchCase, wholeWord }) // searchRevision is bumped on ProseMirror doc edits to trigger recomputation // eslint-disable-next-line react-hooks/exhaustive-deps - }, [editor, isSearchOpen, searchRequestQuery, searchRevision, matchCase, wholeWord]) + }, [editor, findMatches, isSearchOpen, searchRequestQuery, searchRevision, matchCase, wholeWord]) const matchCount = matches.length @@ -78,11 +85,11 @@ export function useRichMarkdownSearch({ } // Why: replace mutates document ranges immediately, so it must use the // current input value instead of the debounced highlight match set. - return findRichMarkdownSearchMatches(editor.state.doc, searchQuery, { + return findMatches(editor.state.doc, searchQuery, { matchCase, wholeWord }) - }, [editor, isSearchOpen, matchCase, searchQuery, wholeWord]) + }, [editor, findMatches, isSearchOpen, matchCase, searchQuery, wholeWord]) // Why: mirror the guard used by replaceCurrentMatch/replaceAllMatches so the // disabled state never disagrees with what a click will actually do during the From 7b44f3c0e33c4f9d1a8ea5d0bae9f64bc80c812f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:02:44 -0700 Subject: [PATCH 13/69] perf: decode fragmented CLI replies without repeated scans (#18909) * perf: decode fragmented CLI replies without rescanning accumulated text * bench: require an explicit CLI framing baseline --- .../benchmark-cli-response-framing.mjs | 128 +++++++++++++++ src/cli/runtime/transport-framing.test.ts | 150 ++++++++++++++++++ src/cli/runtime/transport.ts | 27 ++-- 3 files changed, 296 insertions(+), 9 deletions(-) create mode 100644 config/scripts/benchmark-cli-response-framing.mjs create mode 100644 src/cli/runtime/transport-framing.test.ts diff --git a/config/scripts/benchmark-cli-response-framing.mjs b/config/scripts/benchmark-cli-response-framing.mjs new file mode 100644 index 00000000000..40aab8d08f7 --- /dev/null +++ b/config/scripts/benchmark-cli-response-framing.mjs @@ -0,0 +1,128 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { EventEmitter } from 'node:events' +import { readFileSync } from 'node:fs' +import Module from 'node:module' +import { dirname, resolve } from 'node:path' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +// Run from the worktree root: node config/scripts/benchmark-cli-response-framing.mjs +const sourcePath = 'src/cli/runtime/transport.ts' +const baselineRef = process.argv[2] +assert.ok(baselineRef, 'Pass the pre-change transport revision as base-ref.') +const beforeSource = execFileSync('git', ['show', `${baselineRef}:${sourcePath}`], { + encoding: 'utf8' +}) +let chunks = [] + +async function loadTransport(source) { + const built = await build({ + stdin: { contents: source, loader: 'ts', resolveDir: dirname(resolve(sourcePath)) }, + bundle: true, + platform: 'node', + format: 'cjs', + write: false, + logLevel: 'silent' + }) + const module = new Module(resolve(sourcePath)) + const originalRequire = module.require.bind(module) + module.require = (name) => { + if (name === 'node:crypto') { + return { randomUUID: () => 'benchmark-request' } + } + if (name !== 'node:net') { + return originalRequire(name) + } + return { + createConnection() { + const socket = new EventEmitter() + socket.setEncoding = () => {} + socket.end = () => {} + socket.destroy = () => {} + socket.write = () => { + for (const chunk of chunks) { + socket.emit('data', chunk) + } + } + queueMicrotask(() => socket.emit('connect')) + return socket + } + } + } + module._compile(built.outputFiles[0].text, resolve(sourcePath)) + return module.exports.sendRequest +} + +const before = await loadTransport(beforeSource) +const after = await loadTransport(readFileSync(sourcePath, 'utf8')) +const metadata = { + runtimeId: 'benchmark-runtime', + authToken: 'benchmark-token', + transports: [{ kind: 'unix', endpoint: 'injected-socket' }] +} +const run = (sendRequest) => sendRequest(metadata, 'terminal.read', {}, 30000) + +async function measure(sendRequest, payloadBytes, repetitions) { + const warmup = await run(sendRequest) + assert.equal(warmup.result.data.length, payloadBytes) + const samples = [] + for (let sample = 0; sample < 5; sample++) { + const start = performance.now() + for (let iteration = 0; iteration < repetitions; iteration++) { + await run(sendRequest) + } + samples.push((performance.now() - start) / repetitions) + } + return samples.sort((a, b) => a - b)[2] +} + +async function searchedCharacters(sendRequest) { + const original = String.prototype.indexOf + let searched = 0 + String.prototype.indexOf = function (needle, position) { + if (needle === '\n') { + searched += this.length - (position ?? 0) + } + return original.call(this, needle, position) + } + try { + await run(sendRequest) + } finally { + String.prototype.indexOf = original + } + return searched +} + +const rows = [] +for (const [payloadBytes, chunkChars, repetitions] of [ + [32, 65536, 1000], + [1024 * 1024, 2 * 1024 * 1024, 20], + [1024 * 1024, 65536, 10], + [1024 * 1024, 4096, 5], + [4 * 1024 * 1024, 4096, 2], + [4 * 1024 * 1024, 256, 1] +]) { + const line = `${JSON.stringify({ + id: 'benchmark-request', + ok: true, + result: { data: 'x'.repeat(payloadBytes) }, + _meta: { runtimeId: 'benchmark-runtime' } + })}\n` + chunks = [] + for (let offset = 0; offset < line.length; offset += chunkChars) { + chunks.push(line.slice(offset, offset + chunkChars)) + } + const beforeMs = await measure(before, payloadBytes, repetitions) + const afterMs = await measure(after, payloadBytes, repetitions) + rows.push({ + payloadBytes, + chunkChars, + beforeMs: +beforeMs.toFixed(6), + afterMs: +afterMs.toFixed(6), + speedup: +(beforeMs / afterMs).toFixed(2), + beforeSearchedCharacters: await searchedCharacters(before), + afterSearchedCharacters: await searchedCharacters(after) + }) +} +console.log(JSON.stringify({ node: process.version, baselineRef, rows }, null, 2)) diff --git a/src/cli/runtime/transport-framing.test.ts b/src/cli/runtime/transport-framing.test.ts new file mode 100644 index 00000000000..cf3150f60c9 --- /dev/null +++ b/src/cli/runtime/transport-framing.test.ts @@ -0,0 +1,150 @@ +import { EventEmitter } from 'node:events' +import { StringDecoder } from 'node:string_decoder' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { RuntimeMetadata } from '../../shared/runtime-bootstrap' +import { sendRequest } from './transport' + +const { createConnection } = vi.hoisted(() => ({ createConnection: vi.fn() })) +vi.mock('node:net', () => ({ createConnection })) +vi.mock('node:crypto', () => ({ randomUUID: () => 'request-1' })) + +const metadata: RuntimeMetadata = { + runtimeId: 'runtime-1', + pid: 123, + transports: [{ kind: 'unix', endpoint: 'test-only' }], + authToken: 'token', + startedAt: 1 +} +const reply = (result: unknown) => + `${JSON.stringify({ id: 'request-1', ok: true, result, _meta: { runtimeId: 'runtime-1' } })}\n` + +class TestSocket extends EventEmitter { + setEncoding = vi.fn() + write = vi.fn() + end = vi.fn() + destroy = vi.fn() +} +let socket: TestSocket + +beforeEach(() => { + socket = new TestSocket() + createConnection.mockReturnValue(socket) +}) +afterEach(() => { + vi.restoreAllMocks() + vi.useRealTimers() +}) + +describe('CLI runtime response framing', () => { + it.each([1, 7, 256, 4096])( + 'reads a fragmented response with %i-character chunks', + async (size) => { + const result = { data: '界😀'.repeat(10000) } + const encoded = reply(result) + const pending = sendRequest(metadata, 'terminal.read', {}, 30000) + for (let offset = 0; offset < encoded.length; offset += size) { + socket.emit('data', encoded.slice(offset, offset + size)) + } + await expect(pending).resolves.toMatchObject({ result }) + expect(socket.setEncoding).toHaveBeenCalledExactlyOnceWith('utf8') + expect(socket.end).toHaveBeenCalledOnce() + } + ) + + it('accepts Unicode split across socket bytes using the existing UTF-8 decoder', async () => { + const pending = sendRequest(metadata, 'terminal.read', {}, 30000) + const decoder = new StringDecoder('utf8') + for (const byte of Buffer.from(reply({ data: '界😀é' }))) { + socket.emit('data', decoder.write(Buffer.from([byte]))) + } + socket.emit('data', decoder.end()) + await expect(pending).resolves.toMatchObject({ result: { data: '界😀é' } }) + }) + + it('searches each fragment once without rescanning the accumulated reply', async () => { + const encoded = reply({ data: 'x'.repeat(1024 * 1024) }) + const pending = sendRequest(metadata, 'terminal.read', {}, 30000) + const originalIndexOf = String.prototype.indexOf + let searchedCharacters = 0 + const search = vi + .spyOn(String.prototype, 'indexOf') + .mockImplementation(function (this: string, value, position) { + if (value === '\n') { + searchedCharacters += this.length - (position ?? 0) + } + return originalIndexOf.call(this, value, position) + }) + try { + for (let offset = 0; offset < encoded.length; offset += 256) { + socket.emit('data', encoded.slice(offset, offset + 256)) + } + } finally { + search.mockRestore() + } + await expect(pending).resolves.toMatchObject({ ok: true }) + expect(searchedCharacters).toBe(encoded.length) + }) + + it('refreshes keepalives across chunks and ignores blanks and data after the final frame', async () => { + vi.useFakeTimers() + const pending = sendRequest(metadata, 'terminal.read', {}, 100) + await vi.advanceTimersByTimeAsync(90) + socket.emit('data', ' \r\n{"_keep') + socket.emit('data', 'alive":true}\n\t\n') + await vi.advanceTimersByTimeAsync(90) + socket.emit('data', `${reply({ data: 'done' })}invalid JSON\n`) + socket.emit('data', 'more ignored data') + await expect(pending).resolves.toMatchObject({ result: { data: 'done' } }) + expect(socket.end).toHaveBeenCalledOnce() + expect(vi.getTimerCount()).toBe(0) + }) + + it.each([ + ['broken JSON\n', 'invalid_runtime_response'], + ['{}\n', 'invalid_runtime_response'], + ['{"id":"other","ok":true,"result":{}}\n', 'invalid_runtime_response'], + [ + '{"id":"request-1","ok":true,"result":{},"_meta":{"runtimeId":"other"}}\n', + 'runtime_unavailable' + ] + ])( + 'rejects a fragmented invalid first frame before subsequent valid frames', + async (line, code) => { + const pending = sendRequest(metadata, 'terminal.read', {}, 30000) + socket.emit('data', line.slice(0, 2)) + socket.emit('data', line.slice(2) + reply({ data: 'ignored' })) + await expect(pending).rejects.toMatchObject({ code }) + expect(socket.end).toHaveBeenCalledOnce() + } + ) + + it('preserves terminal failure envelopes', async () => { + const pending = sendRequest(metadata, 'terminal.read', {}, 30000) + socket.emit('data', '{"id":"request-1","ok":false,"error":{"code":"bad","message":"no"}}\n') + await expect(pending).resolves.toMatchObject({ + ok: false, + error: { code: 'bad', message: 'no' } + }) + }) + + it('rejects close with an incomplete frame and does not parse later data', async () => { + const pending = sendRequest(metadata, 'terminal.read', {}, 30000) + socket.emit('data', '{"id":') + socket.emit('close') + socket.emit('data', reply({ data: 'ignored' })) + await expect(pending).rejects.toMatchObject({ code: 'runtime_unavailable' }) + expect(socket.end).toHaveBeenCalledOnce() + }) + + it('destroys a timed out socket holding an incomplete frame', async () => { + vi.useFakeTimers() + const pending = sendRequest(metadata, 'terminal.read', {}, 100) + const rejected = expect(pending).rejects.toMatchObject({ code: 'runtime_timeout' }) + socket.emit('data', '{"id":') + await vi.advanceTimersByTimeAsync(100) + socket.emit('data', reply({ data: 'ignored' })) + await rejected + expect(socket.destroy).toHaveBeenCalledOnce() + expect(vi.getTimerCount()).toBe(0) + }) +}) diff --git a/src/cli/runtime/transport.ts b/src/cli/runtime/transport.ts index 091ca9da0e0..4e0d0a15ec7 100644 --- a/src/cli/runtime/transport.ts +++ b/src/cli/runtime/transport.ts @@ -31,7 +31,7 @@ export async function sendRequest( return } const socket = createConnection(transport.endpoint) - let buffer = '' + let lineSegments: string[] = [] let settled = false const requestId = randomUUID() @@ -40,6 +40,7 @@ export async function sendRequest( return } settled = true + lineSegments = [] socket.destroy() reject( new RuntimeClientError( @@ -56,6 +57,7 @@ export async function sendRequest( return } settled = true + lineSegments = [] clearTimeout(timeout) socket.end() if (result.ok === false) { @@ -89,18 +91,27 @@ export async function sendRequest( }) }) socket.on('data', (chunk: string) => { - buffer += chunk // Why: the server may interleave `{"_keepalive":true}\n` frames with the // final success/failure frame to keep both idle timers alive during a // long-poll (see design doc §3.1). Read frames in a loop until we see a // terminal frame. Each keepalive refreshes the client-side timer so a // 10 min wait doesn't trip the 60 s default ceiling. - let newlineIndex = buffer.indexOf('\n') - while (newlineIndex !== -1 && !settled) { - const line = buffer.slice(0, newlineIndex) - buffer = buffer.slice(newlineIndex + 1) + let cursor = 0 + while (cursor < chunk.length && !settled) { + const newlineIndex = chunk.indexOf('\n', cursor) + if (newlineIndex === -1) { + lineSegments.push(chunk.slice(cursor)) + return + } + const segment = chunk.slice(cursor, newlineIndex) + let line = segment + if (lineSegments.length > 0) { + lineSegments.push(segment) + line = lineSegments.join('') + lineSegments = [] + } + cursor = newlineIndex + 1 if (line.trim().length === 0) { - newlineIndex = buffer.indexOf('\n') continue } @@ -124,7 +135,6 @@ export async function sendRequest( // major). See §7 risk #9. if (isKeepaliveFrame(raw)) { timeout.refresh() - newlineIndex = buffer.indexOf('\n') continue } @@ -150,7 +160,6 @@ export async function sendRequest( const frame = parsed.data if ('_keepalive' in frame) { timeout.refresh() - newlineIndex = buffer.indexOf('\n') continue } From a07d8fe19cca333e5cfcf106f9bbe83ee741ee78 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:02:48 -0700 Subject: [PATCH 14/69] perf: avoid file-to-file scans when selecting deletion roots (#18913) --- ...file-explorer-deletion-roots-benchmark.mjs | 75 +++++++++++++++++++ .../file-explorer-batch-deletion.test.ts | 60 +++++++++++++++ .../file-explorer-batch-deletion.ts | 6 +- 3 files changed, 137 insertions(+), 4 deletions(-) create mode 100644 config/scripts/file-explorer-deletion-roots-benchmark.mjs diff --git a/config/scripts/file-explorer-deletion-roots-benchmark.mjs b/config/scripts/file-explorer-deletion-roots-benchmark.mjs new file mode 100644 index 00000000000..d276964861b --- /dev/null +++ b/config/scripts/file-explorer-deletion-roots-benchmark.mjs @@ -0,0 +1,75 @@ +import assert from 'node:assert/strict' +import { join } from 'node:path' +import { performance } from 'node:perf_hooks' +import { fileURLToPath } from 'node:url' +import { build } from 'esbuild' + +const root = fileURLToPath(new URL('../..', import.meta.url)) +const bundled = await build({ + stdin: { + contents: `export { selectDeletionRoots } from './file-explorer-batch-deletion'; + export { isPathEqualOrDescendant } from './file-explorer-paths';`, + resolveDir: join(root, 'src/renderer/src/components/right-sidebar'), + loader: 'ts' + }, + alias: { '@': join(root, 'src/renderer/src') }, + bundle: true, + platform: 'node', + format: 'esm', + write: false, + logLevel: 'silent' +}) +const { selectDeletionRoots, isPathEqualOrDescendant } = await import( + `data:text/javascript;base64,${Buffer.from(bundled.outputFiles[0].text).toString('base64')}` +) + +// Original production selector; both paths use the same path-comparison implementation. +function original(nodes) { + return nodes.filter( + (n) => + !nodes.some( + (other) => other !== n && other.isDirectory && isPathEqualOrDescendant(n.path, other.path) + ) + ) +} + +function measure(run, nodes) { + for (let index = 0; index < 3; index++) { + run(nodes) + } + const samples = [] + for (let index = 0; index < 11; index++) { + const start = performance.now() + run(nodes) + samples.push(performance.now() - start) + } + return samples.sort((a, b) => a - b)[5] +} + +const results = [] +for (const [fileCount, directoryCount] of [ + [100, 0], + [1000, 0], + [5000, 0], + [5000, 5], + [0, 100] +]) { + const nodes = Array.from({ length: fileCount + directoryCount }, (_, index) => ({ + name: `item-${index}`, + path: `/repo/item-${index}`, + relativePath: `item-${index}`, + isDirectory: index >= fileCount, + depth: 0 + })) + const expected = original(nodes) + const actual = selectDeletionRoots(nodes) + assert.equal(actual.length, expected.length) + actual.forEach((node, index) => assert.equal(node, expected[index])) + results.push({ + fileCount, + directoryCount, + beforeMs: measure(original, nodes), + afterMs: measure(selectDeletionRoots, nodes) + }) +} +console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2)) diff --git a/src/renderer/src/components/right-sidebar/file-explorer-batch-deletion.test.ts b/src/renderer/src/components/right-sidebar/file-explorer-batch-deletion.test.ts index e93928bf4fd..688683bce13 100644 --- a/src/renderer/src/components/right-sidebar/file-explorer-batch-deletion.test.ts +++ b/src/renderer/src/components/right-sidebar/file-explorer-batch-deletion.test.ts @@ -37,6 +37,66 @@ describe('selectDeletionRoots', () => { const other = node('/repo/a.ts/impossible-child') expect(selectDeletionRoots([file, other])).toEqual([file, other]) }) + + it('reads directory membership once per selected node, including file-only selections', () => { + let directoryReads = 0 + const nodes = Array.from({ length: 1_000 }, (_, index) => ({ + ...node(`/repo/file-${index}.ts`), + get isDirectory() { + directoryReads += 1 + return false + } + })) + const selected = selectDeletionRoots(nodes) + expect(directoryReads).toBe(nodes.length) + expect(selected).not.toBe(nodes) + selected.forEach((entry, index) => expect(entry).toBe(nodes[index])) + }) + + it('preserves order, node identity and host ownership when children precede parents', () => { + const child = node('/repo/docs/guide.md') + const first = { ...node('/repo/first.ts'), operationOwner: { kind: 'local' as const } } + const parent = { + ...node('/repo/docs', true), + operationOwner: { kind: 'ssh' as const, connectionId: 'ssh-owner' } + } + const last = { + ...node('/repo/last.ts'), + operationOwner: { kind: 'unresolved' as const } + } + const nodes = [child, first, parent, last] + const selected = selectDeletionRoots(nodes) + expect(selected).toEqual([first, parent, last]) + expect(selected[0]).toBe(first) + expect(selected[1]).toBe(parent) + expect(selected[2]).toBe(last) + expect(nodes).toEqual([child, first, parent, last]) + }) + + it('keeps repeated references but excludes distinct directories with the same path', () => { + const dir = node('/repo/docs', true) + expect(selectDeletionRoots([dir, dir])).toEqual([dir, dir]) + expect(selectDeletionRoots([dir, { ...dir }])).toEqual([]) + const file = node('/repo/docs') + expect(selectDeletionRoots([file, dir])).toEqual([dir]) + }) + + it.each([ + ['/repo/docs/', '/repo/docs/guide.md', '/repo/docs-other/guide.md'], + ['C:\\Repo\\Docs\\', 'c:/repo/docs/guide.md', 'C:/Repo/Docs-other/guide.md'], + ['\\\\Server\\Share\\Docs', '//server/share/docs/guide.md', '//server/other/docs/guide.md'] + ])('retains path boundaries for %s', (parentPath, childPath, outsidePath) => { + const parent = node(parentPath, true) + const child = node(childPath) + const outside = node(outsidePath) + expect(selectDeletionRoots([outside, child, parent])).toEqual([outside, parent]) + }) + + it('keeps case-distinct POSIX paths', () => { + const parent = node('/repo/Docs', true) + const outside = node('/repo/docs/guide.md') + expect(selectDeletionRoots([outside, parent])).toEqual([outside, parent]) + }) }) describe('runBatchDeletion', () => { diff --git a/src/renderer/src/components/right-sidebar/file-explorer-batch-deletion.ts b/src/renderer/src/components/right-sidebar/file-explorer-batch-deletion.ts index 63a5d029739..5f052bec588 100644 --- a/src/renderer/src/components/right-sidebar/file-explorer-batch-deletion.ts +++ b/src/renderer/src/components/right-sidebar/file-explorer-batch-deletion.ts @@ -5,11 +5,9 @@ import type { TreeNode } from './file-explorer-types' // already removes the child, and issuing both requests races on the // now-missing path and produces spurious errors. export function selectDeletionRoots(nodes: TreeNode[]): TreeNode[] { + const directories = nodes.filter((node) => node.isDirectory) return nodes.filter( - (n) => - !nodes.some( - (other) => other !== n && other.isDirectory && isPathEqualOrDescendant(n.path, other.path) - ) + (n) => !directories.some((other) => other !== n && isPathEqualOrDescendant(n.path, other.path)) ) } From e9d9d42ccca165935fe55a128cb911af4b895ad9 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:02:53 -0700 Subject: [PATCH 15/69] perf: find mobile Markdown placeholder prefixes in one scan (#18914) * perf: find mobile Markdown placeholder prefixes in one scan * style: format mobile Markdown benchmark guard * docs: explain collision-free Markdown prefix length --- .../mobile-markdown-placeholder-benchmark.mjs | 58 +++++++++++++++++++ .../mobile-markdown-preview-html.ts | 15 +++-- ...obile-markdown-preview-placeholder.test.ts | 33 +++++++++++ 3 files changed, 102 insertions(+), 4 deletions(-) create mode 100644 config/scripts/mobile-markdown-placeholder-benchmark.mjs create mode 100644 mobile/src/components/mobile-markdown-preview-placeholder.test.ts diff --git a/config/scripts/mobile-markdown-placeholder-benchmark.mjs b/config/scripts/mobile-markdown-placeholder-benchmark.mjs new file mode 100644 index 00000000000..20280e5a8d2 --- /dev/null +++ b/config/scripts/mobile-markdown-placeholder-benchmark.mjs @@ -0,0 +1,58 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { readFileSync } from 'node:fs' +import { dirname, resolve } from 'node:path' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +const sourcePath = 'mobile/src/components/mobile-markdown-preview-html.ts' +const baselineRef = process.argv[2] +if (!baselineRef) { + throw new Error( + 'Usage: node config/scripts/mobile-markdown-placeholder-benchmark.mjs ' + ) +} +async function load(source) { + const result = await build({ + stdin: { contents: source, resolveDir: dirname(resolve(sourcePath)), loader: 'ts' }, + bundle: true, + write: false, + platform: 'node', + format: 'esm' + }) + return ( + await import( + `data:text/javascript;base64,${Buffer.from(result.outputFiles[0].text).toString('base64')}` + ) + ).normalizeMobileMarkdownPreviewHtml +} +const before = await load( + execFileSync('git', ['show', `${baselineRef}:${sourcePath}`], { encoding: 'utf8' }) +) +const after = await load(readFileSync(sourcePath, 'utf8')) +function measure(fn, input, repeats) { + const samples = [] + for (let run = 0; run < repeats; run++) { + const start = performance.now() + fn(input) + samples.push(performance.now() - start) + } + return samples.sort((a, b) => a - b)[Math.floor(samples.length / 2)] +} +const results = [] +for (const [shape, input] of [ + ['ordinary Markdown', '# Hello\n\n

Use `Array` and bold.

'], + ...[2048, 8192, 16384].map((length) => [ + `${length} underscore collision`, + `\uE000ORCA_MD_CODE_${'_'.repeat(length)}0\uE000 and \`Array\`` + ]) +]) { + assert.equal(after(input), before(input)) + results.push({ + shape, + bytes: Buffer.byteLength(input), + beforeMs: measure(before, input, 5), + afterMs: measure(after, input, 15) + }) +} +console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2)) diff --git a/mobile/src/components/mobile-markdown-preview-html.ts b/mobile/src/components/mobile-markdown-preview-html.ts index 8bb5fca0934..a5866b171e6 100644 --- a/mobile/src/components/mobile-markdown-preview-html.ts +++ b/mobile/src/components/mobile-markdown-preview-html.ts @@ -204,11 +204,18 @@ function escapeRegExp(value: string): string { } function codePlaceholderPrefix(content: string): string { - let prefix = CODE_PLACEHOLDER_PREFIX_BASE - while (content.includes(prefix)) { - prefix = `${prefix}_` + let suffixLength = 0 + let cursor = 0 + while ((cursor = content.indexOf(CODE_PLACEHOLDER_PREFIX_BASE, cursor)) !== -1) { + cursor += CODE_PLACEHOLDER_PREFIX_BASE.length + const suffixStart = cursor + while (content[cursor] === '_') { + cursor += 1 + } + // One extra underscore keeps the prefix longer than every authored run. + suffixLength = Math.max(suffixLength, cursor - suffixStart + 1) } - return prefix + return CODE_PLACEHOLDER_PREFIX_BASE + '_'.repeat(suffixLength) } function protectMarkdownCode(content: string): { diff --git a/mobile/src/components/mobile-markdown-preview-placeholder.test.ts b/mobile/src/components/mobile-markdown-preview-placeholder.test.ts new file mode 100644 index 00000000000..f751608af78 --- /dev/null +++ b/mobile/src/components/mobile-markdown-preview-placeholder.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, it } from 'vitest' +import { normalizeMobileMarkdownPreviewHtml } from './mobile-markdown-preview-html' + +const marker = '\uE000ORCA_MD_CODE_' +const suffix = '\uE000' + +describe('mobile Markdown code placeholder collisions', () => { + it.each([0, 1, 2, 15, 128, 16384])('preserves a literal marker with %i underscores', (length) => { + const literal = `${marker}${'_'.repeat(length)}0${suffix}` + const input = `${literal} and \`Array\`\n\n\`\`\`html\n

literal

\n\`\`\`` + expect(normalizeMobileMarkdownPreviewHtml(input)).toBe(input) + }) + + it('handles adjacent markers and repeated maximum suffixes', () => { + const literal = `${marker}${marker}__0${suffix}${marker}__1${suffix}${marker}_2${suffix}` + expect(normalizeMobileMarkdownPreviewHtml(`

${literal} and \`

\`

`)).toBe( + `${literal} and \`
\`` + ) + }) + + it('preserves authored markers across generated suffix orders and HTML islands', () => { + let seed = 173 + for (let sample = 0; sample < 500; sample++) { + const literals: string[] = [] + for (let index = 0; index < 8; index++) { + seed = (Math.imul(seed, 1664525) + 1013904223) >>> 0 + literals.push(`${marker}${'_'.repeat(seed % 32)}${index}${suffix}`) + } + const text = literals.join(' ') + ' and `Array`' + expect(normalizeMobileMarkdownPreviewHtml(`

${text}

`)).toBe(text) + } + }) +}) From 4204bdf7172f9e2a822cbfde7349a8dc63b2f4ae Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:02:57 -0700 Subject: [PATCH 16/69] perf: avoid repeated Quick Open exclusion string allocations (#18916) --- .../quick-open-exclusion-benchmark.mjs | 60 +++++++++++++++++++ src/shared/quick-open-filter.test.ts | 37 ++++++++++++ src/shared/quick-open-filter.ts | 2 +- 3 files changed, 98 insertions(+), 1 deletion(-) create mode 100644 config/scripts/quick-open-exclusion-benchmark.mjs diff --git a/config/scripts/quick-open-exclusion-benchmark.mjs b/config/scripts/quick-open-exclusion-benchmark.mjs new file mode 100644 index 00000000000..399302c5a2b --- /dev/null +++ b/config/scripts/quick-open-exclusion-benchmark.mjs @@ -0,0 +1,60 @@ +import assert from 'node:assert/strict' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +const bundled = await build({ + entryPoints: ['src/shared/quick-open-filter.ts'], + bundle: true, + platform: 'node', + format: 'esm', + write: false, + logLevel: 'silent' +}) +const { shouldExcludeQuickOpenRelPath: after } = await import( + `data:text/javascript;base64,${Buffer.from(bundled.outputFiles[0].text).toString('base64')}` +) +// Original production predicate, including its exact boundary check. +function before(relPath, prefixes) { + for (const prefix of prefixes) { + if (relPath === prefix) { + return true + } + if (relPath.length > prefix.length && relPath.startsWith(`${prefix}/`)) { + return true + } + } + return false +} +const files = Array.from( + { length: 100000 }, + (_, index) => `src/components/group-${index % 100}/file-${index}.tsx` +) +function run(fn, prefixes) { + let excluded = 0 + for (const file of files) { + excluded += Number(fn(file, prefixes)) + } + return excluded +} +function measure(fn, prefixes) { + run(fn, prefixes) + const samples = [] + for (let index = 0; index < 5; index++) { + const start = performance.now() + run(fn, prefixes) + samples.push(performance.now() - start) + } + return samples.sort((a, b) => a - b)[2] +} +const results = [] +for (const count of [0, 10, 100, 500]) { + const prefixes = Array.from({ length: count }, (_, index) => `nested-worktrees/worktree-${index}`) + assert.equal(run(after, prefixes), run(before, prefixes)) + results.push({ + files: files.length, + exclusions: count, + beforeMs: measure(before, prefixes), + afterMs: measure(after, prefixes) + }) +} +console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2)) diff --git a/src/shared/quick-open-filter.test.ts b/src/shared/quick-open-filter.test.ts index 1d96bc1744f..c481a1e8854 100644 --- a/src/shared/quick-open-filter.test.ts +++ b/src/shared/quick-open-filter.test.ts @@ -102,6 +102,43 @@ describe('buildExcludePathPrefixes', () => { }) describe('shouldExcludeQuickOpenRelPath', () => { + it('matches the original filter across boundary and Unicode path combinations', () => { + const paths = [ + '', + '/', + 'a', + 'a/', + 'a//', + 'ab', + 'a/b', + 'a\\b', + 'A/b', + '界/😀', + '界/😀x', + 'a[1]/x', + 'a./x' + ] + for (const prefix of paths) { + for (const relPath of paths) { + const expected = + relPath === prefix || (relPath.length > prefix.length && relPath.startsWith(`${prefix}/`)) + expect(shouldExcludeQuickOpenRelPath(relPath, [prefix])).toBe(expected) + } + } + }) + + it('preserves normalized Windows and UNC exclusion boundaries', () => { + for (const [root, excluded] of [ + ['C:\\Repo', 'C:\\Repo\\trees\\one'], + ['\\\\server\\share\\repo', '\\\\server\\share\\repo\\trees\\one'] + ]) { + const prefixes = buildExcludePathPrefixes(root, [excluded]) + expect(prefixes).toEqual(['trees/one']) + expect(shouldExcludeQuickOpenRelPath('trees/one/file.ts', prefixes)).toBe(true) + expect(shouldExcludeQuickOpenRelPath('trees/one-more/file.ts', prefixes)).toBe(false) + } + }) + it('matches exact and boundary paths only', () => { expect(shouldExcludeQuickOpenRelPath('packages/app', ['packages/app'])).toBe(true) expect(shouldExcludeQuickOpenRelPath('packages/app/x.ts', ['packages/app'])).toBe(true) diff --git a/src/shared/quick-open-filter.ts b/src/shared/quick-open-filter.ts index 9d5bebd80b3..b7592657a11 100644 --- a/src/shared/quick-open-filter.ts +++ b/src/shared/quick-open-filter.ts @@ -119,7 +119,7 @@ export function shouldExcludeQuickOpenRelPath( if (relPath === prefix) { return true } - if (relPath.length > prefix.length && relPath.startsWith(`${prefix}/`)) { + if (relPath[prefix.length] === '/' && relPath.startsWith(prefix)) { return true } } From 295684dc6d4b5d1f777e9fdd65ef614b1d089d5c Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:03:02 -0700 Subject: [PATCH 17/69] perf: skip unrelated shared symlink probes during Git status (#18918) --- src/main/git/source-control/status-read.ts | 17 +++- .../git/status-symlink-probe-budget.test.ts | 81 +++++++++++++++++++ 2 files changed, 96 insertions(+), 2 deletions(-) create mode 100644 src/main/git/status-symlink-probe-budget.test.ts diff --git a/src/main/git/source-control/status-read.ts b/src/main/git/source-control/status-read.ts index 276bf26e646..1ec81694d33 100644 --- a/src/main/git/source-control/status-read.ts +++ b/src/main/git/source-control/status-read.ts @@ -14,7 +14,10 @@ import { } from '../../../shared/git-status-line-stats-cache' import { resolveWorktreeHostPath } from '../../../shared/git-metadata-path' import { gitOptionalLocksDisabledEnv, gitStreamStdout } from '../runner' -import { findExistingWorktreeSymlinkPaths } from '../worktree-symlink-detection' +import { + findExistingWorktreeSymlinkPaths, + getSafeRelativePath +} from '../worktree-symlink-detection' import type { GetStatusOptions } from './get-status-options' import { statusReadLeaseOwner } from './git-read-cache-invalidation' import { detectConflictOperation } from './git-conflict-operation' @@ -89,8 +92,18 @@ async function dropSharedSymlinkUntrackedEntries( if (sharedLinkPaths.length === 0 || !entries.some((entry) => entry.area === 'untracked')) { return } + const untrackedPaths = new Set( + entries.filter((entry) => entry.area === 'untracked').map((entry) => entry.path) + ) + const candidatePaths = sharedLinkPaths.filter((rawPath) => { + const path = getSafeRelativePath(rawPath) + return path.safe && untrackedPaths.has(path.rel) + }) + if (candidatePaths.length === 0) { + return + } const sharedLinks = new Set( - await findExistingWorktreeSymlinkPaths(worktreePath, sharedLinkPaths, { + await findExistingWorktreeSymlinkPaths(worktreePath, candidatePaths, { wslDistro: options.wslDistro }) ) diff --git a/src/main/git/status-symlink-probe-budget.test.ts b/src/main/git/status-symlink-probe-budget.test.ts new file mode 100644 index 00000000000..b312048078b --- /dev/null +++ b/src/main/git/status-symlink-probe-budget.test.ts @@ -0,0 +1,81 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { resolve } from 'node:path' +import { getStatus } from './source-control/status-read' + +const { lstat, stream, conflict } = vi.hoisted(() => ({ + lstat: vi.fn(), + stream: vi.fn(), + conflict: vi.fn() +})) +vi.mock('node:fs/promises', () => ({ lstat })) +vi.mock('./source-control/git-conflict-operation', () => ({ detectConflictOperation: conflict })) +vi.mock('./runner', () => ({ + gitStreamStdout: stream, + gitOptionalLocksDisabledEnv: () => ({ GIT_OPTIONAL_LOCKS: '0' }) +})) + +beforeEach(() => { + vi.clearAllMocks() + conflict.mockResolvedValue(undefined) + lstat.mockResolvedValue({ isSymbolicLink: () => true }) + stream.mockImplementation(async (_args, options) => { + options.onStdout('? unrelated.txt\n') + return { stoppedEarly: false } + }) +}) + +function status(sharedLinkPaths: string[]) { + return getStatus('/repo', { sharedLinkPaths, includeLineStats: false }) +} + +describe('status shared symlink probe budget', () => { + it.each([1, 8, 32])( + 'does no unrelated symlink probes for %i configured paths over 100 refreshes', + async (count) => { + const paths = Array.from({ length: count }, (_, index) => `shared-${index}`) + for (let refresh = 0; refresh < 100; refresh++) { + expect((await status(paths)).entries).toEqual([ + { path: 'unrelated.txt', status: 'untracked', area: 'untracked' } + ]) + } + expect(lstat).not.toHaveBeenCalled() + expect(stream).toHaveBeenCalledTimes(100) + } + ) + + it('probes matching normalized paths, retaining duplicate probes and original order', async () => { + stream.mockImplementation(async (_args, options) => { + options.onStdout('? link\n? 日本 語\n? unrelated.txt\n') + return { stoppedEarly: false } + }) + const result = await status(['absent', ' /link ', '\\日本 語', 'link', '../link', 'C:link']) + expect(lstat.mock.calls.map(([path]) => path)).toEqual([ + resolve('/repo', 'link'), + resolve('/repo', '日本 語'), + resolve('/repo', 'link') + ]) + expect(result.entries.map((entry) => entry.path)).toEqual(['unrelated.txt']) + }) + + it('rechecks matching paths after the filesystem changes and preserves unreadable paths', async () => { + const paths = ['unrelated.txt'] + expect((await status(paths)).entries).toEqual([]) + lstat.mockResolvedValueOnce({ isSymbolicLink: () => false }) + expect((await status(paths)).entries).toHaveLength(1) + lstat.mockRejectedValueOnce(new Error('EACCES')) + expect((await status(paths)).entries).toHaveLength(1) + expect(lstat).toHaveBeenCalledTimes(3) + }) + + it('does not broaden exact path matching to descendants or case variants', async () => { + stream.mockImplementation(async (_args, options) => { + options.onStdout('? link/child\n? LINK\n') + return { stoppedEarly: false } + }) + expect((await status(['link'])).entries.map((entry) => entry.path)).toEqual([ + 'link/child', + 'LINK' + ]) + expect(lstat).not.toHaveBeenCalled() + }) +}) From 4e8e14424d108caa3e38923a80cfce16587f5970 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:03:08 -0700 Subject: [PATCH 18/69] perf: avoid splitting every path during file autocomplete (#18919) --- .../scripts/mobile-file-ranking-benchmark.mjs | 53 +++++++++++++++++++ .../mobile-native-chat-autocomplete.ts | 2 +- .../runtime-mobile-file-path-search.test.ts | 31 +++++++++++ .../runtime-mobile-file-path-search.ts | 2 +- 4 files changed, 86 insertions(+), 2 deletions(-) create mode 100644 config/scripts/mobile-file-ranking-benchmark.mjs diff --git a/config/scripts/mobile-file-ranking-benchmark.mjs b/config/scripts/mobile-file-ranking-benchmark.mjs new file mode 100644 index 00000000000..68ac5b9d977 --- /dev/null +++ b/config/scripts/mobile-file-ranking-benchmark.mjs @@ -0,0 +1,53 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { readFileSync } from 'node:fs' +import { stripTypeScriptTypes } from 'node:module' +import { performance } from 'node:perf_hooks' + +const baseline = process.argv[2] +if (!baseline) { + throw new Error('Usage: node config/scripts/mobile-file-ranking-benchmark.mjs ') +} +async function load(source) { + const js = stripTypeScriptTypes(source, { mode: 'transform' }) + return await import(`data:text/javascript;base64,${Buffer.from(js).toString('base64')}`) +} +function measure(fn, paths, query) { + for (let warmup = 0; warmup < 10; warmup++) { + fn(paths, query, 16) + } + const samples = [] + for (let i = 0; i < 9; i++) { + const start = performance.now() + fn(paths, query, 16) + samples.push(performance.now() - start) + } + return samples.sort((a, b) => a - b)[4] +} +const results = [] +for (const [file, name] of [ + ['src/main/runtime/runtime-mobile-file-path-search.ts', 'rankRuntimeMobileFilePaths'], + ['mobile/src/session/mobile-native-chat-autocomplete.ts', 'rankSuggestions'] +]) { + const before = ( + await load(execFileSync('git', ['show', `${baseline}:${file}`], { encoding: 'utf8' })) + )[name] + const after = (await load(readFileSync(file, 'utf8')))[name] + for (const count of [100, 100000]) { + const paths = Array.from( + { length: count }, + (_, i) => `src/components/workspace/group-${i % 100}/file-${i}.tsx` + ) + for (const query of ['file-9', 'missing', 'workspace']) { + assert.deepEqual(after(paths, query, 16), before(paths, query, 16)) + results.push({ + function: name, + paths: count, + query, + beforeMs: measure(before, paths, query), + afterMs: measure(after, paths, query) + }) + } + } +} +console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2)) diff --git a/mobile/src/session/mobile-native-chat-autocomplete.ts b/mobile/src/session/mobile-native-chat-autocomplete.ts index 548d0614837..8de107c9fb0 100644 --- a/mobile/src/session/mobile-native-chat-autocomplete.ts +++ b/mobile/src/session/mobile-native-chat-autocomplete.ts @@ -81,7 +81,7 @@ export function rankSuggestions(candidates: readonly string[], query: string, li const substring: string[] = [] for (const candidate of candidates) { const lower = candidate.toLowerCase() - const base = lower.split('/').pop() ?? lower + const base = lower.slice(lower.lastIndexOf('/') + 1) if (lower.startsWith(q) || base.startsWith(q)) { prefix.push(candidate) } else if (lower.includes(q)) { diff --git a/src/main/runtime/runtime-mobile-file-path-search.test.ts b/src/main/runtime/runtime-mobile-file-path-search.test.ts index 495a7933842..0be5ecad50e 100644 --- a/src/main/runtime/runtime-mobile-file-path-search.test.ts +++ b/src/main/runtime/runtime-mobile-file-path-search.test.ts @@ -6,6 +6,37 @@ import { } from './runtime-mobile-file-path-search' describe('rankRuntimeMobileFilePaths', () => { + it('preserves basename matching, ordering and total counts for unusual paths', () => { + const paths = [ + '', + '/', + 'a/', + 'a//b.ts', + 'b.ts', + 'B.TS', + 'a\\b.ts', + '界/😀.ts', + '.hidden', + 'a/./b.ts' + ] + for (const query of ['', ' ', 'b', '.ts', '😀', '/', 'a\\', 'missing']) { + for (const limit of [0, 1, 3, 100]) { + const q = query.trim().toLowerCase() + const prefix = paths.filter((path) => { + const lower = path.toLowerCase() + return lower.startsWith(q) || (lower.split('/').pop() ?? lower).startsWith(q) + }) + const other = paths.filter( + (path) => !prefix.includes(path) && path.toLowerCase().includes(q) + ) + expect(rankRuntimeMobileFilePaths(paths, query, limit)).toEqual({ + paths: [...prefix, ...other].slice(0, limit), + totalCount: prefix.length + other.length + }) + } + } + }) + it('ranks path and basename prefixes before substrings and caps output', () => { expect( rankRuntimeMobileFilePaths( diff --git a/src/main/runtime/runtime-mobile-file-path-search.ts b/src/main/runtime/runtime-mobile-file-path-search.ts index 13213d6dfcf..1455028c798 100644 --- a/src/main/runtime/runtime-mobile-file-path-search.ts +++ b/src/main/runtime/runtime-mobile-file-path-search.ts @@ -76,7 +76,7 @@ export function rankRuntimeMobileFilePaths( let totalCount = 0 for (const path of paths) { const lower = path.toLowerCase() - const basename = lower.split('/').pop() ?? lower + const basename = lower.slice(lower.lastIndexOf('/') + 1) if (lower.startsWith(normalizedQuery) || basename.startsWith(normalizedQuery)) { totalCount++ if (prefix.length < limit) { From 388e9fb77667d46e5732201bb8000da23d145cbf Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:03:14 -0700 Subject: [PATCH 19/69] perf: avoid rescanning emitted source in analysis guards (#18920) --- .../source-string-blanking-benchmark.mjs | 79 +++++++++++++++++++ .../source-scan/source-tree-scan.test.ts | 7 ++ src/shared/source-scan/source-tree-scan.ts | 13 ++- 3 files changed, 96 insertions(+), 3 deletions(-) create mode 100644 config/scripts/source-string-blanking-benchmark.mjs diff --git a/config/scripts/source-string-blanking-benchmark.mjs b/config/scripts/source-string-blanking-benchmark.mjs new file mode 100644 index 00000000000..de677b9baa9 --- /dev/null +++ b/config/scripts/source-string-blanking-benchmark.mjs @@ -0,0 +1,79 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { stripTypeScriptTypes } from 'node:module' +import { performance } from 'node:perf_hooks' +import { blankStringContents as after } from '../../src/shared/source-scan/source-tree-scan.ts' + +const ref = process.argv[2] +if (!ref) { + throw new Error('Usage: node config/scripts/source-string-blanking-benchmark.mjs ') +} +const source = execFileSync('git', ['show', `${ref}:src/shared/source-scan/source-tree-scan.ts`], { + encoding: 'utf8' +}) +const { blankStringContents: before } = await import( + `data:text/javascript;base64,${Buffer.from(stripTypeScriptTypes(source)).toString('base64')}` +) +const tokens = [ + 'a', + '/', + '*', + ' ', + '\n', + '\r', + '\t', + '\u00a0', + '\u2028', + '"', + "'", + '`', + '${', + '}', + '{', + '\\', + '(', + ')', + '[', + ']', + '=', + '+', + '-', + ';' +] +let seed = 173 +for (let sample = 0; sample < 3000; sample++) { + let input = '' + for (let token = 0; token < 40; token++) { + seed = (Math.imul(seed, 1664525) + 1013904223) >>> 0 + input += tokens[seed % tokens.length] + } + assert.equal(after(input), before(input), JSON.stringify(input)) + assert.equal(after(input, true), before(input, true), JSON.stringify(input)) +} +function measure(fn, input) { + const samples = [] + for (let run = 0; run < 3; run++) { + const start = performance.now() + fn(input) + samples.push(performance.now() - start) + } + return samples.sort((a, b) => a - b)[1] +} +const results = [] +for (const lines of [100, 1000, 5000, 10000]) { + const input = 'const x = value / 2;\n'.repeat(lines) + assert.equal(after(input), before(input)) + results.push({ + lines, + bytes: Buffer.byteLength(input), + beforeMs: measure(before, input), + afterMs: measure(after, input) + }) +} +console.log( + JSON.stringify( + { node: process.version, platform: process.platform, differentialCases: 3000, results }, + null, + 2 + ) +) diff --git a/src/shared/source-scan/source-tree-scan.test.ts b/src/shared/source-scan/source-tree-scan.test.ts index be4b72e4ce9..07ddb52a845 100644 --- a/src/shared/source-scan/source-tree-scan.test.ts +++ b/src/shared/source-scan/source-tree-scan.test.ts @@ -47,6 +47,13 @@ describe('stripComments', () => { }) describe('blankStringContents', () => { + it('does not rescan the accumulated source for each division operator', () => { + const source = 'const x = value / 2;\n'.repeat(10000) + const started = performance.now() + expect(blankStringContents(source)).toBe(source) + expect(performance.now() - started).toBeLessThan(200) + }) + it('neutralises parentheses inside a string so a call is matched whole', () => { // A shell script embedded as a string closed the call early, so the options // object fell outside the match and its flags read as absent. diff --git a/src/shared/source-scan/source-tree-scan.ts b/src/shared/source-scan/source-tree-scan.ts index dce7ae1a274..937a2fc39db 100644 --- a/src/shared/source-scan/source-tree-scan.ts +++ b/src/shared/source-scan/source-tree-scan.ts @@ -179,8 +179,7 @@ export function blankStringContentsDesynced(source: string): boolean { * each also has a prefix reading: `!` (non-null assertion vs `!/re/.test(x)`), * `+` `-` `*` `%` `^` `~` (postfix `--`/`++`), and `>` `}` (JSX close). */ -function startsRegexLiteral(emitted: string): boolean { - const prev = emitted.replace(/\s+$/, '').at(-1) +function startsRegexLiteral(prev: string | undefined): boolean { return prev === undefined || '(,=:[&|?;'.includes(prev) } @@ -209,6 +208,7 @@ function findRegexLiteralEnd(source: string, start: number): number { export function blankStringContents(source: string, reportDesync = false): string { let out = '' + let lastSignificantChar: string | undefined let index = 0 let quote: string | null = null // Brace depth per interpolation, so a `}` inside `${ { a: 1 } }` does not @@ -220,6 +220,7 @@ export function blankStringContents(source: string, reportDesync = false): strin templates.push(0) quote = null out += '${' + lastSignificantChar = '{' index += 2 continue } @@ -232,6 +233,7 @@ export function blankStringContents(source: string, reportDesync = false): strin templates.pop() quote = '`' out += char + lastSignificantChar = char index += 1 continue } @@ -256,6 +258,7 @@ export function blankStringContents(source: string, reportDesync = false): strin if (char === quote) { quote = null out += char + lastSignificantChar = char } else { out += char === '\n' ? char : ' ' } @@ -270,10 +273,11 @@ export function blankStringContents(source: string, reportDesync = false): strin // comments first, but this runs standalone too, and at index 0 a file // starting with a banner comment read as one giant regex. const next = source[index + 1] - if (char === '/' && next !== '/' && next !== '*' && startsRegexLiteral(out)) { + if (char === '/' && next !== '/' && next !== '*' && startsRegexLiteral(lastSignificantChar)) { const end = findRegexLiteralEnd(source, index) if (end !== -1) { out += `/${' '.repeat(end - index - 1)}` + lastSignificantChar = '/' index = end continue } @@ -282,6 +286,9 @@ export function blankStringContents(source: string, reportDesync = false): strin quote = char } out += char + if (/\S/.test(char)) { + lastSignificantChar = char + } index += 1 } if (reportDesync) { From fb7b75d55dc57a4b6c5ce9437869e6505f63178a Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:03:19 -0700 Subject: [PATCH 20/69] perf(cli): skip feature formatters during help and error startup (#18923) * perf(cli): load error reporting without feature formatters * test(cli): follow extracted error reporter in import guard * chore(cli): track cli-error.ts in deferral equivalence baseline The equivalence script restores TOUCHED files from the baseline rev to rebuild the pre-deferral CLI. reportCliError/formatCliError moved from format.ts into cli-error.ts, so the baseline arm must also drop cli-error.ts (absent at older revs) or the old tree would still compile against the new module. --- .../scripts/benchmark-cli-error-imports.mjs | 121 ++++++++++++++ ...li-runtime-client-deferral-equivalence.mjs | 23 ++- src/cli/cli-error.ts | 144 +++++++++++++++++ src/cli/format.ts | 147 +----------------- src/cli/index.ts | 2 +- src/cli/runtime-client-deferral.test.ts | 7 +- 6 files changed, 289 insertions(+), 155 deletions(-) create mode 100644 config/scripts/benchmark-cli-error-imports.mjs create mode 100644 src/cli/cli-error.ts diff --git a/config/scripts/benchmark-cli-error-imports.mjs b/config/scripts/benchmark-cli-error-imports.mjs new file mode 100644 index 00000000000..a4648f84aec --- /dev/null +++ b/config/scripts/benchmark-cli-error-imports.mjs @@ -0,0 +1,121 @@ +import assert from 'node:assert/strict' +import { createRequire } from 'node:module' +import { existsSync, realpathSync } from 'node:fs' +import { delimiter, join, resolve } from 'node:path' + +// Emit each revision with tsc -p config/tsconfig.cli.json --outDir --composite false --incremental false. +// Run: node config/scripts/benchmark-cli-error-imports.mjs +const [beforeDir, afterDir] = process.argv.slice(2) +assert.ok(beforeDir && afterDir, 'Pass distinct before and after TypeScript output directories.') +assert.notEqual( + realpathSync(beforeDir), + realpathSync(afterDir), + 'Do not compare a build to itself.' +) +const entries = { + before: join(resolve(beforeDir), 'cli', 'index.js'), + after: join(resolve(afterDir), 'cli', 'index.js') +} +for (const entry of Object.values(entries)) { + assert.ok(existsSync(entry), `Missing emitted CLI: ${entry}`) +} + +const { runProcessSync } = createRequire(import.meta.url)( + join(resolve(afterDir), 'shared', 'child-process', 'run-process.js') +) + +const child = String.raw` + const { performance } = require('node:perf_hooks') + const { writeSync } = require('node:fs') + const { createHash } = require('node:crypto') + const { basename } = require('node:path') + let stdout = '', stderr = '' + process.stdout.write = (text) => { stdout += text; return true } + process.stderr.write = (text) => { stderr += text; return true } + const started = performance.now() + const cli = require(process.argv[1]) + const importMs = performance.now() - started + cli.main(JSON.parse(process.argv[2])).then(() => { + const totalMs = performance.now() - started + const modules = Object.keys(require.cache) + writeSync(1, JSON.stringify({ + importMs, totalMs, modules: modules.length, + featureFormatters: modules.filter((file) => ['browser', 'terminal', 'project', 'automation', 'workspace', 'computer'].some((name) => basename(file) === name + '-format.js')), + stdout: createHash('sha256').update(stdout).digest('hex'), + stderr: createHash('sha256').update(stderr).digest('hex'), + exitCode: process.exitCode || 0 + })) + process.exitCode = 0 + }).catch((error) => { writeSync(2, String(error)); process.exitCode = 1 }) +` +const cases = [ + ['--help'], + ['help', 'terminal', 'read'], + ['does-not-exist'], + ['computer', 'click', '--does-not-exist'], + ['does-not-exist', '--json'] +] +const median = (values) => [...values].sort((a, b) => a - b)[Math.floor(values.length / 2)] +const summarize = (samples) => ({ + importMs: median(samples.map((sample) => sample.importMs)), + totalMs: median(samples.map((sample) => sample.totalMs)), + modules: samples[0].modules +}) +const rows = [] +for (const args of cases) { + const samples = { before: [], after: [] } + let expected + for (let run = 0; run < 22; run++) { + for (const variant of run % 2 ? ['after', 'before'] : ['before', 'after']) { + const result = runProcessSync({ + program: process.execPath, + args: ['-e', child, entries[variant], JSON.stringify(args)], + timeoutMs: 30_000, + env: { + ...process.env, + NODE_PATH: [resolve('node_modules'), process.env.NODE_PATH] + .filter(Boolean) + .join(delimiter) + } + }) + assert.equal(result.timedOut, false, 'CLI child timed out.') + assert.equal(result.code, 0, result.stderr) + const sample = JSON.parse(result.stdout) + const output = { stdout: sample.stdout, stderr: sample.stderr, exitCode: sample.exitCode } + expected ??= output + assert.deepEqual(output, expected, `${variant} output changed for ${args.join(' ')}`) + if (variant === 'after') { + assert.deepEqual( + sample.featureFormatters, + [], + 'Help and syntax errors must skip feature formatters.' + ) + } + if (run >= 2) { + samples[variant].push(sample) + } + } + } + assert.ok(samples.after[0].modules < samples.before[0].modules, 'Expected fewer loaded modules.') + rows.push({ + args, + before: summarize(samples.before), + after: summarize(samples.after), + output: expected, + samples + }) +} +console.log( + JSON.stringify( + { + node: process.version, + platform: process.platform, + measurement: + 'Fresh-process import + main; excludes process creation; warmed filesystem; 2 warmups and 20 samples per variant, alternating order.', + entries, + rows + }, + null, + 2 + ) +) diff --git a/config/scripts/cli-runtime-client-deferral-equivalence.mjs b/config/scripts/cli-runtime-client-deferral-equivalence.mjs index f443bf3b9f9..a231b5ba75a 100644 --- a/config/scripts/cli-runtime-client-deferral-equivalence.mjs +++ b/config/scripts/cli-runtime-client-deferral-equivalence.mjs @@ -2,7 +2,7 @@ // Equivalence check for deferring the RuntimeClient module graph in the CLI. // // Builds the CLI twice with the REAL tsc emit — once from the working tree and -// once with the seven touched files restored from git HEAD~ (the pre-deferral +// once with the touched files restored from git HEAD~ (the pre-deferral // implementation) — then compares stdout, stderr and exit code BYTE FOR BYTE // across a matrix of invocations. // @@ -13,7 +13,7 @@ // // Usage: node config/scripts/cli-runtime-client-deferral-equivalence.mjs [--baseline ] import { execFileSync, spawnSync } from 'node:child_process' -import { mkdirSync, mkdtempSync, rmSync, writeFileSync, readFileSync } from 'node:fs' +import { existsSync, mkdirSync, mkdtempSync, rmSync, writeFileSync, readFileSync } from 'node:fs' import { join, resolve } from 'node:path' import { fileURLToPath } from 'node:url' @@ -21,8 +21,11 @@ const REPO = fileURLToPath(new URL('../..', import.meta.url)) // The files this change touches. Restoring exactly these from the baseline rev // reconstructs the old implementation without disturbing anything else. +// Files absent at the baseline (e.g. cli-error.ts, split out of format.ts +// later) are removed for the baseline build and put back afterwards. const TOUCHED = [ 'src/cli/args.ts', + 'src/cli/cli-error.ts', 'src/cli/dispatch.ts', 'src/cli/flags.ts', 'src/cli/format.ts', @@ -72,12 +75,16 @@ function buildTree(label, baselineRev) { if (baselineRev) { for (const file of TOUCHED) { const path = join(REPO, file) - restored.push([path, readFileSync(path)]) - const old = execFileSync('git', ['show', `${baselineRev}:${file}`], { + restored.push([path, existsSync(path) ? readFileSync(path) : null]) + const old = spawnSync('git', ['show', `${baselineRev}:${file}`], { cwd: REPO, maxBuffer: 64 * 1024 * 1024 }) - writeFileSync(path, old) + if (old.status === 0) { + writeFileSync(path, old.stdout) + } else { + rmSync(path, { force: true }) + } } } execFileSync( @@ -97,7 +104,11 @@ function buildTree(label, baselineRev) { ) } finally { for (const [path, contents] of restored) { - writeFileSync(path, contents) + if (contents === null) { + rmSync(path, { force: true }) + } else { + writeFileSync(path, contents) + } } } return join(outDir, 'cli/index.js') diff --git a/src/cli/cli-error.ts b/src/cli/cli-error.ts new file mode 100644 index 00000000000..6a87f149079 --- /dev/null +++ b/src/cli/cli-error.ts @@ -0,0 +1,144 @@ +import { computerUseErrorRecoveryData } from '../shared/computer-use-error-recovery' +import { + matchAutomationOwnerConflict, + stripAutomationOwnerConflictCode +} from '../shared/automation-owner-conflict' +import { automationOwnerConflictRecovery } from './automation-owner-conflict-recovery' +import type { RuntimeRpcFailure } from './runtime-client' +import { RuntimeClientError, RuntimeRpcFailureError } from './runtime/types' + +type CliErrorContext = { + commandPath?: readonly string[] +} + +export function formatCliError(error: unknown, context: CliErrorContext = {}): string { + const message = error instanceof Error ? error.message : String(error) + if (error instanceof RuntimeClientError && error.code === 'runtime_unavailable') { + if (hasOrchestrationRequestId(error.data)) { + return message + } + return `${message}\nOrca is not running. Run 'orca open' first.` + } + // Why: error-specific recovery must win over the generic computer fallback. + // Classified from the whole error, not just `.code`: a hop that flattens the class leaves only the token. + const conflict = automationOwnerConflictRecovery(matchAutomationOwnerConflict(error)) + if (conflict) { + return formatMessageWithNextSteps(stripAutomationOwnerConflictCode(message), conflict.nextSteps) + } + if (error instanceof RuntimeClientError) { + const nextSteps = nextStepsFromData(error.data) + if (nextSteps.length > 0) { + return formatMessageWithNextSteps(message, nextSteps) + } + if (error.code === 'invalid_argument' && context.commandPath?.[0] === 'computer') { + return formatMessageWithNextSteps( + message, + computerUseErrorRecoveryData('invalid_argument')?.nextSteps ?? [] + ) + } + } + if ( + error instanceof RuntimeRpcFailureError && + error.response.error.code === 'runtime_unavailable' + ) { + return `${message}\nOrca is not running. Run 'orca open' first.` + } + if (error instanceof RuntimeRpcFailureError) { + return formatMessageWithNextSteps(message, nextStepsFromData(error.response.error.data)) + } + return message +} + +function hasOrchestrationRequestId(data: unknown): boolean { + return ( + data !== null && + typeof data === 'object' && + typeof (data as { orchestrationRequestId?: unknown }).orchestrationRequestId === 'string' + ) +} + +export function reportCliError(error: unknown, json: boolean, context: CliErrorContext = {}): void { + if (json) { + if (error instanceof RuntimeRpcFailureError) { + console.log(JSON.stringify(withAutomationOwnerConflictRecovery(error.response), null, 2)) + } else { + const response: RuntimeRpcFailure = { + id: 'local', + ok: false, + error: { + code: + matchAutomationOwnerConflict(error) ?? + (error instanceof RuntimeClientError ? error.code : 'runtime_error'), + message: stripAutomationOwnerConflictCode( + error instanceof Error ? error.message : String(error) + ), + data: localCliErrorData(error, context) + }, + _meta: { + runtimeId: null + } + } + console.log(JSON.stringify(response, null, 2)) + } + } else { + console.error(formatCliError(error, context)) + } +} + +/** Machine-readable half of the same recovery the human message carries. */ +function withAutomationOwnerConflictRecovery(response: RuntimeRpcFailure): RuntimeRpcFailure { + const code = matchAutomationOwnerConflict(response) + const conflict = automationOwnerConflictRecovery(code) + if (!conflict || !code) { + return response + } + return { + ...response, + error: { + ...response.error, + // Restores the classification a flattening hop dropped, so --json consumers read the conflict, not the transport. + code, + message: stripAutomationOwnerConflictCode(response.error.message), + data: response.error.data ?? conflict + } + } +} + +function formatMessageWithNextSteps(message: string, nextSteps: readonly string[]): string { + if (nextSteps.length === 0) { + return message + } + return `${message}\n${nextSteps.map((step) => `Next step: ${step}`).join('\n')}` +} + +function nextStepsFromData(data: unknown): string[] { + if ( + data && + typeof data === 'object' && + Array.isArray((data as { nextSteps?: unknown }).nextSteps) + ) { + return (data as { nextSteps: unknown[] }).nextSteps.filter( + (step): step is string => typeof step === 'string' + ) + } + return [] +} + +function localCliErrorData(error: unknown, context: CliErrorContext): unknown { + // Why: error-specific recovery must win over the generic computer fallback. + if (error instanceof RuntimeClientError && error.data !== undefined) { + return error.data + } + const conflict = automationOwnerConflictRecovery(matchAutomationOwnerConflict(error)) + if (conflict) { + return conflict + } + if ( + error instanceof RuntimeClientError && + error.code === 'invalid_argument' && + context.commandPath?.[0] === 'computer' + ) { + return computerUseErrorRecoveryData('invalid_argument') + } + return undefined +} diff --git a/src/cli/format.ts b/src/cli/format.ts index 1487a69eea0..dd6b7b739c7 100644 --- a/src/cli/format.ts +++ b/src/cli/format.ts @@ -1,13 +1,8 @@ import type { CliStatusResult } from '../shared/runtime-types' -import { computerUseErrorRecoveryData } from '../shared/computer-use-error-recovery' -import { - matchAutomationOwnerConflict, - stripAutomationOwnerConflictCode -} from '../shared/automation-owner-conflict' -import { automationOwnerConflictRecovery } from './automation-owner-conflict-recovery' import { prepareComputerCliJsonResult } from './computer-format' -import type { RuntimeRpcFailure, RuntimeRpcSuccess } from './runtime-client' -import { RuntimeClientError, RuntimeRpcFailureError } from './runtime/types' +import type { RuntimeRpcSuccess } from './runtime-client' + +export { formatCliError, reportCliError } from './cli-error' export { formatBrowserProfileList, @@ -67,10 +62,6 @@ export { formatWorktreeShow } from './workspace-format' -type CliErrorContext = { - commandPath?: readonly string[] -} - export function printResult( response: RuntimeRpcSuccess, json: boolean, @@ -83,138 +74,6 @@ export function printResult( console.log(formatter(response.result)) } -export function formatCliError(error: unknown, context: CliErrorContext = {}): string { - const message = error instanceof Error ? error.message : String(error) - if (error instanceof RuntimeClientError && error.code === 'runtime_unavailable') { - if (hasOrchestrationRequestId(error.data)) { - return message - } - return `${message}\nOrca is not running. Run 'orca open' first.` - } - // Why: error-specific recovery must win over the generic computer fallback. - // Classified from the whole error, not just `.code`: a hop that flattens the class leaves only the token. - const conflict = automationOwnerConflictRecovery(matchAutomationOwnerConflict(error)) - if (conflict) { - return formatMessageWithNextSteps(stripAutomationOwnerConflictCode(message), conflict.nextSteps) - } - if (error instanceof RuntimeClientError) { - const nextSteps = nextStepsFromData(error.data) - if (nextSteps.length > 0) { - return formatMessageWithNextSteps(message, nextSteps) - } - if (error.code === 'invalid_argument' && context.commandPath?.[0] === 'computer') { - return formatMessageWithNextSteps( - message, - computerUseErrorRecoveryData('invalid_argument')?.nextSteps ?? [] - ) - } - } - if ( - error instanceof RuntimeRpcFailureError && - error.response.error.code === 'runtime_unavailable' - ) { - return `${message}\nOrca is not running. Run 'orca open' first.` - } - if (error instanceof RuntimeRpcFailureError) { - return formatMessageWithNextSteps(message, nextStepsFromData(error.response.error.data)) - } - return message -} - -function hasOrchestrationRequestId(data: unknown): boolean { - return ( - data !== null && - typeof data === 'object' && - typeof (data as { orchestrationRequestId?: unknown }).orchestrationRequestId === 'string' - ) -} - -export function reportCliError(error: unknown, json: boolean, context: CliErrorContext = {}): void { - if (json) { - if (error instanceof RuntimeRpcFailureError) { - console.log(JSON.stringify(withAutomationOwnerConflictRecovery(error.response), null, 2)) - } else { - const response: RuntimeRpcFailure = { - id: 'local', - ok: false, - error: { - code: - matchAutomationOwnerConflict(error) ?? - (error instanceof RuntimeClientError ? error.code : 'runtime_error'), - message: stripAutomationOwnerConflictCode( - error instanceof Error ? error.message : String(error) - ), - data: localCliErrorData(error, context) - }, - _meta: { - runtimeId: null - } - } - console.log(JSON.stringify(response, null, 2)) - } - } else { - console.error(formatCliError(error, context)) - } -} - -/** Machine-readable half of the same recovery the human message carries. */ -function withAutomationOwnerConflictRecovery(response: RuntimeRpcFailure): RuntimeRpcFailure { - const code = matchAutomationOwnerConflict(response) - const conflict = automationOwnerConflictRecovery(code) - if (!conflict || !code) { - return response - } - return { - ...response, - error: { - ...response.error, - // Restores the classification a flattening hop dropped, so --json consumers read the conflict, not the transport. - code, - message: stripAutomationOwnerConflictCode(response.error.message), - data: response.error.data ?? conflict - } - } -} - -function formatMessageWithNextSteps(message: string, nextSteps: readonly string[]): string { - if (nextSteps.length === 0) { - return message - } - return `${message}\n${nextSteps.map((step) => `Next step: ${step}`).join('\n')}` -} - -function nextStepsFromData(data: unknown): string[] { - if ( - data && - typeof data === 'object' && - Array.isArray((data as { nextSteps?: unknown }).nextSteps) - ) { - return (data as { nextSteps: unknown[] }).nextSteps.filter( - (step): step is string => typeof step === 'string' - ) - } - return [] -} - -function localCliErrorData(error: unknown, context: CliErrorContext): unknown { - // Why: error-specific recovery must win over the generic computer fallback. - if (error instanceof RuntimeClientError && error.data !== undefined) { - return error.data - } - const conflict = automationOwnerConflictRecovery(matchAutomationOwnerConflict(error)) - if (conflict) { - return conflict - } - if ( - error instanceof RuntimeClientError && - error.code === 'invalid_argument' && - context.commandPath?.[0] === 'computer' - ) { - return computerUseErrorRecoveryData('invalid_argument') - } - return undefined -} - export type HostListEntry = { kind: 'local' | 'ssh' | 'environment' name: string diff --git a/src/cli/index.ts b/src/cli/index.ts index 9389113b195..a0e1354307f 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -15,7 +15,7 @@ import { resolveHostFlagEnvironmentId } from './execution-host-flag' import { listSshTargets } from './host-selector-alternatives' -import { reportCliError } from './format' +import { reportCliError } from './cli-error' import { printHelp } from './help' import type { RuntimeClient } from './runtime-client' import { COMMAND_SPECS } from './specs' diff --git a/src/cli/runtime-client-deferral.test.ts b/src/cli/runtime-client-deferral.test.ts index 658cc60f0a4..5fcdf8b9686 100644 --- a/src/cli/runtime-client-deferral.test.ts +++ b/src/cli/runtime-client-deferral.test.ts @@ -84,14 +84,12 @@ describe('RuntimeClient module-graph deferral', () => { process.exitCode = 0 }) - // Why: the whole point of the change. These six modules load on EVERY - // invocation, so a value-import of the barrel from any of them drags the - // RuntimeClient graph (zod, ws, tweetnacl) back onto the --help path. + // These eager modules must not pull the RuntimeClient dependency graph into help. it.each([ 'args.ts', 'flags.ts', 'dispatch.ts', - 'format.ts', + 'cli-error.ts', 'selectors.ts', 'execution-host-flag.ts' ])('%s imports error classes from ./runtime/types, not the barrel', (file) => { @@ -110,6 +108,7 @@ describe('RuntimeClient module-graph deferral', () => { expect(source).toContain("import type { RuntimeClient } from './runtime-client'") expect(source).not.toMatch(/^import \{[^}]*RuntimeClient[^}]*\} from '\.\/runtime-client'/m) expect(source).toContain("await import('./runtime-client.js')") + expect(source).toContain("import { reportCliError } from './cli-error'") }) it('constructs no client for --help', async () => { From 9b76ff9217461bdb81f9acb4af55b042dcf70301 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:03:24 -0700 Subject: [PATCH 21/69] perf(explorer): avoid redundant dotfile path filtering (#18929) --- .../benchmark-explorer-dotfile-filter.mjs | 165 ++++++++++++++++++ .../right-sidebar/file-explorer-entries.ts | 5 +- 2 files changed, 166 insertions(+), 4 deletions(-) create mode 100644 config/scripts/benchmark-explorer-dotfile-filter.mjs diff --git a/config/scripts/benchmark-explorer-dotfile-filter.mjs b/config/scripts/benchmark-explorer-dotfile-filter.mjs new file mode 100644 index 00000000000..e66a0ceb5af --- /dev/null +++ b/config/scripts/benchmark-explorer-dotfile-filter.mjs @@ -0,0 +1,165 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import Module from 'node:module' +import { resolve } from 'node:path' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +// Pass the pre-change file-explorer-entries.ts snapshot as the only argument. +const baselinePath = process.argv[2] +assert.ok(baselinePath, 'Pass a pre-change file-explorer-entries.ts snapshot.') +const entry = 'src/renderer/src/components/right-sidebar/file-explorer-entries.ts' +const baseline = readFileSync(baselinePath, 'utf8') +assert.notEqual(baseline, readFileSync(entry, 'utf8'), 'Do not compare the source to itself.') + +async function load(useBaseline) { + const result = await build({ + stdin: { + contents: `export { isDotfileRelativePath } from './${entry}'; +export { createNameFilteredFileExplorerProjection } from './src/renderer/src/components/right-sidebar/file-explorer-name-filter-projection.ts';`, + resolveDir: process.cwd(), + loader: 'ts' + }, + bundle: true, + platform: 'node', + format: 'cjs', + write: false, + logLevel: 'silent', + alias: { '@': resolve('src/renderer/src') }, + plugins: useBaseline + ? [ + { + name: 'baseline-dotfile-predicate', + setup(builder) { + builder.onLoad({ filter: /file-explorer-entries\.ts$/ }, () => ({ + contents: baseline, + loader: 'ts' + })) + } + } + ] + : [] + }) + const module = new Module(resolve('dotfile-benchmark.cjs')) + module.paths = Module._nodeModulePaths(process.cwd()) + module._compile(result.outputFiles[0].text, module.id) + return module.exports +} + +const versions = [await load(true), await load(false)] +let parityCases = 0 +function check(path, depth) { + assert.equal( + versions[0].isDotfileRelativePath(path), + versions[1].isDotfileRelativePath(path), + path + ) + parityCases++ + if (depth > 0) { + for (const character of ['.', '/', '\\', 'a', '\n']) { + check(path + character, depth - 1) + } + } +} +check('', 8) + +function measure(functions, iterations = 1) { + let sink = 0 + const run = (fn) => { + for (let i = 0; i < iterations; i++) { + sink += Number(fn()) + } + } + for (const fn of functions) { + for (let warmup = 0; warmup < 3; warmup++) { + run(fn) + } + } + const samples = [[], []] + for (let round = 0; round < 11; round++) { + for (const variant of round % 2 ? [1, 0] : [0, 1]) { + const start = performance.now() + run(functions[variant]) + samples[variant].push(performance.now() - start) + } + } + return { + beforeMs: samples[0].sort((a, b) => a - b)[5], + afterMs: samples[1].sort((a, b) => a - b)[5], + iterations, + sink + } +} + +const predicates = [] +for (const path of [ + 'a', + '.env', + 'packages/pkg/src/file.tsx', + `a${'.'.repeat(254)}`, + `${'/'.repeat(4096)}.`, + `${'../'.repeat(1000)}file.ts`, + '😀/.你好', + '\n/.\n' +]) { + check(path, 0) + predicates.push({ + pathLength: path.length, + prefix: path.slice(0, 40), + ...measure( + versions.map((version) => () => version.isDotfileRelativePath(path)), + 10_000 + ) + }) +} + +const projections = [] +for (const count of [1000, 10_000, 100_000]) { + for (const query of ['nonmatching-needle', 'file-42']) { + const args = { + ignoredSet: new Set(['unrelated']), + nameFilter: { + query, + relativePaths: Array.from( + { length: count }, + (_, i) => `packages/package-${i % 50}/src/components/section-${i % 10}/file-${i}.tsx` + ) + }, + showDotfiles: false, + showGitIgnoredFiles: false, + worktreePath: '/workspace' + } + const functions = versions.map( + (version) => () => version.createNameFilteredFileExplorerProjection(args) + ) + const rows = functions.map((fn) => { + const projection = fn() + return Array.from({ length: projection.getVisibleCount() }, (_, i) => + projection.getRowAtIndex(i) + ) + }) + assert.deepEqual(rows[0], rows[1]) + projections.push({ + count, + query, + visibleRows: rows[0].length, + ...measure(functions.map((fn) => () => fn().getVisibleCount())) + }) + } +} +console.log( + JSON.stringify( + { + node: process.version, + platform: process.platform, + baselinePath: resolve(baselinePath), + parityCases, + samples: 11, + warmups: 3, + predicates, + projections + }, + null, + 2 + ) +) diff --git a/src/renderer/src/components/right-sidebar/file-explorer-entries.ts b/src/renderer/src/components/right-sidebar/file-explorer-entries.ts index 4e5b6e7b423..f21116c7822 100644 --- a/src/renderer/src/components/right-sidebar/file-explorer-entries.ts +++ b/src/renderer/src/components/right-sidebar/file-explorer-entries.ts @@ -9,8 +9,5 @@ function isDotfileSegment(segment: string): boolean { } export function isDotfileRelativePath(relativePath: string): boolean { - return relativePath - .split(/[\\/]+/) - .filter(Boolean) - .some(isDotfileSegment) + return relativePath.split(/[\\/]+/).some(isDotfileSegment) } From d1e62419b6af046a337e1d9a4dee0c7f79ff4508 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:03:28 -0700 Subject: [PATCH 22/69] perf(watcher): stop admitting stats after batch cancellation (#18931) --- .../filesystem-watcher-local-events.test.ts | 30 +++++++++++++++++++ .../ipc/filesystem-watcher-local-events.ts | 5 +++- 2 files changed, 34 insertions(+), 1 deletion(-) diff --git a/src/main/ipc/filesystem-watcher-local-events.test.ts b/src/main/ipc/filesystem-watcher-local-events.test.ts index e08145e7ad2..1907383bd6b 100644 --- a/src/main/ipc/filesystem-watcher-local-events.test.ts +++ b/src/main/ipc/filesystem-watcher-local-events.test.ts @@ -12,6 +12,7 @@ vi.mock('fs/promises', () => ({ stat: statMock })) vi.mock('./parcel-watcher-process', () => ({ subscribeViaWatcherProcess: subscribeMock })) import { createLocalWatcher } from './filesystem-watcher-local-events' +import { cancelLocalBatchFlush } from './filesystem-watcher-batch-control' function deferred(): { promise: Promise; resolve: (value: T) => void } { let resolve!: (value: T) => void @@ -170,6 +171,35 @@ describe('local filesystem watcher flush serialization', () => { ) }) + it('starts no further stats when a full inflight batch is cancelled', async () => { + const eventCount = 5_000 + const pendingStats = deferred<{ isDirectory: () => boolean }>() + statMock.mockReturnValue(pendingStats.promise) + const root = await createLocalWatcher('/repo', '/repo') + root.listeners.set(1, sender as never) + + watcherCallback?.( + null, + Array.from({ length: eventCount }, (_, index) => ({ + type: 'update' as const, + path: `/repo/file-${index}.ts` + })) + ) + vi.advanceTimersByTime(WATCH_BATCH_TRAILING_MS) + await flushMicrotasks() + expect(statMock).toHaveBeenCalledTimes(8) + + cancelLocalBatchFlush(root) + pendingStats.resolve({ isDirectory: () => false }) + for (let i = 0; i < eventCount * 4 && root.batch.flushInFlight; i++) { + await Promise.resolve() + } + + expect(root.batch.flushInFlight).toBe(false) + expect(statMock).toHaveBeenCalledTimes(8) + expect(sender.send).not.toHaveBeenCalled() + }) + it('leaves an open debounce window to the armed timer instead of draining early', async () => { const firstStat = deferred<{ isDirectory: () => boolean }>() const secondStat = deferred<{ isDirectory: () => boolean }>() diff --git a/src/main/ipc/filesystem-watcher-local-events.ts b/src/main/ipc/filesystem-watcher-local-events.ts index 57ef7da4e1f..9f31adde4bd 100644 --- a/src/main/ipc/filesystem-watcher-local-events.ts +++ b/src/main/ipc/filesystem-watcher-local-events.ts @@ -139,7 +139,10 @@ async function flushBatch(root: WatchedRoot): Promise { DIRECTORY_STAT_CONCURRENCY, async (evt) => { // Why: a deleted path can't be stat'd; leave isDirectory undefined and let the renderer infer from dirCache. - const isDirectory = evt.type === 'delete' ? undefined : await tryStatIsDirectory(evt.path) + const isDirectory = + root.batch.cancelled || evt.type === 'delete' + ? undefined + : await tryStatIsDirectory(evt.path) return { kind: evt.type, From 6f28e019b5f52e858ee11286956982c975030af7 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:03:33 -0700 Subject: [PATCH 23/69] perf(hooks): use native reverse search for transcript lines (#18936) --- .../benchmark-transcript-reverse-lines.mjs | 124 ++++++++++++++++++ .../transcript-reader.test.ts | 70 ++++++++++ .../agent-hook-listener/transcript-reader.ts | 6 +- 3 files changed, 196 insertions(+), 4 deletions(-) create mode 100644 config/scripts/benchmark-transcript-reverse-lines.mjs create mode 100644 src/shared/agent-hook-listener/transcript-reader.test.ts diff --git a/config/scripts/benchmark-transcript-reverse-lines.mjs b/config/scripts/benchmark-transcript-reverse-lines.mjs new file mode 100644 index 00000000000..e9d370d1839 --- /dev/null +++ b/config/scripts/benchmark-transcript-reverse-lines.mjs @@ -0,0 +1,124 @@ +import assert from 'node:assert/strict' +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import Module from 'node:module' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +const entry = 'src/shared/agent-hook-listener/transcript-reader.ts' +assert.ok(process.argv[2], 'Pass a pre-change transcript-reader.ts snapshot.') +const baseline = readFileSync(process.argv[2], 'utf8') +assert.notEqual(baseline, readFileSync(entry, 'utf8'), 'Do not compare the source to itself.') + +async function load(useBaseline) { + const result = await build({ + stdin: { + contents: `export * from './${entry}'; +export { extractAssistantTextFromLine } from './src/shared/agent-hook-listener/transcript-entry-text.ts';`, + resolveDir: process.cwd(), + loader: 'ts' + }, + bundle: true, + platform: 'node', + format: 'cjs', + write: false, + logLevel: 'silent', + plugins: useBaseline + ? [ + { + name: 'baseline-transcript-reader', + setup(builder) { + builder.onLoad({ filter: /transcript-reader\.ts$/ }, () => ({ + contents: baseline, + loader: 'ts' + })) + } + } + ] + : [] + }) + const module = new Module(resolve('transcript-benchmark.cjs')) + module.paths = Module._nodeModulePaths(process.cwd()) + module._compile(result.outputFiles[0].text, module.id) + return module.exports +} + +const versions = [await load(true), await load(false)] +function measure(functions, iterations) { + let sink = 0 + const run = (fn) => { + for (let i = 0; i < iterations; i++) { + sink += fn()?.length ?? 0 + } + } + for (const fn of functions) { + for (let i = 0; i < 3; i++) { + run(fn) + } + } + const samples = [[], []] + for (let round = 0; round < 11; round++) { + for (const index of round % 2 ? [1, 0] : [0, 1]) { + const start = performance.now() + run(functions[index]) + samples[index].push((performance.now() - start) / iterations) + } + } + return { + beforeMs: samples[0].sort((a, b) => a - b)[5], + afterMs: samples[1].sort((a, b) => a - b)[5], + iterations, + sink + } +} + +const cases = [ + ['tiny', `${JSON.stringify({ role: 'assistant', content: 'hello' })}\n`, 10000], + ['64KiB line', `${JSON.stringify({ role: 'assistant', content: 'x'.repeat(65500) })}\n`, 100], + [ + '4MiB line', + `${JSON.stringify({ role: 'assistant', content: 'x'.repeat(4 * 1024 * 1024 - 40) })}\n`, + 10 + ], + [ + '1000 short tool lines', + Array.from({ length: 1000 }, () => + JSON.stringify({ role: 'tool', content: 'x'.repeat(100) }) + ).join('\n'), + 50 + ], + [ + 'Unicode line', + `${JSON.stringify({ role: 'assistant', content: '😀漢字'.repeat(16000) })}\n`, + 100 + ], + [ + 'leading and trailing blank lines', + `\n\r\n${JSON.stringify({ role: 'assistant', content: 'hello' })}\n\n`, + 10000 + ] +] +const directory = mkdtempSync(join(tmpdir(), 'orca-transcript-benchmark-')) +try { + for (const [name, text, iterations] of cases) { + const file = join(directory, 'transcript.jsonl') + writeFileSync(file, text) + const scanners = versions.map( + (v) => () => v.findLastExtractedTranscriptLineText(text, v.extractAssistantTextFromLine) + ) + const readers = versions.map((v) => () => v.readLastAssistantFromTranscriptOnce(file)) + assert.equal(scanners[0](), scanners[1](), name) + assert.equal(readers[0](), readers[1](), name) + console.log( + JSON.stringify({ + name, + bytes: Buffer.byteLength(text), + scanner: measure(scanners, iterations), + warmFileReader: measure(readers, Math.min(iterations, 100)) + }) + ) + } +} finally { + rmSync(directory, { recursive: true, force: true }) +} diff --git a/src/shared/agent-hook-listener/transcript-reader.test.ts b/src/shared/agent-hook-listener/transcript-reader.test.ts new file mode 100644 index 00000000000..83780b08388 --- /dev/null +++ b/src/shared/agent-hook-listener/transcript-reader.test.ts @@ -0,0 +1,70 @@ +import { describe, expect, it } from 'vitest' +import { findLastExtractedTranscriptLineText } from './transcript-reader' +import { extractAssistantTextFromLine } from './transcript-entry-text' + +function expectedLines(text: string): string[] { + return text + .split('\n') + .toReversed() + .map((line) => line.trim()) + .filter((line) => line.length > 0) +} + +describe('backward transcript line extraction', () => { + it.each(['', '\n', '\n\n', '\r\n', ' \t\r\n', '\na\n', 'a\nb', '😀\r\n漢字'])( + 'visits each nonblank line once in reverse order for %j', + (text) => { + const seen: string[] = [] + expect( + findLastExtractedTranscriptLineText(text, (line) => { + seen.push(line) + return undefined + }) + ).toBeUndefined() + expect(seen).toEqual(expectedLines(text)) + } + ) + + it('preserves line order and early return across generated delimiters', () => { + let seed = 29 + const fragments = ['a', '\n', '\r\n', ' ', '\t', '😀', '\u2028', '\0'] + for (let sample = 0; sample < 1000; sample++) { + let text = '' + for (let i = 0; i < sample % 100; i++) { + seed = (Math.imul(seed, 1664525) + 1013904223) >>> 0 + text += fragments[seed % fragments.length] + } + const lines = expectedLines(text) + const stop = sample % (lines.length + 1) + const seen: string[] = [] + const result = findLastExtractedTranscriptLineText(text, (line) => { + seen.push(line) + return seen.length === stop + 1 ? line : undefined + }) + expect(seen).toEqual(lines.slice(0, stop + 1)) + expect(result).toBe(lines[stop]) + } + }) + + it('returns the newest assistant message behind a long tool line', () => { + const message = JSON.stringify({ role: 'assistant', content: 'latest 😀' }) + const tool = JSON.stringify({ role: 'tool', content: 'x'.repeat(256 * 1024) }) + expect( + findLastExtractedTranscriptLineText( + `\n{"role":"assistant","content":"older"}\r\n${message}\r\n${tool}\n`, + extractAssistantTextFromLine + ) + ).toBe('latest 😀') + }) + + it('treats an empty extracted string as a result and stops before older lines', () => { + const seen: string[] = [] + expect( + findLastExtractedTranscriptLineText('older\nlatest\n', (line) => { + seen.push(line) + return '' + }) + ).toBe('') + expect(seen).toEqual(['latest']) + }) +}) diff --git a/src/shared/agent-hook-listener/transcript-reader.ts b/src/shared/agent-hook-listener/transcript-reader.ts index f9b1472384e..89df71da2df 100644 --- a/src/shared/agent-hook-listener/transcript-reader.ts +++ b/src/shared/agent-hook-listener/transcript-reader.ts @@ -88,10 +88,8 @@ export function findLastExtractedTranscriptLineText( ): string | undefined { let lineEnd = text.length - for (let index = text.length - 1; index >= -1; index--) { - if (index >= 0 && text.charCodeAt(index) !== 10) { - continue - } + while (lineEnd > 0) { + const index = text.lastIndexOf('\n', lineEnd - 1) const line = text.slice(index + 1, lineEnd).trim() if (line.length > 0) { From bf073b833e648b235d0d0ae49e29f87f5146d0ac Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:03:37 -0700 Subject: [PATCH 24/69] perf(skills): skip symlink probes beyond discovery depth (#18937) --- config/scripts/benchmark-skill-depth.mjs | 122 +++++++++++++++++++ src/main/skills/skill-root-file-walk.test.ts | 28 +++++ src/main/skills/skill-root-file-walk.ts | 14 ++- 3 files changed, 159 insertions(+), 5 deletions(-) create mode 100644 config/scripts/benchmark-skill-depth.mjs diff --git a/config/scripts/benchmark-skill-depth.mjs b/config/scripts/benchmark-skill-depth.mjs new file mode 100644 index 00000000000..1ebb606f93c --- /dev/null +++ b/config/scripts/benchmark-skill-depth.mjs @@ -0,0 +1,122 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import * as fs from 'node:fs/promises' +import Module from 'node:module' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +// Pass a pre-change skill-root-file-walk.ts snapshot as the only argument. +const baselinePath = process.argv[2] +const brokenLinks = process.argv.includes('--broken') +assert.ok(baselinePath, 'Pass a pre-change skill-root-file-walk.ts snapshot.') +const entry = 'src/main/skills/skill-root-file-walk.ts' +const baseline = readFileSync(baselinePath, 'utf8') +assert.notEqual(baseline, readFileSync(entry, 'utf8'), 'Do not compare the source to itself.') +let statCalls = 0 + +async function load(useBaseline) { + const result = await build({ + entryPoints: [entry], + bundle: true, + platform: 'node', + format: 'cjs', + write: false, + logLevel: 'silent', + plugins: useBaseline + ? [ + { + name: 'baseline-skill-depth', + setup(builder) { + builder.onLoad({ filter: /skill-root-file-walk\.ts$/ }, () => ({ + contents: baseline, + loader: 'ts' + })) + } + } + ] + : [] + }) + const module = new Module(resolve('skill-depth-benchmark.cjs')) + module.paths = Module._nodeModulePaths(process.cwd()) + const originalRequire = module.require.bind(module) + module.require = (name) => + name === 'node:fs/promises' + ? { + ...fs, + stat: (...args) => { + statCalls++ + return fs.stat(...args) + } + } + : originalRequire(name) + module._compile(result.outputFiles[0].text, module.id) + return module.exports.findSkillFiles +} + +const before = await load(true) +const after = await load(false) +const median = (values) => values.sort((a, b) => a - b)[Math.floor(values.length / 2)] +const temporaryRoot = await fs.mkdtemp(join(tmpdir(), 'orca-skill-depth-benchmark-')) +try { + for (const links of [0, 8, 100, 1000]) { + const root = join(temporaryRoot, String(links)) + const edge = join(root, 'a', 'b', 'c', 'd') + const target = join(temporaryRoot, 'target') + await fs.mkdir(edge, { recursive: true }) + await fs.mkdir(target, { recursive: true }) + await fs.writeFile(join(target, 'SKILL.md'), 'skill') + await fs.writeFile(join(edge, 'SKILL.md'), 'edge') + for (let index = 0; index < links; index++) { + await fs.symlink( + brokenLinks ? join(target, 'missing') : target, + join(edge, `link${index}`), + process.platform === 'win32' ? 'junction' : 'dir' + ) + } + for (const depth of [4, 5]) { + const timings = { before: [], after: [] } + const counts = {} + let rows + for (let sample = 0; sample < 13; sample++) { + const versions = + sample % 2 + ? [ + ['after', after], + ['before', before] + ] + : [ + ['before', before], + ['after', after] + ] + for (const [name, walk] of versions) { + statCalls = 0 + const start = performance.now() + const result = await walk(root, depth) + const elapsed = performance.now() - start + if (rows) { + assert.deepEqual(result, rows) + } + rows = result + counts[name] = statCalls + if (sample >= 2) { + timings[name].push(elapsed) + } + } + } + console.log( + JSON.stringify({ + links, + brokenLinks, + depth, + statCalls: counts, + rows: rows.length, + medianMs: { before: median(timings.before), after: median(timings.after) } + }) + ) + } + } +} finally { + await fs.rm(temporaryRoot, { recursive: true, force: true }) +} diff --git a/src/main/skills/skill-root-file-walk.test.ts b/src/main/skills/skill-root-file-walk.test.ts index 3ca80bdb495..ad84eced6b6 100644 --- a/src/main/skills/skill-root-file-walk.test.ts +++ b/src/main/skills/skill-root-file-walk.test.ts @@ -45,6 +45,34 @@ describe('findSkillFiles', () => { expect(found).toEqual([join(root, 'near', 'SKILL.md')]) }) + it('does not stat directory links beyond the depth bound but still follows in-bound links', async () => { + const base = await makeTree() + const root = join(base, 'skills') + const edge = join(root, 'a', 'b', 'c', 'd') + const target = join(base, 'linked') + await writeFileAt(join(edge, 'SKILL.md')) + await writeFileAt(join(target, 'SKILL.md')) + for (let index = 0; index < 32; index += 1) { + await symlink( + target, + join(edge, `link${index.toString().padStart(2, '0')}`), + process.platform === 'win32' ? 'junction' : 'dir' + ) + } + const statPaths: string[] = [] + onStat = async (path) => { + statPaths.push(path) + } + + expect(await findSkillFiles(root, 4)).toEqual([join(edge, 'SKILL.md')]) + expect(statPaths).toEqual([]) + expect(await findSkillFiles(root, 5)).toEqual([ + join(edge, 'SKILL.md'), + join(edge, 'link00', 'SKILL.md') + ]) + expect(statPaths).toHaveLength(32) + }) + it('returns nothing for a missing root rather than throwing', async () => { expect(await findSkillFiles(join(await makeTree(), 'absent'), 4)).toEqual([]) }) diff --git a/src/main/skills/skill-root-file-walk.ts b/src/main/skills/skill-root-file-walk.ts index 261a650a967..4be1843f602 100644 --- a/src/main/skills/skill-root-file-walk.ts +++ b/src/main/skills/skill-root-file-walk.ts @@ -43,9 +43,6 @@ export async function findSkillFiles( // indistinguishable from a genuinely small root, and a caller that cached it // would publish "these skills no longer exist". signal?.throwIfAborted() - if (!isWithinDepth(rootPath, dirPath, maxDepth)) { - return - } let resolvedDirPath: string try { resolvedDirPath = await realpath(dirPath) @@ -61,6 +58,8 @@ export async function findSkillFiles( if (!entries) { return } + // Directory entry names add one segment, so siblings share the depth verdict. + let childrenWithinDepth: boolean | undefined for (const entry of entries) { signal?.throwIfAborted() // Why: a staged sibling sits directly in a scanned root, so without this a @@ -86,10 +85,15 @@ export async function findSkillFiles( continue } if (entry.isDirectory()) { - await visit(entryPath) + if ((childrenWithinDepth ??= isWithinDepth(rootPath, entryPath, maxDepth))) { + await visit(entryPath) + } continue } - if (entry.isSymbolicLink()) { + if ( + entry.isSymbolicLink() && + (childrenWithinDepth ??= isWithinDepth(rootPath, entryPath, maxDepth)) + ) { // Why: users commonly symlink agent skill dirs across providers; follow // directory links but guard by realpath so recursive links cannot loop. let linksToDirectory = false From 992360f12126cb4c387273fe805a99baa6fccb4f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:03:42 -0700 Subject: [PATCH 25/69] perf(mobile): cancel direct probes when their owner stops (#18940) * perf(mobile): cancel direct probes when their owner stops * fix(mobile): fence direct migration after supervisor stop --- .../transport/mobile-direct-endpoint-probe.ts | 28 ++- .../mobile-direct-probe-stop-budget.test.ts | 175 ++++++++++++++++++ .../transport/mobile-direct-return-probe.ts | 43 ++++- .../transport/mobile-endpoint-supervisor.ts | 3 +- 4 files changed, 241 insertions(+), 8 deletions(-) create mode 100644 mobile/src/transport/mobile-direct-probe-stop-budget.test.ts diff --git a/mobile/src/transport/mobile-direct-endpoint-probe.ts b/mobile/src/transport/mobile-direct-endpoint-probe.ts index 03264d5f7e5..038f73b93b8 100644 --- a/mobile/src/transport/mobile-direct-endpoint-probe.ts +++ b/mobile/src/transport/mobile-direct-endpoint-probe.ts @@ -31,7 +31,14 @@ export function directPathForEndpoint( // instead of holding the supervisor's operation mutex for the full outer bound. const RECONNECT_GRACE_MS = 2_000 -function waitForAuthenticatedSession(session: RpcClient, timeoutMs: number): Promise { +function waitForAuthenticatedSession( + session: RpcClient, + timeoutMs: number, + signal?: AbortSignal +): Promise { + if (signal?.aborted) { + return Promise.reject(new Error('probe cancelled')) + } if (session.getState() === 'connected') { return Promise.resolve() } @@ -72,7 +79,13 @@ function waitForAuthenticatedSession(session: RpcClient, timeoutMs: number): Pro finish() reject(new Error('probe session authentication timed out')) }, timeoutMs) + const onAbort = (): void => { + finish() + reject(new Error('probe cancelled')) + } + signal?.addEventListener('abort', onAbort, { once: true }) function finish(): void { + signal?.removeEventListener('abort', onAbort) if (timer) { clearTimeout(timer) } @@ -87,8 +100,12 @@ function waitForAuthenticatedSession(session: RpcClient, timeoutMs: number): Pro export async function openAuthenticatedDirectEndpoint( host: HostProfile, openDirect: (endpoint: string) => RpcClient, - timeoutMs: number + timeoutMs: number, + signal?: AbortSignal ): Promise<{ client: RpcClient; path: Exclude } | null> { + if (signal?.aborted) { + return null + } const endpoints = directEndpointUrls(host) return await new Promise((resolve) => { const clients = new Set() @@ -110,8 +127,13 @@ export async function openAuthenticatedDirectEndpoint( continue } clients.add(client) - void waitForAuthenticatedSession(client, timeoutMs).then( + void waitForAuthenticatedSession(client, timeoutMs, signal).then( () => { + if (signal?.aborted) { + client.close() + rejectCandidate() + return + } if (settled) { client.close() return diff --git a/mobile/src/transport/mobile-direct-probe-stop-budget.test.ts b/mobile/src/transport/mobile-direct-probe-stop-budget.test.ts new file mode 100644 index 00000000000..535790ed480 --- /dev/null +++ b/mobile/src/transport/mobile-direct-probe-stop-budget.test.ts @@ -0,0 +1,175 @@ +import { expect, it, vi } from 'vitest' +import { + dependencies, + FakeLogicalClient, + FakeSession, + host +} from './mobile-endpoint-supervisor-test-fakes' +import { MobileEndpointHysteresis } from './mobile-endpoint-hysteresis' +import { createStableLogicalRpcClient } from './stable-logical-rpc-client' +import { MobileEndpointSupervisor } from './mobile-endpoint-supervisor' +vi.mock('react-native', () => ({ Platform: { OS: 'ios' } })) +vi.mock('expo-secure-store', () => ({ WHEN_UNLOCKED_THIS_DEVICE_ONLY: 'when-unlocked' })) +vi.mock('expo-crypto', () => ({ getRandomBytes: (length: number) => new Uint8Array(length) })) +it('closes in-flight candidates and clears their timeout when the owner stops', async () => { + vi.useFakeTimers() + try { + const candidate = new FakeSession('connecting') + const logical = new FakeLogicalClient('connected', 'relay') + const deps = dependencies({ openDirect: vi.fn(() => candidate) }) + const supervisor = new MobileEndpointSupervisor(logical, host, deps) + await supervisor.start() + await vi.advanceTimersByTimeAsync(15_000) + expect(deps.openDirect).toHaveBeenCalledOnce() + supervisor.stop() + await vi.advanceTimersByTimeAsync(0) + expect(candidate.close).toHaveBeenCalledOnce() + expect(vi.getTimerCount()).toBe(0) + await vi.advanceTimersByTimeAsync(12_000) + expect(vi.getTimerCount()).toBe(0) + expect(candidate.close).toHaveBeenCalledOnce() + expect(logical.migrateTo).not.toHaveBeenCalled() + expect(deps.openDirect).toHaveBeenCalledOnce() + } finally { + vi.restoreAllMocks() + vi.useRealTimers() + } +}) + +it('closes an authenticated candidate when stop races its completion', async () => { + vi.useFakeTimers() + try { + const candidate = new FakeSession('connecting') + const logical = new FakeLogicalClient('connected', 'relay') + const deps = dependencies({ openDirect: vi.fn(() => candidate) }) + const supervisor = new MobileEndpointSupervisor(logical, host, deps) + await supervisor.start() + await vi.advanceTimersByTimeAsync(15_000) + candidate.publishState('connected') + supervisor.stop() + await vi.advanceTimersByTimeAsync(0) + expect(candidate.close).toHaveBeenCalledOnce() + expect(logical.migrateTo).not.toHaveBeenCalled() + expect(vi.getTimerCount()).toBe(0) + } finally { + vi.restoreAllMocks() + vi.useRealTimers() + } +}) + +it('preserves an in-flight probe across a transient background pause', async () => { + vi.useFakeTimers() + try { + const candidate = new FakeSession('connecting') + const logical = new FakeLogicalClient('connected', 'relay') + const deps = dependencies({ openDirect: vi.fn(() => candidate) }) + const supervisor = new MobileEndpointSupervisor(logical, host, deps) + await supervisor.start() + await vi.advanceTimersByTimeAsync(15_000) + supervisor.setForeground(false) + await vi.advanceTimersByTimeAsync(0) + expect(candidate.close).not.toHaveBeenCalled() + supervisor.stop() + await vi.advanceTimersByTimeAsync(0) + expect(candidate.close).toHaveBeenCalledOnce() + expect(vi.getTimerCount()).toBe(0) + } finally { + vi.restoreAllMocks() + vi.useRealTimers() + } +}) + +it('releases every candidate when multiple endpoint probes are pending', async () => { + vi.useFakeTimers() + try { + const candidates: FakeSession[] = [] + const logical = new FakeLogicalClient('connected', 'relay') + const deps = dependencies({ + openDirect: vi.fn(() => { + const candidate = new FakeSession('connecting') + candidates.push(candidate) + return candidate + }) + }) + const supervisor = new MobileEndpointSupervisor( + logical, + { + ...host, + endpoints: [{ id: 'alternate', kind: 'tailscale', url: 'ws://100.64.0.2:6768' }] + }, + deps + ) + await supervisor.start() + await vi.advanceTimersByTimeAsync(15_000) + expect(candidates).toHaveLength(2) + supervisor.stop() + await vi.advanceTimersByTimeAsync(0) + expect(vi.getTimerCount()).toBe(0) + for (const candidate of candidates) { + expect(candidate.close).toHaveBeenCalledOnce() + candidate.publishState('connected') + } + await vi.advanceTimersByTimeAsync(60_000) + expect(logical.migrateTo).not.toHaveBeenCalled() + expect(deps.openDirect).toHaveBeenCalledTimes(2) + expect(vi.getTimerCount()).toBe(0) + } finally { + vi.restoreAllMocks() + vi.useRealTimers() + } +}) + +it.each([false, true])( + 'fences migration finishing after stop (already swapped: %s)', + async (alreadySwapped) => { + vi.useFakeTimers() + try { + const recordedMigration = vi.spyOn(MobileEndpointHysteresis.prototype, 'recordMigration') + const relay = new FakeSession('connected') + const logical = createStableLogicalRpcClient(relay, 'relay') + const candidates: FakeSession[] = [] + const deps = dependencies({ + openDirect: vi.fn(() => { + const candidate = new FakeSession('connected') + candidates.push(candidate) + return candidate + }) + }) + const supervisor = new MobileEndpointSupervisor(logical, host, deps) + const migrate = logical.migrateTo.bind(logical) + let release!: () => void + const pending = new Promise((resolve) => { + release = resolve + }) + const migration = vi.spyOn(logical, 'migrateTo').mockImplementation(async (...args) => { + if (alreadySwapped) { + await migrate(...args) + } + await pending + if (!alreadySwapped) { + await migrate(...args) + } + }) + await supervisor.start() + await vi.advanceTimersByTimeAsync(60_000) + expect(migration).toHaveBeenCalledOnce() + const requestsBeforeStop = relay.sendRequest.mock.calls.length + const candidateRequestsBeforeStop = candidates[3].sendRequest.mock.calls.length + const migrationsBeforeStop = recordedMigration.mock.calls.length + supervisor.stop() + release() + await vi.advanceTimersByTimeAsync(0) + expect(logical.getActivePath()).toBe(alreadySwapped ? 'lan' : 'relay') + expect(logical.getGeneration()).toBe(alreadySwapped ? 2 : 1) + expect(relay.sendRequest).toHaveBeenCalledTimes(requestsBeforeStop) + expect(candidates[3].sendRequest).toHaveBeenCalledTimes(candidateRequestsBeforeStop) + expect(recordedMigration).toHaveBeenCalledTimes(migrationsBeforeStop) + expect(candidates[3].close).toHaveBeenCalledTimes(alreadySwapped ? 0 : 1) + expect(vi.getTimerCount()).toBe(0) + logical.close() + } finally { + vi.restoreAllMocks() + vi.useRealTimers() + } + } +) diff --git a/mobile/src/transport/mobile-direct-return-probe.ts b/mobile/src/transport/mobile-direct-return-probe.ts index dfb0572aa38..3ae31edd07f 100644 --- a/mobile/src/transport/mobile-direct-return-probe.ts +++ b/mobile/src/transport/mobile-direct-return-probe.ts @@ -11,6 +11,9 @@ const DIRECT_PROBE_INTERVAL_MS = 15_000 export class DirectReturnProbe { private timer: ReturnType | null = null + private stopped = false + private activeProbe: AbortController | null = null + constructor( private readonly deps: { now: () => number @@ -24,14 +27,18 @@ export class DirectReturnProbe { canSchedule: () => boolean canAttempt: () => boolean beginOperation: () => void - migrate: (client: RpcClient, path: MobileConnectionPath) => Promise + migrate: ( + client: RpcClient, + path: MobileConnectionPath, + shouldAbort: () => boolean + ) => Promise onDirectMigrated: () => Promise afterProbe: () => void } ) {} schedule(delayMs = DIRECT_PROBE_INTERVAL_MS): void { - if (!this.hooks.canSchedule() || this.timer) { + if (this.stopped || !this.hooks.canSchedule() || this.timer) { return } this.timer = this.deps.setTimer(() => { @@ -47,19 +54,34 @@ export class DirectReturnProbe { } } + stop(): void { + this.stopped = true + this.clear() + this.activeProbe?.abort() + } + private async probe(): Promise { + if (this.stopped) { + return + } if (!this.hooks.canAttempt() || !this.hooks.hysteresis.canProbe(this.deps.now())) { this.schedule() return } + const controller = new AbortController() + this.activeProbe = controller this.hooks.beginOperation() let successful: Awaited> = null try { successful = await openAuthenticatedDirectEndpoint( this.hooks.host(), this.deps.openDirect, - 12_000 + 12_000, + controller.signal ) + if (this.stopped) { + return + } if (!successful) { this.hooks.hysteresis.recordDirectFailure(this.deps.now()) return @@ -68,11 +90,24 @@ export class DirectReturnProbe { successful.client.close() return } - await this.hooks.migrate(successful.client, successful.path) + const candidate = successful + // Migration owns the candidate, including closing it if cutover is canceled. successful = null + try { + await this.hooks.migrate(candidate.client, candidate.path, () => this.stopped) + } catch (error) { + if (this.stopped) { + return + } + throw error + } + if (this.stopped) { + return + } this.hooks.hysteresis.recordMigration(this.deps.now()) await this.hooks.onDirectMigrated() } finally { + this.activeProbe = null successful?.client.close() // Why: a relay drop or backoff timer can arrive while the probe owns the // operation mutex; afterProbe releases it and replays deferred recovery. diff --git a/mobile/src/transport/mobile-endpoint-supervisor.ts b/mobile/src/transport/mobile-endpoint-supervisor.ts index 6fca6c0cc17..9ba12f35112 100644 --- a/mobile/src/transport/mobile-endpoint-supervisor.ts +++ b/mobile/src/transport/mobile-endpoint-supervisor.ts @@ -120,7 +120,7 @@ export class MobileEndpointSupervisor { canSchedule: () => this.isActive() && this.logical.getActivePath() === 'relay', canAttempt: () => this.isActive() && !this.operationInFlight, beginOperation: () => (this.operationInFlight = true), - migrate: (client, path) => this.logical.migrateTo(client, path), + migrate: (client, path, abort) => this.logical.migrateTo(client, path, undefined, abort), onDirectMigrated: async () => { this.leaseRotation.clear() this.relayRotationPending = false @@ -195,6 +195,7 @@ export class MobileEndpointSupervisor { stop(): void { this.stopped = true + this.directProbe.stop() this.unsubscribeState?.() this.unsubscribeState = null this.backgroundGrace.stop() From d969af9ecc2d98d2fc52af4d214e7e8c630f7c1f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:03:48 -0700 Subject: [PATCH 26/69] perf(jira): preserve replacement attachment download singleflight (#18944) --- .../attachment-image-cache-generation.test.ts | 57 +++++++++++++++++++ src/main/jira/attachment-image-cache.ts | 5 +- 2 files changed, 61 insertions(+), 1 deletion(-) create mode 100644 src/main/jira/attachment-image-cache-generation.test.ts diff --git a/src/main/jira/attachment-image-cache-generation.test.ts b/src/main/jira/attachment-image-cache-generation.test.ts new file mode 100644 index 00000000000..e92ca5c5a53 --- /dev/null +++ b/src/main/jira/attachment-image-cache-generation.test.ts @@ -0,0 +1,57 @@ +import { beforeEach, describe, expect, it } from 'vitest' +import { + _resetAttachmentImageCache, + clearAttachmentImagesForSite, + getCachedAttachmentDataUrl, + loadAttachmentDataUrlWithCache +} from './attachment-image-cache' + +type Image = { dataUrl: string; byteSize: number } | null +function deferredImage() { + let resolve!: (image: Image) => void + let reject!: (error: Error) => void + const promise = new Promise((done, fail) => { + resolve = done + reject = fail + }) + return { promise, resolve, reject } +} + +beforeEach(_resetAttachmentImageCache) + +describe.each(['site', 'all'] as const)('attachment download after clearing %s', (scope) => { + it.each(['success', 'empty', 'failure'] as const)( + 'keeps the replacement singleflight when the old download completes with %s', + async (outcome) => { + const old = deferredImage() + const replacement = deferredImage() + let downloads = 0 + const load = () => { + downloads += 1 + return downloads === 1 ? old.promise : replacement.promise + } + const args = { siteId: 'site-a', attachmentId: 'image-1', load } + const first = loadAttachmentDataUrlWithCache(args).catch(() => 'old failure') + clearAttachmentImagesForSite(scope === 'site' ? 'site-a' : undefined) + const second = loadAttachmentDataUrlWithCache(args) + expect(downloads).toBe(2) + + if (outcome === 'failure') { + old.reject(new Error('old failure')) + } else { + old.resolve(outcome === 'empty' ? null : { dataUrl: 'old image', byteSize: 3 }) + } + expect(await first).toBe( + outcome === 'failure' ? 'old failure' : outcome === 'empty' ? null : 'old image' + ) + expect(getCachedAttachmentDataUrl('site-a', 'image-1')).toBeNull() + + const third = loadAttachmentDataUrlWithCache(args) + expect(downloads).toBe(2) + replacement.resolve({ dataUrl: 'new image', byteSize: 3 }) + expect(await second).toBe('new image') + expect(await third).toBe('new image') + expect(getCachedAttachmentDataUrl('site-a', 'image-1')).toBe('new image') + } + ) +}) diff --git a/src/main/jira/attachment-image-cache.ts b/src/main/jira/attachment-image-cache.ts index f9fdbfe3f7d..9d118ffdc9c 100644 --- a/src/main/jira/attachment-image-cache.ts +++ b/src/main/jira/attachment-image-cache.ts @@ -133,7 +133,10 @@ export async function loadAttachmentDataUrlWithCache(args: { } return loaded.dataUrl } finally { - inFlight.delete(key) + // A cleared generation no longer owns the current download's singleflight slot. + if (currentEpoch(args.siteId) === epochAtStart) { + inFlight.delete(key) + } } })() From 445c1aeaaf7c2cd43a175658c0b8a59117cb292f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:03:53 -0700 Subject: [PATCH 27/69] perf(speech): reuse the model download idle timer (#18945) --- .../model-manager-stream-cleanup.test.ts | 24 ++++++++++++++----- src/main/speech/speech-model-http-download.ts | 7 ++++-- 2 files changed, 23 insertions(+), 8 deletions(-) diff --git a/src/main/speech/model-manager-stream-cleanup.test.ts b/src/main/speech/model-manager-stream-cleanup.test.ts index dcebaca37a3..34b7aa893d8 100644 --- a/src/main/speech/model-manager-stream-cleanup.test.ts +++ b/src/main/speech/model-manager-stream-cleanup.test.ts @@ -1,4 +1,4 @@ -import { mkdtempSync, rmSync } from 'node:fs' +import { mkdtempSync, readFileSync, rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' import { PassThrough } from 'node:stream' @@ -34,15 +34,17 @@ describe('ModelManager stream cleanup', () => { netRequestMock.mockReset() }) - it('removes response progress listeners after a model download finishes', async () => { + it('reuses the idle timer and removes progress listeners after a fragmented download', async () => { const dir = mkdtempSync(join(tmpdir(), 'orca-model-manager-')) + vi.useFakeTimers({ toFake: ['setTimeout', 'clearTimeout'] }) + const timeoutSpy = vi.spyOn(globalThis, 'setTimeout') try { const response = new PassThrough() as PassThrough & { statusCode: number headers: Record } response.statusCode = 200 - response.headers = { 'content-length': '4' } + response.headers = { 'content-length': '1000' } const responseHandlers: ((response: unknown) => void)[] = [] const request = { abort: vi.fn(() => request), @@ -66,16 +68,26 @@ describe('ModelManager stream cleanup', () => { const download = manager.downloadFile( 'https://example.com/model.bin', join(dir, 'model.bin'), - 4, + 1000, 'm', () => false ) - response.write(Buffer.from('ab')) - response.end(Buffer.from('cd')) + await vi.advanceTimersByTimeAsync(60_000) + for (let index = 0; index < 1000; index += 1) { + response.write(Buffer.from('a')) + } + await vi.advanceTimersByTimeAsync(119_999) + expect(request.abort).not.toHaveBeenCalled() + response.end() await expect(download).resolves.toBeUndefined() expect(response.listenerCount('data')).toBe(0) + expect(vi.getTimerCount()).toBe(0) + expect(readFileSync(join(dir, 'model.bin'), 'utf8')).toBe('a'.repeat(1000)) + expect(timeoutSpy.mock.calls.filter(([, delay]) => delay === 120_000)).toHaveLength(1) } finally { + timeoutSpy.mockRestore() + vi.useRealTimers() rmSync(dir, { recursive: true, force: true }) } }) diff --git a/src/main/speech/speech-model-http-download.ts b/src/main/speech/speech-model-http-download.ts index ae2bae9648e..3bd2f24ab97 100644 --- a/src/main/speech/speech-model-http-download.ts +++ b/src/main/speech/speech-model-http-download.ts @@ -71,8 +71,11 @@ export abstract class SpeechModelHttpDownload { request = null } const resetIdleTimeout = (): void => { - clearIdleTimeout() - idleTimeout = setTimeout(onRequestTimeout, DOWNLOAD_IDLE_TIMEOUT_MS) + if (idleTimeout) { + idleTimeout.refresh() + } else { + idleTimeout = setTimeout(onRequestTimeout, DOWNLOAD_IDLE_TIMEOUT_MS) + } } const resolveOnce = (): void => { if (settled) { From 56626e7daad08b554ad124b586d6082629d886cc Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:03:58 -0700 Subject: [PATCH 28/69] perf(ssh): reuse and release relay startup buffers (#18953) * perf(ssh): reuse the searched relay startup prefix * perf(ssh): release startup banners after relay readiness --- .../scripts/benchmark-sentinel-retention.mjs | 72 +++++++++++++++++++ src/main/ssh/ssh-relay-deploy-helpers.ts | 3 +- .../ssh-relay-sentinel-copy-budget.test.ts | 62 ++++++++++++++++ 3 files changed, 136 insertions(+), 1 deletion(-) create mode 100644 config/scripts/benchmark-sentinel-retention.mjs create mode 100644 src/main/ssh/ssh-relay-sentinel-copy-budget.test.ts diff --git a/config/scripts/benchmark-sentinel-retention.mjs b/config/scripts/benchmark-sentinel-retention.mjs new file mode 100644 index 00000000000..93564eeac01 --- /dev/null +++ b/config/scripts/benchmark-sentinel-retention.mjs @@ -0,0 +1,72 @@ +import { strict as assert } from 'node:assert' +import { EventEmitter } from 'node:events' +import { mkdtemp, rm } from 'node:fs/promises' +import { createRequire } from 'node:module' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { build } from 'esbuild' + +if (!global.gc) { + throw new Error('Run with node --expose-gc') +} +const root = resolve(import.meta.dirname, '../..') +const directory = await mkdtemp(join(tmpdir(), 'orca-sentinel-retention-')) +const output = join(directory, 'sentinel.cjs') +try { + await build({ + stdin: { + contents: `export {waitForSentinel} from './src/main/ssh/ssh-relay-deploy-helpers'; +export {RELAY_SENTINEL} from './src/main/ssh/relay-protocol';`, + resolveDir: root, + loader: 'ts' + }, + bundle: true, + platform: 'node', + format: 'cjs', + packages: 'external', + banner: { + js: `var require = require('node:module').createRequire(${JSON.stringify(join(root, 'package.json'))});` + }, + outfile: output + }) + const { waitForSentinel, RELAY_SENTINEL } = createRequire(import.meta.url)(output) + const held = [] + const banners = [] + for (let i = 0; i < 100; i++) { + const channel = Object.assign(new EventEmitter(), { + stderr: new EventEmitter(), + stdin: { write: () => true }, + close: () => {} + }) + const pending = waitForSentinel(channel) + banners.push(feedBanner(channel)) + channel.emit('data', Buffer.from(RELAY_SENTINEL)) + const transport = await pending + const received = [] + transport.onData((bytes) => received.push(bytes.toString())) + channel.emit('data', Buffer.from('frame')) + assert.deepEqual(received, ['frame']) + held.push({ channel, transport }) + } + await new Promise((resolve) => setImmediate(resolve)) + for (let i = 0; i < 5; i++) { + global.gc() + } + const retained = banners.filter((reference) => reference.deref() !== undefined).length + console.log( + JSON.stringify({ + connections: held.length, + bannerBytes: 65536, + retainedBannerBuffers: retained, + retainedBannerBytes: retained * 65536 + }) + ) +} finally { + await rm(directory, { recursive: true, force: true }) +} + +function feedBanner(channel) { + const banner = Buffer.alloc(65536, 120) + channel.emit('data', banner) + return new WeakRef(banner.buffer) +} diff --git a/src/main/ssh/ssh-relay-deploy-helpers.ts b/src/main/ssh/ssh-relay-deploy-helpers.ts index a035133ddc5..f9a8f167752 100644 --- a/src/main/ssh/ssh-relay-deploy-helpers.ts +++ b/src/main/ssh/ssh-relay-deploy-helpers.ts @@ -209,6 +209,7 @@ export function waitForSentinel( const afterSentinelOffset = sentinelIdx + RELAY_SENTINEL_BUFFER.length - bufferedStdout.length const afterSentinel = data.subarray(Math.max(0, afterSentinelOffset)) + bufferedStdout = Buffer.alloc(0) if (afterSentinel.length > 0) { pendingAfterSentinel = afterSentinel @@ -258,7 +259,7 @@ export function waitForSentinel( return } - bufferedStdout = bufferedStdout.length === 0 ? data : Buffer.concat([bufferedStdout, data]) + bufferedStdout = startupStdout }) }) } diff --git a/src/main/ssh/ssh-relay-sentinel-copy-budget.test.ts b/src/main/ssh/ssh-relay-sentinel-copy-budget.test.ts new file mode 100644 index 00000000000..b05c8a71d92 --- /dev/null +++ b/src/main/ssh/ssh-relay-sentinel-copy-budget.test.ts @@ -0,0 +1,62 @@ +import { EventEmitter } from 'node:events' +import type { ClientChannel } from 'ssh2' +import { expect, it, vi } from 'vitest' +import { RELAY_SENTINEL } from './relay-protocol' +import { waitForSentinel } from './ssh-relay-deploy-helpers' + +it.each([1, 256])('copies each startup prefix once across %i chunks', async (chunks) => { + const channel = Object.assign(new EventEmitter(), { + stderr: new EventEmitter(), + stdin: { write: vi.fn(() => true) }, + close: vi.fn(), + pause: vi.fn(), + resume: vi.fn() + }) + const pending = waitForSentinel(channel as unknown as ClientChannel) + const chunk = Buffer.alloc((64 * 1024) / chunks, 120) + const concat = vi.spyOn(Buffer, 'concat') + let calls = 0 + let copied = 0 + try { + for (let i = 0; i < chunks; i++) { + channel.emit('data', chunk) + } + calls = concat.mock.calls.length + copied = concat.mock.calls.reduce( + (sum, [buffers]) => sum + buffers.reduce((bytes, buffer) => bytes + buffer.length, 0), + 0 + ) + } finally { + concat.mockRestore() + } + channel.emit('data', Buffer.from(`${RELAY_SENTINEL}first-frame`)) + const transport = await pending + const received: string[] = [] + transport.onData((bytes) => received.push(bytes.toString())) + expect(received).toEqual(['first-frame']) + expect(channel.close).not.toHaveBeenCalled() + expect(calls).toBe(chunks - 1) + expect(copied).toBe(chunk.length * ((chunks * (chunks + 1)) / 2 - 1)) +}) + +it.each(Array.from({ length: RELAY_SENTINEL.length + 1 }, (_, i) => i))( + 'preserves the marker and binary payload when split at byte %i', + async (split) => { + const channel = Object.assign(new EventEmitter(), { + stderr: new EventEmitter(), + stdin: { write: vi.fn(() => true) }, + close: vi.fn() + }) + const pending = waitForSentinel(channel as unknown as ClientChannel) + const marker = Buffer.from(RELAY_SENTINEL) + const payload = Buffer.from([0, 255, 128, 10, 13, 1]) + channel.emit('data', Buffer.alloc(63 * 1024, 120)) + channel.emit('data', marker.subarray(0, split)) + channel.emit('data', Buffer.concat([marker.subarray(split), payload])) + const transport = await pending + const received: Buffer[] = [] + transport.onData((bytes) => received.push(bytes)) + expect(Buffer.concat(received)).toEqual(payload) + expect(channel.close).not.toHaveBeenCalled() + } +) From 5cc432eead0729f711cf9fde977dfeef2b46dda7 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:04:03 -0700 Subject: [PATCH 29/69] perf(ssh): reuse streamed response idle timers (#18956) --- ...sh-file-stream-inactivity-deadline.test.ts | 87 +++++++++++++++++++ .../ssh-file-stream-inactivity-deadline.ts | 5 +- .../ssh/ssh-git-response-stream-reader.ts | 5 +- .../ssh/ssh-git-stream-idle-timer.test.ts | 80 +++++++++++++++++ 4 files changed, 175 insertions(+), 2 deletions(-) create mode 100644 src/main/ssh/ssh-file-stream-inactivity-deadline.test.ts create mode 100644 src/main/ssh/ssh-git-stream-idle-timer.test.ts diff --git a/src/main/ssh/ssh-file-stream-inactivity-deadline.test.ts b/src/main/ssh/ssh-file-stream-inactivity-deadline.test.ts new file mode 100644 index 00000000000..f87d1e6f778 --- /dev/null +++ b/src/main/ssh/ssh-file-stream-inactivity-deadline.test.ts @@ -0,0 +1,87 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { createSshFileStreamInactivityDeadline } from './ssh-file-stream-inactivity-deadline' +import type { SystemPowerLifecycleListener } from '../system-power-lifecycle' + +afterEach(() => { + vi.restoreAllMocks() + vi.useRealTimers() +}) + +describe('SSH file stream inactivity timer', () => { + it('reuses one timer while retaining the deadline of the latest chunk', () => { + vi.useFakeTimers() + const allocate = vi.spyOn(globalThis, 'setTimeout') + const onTimeout = vi.fn() + const unsubscribe = vi.fn() + const deadline = createSshFileStreamInactivityDeadline(onTimeout, (listener) => { + listener.onResume() + return unsubscribe + }) + deadline.reset() + vi.advanceTimersByTime(30_000) + for (let chunk = 0; chunk < 1000; chunk += 1) { + deadline.reset() + } + expect(allocate).toHaveBeenCalledTimes(1) + vi.advanceTimersByTime(59_999) + expect(onTimeout).not.toHaveBeenCalled() + vi.advanceTimersByTime(1) + expect(onTimeout).toHaveBeenCalledTimes(1) + deadline.clear() + expect(unsubscribe).toHaveBeenCalledTimes(1) + expect(vi.getTimerCount()).toBe(0) + }) + + it('releases on suspend and creates a fresh timer on resume', () => { + vi.useFakeTimers() + const allocate = vi.spyOn(globalThis, 'setTimeout') + const onTimeout = vi.fn() + let power!: SystemPowerLifecycleListener + const deadline = createSshFileStreamInactivityDeadline(onTimeout, (listener) => { + power = listener + listener.onResume() + return vi.fn() + }) + deadline.reset() + vi.advanceTimersByTime(30_000) + power.onSuspend() + for (let chunk = 0; chunk < 1000; chunk += 1) { + deadline.reset() + } + expect(vi.getTimerCount()).toBe(0) + vi.advanceTimersByTime(120_000) + expect(onTimeout).not.toHaveBeenCalled() + power.onResume() + expect(allocate).toHaveBeenCalledTimes(2) + vi.advanceTimersByTime(59_999) + expect(onTimeout).not.toHaveBeenCalled() + vi.advanceTimersByTime(1) + expect(onTimeout).toHaveBeenCalledTimes(1) + deadline.clear() + expect(vi.getTimerCount()).toBe(0) + }) + + it('clears the timer and subscription and supports a later reset', () => { + vi.useFakeTimers() + const onTimeout = vi.fn() + const unsubscribe = vi.fn() + const subscribe = vi.fn((listener: SystemPowerLifecycleListener) => { + listener.onResume() + return unsubscribe + }) + const deadline = createSshFileStreamInactivityDeadline(onTimeout, subscribe) + deadline.reset() + deadline.clear() + deadline.clear() + expect(unsubscribe).toHaveBeenCalledTimes(1) + expect(vi.getTimerCount()).toBe(0) + vi.advanceTimersByTime(120_000) + expect(onTimeout).not.toHaveBeenCalled() + deadline.reset() + expect(subscribe).toHaveBeenCalledTimes(2) + expect(vi.getTimerCount()).toBe(1) + deadline.clear() + expect(unsubscribe).toHaveBeenCalledTimes(2) + expect(vi.getTimerCount()).toBe(0) + }) +}) diff --git a/src/main/ssh/ssh-file-stream-inactivity-deadline.ts b/src/main/ssh/ssh-file-stream-inactivity-deadline.ts index 5470e8121fe..cbb9839f9b3 100644 --- a/src/main/ssh/ssh-file-stream-inactivity-deadline.ts +++ b/src/main/ssh/ssh-file-stream-inactivity-deadline.ts @@ -24,10 +24,13 @@ export function createSshFileStreamInactivityDeadline( } } const arm = (): void => { - clearTimer() if (suspended) { return } + if (timer) { + timer.refresh() + return + } timer = setTimeout(onTimeout, SSH_FILE_STREAM_INACTIVITY_TIMEOUT_MS) timer.unref?.() } diff --git a/src/main/ssh/ssh-git-response-stream-reader.ts b/src/main/ssh/ssh-git-response-stream-reader.ts index 6a50dc28a72..0a8b26aa779 100644 --- a/src/main/ssh/ssh-git-response-stream-reader.ts +++ b/src/main/ssh/ssh-git-response-stream-reader.ts @@ -90,7 +90,10 @@ export function requestGitStreamable( // killed, but a wedged stream (no frames arriving) rejects instead of // hanging the caller forever. const armInactivity = (): void => { - clearInactivity() + if (inactivityTimer) { + inactivityTimer.refresh() + return + } inactivityTimer = setTimeout(() => { fail( new GitResponseStreamError( diff --git a/src/main/ssh/ssh-git-stream-idle-timer.test.ts b/src/main/ssh/ssh-git-stream-idle-timer.test.ts new file mode 100644 index 00000000000..7cd5207c63a --- /dev/null +++ b/src/main/ssh/ssh-git-stream-idle-timer.test.ts @@ -0,0 +1,80 @@ +import { expect, it, vi } from 'vitest' +import type { SshChannelMultiplexer } from './ssh-channel-multiplexer' +import { requestGitStreamable } from './ssh-git-response-stream-reader' + +it.each(['end', 'abort', 'timeout'] as const)( + 'reuses the idle deadline across 1000 chunks and cleans up on %s', + async (finish) => { + vi.useFakeTimers() + const setTimer = vi.spyOn(globalThis, 'setTimeout') + try { + const listeners = new Map) => void>() + const controller = new AbortController() + const content = 'x'.repeat(998) + const encoded = Buffer.from(JSON.stringify(content)) + const notify = vi.fn() + const mux = { + request: vi.fn(async () => ({ + __orcaGitResponseStream: { streamId: 7, totalBytes: encoded.length, chunkCount: 1000 } + })), + isDisposed: () => false, + notify, + onDispose: () => () => {}, + onNotificationByMethod: ( + method: string, + callback: (params: Record) => void + ) => { + listeners.set(method, callback) + return () => listeners.delete(method) + } + } + const promise = requestGitStreamable( + mux as unknown as SshChannelMultiplexer, + 'git.diff', + {}, + { + signal: controller.signal + } + ) + const outcome = promise.then( + (value) => ({ value }), + (error: Error) => ({ error: error.message }) + ) + await vi.advanceTimersByTimeAsync(15_000) + for (let seq = 0; seq < encoded.length; seq++) { + listeners.get('git.responseChunk')!({ + streamId: 7, + seq, + data: encoded.subarray(seq, seq + 1).toString('base64') + }) + } + await vi.advanceTimersByTimeAsync(29_999) + expect(listeners.size).toBe(3) + expect(notify.mock.calls.filter(([method]) => method === 'git.responseAck')).toHaveLength( + 1000 + ) + const allocations = setTimer.mock.calls.filter(([, delay]) => delay === 30_000).length + if (finish === 'end') { + listeners.get('git.responseEnd')!({ streamId: 7 }) + expect(await outcome).toEqual({ value: content }) + } else if (finish === 'abort') { + controller.abort() + expect(await outcome).toEqual({ error: 'Request was cancelled' }) + } else { + await vi.advanceTimersByTimeAsync(1) + expect(await outcome).toEqual({ + error: 'Git response stream stalled (>30000ms without data)' + }) + } + expect(allocations).toBe(1) + expect(vi.getTimerCount()).toBe(0) + expect(listeners.size).toBe(0) + expect( + notify.mock.calls.filter(([method]) => method === 'git.cancelResponseStream') + ).toHaveLength(finish === 'end' ? 0 : 1) + } finally { + setTimer.mockRestore() + vi.useRealTimers() + } + } +) From 0a573ceac88e05bc729ebcee925ec9e238265f33 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:04:07 -0700 Subject: [PATCH 30/69] perf(browser): reuse decoded single-chunk upload buffers (#18960) --- .../browser-client-upload-transfer.test.ts | 37 ++++++++++++++++++- .../browser/browser-client-upload-transfer.ts | 2 +- 2 files changed, 37 insertions(+), 2 deletions(-) diff --git a/src/main/browser/browser-client-upload-transfer.test.ts b/src/main/browser/browser-client-upload-transfer.test.ts index fdc129618e8..84bb649abd3 100644 --- a/src/main/browser/browser-client-upload-transfer.test.ts +++ b/src/main/browser/browser-client-upload-transfer.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from 'vitest' +import { describe, expect, it, vi } from 'vitest' import type { BrowserClientHostCommandEvent } from '../../shared/browser-client-host-protocol' import { @@ -102,3 +102,38 @@ describe('readBrowserClientUploadPaths', () => { ) }) }) + +it.each([0, 1, 128 * 1024])( + 'avoids recopying 16 single-chunk uploads of %i bytes', + async (size) => { + const source = Buffer.alloc(size, 171) + const response = { + contentBase64: source.toString('base64'), + bytesRead: size, + totalBytes: size, + eof: true + } + const remotePaths = Array.from({ length: 16 }, (_, i) => `file-${i}.bin`) + const request = vi.fn(async () => response) + const concat = vi.spyOn(Buffer, 'concat') + let copies = 0 + let files: Awaited> + try { + files = await fetchBrowserClientUploadFiles({ request, event, remotePaths }) + copies = concat.mock.calls.length + } finally { + concat.mockRestore() + } + expect(copies).toBe(0) + expect(request).toHaveBeenCalledTimes(16) + expect(files.map((file) => file.remotePath)).toEqual(remotePaths) + for (const file of files) { + expect(file.contents).toEqual(source) + } + if (size > 0) { + files[0].contents[0] = 0 + expect(files[1].contents[0]).toBe(171) + expect(source[0]).toBe(171) + } + } +) diff --git a/src/main/browser/browser-client-upload-transfer.ts b/src/main/browser/browser-client-upload-transfer.ts index 1863f7b9f75..f85af071633 100644 --- a/src/main/browser/browser-client-upload-transfer.ts +++ b/src/main/browser/browser-client-upload-transfer.ts @@ -74,7 +74,7 @@ export async function fetchBrowserClientUploadFiles(options: { throw new Error('browser_client_upload_transfer_stalled') } } - files.push({ remotePath, contents: Buffer.concat(chunks) }) + files.push({ remotePath, contents: chunks.length === 1 ? chunks[0] : Buffer.concat(chunks) }) } return files } From 8ed81ceb8d53d5d057381fb1cee0fc41916dbc9d Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:04:11 -0700 Subject: [PATCH 31/69] perf(tabs): index saved tab order during hydration repair (#18964) --- config/scripts/benchmark-tab-group-repair.mjs | 80 +++++++++++++++++++ .../slices/tab-group-reference-repair.test.ts | 54 +++++++++++++ .../slices/tab-group-reference-repair.ts | 3 +- 3 files changed, 136 insertions(+), 1 deletion(-) create mode 100644 config/scripts/benchmark-tab-group-repair.mjs create mode 100644 src/renderer/src/store/slices/tab-group-reference-repair.test.ts diff --git a/config/scripts/benchmark-tab-group-repair.mjs b/config/scripts/benchmark-tab-group-repair.mjs new file mode 100644 index 00000000000..17a1161fc4c --- /dev/null +++ b/config/scripts/benchmark-tab-group-repair.mjs @@ -0,0 +1,80 @@ +import { strict as assert } from 'node:assert' +import { mkdtemp, readFile, rm } from 'node:fs/promises' +import { createRequire } from 'node:module' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { performance } from 'node:perf_hooks' +import { build } from 'esbuild' + +const root = resolve(import.meta.dirname, '../..') +const source = join(root, 'src/renderer/src/store/slices/tab-group-reference-repair.ts') +const directory = await mkdtemp(join(tmpdir(), 'orca-tab-repair-')) +const current = await readFile(source, 'utf8') +const indexed = `const orderedTabIds = new Set(group.tabOrder) + const missingTabIds = ownedTabIds.filter((tabId) => !orderedTabIds.has(tabId))` +assert(current.includes(indexed), 'Expected indexed implementation') +try { + const implementations = [] + for (const baseline of [true, false]) { + const outfile = join(directory, baseline ? 'before.cjs' : 'after.cjs') + await build({ + stdin: { + contents: baseline + ? current.replace( + indexed, + 'const missingTabIds = ownedTabIds.filter((tabId) => !group.tabOrder.includes(tabId))' + ) + : current, + resolveDir: resolve(source, '..'), + loader: 'ts' + }, + bundle: true, + platform: 'node', + format: 'cjs', + outfile, + alias: { '@': join(root, 'src/renderer/src') } + }) + implementations.push(createRequire(import.meta.url)(outfile).appendOwnedTabIdsToGroups) + } + const rows = [] + for (const count of [1, 10, 100, 1_000, 10_000]) { + for (const missing of [false, true]) { + const ids = Array.from({ length: count }, (_, i) => `tab-${i}`) + const groups = [ + { id: 'group', worktreeId: 'workspace', activeTabId: null, tabOrder: ids, recentTabIds: [] } + ] + const owners = new Map(ids.map((id) => [missing ? `missing-${id}` : id, 'group'])) + assert.deepEqual(implementations[0](groups, owners), implementations[1](groups, owners)) + const iterations = Math.max(1, Math.floor(10_000 / count)) + const samples = [[], []] + for (let sample = -3; sample < 11; sample++) { + for (const index of sample % 2 === 0 ? [0, 1] : [1, 0]) { + const start = performance.now() + for (let i = 0; i < iterations; i++) { + implementations[index](groups, owners) + } + const elapsed = (performance.now() - start) / iterations + if (sample >= 0) { + samples[index].push(elapsed) + } + } + } + rows.push({ + count, + missing, + iterations, + beforeMs: samples[0].sort((a, b) => a - b)[5], + afterMs: samples[1].sort((a, b) => a - b)[5] + }) + } + } + console.log( + JSON.stringify( + { node: process.version, platform: process.platform, samples: 11, warmups: 3, rows }, + null, + 2 + ) + ) +} finally { + await rm(directory, { recursive: true, force: true }) +} diff --git a/src/renderer/src/store/slices/tab-group-reference-repair.test.ts b/src/renderer/src/store/slices/tab-group-reference-repair.test.ts new file mode 100644 index 00000000000..a7cb3dca125 --- /dev/null +++ b/src/renderer/src/store/slices/tab-group-reference-repair.test.ts @@ -0,0 +1,54 @@ +import { describe, expect, it } from 'vitest' +import type { TabGroup } from '../../../../shared/tab-types' +import { appendOwnedTabIdsToGroups } from './tab-group-reference-repair' + +function group(id: string, tabOrder: string[]): TabGroup { + return { id, worktreeId: 'workspace', activeTabId: null, tabOrder, recentTabIds: [] } +} + +describe('appendOwnedTabIdsToGroups', () => { + it('preserves existing order, duplicates, and untouched group identities', () => { + const complete = group('complete', ['b', 'a', 'a']) + const missing = group('missing', ['stale', 'c']) + const unowned = group('unowned', ['external']) + const owners = new Map([ + ['a', 'complete'], + ['b', 'complete'], + ['d', 'missing'], + ['c', 'missing'], + ['e', 'missing'], + ['elsewhere', 'absent'] + ]) + const result = appendOwnedTabIdsToGroups([complete, missing, unowned], owners) + expect(result).toEqual([complete, { ...missing, tabOrder: ['stale', 'c', 'd', 'e'] }, unowned]) + expect(result[0]).toBe(complete) + expect(result[2]).toBe(unowned) + expect(missing.tabOrder).toEqual(['stale', 'c']) + }) + + it.each([false, true])('bounds saved-order reads with missing tabs: %s', (missing) => { + const count = 1_000 + const ids = Array.from({ length: count }, (_, i) => `tab-${i}`) + let reads = 0 + const order = new Proxy(ids, { + get(target, property, receiver) { + if (typeof property === 'string' && /^\d+$/.test(property)) { + reads++ + } + return Reflect.get(target, property, receiver) + } + }) + const original = group('group', order) + const ownedIds = missing ? ids.map((id) => `missing-${id}`) : ids + const result = appendOwnedTabIdsToGroups( + [original], + new Map(ownedIds.map((id) => [id, original.id])) + ) + const repairReads = reads + expect(result[0].tabOrder).toEqual(missing ? [...ids, ...ownedIds] : ids) + if (!missing) { + expect(result[0]).toBe(original) + } + expect(repairReads).toBeLessThanOrEqual(count * 2) + }) +}) diff --git a/src/renderer/src/store/slices/tab-group-reference-repair.ts b/src/renderer/src/store/slices/tab-group-reference-repair.ts index 2bf1c468093..77d6dd38c81 100644 --- a/src/renderer/src/store/slices/tab-group-reference-repair.ts +++ b/src/renderer/src/store/slices/tab-group-reference-repair.ts @@ -72,7 +72,8 @@ export function appendOwnedTabIdsToGroups( if (!ownedTabIds) { return group } - const missingTabIds = ownedTabIds.filter((tabId) => !group.tabOrder.includes(tabId)) + const orderedTabIds = new Set(group.tabOrder) + const missingTabIds = ownedTabIds.filter((tabId) => !orderedTabIds.has(tabId)) return missingTabIds.length > 0 ? { ...group, tabOrder: [...group.tabOrder, ...missingTabIds] } : group From 78e3721c2331ba54bbfa2dbb3065bfa019fa5b58 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:04:16 -0700 Subject: [PATCH 32/69] perf(palette): reuse allowed quality arrays during matching (#18966) --- .../match-field-allocation.test.ts | 90 +++++++++++++++++++ .../src/lib/palette-match/match-field.ts | 26 +++--- 2 files changed, 103 insertions(+), 13 deletions(-) create mode 100644 src/renderer/src/lib/palette-match/match-field-allocation.test.ts diff --git a/src/renderer/src/lib/palette-match/match-field-allocation.test.ts b/src/renderer/src/lib/palette-match/match-field-allocation.test.ts new file mode 100644 index 00000000000..5b0021d2e71 --- /dev/null +++ b/src/renderer/src/lib/palette-match/match-field-allocation.test.ts @@ -0,0 +1,90 @@ +import { describe, expect, it, vi } from 'vitest' +import { + indexPaletteField, + type PaletteIdentifierKind, + type PaletteFieldProfile +} from './indexed-field' +import { matchPaletteField } from './match-field' +import { createPaletteQueryToken } from './palette-query' + +describe('palette field quality allocation', () => { + it.each(['scan', 's', '123', 'scna', 'zzz'])( + 'does not allocate a Set per field for %s', + (query) => { + const profiles: PaletteFieldProfile[] = [ + 'structured-label', + 'identifier', + 'path', + 'prose', + 'exact-alias' + ] + const fields = Array.from({ length: 1_000 }, (_, i) => + indexPaletteField({ + id: String(i), + profile: profiles[i % profiles.length], + text: 'scan daily 1234 workspace', + ...(i % 2 === 0 ? { identifier: { kind: 'number' as const } } : {}) + })! + ) + const token = createPaletteQueryToken(query, 0) + let allocations = 0 + const NativeSet = globalThis.Set + class CountedSet extends NativeSet { + constructor(values?: Iterable | null) { + super(values) + allocations++ + } + } + vi.stubGlobal('Set', CountedSet) + try { + for (const field of fields) { + matchPaletteField(field, token) + } + } finally { + vi.unstubAllGlobals() + } + expect(allocations).toBe(0) + } + ) +}) + +describe('palette quality restrictions remain local to each match', () => { + it.each(['number', 'version', 'date', 'port', 'sha', 'key'])( + 'preserves prefix permissions for %s', + (kind) => { + const field = indexPaletteField({ + id: 'id', + profile: 'identifier', + text: '12345', + identifier: { kind } + })! + const prefix = createPaletteQueryToken('123', 0) + const exact = createPaletteQueryToken('12345', 0) + const expected = ['port', 'sha', 'key'].includes(kind) + ? { quality: 'field-prefix', ranges: [{ start: 0, end: 3 }] } + : null + expect(matchPaletteField(field, prefix)).toEqual(expected) + expect(matchPaletteField(field, exact)).toEqual({ + quality: 'field-exact', + ranges: [{ start: 0, end: 5 }] + }) + expect(matchPaletteField(field, prefix)).toEqual(expected) + } + ) + + it.each(['structured-label', 'identifier', 'path', 'prose', 'exact-alias'])( + 'preserves typo restrictions for %s without mutating the profile', + (profile) => { + const field = indexPaletteField({ id: 'id', profile, text: 'scan' })! + expect(matchPaletteField(field, createPaletteQueryToken('s', 0))).toEqual({ + quality: 'field-prefix', + ranges: [{ start: 0, end: 1 }] + }) + expect(matchPaletteField(field, createPaletteQueryToken('scam', 0))).toEqual( + ['structured-label', 'prose'].includes(profile) + ? { quality: 'typo', ranges: [{ start: 0, end: 4 }] } + : null + ) + } + ) +}) diff --git a/src/renderer/src/lib/palette-match/match-field.ts b/src/renderer/src/lib/palette-match/match-field.ts index f9015ec49a5..1744219524c 100644 --- a/src/renderer/src/lib/palette-match/match-field.ts +++ b/src/renderer/src/lib/palette-match/match-field.ts @@ -32,7 +32,7 @@ const SIGILS = new Set(['#', '!']) function allowedQualities( field: PaletteIndexedField, token: PaletteQueryToken -): ReadonlySet { +): readonly PaletteMatchQuality[] { let qualities = paletteProfileAllowedQualities(field.profile) if (field.identifier && !identifierKindAllowsPrefix(field.identifier.kind)) { qualities = qualities.filter((quality) => !PREFIX_QUALITIES.has(quality)) @@ -43,7 +43,7 @@ function allowedQualities( if (token.isIdentifierLike) { qualities = qualities.filter((quality) => quality !== 'typo') } - return new Set(qualities) + return qualities } /** `#123` must not reach a GitLab MR, and `!123` must not reach a GitHub PR. */ @@ -83,15 +83,15 @@ function toRanges(field: PaletteIndexedField, start: number, end: number): reado function matchLiteral( field: PaletteIndexedField, token: PaletteQueryToken, - qualities: ReadonlySet + qualities: readonly PaletteMatchQuality[] ): PaletteFieldMatch | null { const normalized = field.text.normalized const text = token.text - if (qualities.has('field-exact') && normalized === text) { + if (qualities.includes('field-exact') && normalized === text) { return { quality: 'field-exact', ranges: toRanges(field, 0, normalized.length) } } - if (qualities.has('word-exact')) { + if (qualities.includes('word-exact')) { const word = field.words.find((entry) => entry.text === text) if (word) { return { quality: 'word-exact', ranges: toRanges(field, word.start, word.end) } @@ -101,10 +101,10 @@ function matchLiteral( return { quality: 'word-exact', ranges: toRanges(field, atom.start, atom.end) } } } - if (qualities.has('field-prefix') && normalized.startsWith(text)) { + if (qualities.includes('field-prefix') && normalized.startsWith(text)) { return { quality: 'field-prefix', ranges: toRanges(field, 0, text.length) } } - if (qualities.has('word-prefix')) { + if (qualities.includes('word-prefix')) { const word = field.words.find((entry) => entry.text.startsWith(text)) const atom = field.atoms.find((entry) => normalized.startsWith(text, entry.start)) const start = word && atom ? Math.min(word.start, atom.start) : (word?.start ?? atom?.start) @@ -117,13 +117,13 @@ function matchLiteral( if (literalIndex === -1) { return null } - if (qualities.has('boundary-substring') && isWordStart(field, literalIndex)) { + if (qualities.includes('boundary-substring') && isWordStart(field, literalIndex)) { return { quality: 'boundary-substring', ranges: toRanges(field, literalIndex, literalIndex + text.length) } } - if (qualities.has('literal-substring')) { + if (qualities.includes('literal-substring')) { return { quality: 'literal-substring', ranges: toRanges(field, literalIndex, literalIndex + text.length) @@ -135,9 +135,9 @@ function matchLiteral( function matchCompact( field: PaletteIndexedField, token: PaletteQueryToken, - qualities: ReadonlySet + qualities: readonly PaletteMatchQuality[] ): PaletteFieldMatch | null { - if (!qualities.has('compact') || token.compact.length < MIN_COMPACT_LENGTH) { + if (!qualities.includes('compact') || token.compact.length < MIN_COMPACT_LENGTH) { return null } for (const atom of field.atoms) { @@ -152,9 +152,9 @@ function matchCompact( function matchTypo( field: PaletteIndexedField, token: PaletteQueryToken, - qualities: ReadonlySet + qualities: readonly PaletteMatchQuality[] ): PaletteFieldMatch | null { - if (!qualities.has('typo') || !token.isLetterOnly || !isPaletteTypoCandidate(token.text)) { + if (!qualities.includes('typo') || !token.isLetterOnly || !isPaletteTypoCandidate(token.text)) { return null } for (const word of field.words) { From 37427bfd1a7f88b015b50178318733a75f8ffd9f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:04:21 -0700 Subject: [PATCH 33/69] perf: remember equivalent session tab source identities (#18976) --- ...ession-write-subscriber-allocation.test.ts | 28 +++++++++++++++++++ .../src/lib/session-write-subscriber.ts | 5 ++-- 2 files changed, 31 insertions(+), 2 deletions(-) diff --git a/src/renderer/src/lib/session-write-subscriber-allocation.test.ts b/src/renderer/src/lib/session-write-subscriber-allocation.test.ts index 9681af5a973..74c9c6db968 100644 --- a/src/renderer/src/lib/session-write-subscriber-allocation.test.ts +++ b/src/renderer/src/lib/session-write-subscriber-allocation.test.ts @@ -90,6 +90,34 @@ afterEach(() => { }) describe('session write subscriber allocation', () => { + it.each(['tabsByWorktree', 'unifiedTabsByWorktree'] as const)( + 'remembers an equivalent %s source before unrelated writes', + (field) => { + const harness = createHarness() + try { + harness.write(() => ({ [field]: { 'wt-1': [] } })) + vi.advanceTimersByTime(500) + harness.persisted.length = 0 + + harness.write(() => ({ [field]: { 'wt-1': [] } })) + const calls = countFilterCalls(() => { + for (let write = 0; write < 200; write += 1) { + harness.write(() => ({ runtimePaneTitlesByTabId: {} })) + } + }) + expect(calls).toBe(0) + vi.advanceTimersByTime(500) + expect(harness.persisted).toHaveLength(0) + + harness.write(() => ({ activeTabId: 'next-tab' })) + vi.advanceTimersByTime(500) + expect(harness.persisted).toHaveLength(1) + } finally { + harness.dispose() + } + } + ) + it('allocates nothing for store writes that touch no session field', () => { const harness = createHarness() try { diff --git a/src/renderer/src/lib/session-write-subscriber.ts b/src/renderer/src/lib/session-write-subscriber.ts index e0e366ef6ba..d54675e15ab 100644 --- a/src/renderer/src/lib/session-write-subscriber.ts +++ b/src/renderer/src/lib/session-write-subscriber.ts @@ -233,12 +233,13 @@ export function createSessionWriteSubscriber({ prev === null ? [...SESSION_RELEVANT_FIELDS] : SESSION_RELEVANT_FIELDS.filter((key) => prev?.[key] !== next[key]) + // Equivalent projections still consume the new source identities. + prevTabsSource = state.tabsByWorktree + prevUnifiedTabsSource = state.unifiedTabsByWorktree if (changedFields.length === 0 && pendingChangedFields.size === 0) { return } prev = next - prevTabsSource = state.tabsByWorktree - prevUnifiedTabsSource = state.unifiedTabsByWorktree for (const field of changedFields) { pendingChangedFields.add(field) } From 64374d5dffb79e6c3b2407b76db331cf7ac8217f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:04:26 -0700 Subject: [PATCH 34/69] perf(cli): skip impossible typo distance comparisons (#18977) --- src/cli/command-suggestion-budget.test.ts | 44 +++++++++++++++++++++++ src/cli/command-suggestion.ts | 19 +++++++--- 2 files changed, 58 insertions(+), 5 deletions(-) create mode 100644 src/cli/command-suggestion-budget.test.ts diff --git a/src/cli/command-suggestion-budget.test.ts b/src/cli/command-suggestion-budget.test.ts new file mode 100644 index 00000000000..7902ad7fdba --- /dev/null +++ b/src/cli/command-suggestion-budget.test.ts @@ -0,0 +1,44 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import * as distance from '../shared/edit-distance' +import { suggestCommands, unknownFlagData } from './command-suggestion' +import type { CommandSpec } from './command-spec' + +const specs: CommandSpec[] = [ + { path: ['list'], summary: '', usage: '', allowedFlags: [] }, + { path: ['remove'], summary: '', usage: '', allowedFlags: [], destructive: true } +] + +afterEach(() => vi.restoreAllMocks()) + +describe('suggestion distance work', () => { + it('does no distance calculations for a long command, including destructive intent', () => { + const spy = vi.spyOn(distance, 'levenshtein') + expect(suggestCommands(specs, ['x'.repeat(32_768)])).toEqual([]) + expect(spy).not.toHaveBeenCalled() + }) + + it('does no distance calculations for a long flag but still lists valid flags', () => { + const spy = vi.spyOn(distance, 'levenshtein') + expect(unknownFlagData('x'.repeat(32_768), ['worktree', 'json'])).toEqual({ + validFlags: ['json', 'worktree'], + suggestions: [], + nextSteps: ['Valid flags: --json, --worktree'] + }) + expect(spy).not.toHaveBeenCalled() + }) + + it('keeps the inclusive three-edit suggestion boundary', () => { + expect(suggestCommands(specs, ['listxxx'])).toEqual(['list']) + expect(unknownFlagData('jsonxxx', ['json']).suggestions).toEqual(['json']) + }) + + it('keeps the inclusive one-edit destructive intent boundary', () => { + expect(suggestCommands(specs, ['remov'])).toEqual(['remove']) + expect(suggestCommands(specs, ['remo'])).toEqual([]) + }) + + it('retains UTF-16 distance semantics at the length boundary', () => { + expect(unknownFlagData('json😀x', ['json']).suggestions).toEqual(['json']) + expect(unknownFlagData('json😀😀', ['json']).suggestions).toEqual([]) + }) +}) diff --git a/src/cli/command-suggestion.ts b/src/cli/command-suggestion.ts index 7b80138e2f3..9c935f694fa 100644 --- a/src/cli/command-suggestion.ts +++ b/src/cli/command-suggestion.ts @@ -37,7 +37,10 @@ function destructiveVerbs(specs: CommandSpec[]): Set { // input token is itself a near-miss of a destructive verb. #6303 function intendsDestruction(inputToken: string, verbs: Set): boolean { for (const verb of verbs) { - if (levenshtein(inputToken, verb) <= DESTRUCTIVE_INTENT_THRESHOLD) { + if ( + Math.abs(inputToken.length - verb.length) <= DESTRUCTIVE_INTENT_THRESHOLD && + levenshtein(inputToken, verb) <= DESTRUCTIVE_INTENT_THRESHOLD + ) { return true } } @@ -85,7 +88,9 @@ export function suggestCommands(specs: CommandSpec[], commandPath: string[]): st continue } seen.add(joined) - scored.push({ label: joined, distance: levenshtein(input, joined) }) + if (Math.abs(input.length - joined.length) <= SUGGESTION_THRESHOLD) { + scored.push({ label: joined, distance: levenshtein(input, joined) }) + } } } return rankByDistance(scored) @@ -106,9 +111,13 @@ export type FlagErrorData = { } function suggestFlags(flag: string, validFlags: string[]): string[] { - return rankByDistance( - validFlags.map((candidate) => ({ label: candidate, distance: levenshtein(flag, candidate) })) - ) + const scored: { label: string; distance: number }[] = [] + for (const candidate of validFlags) { + if (Math.abs(flag.length - candidate.length) <= SUGGESTION_THRESHOLD) { + scored.push({ label: candidate, distance: levenshtein(flag, candidate) }) + } + } + return rankByDistance(scored) } // Why: include the accepted set so agents can recover without another help call. From 97526f65adb590dc3790f00b646d5fc66c98914f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:59:33 -0700 Subject: [PATCH 35/69] fix(tests): stabilize divider viewport and pointer-capture event ordering (#19004) * fix(tests): size divider capture-loss viewport deterministically * test: advance pointer events before awaiting capture loss --- ...terminal-pane-divider-capture-loss.spec.ts | 40 +++++-------------- 1 file changed, 9 insertions(+), 31 deletions(-) diff --git a/tests/e2e/terminal-pane-divider-capture-loss.spec.ts b/tests/e2e/terminal-pane-divider-capture-loss.spec.ts index 5f3bfca177b..8120c2a60e7 100644 --- a/tests/e2e/terminal-pane-divider-capture-loss.spec.ts +++ b/tests/e2e/terminal-pane-divider-capture-loss.spec.ts @@ -1,4 +1,4 @@ -import type { ElectronApplication, Page } from '@stablyai/playwright-test' +import type { Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' import { splitActiveTerminalPane, @@ -22,32 +22,6 @@ type DividerGeometry = { test.use({ seedTestRepo: false }) -async function setFullscreen(electronApp: ElectronApplication, page: Page): Promise { - await expect - .poll(async () => { - try { - return await electronApp.evaluate(({ BrowserWindow }) => { - const window = BrowserWindow.getAllWindows()[0] - if (!window) { - return false - } - if (window.isMinimized()) { - window.restore() - } - window.show() - window.focus() - window.setFullScreen(true) - return window.isFullScreen() - }) - } catch { - return false - } - }) - .toBe(true) - await expect.poll(() => page.evaluate(() => innerWidth >= 1000 && innerHeight >= 700)).toBe(true) - await page.waitForTimeout(1200) -} - async function addTestRepo(page: Page, repoPath: string): Promise { const repoId = await page.evaluate(async (path) => { const result = await window.api.repos.add({ path }) @@ -122,17 +96,21 @@ function gridsMatch(geometry: DividerGeometry): boolean { } test('@headful keeps resizing after the divider loses pointer capture', async ({ - electronApp, orcaPage, testRepoPath }, testInfo) => { - await setFullscreen(electronApp, orcaPage) + // Keep the 260px drag above the fit floor regardless of the CI display resolution. + await orcaPage.setViewportSize({ width: 1600, height: 1000 }) await addTestRepo(orcaPage, testRepoPath) await ensureTerminalVisible(orcaPage, 30_000) await waitForActiveTerminalManager(orcaPage, 30_000) await splitActiveTerminalPane(orcaPage, 'vertical') await waitForPaneCount(orcaPage, 2, 30_000) + await expect + .poll(async () => (await readDividerGeometry(orcaPage)).second.width) + .toBeGreaterThan(400) + const divider = orcaPage.locator('.pane-divider.is-vertical').first() await expect(divider).toBeVisible() const box = await divider.boundingBox() @@ -170,11 +148,11 @@ test('@headful keeps resizing after the divider loses pointer capture', async ({ } element.releasePointerCapture(pointerId) }) + // Pending capture changes are dispatched with the next pointer event. + await orcaPage.mouse.move(startX + 260, startY, { steps: 10 }) await expect .poll(() => divider.evaluate((element) => Number(element.dataset.captureLossCount ?? '0'))) .toBe(1) - - await orcaPage.mouse.move(startX + 260, startY, { steps: 10 }) await orcaPage.mouse.up() await expect.poll(async () => gridsMatch(await readDividerGeometry(orcaPage))).toBe(true) const after = await readDividerGeometry(orcaPage) From 09ee4c1b1857484eddfb93d357163b3731d5c8f2 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 21:07:46 -0700 Subject: [PATCH 36/69] fix(mobile): stop host streams after relay subscription cancellation (#18926) * fix(mobile): release cancelled relay stream subscriptions * fix(mobile): keep shared-token relay siblings live on unsubscribe nativeChat and terminal unsubscribe tokens are deterministic per view target, and the host evicts on duplicate registration. Skip the unsubscribe RPC while a live sibling on the same connection still owns that token. --- ...mobile-relay-browser-cancel-budget.test.ts | 109 ++++++++ .../mobile-relay-rpc-session.test.ts | 30 ++ .../src/transport/mobile-relay-rpc-session.ts | 1 + ...bile-relay-rpc-stream-cancellation.test.ts | 259 ++++++++++++++++++ .../src/transport/mobile-relay-rpc-streams.ts | 119 +++++++- .../rpc-client-terminal-subscription.ts | 8 +- 6 files changed, 509 insertions(+), 17 deletions(-) create mode 100644 mobile/src/transport/mobile-relay-browser-cancel-budget.test.ts create mode 100644 mobile/src/transport/mobile-relay-rpc-stream-cancellation.test.ts diff --git a/mobile/src/transport/mobile-relay-browser-cancel-budget.test.ts b/mobile/src/transport/mobile-relay-browser-cancel-budget.test.ts new file mode 100644 index 00000000000..59426fc42da --- /dev/null +++ b/mobile/src/transport/mobile-relay-browser-cancel-budget.test.ts @@ -0,0 +1,109 @@ +import { describe, expect, it } from 'vitest' +import { RuntimeBrowserScreencastController } from '../../../src/main/runtime/runtime-browser-screencast-controller' +import type { RuntimeBrowserCommands } from '../../../src/main/runtime/orca-runtime-browser' +import type { BrowserScreencastResult } from '../../../src/shared/runtime-types' +import { MobileRelayRpcStreams } from './mobile-relay-rpc-streams' +import type { RpcResponse } from './types' + +describe('relay browser cancellation resource budget', () => { + it.each([false, true])('stops host frames when cancellation precedes ready=%s', async (early) => { + const subscriptions = new Map void | Promise>() + const done = Promise.withResolvers() + const ready = Promise.withResolvers() + let sequence = 0 + let stopped = false + let frameSends = 0 + let frameBytes = 0 + let sendBinary: (bytes: Uint8Array) => boolean | void = () => false + let hostRun: Promise | undefined + const methods: string[] = [] + const cleanup = (id: string): void => { + const release = subscriptions.get(id) + subscriptions.delete(id) + void release?.() + } + const host = new RuntimeBrowserScreencastController({ + getCommands: () => + ({ + browserScreencast: async (_params, stream) => { + sendBinary = stream.sendBinary + return { + subscriptionId: 'server-stream', + ready: { type: 'ready', subscriptionId: 'server-stream', browserPageId: 'page' }, + session: { + done: done.promise, + stop: () => { + stopped = true + done.resolve() + } + }, + flushPendingFrame: () => {} + } + } + }) as RuntimeBrowserCommands, + registerSubscriptionCleanup: (id, release) => subscriptions.set(id, release), + cleanupSubscription: cleanup, + getDriver: () => ({ kind: 'idle' }), + setDriver: () => {}, + notifyRemoteViewersChanged: () => {} + }) + const streams = new MobileRelayRpcStreams({ + nextId: () => `request-${++sequence}`, + waitForConnected: async () => {}, + sendFrame: (request) => { + methods.push(request.method) + if (request.method === 'browser.screencast' && (request.params as { page?: string }).page) { + hostRun = host.start(request.params as Parameters[0], { + connectionId: 'relay-connection', + sendBinary: (bytes) => { + frameSends++ + frameBytes += bytes.byteLength + return true + }, + emit: (result: BrowserScreencastResult) => { + if (result.type === 'ready') { + ready.resolve({ + id: request.id, + ok: true, + streaming: true, + result, + _meta: { runtimeId: 'host' } + }) + } + } + }) + } else if (request.method === 'browser.screencast.unsubscribe') { + cleanup((request.params as { subscriptionId: string }).subscriptionId) + } + return true + } + }) + const cancel = streams.subscribe('browser.screencast', { page: 'page' }, () => {}) + try { + const response = await ready.promise + if (early) { + cancel() + } + streams.handleResponse(response) + if (!early) { + cancel() + } + for (let frame = 0; frame < 100; frame++) { + if (!stopped) { + sendBinary(new Uint8Array(65_536)) + } + } + expect({ stopped, subscriptions: subscriptions.size, frameSends, frameBytes }).toEqual({ + stopped: true, + subscriptions: 0, + frameSends: 0, + frameBytes: 0 + }) + expect(methods).toEqual(['browser.screencast', 'browser.screencast.unsubscribe']) + } finally { + cleanup('server-stream') + await hostRun + streams.clear() + } + }) +}) diff --git a/mobile/src/transport/mobile-relay-rpc-session.test.ts b/mobile/src/transport/mobile-relay-rpc-session.test.ts index 5887dffc73d..4bf617faf50 100644 --- a/mobile/src/transport/mobile-relay-rpc-session.test.ts +++ b/mobile/src/transport/mobile-relay-rpc-session.test.ts @@ -136,6 +136,36 @@ describe('mobile relay RPC session', () => { }) afterEach(() => vi.useRealTimers()) + it('releases stream listeners on failure even when close follows it', async () => { + const { session } = await authenticateSession() + const listener = vi.fn() + session.subscribe('runtime.clientEvents.subscribe', {}, listener) + await Promise.resolve() + const request = JSON.parse(fakes.sendText.mock.calls[0]![0] as string) as { id: string } + fakes.linkOptions!.onText( + JSON.stringify({ + id: request.id, + ok: true, + streaming: true, + result: { type: 'ready', subscriptionId: 'server-events' }, + _meta: { runtimeId: 'runtime-1' } + }) + ) + expect(listener).toHaveBeenCalledTimes(1) + fakes.linkOptions!.onError(new Error('relay lost')) + session.close() + fakes.linkOptions!.onText( + JSON.stringify({ + id: request.id, + ok: true, + streaming: true, + result: { type: 'event' }, + _meta: { runtimeId: 'runtime-1' } + }) + ) + expect(listener).toHaveBeenCalledTimes(1) + }) + it('requires exact resume observations and confirms by request ID before becoming connected', async () => { const { session, confirmationRequest, capabilityRequest } = await authenticateSession() diff --git a/mobile/src/transport/mobile-relay-rpc-session.ts b/mobile/src/transport/mobile-relay-rpc-session.ts index 203a0329192..67b50ea591e 100644 --- a/mobile/src/transport/mobile-relay-rpc-session.ts +++ b/mobile/src/transport/mobile-relay-rpc-session.ts @@ -294,6 +294,7 @@ export function connectMobileRelayRpcSession(args: { closed = true failure = error livenessWatchdog.stop(livenessIdentity) + streams.clear() link.close() pending.rejectAll(error) publishState(error instanceof MobileE2EEAuthenticationError ? 'auth-failed' : 'disconnected') diff --git a/mobile/src/transport/mobile-relay-rpc-stream-cancellation.test.ts b/mobile/src/transport/mobile-relay-rpc-stream-cancellation.test.ts new file mode 100644 index 00000000000..0a7a25dfbd7 --- /dev/null +++ b/mobile/src/transport/mobile-relay-rpc-stream-cancellation.test.ts @@ -0,0 +1,259 @@ +import { describe, expect, it, vi } from 'vitest' +import { MobileRelayRpcStreams } from './mobile-relay-rpc-streams' +import type { RpcResponse } from './types' + +function createStreams(waitForConnected = async () => {}) { + let sequence = 0 + const sendFrame = vi.fn((_request: { id: string; method: string; params?: unknown }) => true) + const streams = new MobileRelayRpcStreams({ + nextId: () => `request-${++sequence}`, + sendFrame, + waitForConnected + }) + return { streams, sendFrame } +} + +function response(id: string, result: unknown): RpcResponse { + return { id, ok: true, streaming: true, result, _meta: { runtimeId: 'test' } } +} + +const serverSubscriptions = [ + ['browser.screencast', 'browser.screencast.unsubscribe'], + ['runtime.clientEvents.subscribe', 'runtime.clientEvents.unsubscribe'] +] as const + +describe('mobile relay subscription cancellation', () => { + it.each(serverSubscriptions)('cleans up ready %s exactly once', async (method, unsubscribe) => { + const { streams, sendFrame } = createStreams() + const listener = vi.fn() + const cancel = streams.subscribe(method, {}, listener) + await Promise.resolve() + streams.handleResponse(response('request-1', { type: 'ready', subscriptionId: 'server-1' })) + cancel() + cancel() + expect(sendFrame.mock.calls).toEqual([ + [{ id: 'request-1', method, params: {} }], + [{ id: 'request-2', method: unsubscribe, params: { subscriptionId: 'server-1' } }] + ]) + expect(streams.handleResponse(response('request-1', { type: 'end' }))).toBe(false) + expect(listener).toHaveBeenCalledTimes(1) + }) + + it.each(serverSubscriptions)( + 'cleans up late-ready %s without calling disposed listeners', + async (method, unsubscribe) => { + const { streams, sendFrame } = createStreams() + const listener = vi.fn() + const cancel = streams.subscribe(method, {}, listener) + await Promise.resolve() + cancel() + cancel() + expect(sendFrame).toHaveBeenCalledTimes(1) + expect(streams.handleResponse(response('request-1', { type: 'starting' }))).toBe(true) + streams.handleResponse(response('request-1', { type: 'ready', subscriptionId: 'server-1' })) + expect(sendFrame).toHaveBeenLastCalledWith({ + id: 'request-2', + method: unsubscribe, + params: { subscriptionId: 'server-1' } + }) + expect( + streams.handleResponse(response('request-1', { type: 'ready', subscriptionId: 'server-1' })) + ).toBe(false) + expect(listener).not.toHaveBeenCalled() + } + ) + + it.each(['error', 'end', 'disconnect', 'completed'])( + 'forgets cancelled cleanup routes on %s', + async (ending) => { + const { streams, sendFrame } = createStreams() + const cancel = streams.subscribe('browser.screencast', {}, vi.fn()) + await Promise.resolve() + cancel() + if (ending === 'disconnect') { + streams.clear() + } else if (ending === 'completed') { + streams.handleResponse({ + id: 'request-1', + ok: true, + result: null, + _meta: { runtimeId: 'test' } + }) + } else if (ending === 'error') { + streams.handleResponse({ + id: 'request-1', + ok: false, + error: { code: 'unsupported', message: 'failed' }, + _meta: { runtimeId: 'test' } + }) + } else { + streams.handleResponse(response('request-1', { type: 'end', subscriptionId: 'server-1' })) + } + expect( + streams.handleResponse(response('request-1', { type: 'ready', subscriptionId: 'server-1' })) + ).toBe(false) + expect(sendFrame).toHaveBeenCalledTimes(1) + } + ) + + it.each([ + [ + 'terminal.subscribe', + { terminal: 'term', client: { id: 'phone' } }, + 'terminal.unsubscribe', + { subscriptionId: 'term:phone', client: { id: 'phone' } } + ], + [ + 'session.tabs.subscribe', + { worktree: 'id:workspace' }, + 'session.tabs.unsubscribe', + { worktree: 'id:workspace', subscriptionId: 'request-1' } + ], + [ + 'nativeChat.subscribe', + { subscriptionId: 'chat' }, + 'nativeChat.unsubscribe', + { subscriptionId: 'chat' } + ] + ])( + 'cancels %s using its request cleanup identity', + async (method, params, unsubscribe, unsubscribeParams) => { + const { streams, sendFrame } = createStreams() + const cancel = streams.subscribe(method as string, params, vi.fn()) + await Promise.resolve() + if (method === 'session.tabs.subscribe') { + streams.handleResponse(response('request-1', { type: 'snapshot' })) + } + cancel() + expect(sendFrame).toHaveBeenLastCalledWith({ + id: 'request-2', + method: unsubscribe, + params: unsubscribeParams + }) + } + ) + + it.each([ + 'terminal.subscribe', + 'browser.screencast', + 'runtime.clientEvents.subscribe', + 'session.tabs.subscribe', + 'nativeChat.subscribe' + ])('does not unsubscribe an unsent %s', async (method) => { + const wait = Promise.withResolvers() + const { streams, sendFrame } = createStreams(() => wait.promise) + const cancel = streams.subscribe( + method, + { terminal: 'term', worktree: 'id:workspace', subscriptionId: 'chat' }, + vi.fn() + ) + cancel() + wait.resolve() + await Promise.resolve() + expect(sendFrame).not.toHaveBeenCalled() + expect( + streams.handleResponse(response('request-1', { type: 'ready', subscriptionId: 'server-1' })) + ).toBe(false) + }) + + it.each([false, true])( + 'preserves a same-worktree sibling when cancellation precedes snapshot=%s', + async (early) => { + const { streams, sendFrame } = createStreams() + const first = vi.fn() + const second = vi.fn() + const cancel = streams.subscribe( + 'session.tabs.subscribe', + { worktree: 'id:workspace' }, + first + ) + streams.subscribe('session.tabs.subscribe', { worktree: 'id:workspace' }, second) + await Promise.resolve() + if (early) { + cancel() + } + expect(sendFrame).toHaveBeenCalledTimes(2) + streams.handleResponse(response('request-1', { type: 'snapshot' })) + if (!early) { + cancel() + } + expect(sendFrame).toHaveBeenLastCalledWith({ + id: 'request-3', + method: 'session.tabs.unsubscribe', + params: { worktree: 'id:workspace', subscriptionId: 'request-1' } + }) + streams.handleResponse(response('request-2', { type: 'snapshot' })) + streams.handleResponse(response('request-2', { type: 'updated' })) + expect(second).toHaveBeenCalledTimes(2) + expect(first).toHaveBeenCalledTimes(early ? 0 : 1) + expect(streams.handleResponse(response('request-1', { type: 'updated' }))).toBe(false) + } + ) + + it.each([ + ['nativeChat.subscribe', { agent: 'claude', sessionId: 's1', subscriptionId: 'claude:s1' }], + ['terminal.subscribe', { terminal: 'term', client: { id: 'phone' } }] + ])( + 'keeps the newer %s live when an older same-token subscription unmounts', + async (method, params) => { + const { streams, sendFrame } = createStreams() + const older = vi.fn() + const newer = vi.fn() + const cancelOlder = streams.subscribe(method, params, older) + const cancelNewer = streams.subscribe(method, { ...params }, newer) + await Promise.resolve() + expect(sendFrame).toHaveBeenCalledTimes(2) + cancelOlder() + // The host keys cleanup by the deterministic token, so unsubscribing would evict the newer. + expect(sendFrame).toHaveBeenCalledTimes(2) + streams.handleResponse(response('request-2', { type: 'snapshot' })) + expect(newer).toHaveBeenCalledTimes(1) + expect(streams.handleResponse(response('request-1', { type: 'snapshot' }))).toBe(false) + expect(older).not.toHaveBeenCalled() + cancelNewer() + expect(sendFrame).toHaveBeenCalledTimes(3) + expect(sendFrame).toHaveBeenLastCalledWith( + expect.objectContaining({ method: method.replace(/\.subscribe$/, '.unsubscribe') }) + ) + } + ) + + it('still unsubscribes a shared-token nativeChat stream when the sibling is unsent', async () => { + const wait = Promise.withResolvers() + let connected = false + const { streams, sendFrame } = createStreams(() => + connected ? Promise.resolve() : wait.promise + ) + const params = { agent: 'claude', sessionId: 's1', subscriptionId: 'claude:s1' } + connected = true + const cancelOlder = streams.subscribe('nativeChat.subscribe', params, vi.fn()) + await Promise.resolve() + connected = false + streams.subscribe('nativeChat.subscribe', params, vi.fn()) + cancelOlder() + expect(sendFrame).toHaveBeenCalledTimes(2) + expect(sendFrame).toHaveBeenLastCalledWith({ + id: 'request-3', + method: 'nativeChat.unsubscribe', + params: { subscriptionId: 'claude:s1' } + }) + }) + + it('cleans up every cancelled server subscription across repeated late-ready cycles', async () => { + const { streams, sendFrame } = createStreams() + const listener = vi.fn() + for (let i = 0; i < 100; i++) { + const cancel = streams.subscribe('runtime.clientEvents.subscribe', {}, listener) + await Promise.resolve() + const requestId = `request-${2 * i + 1}` + cancel() + streams.handleResponse(response(requestId, { type: 'ready', subscriptionId: `server-${i}` })) + } + expect( + sendFrame.mock.calls.filter( + ([request]) => (request as { method: string }).method === 'runtime.clientEvents.unsubscribe' + ) + ).toHaveLength(100) + expect(listener).not.toHaveBeenCalled() + }) +}) diff --git a/mobile/src/transport/mobile-relay-rpc-streams.ts b/mobile/src/transport/mobile-relay-rpc-streams.ts index 2eacbd2168f..abb2c12564b 100644 --- a/mobile/src/transport/mobile-relay-rpc-streams.ts +++ b/mobile/src/transport/mobile-relay-rpc-streams.ts @@ -4,9 +4,11 @@ import { type TerminalSnapshotState } from './rpc-client-terminal-binary-frame' import { + buildStreamUnsubscribe, buildTerminalUnsubscribeParams, updateTerminalSubscriptionViewport } from './rpc-client-terminal-subscription' +import { buildReadyStreamUnsubscribe } from './rpc-client-server-subscription' import type { RpcClient } from './rpc-client' import type { RpcResponse, RpcSuccess } from './types' @@ -22,6 +24,23 @@ type StreamRecord = { streamIds: Set subscriptionId?: string cancelled: boolean + sent: boolean + receivedSnapshot?: boolean +} + +type StreamUnsubscribe = { method: string; params: unknown } + +/** Unsubscribe derived from the subscribe params alone (no server-assigned id). */ +function buildParamsUnsubscribe( + method: string, + params: unknown, + requestId: string +): StreamUnsubscribe | null { + if (method === 'terminal.subscribe') { + const unsubscribeParams = buildTerminalUnsubscribeParams(params) + return unsubscribeParams ? { method: 'terminal.unsubscribe', params: unsubscribeParams } : null + } + return buildStreamUnsubscribe(method, params, requestId) } type StreamManagerOptions = { @@ -32,6 +51,10 @@ type StreamManagerOptions = { export class MobileRelayRpcStreams { private readonly streams = new Map() + private readonly cancelledSubscriptions = new Map< + string, + { method: string; unsubscribe?: StreamUnsubscribe } + >() private readonly terminalListeners = new Map void>() private readonly terminalSnapshots = new Map() private activeBrowserStream: StreamRecord | null = null @@ -51,13 +74,15 @@ export class MobileRelayRpcStreams { listener, onBinaryFrame: subscribeOptions?.onBinaryFrame, streamIds: new Set(), - cancelled: false + cancelled: false, + sent: false } this.streams.set(id, stream) void this.options .waitForConnected() .then(() => { if (!stream.cancelled) { + stream.sent = true if (!this.options.sendFrame({ id, method, params: stream.params })) { this.fail(id, stream, 'Connection interrupted') } @@ -75,6 +100,30 @@ export class MobileRelayRpcStreams { } handleResponse(response: RpcResponse): boolean { + const cancelled = this.cancelledSubscriptions.get(response.id) + if (cancelled) { + if (!response.ok) { + this.cancelledSubscriptions.delete(response.id) + } else if (response.result && typeof response.result === 'object') { + const result = response.result as { subscriptionId?: unknown; type?: unknown } + if (result.type === 'end') { + this.cancelledSubscriptions.delete(response.id) + } else if (result.type === 'snapshot' && cancelled.unsubscribe) { + this.cancelledSubscriptions.delete(response.id) + this.options.sendFrame({ id: this.options.nextId(), ...cancelled.unsubscribe }) + } else if (typeof result.subscriptionId === 'string') { + this.cancelledSubscriptions.delete(response.id) + const unsubscribe = buildReadyStreamUnsubscribe(cancelled.method, result.subscriptionId) + if (unsubscribe) { + this.options.sendFrame({ id: this.options.nextId(), ...unsubscribe }) + } + } + } + if (response.ok && response.streaming !== true) { + this.cancelledSubscriptions.delete(response.id) + } + return true + } const stream = this.streams.get(response.id) if (!stream) { return false @@ -86,6 +135,9 @@ export class MobileRelayRpcStreams { const result = (response as RpcSuccess).result if (result && typeof result === 'object') { const metadata = result as { subscriptionId?: unknown; streamId?: unknown; type?: unknown } + if (stream.method === 'session.tabs.subscribe' && metadata.type === 'snapshot') { + stream.receivedSnapshot = true + } if (typeof metadata.subscriptionId === 'string') { stream.subscriptionId = metadata.subscriptionId } @@ -125,6 +177,7 @@ export class MobileRelayRpcStreams { stream.cancelled = true } this.streams.clear() + this.cancelledSubscriptions.clear() this.terminalListeners.clear() this.terminalSnapshots.clear() this.activeBrowserStream = null @@ -136,25 +189,61 @@ export class MobileRelayRpcStreams { return } stream.cancelled = true - if (stream.method === 'terminal.subscribe') { - const params = buildTerminalUnsubscribeParams(stream.params) - if (params) { - this.options.sendFrame({ - id: this.options.nextId(), - method: 'terminal.unsubscribe', - params - }) + if (stream.sent) { + const byParams = buildParamsUnsubscribe(stream.method, stream.params, id) + if (stream.method === 'terminal.subscribe') { + if (byParams) { + this.sendUnsubscribe(byParams) + } + } else { + const unsubscribe = stream.subscriptionId + ? buildReadyStreamUnsubscribe(stream.method, stream.subscriptionId) + : null + if (byParams && stream.method === 'session.tabs.subscribe' && !stream.receivedSnapshot) { + // The host registers cleanup only after resolving the initial snapshot. + this.cancelledSubscriptions.set(id, { method: stream.method, unsubscribe: byParams }) + } else if (unsubscribe || byParams) { + this.sendUnsubscribe((unsubscribe ?? byParams)!) + } else if ( + stream.method === 'browser.screencast' || + stream.method === 'runtime.clientEvents.subscribe' + ) { + // Keep only the cleanup route while the server assigns its subscription ID. + this.cancelledSubscriptions.set(id, { method: stream.method }) + } else if (stream.subscriptionId) { + this.sendUnsubscribe({ + method: stream.method.replace(/\.subscribe$/, '.unsubscribe'), + params: { subscriptionId: stream.subscriptionId } + }) + } } - } else if (stream.subscriptionId) { - this.options.sendFrame({ - id: this.options.nextId(), - method: stream.method.replace(/\.subscribe$/, '.unsubscribe'), - params: { subscriptionId: stream.subscriptionId } - }) } this.remove(id) } + /** Skip the unsubscribe when a live sibling shares the host cleanup token (e.g. nativeChat's + * deterministic `agent:sessionId`), since the host would evict the sibling's registration. */ + private sendUnsubscribe(unsubscribe: StreamUnsubscribe): void { + if (this.hasLiveOwner(unsubscribe)) { + return + } + this.options.sendFrame({ id: this.options.nextId(), ...unsubscribe }) + } + + private hasLiveOwner(unsubscribe: StreamUnsubscribe): boolean { + const token = JSON.stringify(unsubscribe) + for (const [siblingId, sibling] of this.streams) { + if (sibling.cancelled || !sibling.sent) { + continue + } + const siblingUnsubscribe = buildParamsUnsubscribe(sibling.method, sibling.params, siblingId) + if (siblingUnsubscribe && JSON.stringify(siblingUnsubscribe) === token) { + return true + } + } + return false + } + private remove(id: string): void { const stream = this.streams.get(id) if (!stream) { diff --git a/mobile/src/transport/rpc-client-terminal-subscription.ts b/mobile/src/transport/rpc-client-terminal-subscription.ts index 471bc3c4a4e..405f7f1d9e0 100644 --- a/mobile/src/transport/rpc-client-terminal-subscription.ts +++ b/mobile/src/transport/rpc-client-terminal-subscription.ts @@ -38,7 +38,8 @@ export function updateTerminalSubscriptionViewport( * the per-method echo logic out of the rpc-client teardown closure. */ export function buildStreamUnsubscribe( method: string | undefined, - params: unknown + params: unknown, + requestId?: string ): { method: string; params: Record } | null { if (!params || typeof params !== 'object') { return null @@ -46,7 +47,10 @@ export function buildStreamUnsubscribe( if (method === 'session.tabs.subscribe') { const worktree = (params as { worktree?: unknown }).worktree return typeof worktree === 'string' - ? { method: 'session.tabs.unsubscribe', params: { worktree } } + ? { + method: 'session.tabs.unsubscribe', + params: { worktree, ...(requestId ? { subscriptionId: requestId } : {}) } + } : null } if (method === 'nativeChat.subscribe') { From eebedf206f7b23f94859f78ddcf29deab4c36418 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 21:08:18 -0700 Subject: [PATCH 37/69] fix(tests): provide a window manager for Linux Electron CI (#19007) --- .github/scripts/e2e-with-window-manager.sh | 26 +++++++++++++++++++ .github/workflows/e2e.yml | 16 ++++++------ ...red-remote-terminal-stall-recovery.spec.ts | 22 +++++++++++++--- 3 files changed, 53 insertions(+), 11 deletions(-) create mode 100644 .github/scripts/e2e-with-window-manager.sh diff --git a/.github/scripts/e2e-with-window-manager.sh b/.github/scripts/e2e-with-window-manager.sh new file mode 100644 index 00000000000..d431a039809 --- /dev/null +++ b/.github/scripts/e2e-with-window-manager.sh @@ -0,0 +1,26 @@ +#!/usr/bin/env bash +set -euo pipefail +openbox --sm-disable > /tmp/orca-e2e-window-manager.log 2>&1 & +wm_pid=$! +cleanup() { + kill "$wm_pid" 2>/dev/null || true + wait "$wm_pid" 2>/dev/null || true +} +trap cleanup EXIT +ready=false +for attempt in {1..100}; do + if xprop -root _NET_SUPPORTING_WM_CHECK 2>/dev/null | rg -q 'window id # 0x[1-9a-fA-F]'; then + ready=true + break + fi + if ! kill -0 "$wm_pid" 2>/dev/null; then + cat /tmp/orca-e2e-window-manager.log + exit 1 + fi + sleep 0.1 +done +if [ "$ready" != true ]; then + echo 'Window manager did not acquire the Xvfb root window' >&2 + exit 1 +fi +"$@" diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index 5f80c2090ad..a94a7ea2ba5 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -150,7 +150,7 @@ jobs: # Native cache misses need the compiler, Electron needs Xvfb, and paired # Quick Open needs ripgrep. Install them in one apt transaction per shard. - name: Install native build and headless UI tools - run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk python3 ripgrep xvfb zsh + run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk python3 ripgrep xvfb zsh openbox x11-utils - uses: ./.github/actions/install-node-dependencies with: @@ -171,7 +171,7 @@ jobs: # ORCA_E2E_FORWARD_APP_LOGS keeps startup failures visible when Electron # launches but never creates a BrowserWindow. - name: Run E2E tests (${{ matrix.shard_name }}) - run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 ORCA_E2E_WEB_CLIENT=1 ORCA_RELAY_PATH="$GITHUB_WORKSPACE/out/relay" pnpm run test:e2e --shard=${{ matrix.shard }} + run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 ORCA_E2E_WEB_CLIENT=1 ORCA_RELAY_PATH="$GITHUB_WORKSPACE/out/relay" pnpm run test:e2e --shard=${{ matrix.shard }} # Why: Playwright retains traces/screenshots only on failure. Uploading # them as an artifact makes post-mortem debugging on CI possible without @@ -205,7 +205,7 @@ jobs: # unbounded inventory fallback; the paired fixture exercises that real boundary. # Why openssh-client: the Docker-SSH fixture shells out to ssh/ssh-keygen, and this # lane now receives those specs from pr.yml's SSH source mapping. - run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh + run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh openbox x11-utils - uses: ./.github/actions/install-node-dependencies with: @@ -245,7 +245,7 @@ jobs: if grep -l '@headful' "${TEST_FILES[@]}" >/dev/null; then E2E_PROJECT_ARGS+=(--project=electron-headful) fi - xvfb-run --auto-servernum env "${E2E_ENV[@]}" \ + xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env "${E2E_ENV[@]}" \ pnpm run test:e2e "${TEST_FILES[@]}" --workers=1 "${E2E_PROJECT_ARGS[@]}" - name: Upload Playwright traces @@ -282,7 +282,7 @@ jobs: ref: ${{ inputs.ref || github.ref }} - name: Install native build and headless UI tools - run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 xvfb zsh + run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh openbox x11-utils - uses: ./.github/actions/install-node-dependencies with: @@ -297,7 +297,7 @@ jobs: # Why: this is the release-path proof that the deployed Linux relay keeps # its PTY and explorer live across a real watcher SIGSEGV. - name: Run Docker SSH watcher isolation E2E - run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-watcher-isolation + run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-watcher-isolation # Why: Playwright empties test-results/ when it starts, so each step here used to # destroy the previous step's traces. Only the last lane's failure was ever @@ -314,7 +314,7 @@ jobs: # readiness across live SSH, headed paired, and headless serve topologies. - name: Run Docker SSH terminal parking + startup readiness E2E if: always() - run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-terminal-parking + run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-terminal-parking - name: Keep terminal-parking traces if: always() @@ -330,7 +330,7 @@ jobs: # legible as an SSH-named failure. - name: Run remaining Docker SSH E2E if: always() - run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker + run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker - name: Keep remaining-ssh-docker traces if: always() diff --git a/tests/e2e/paired-remote-terminal-stall-recovery.spec.ts b/tests/e2e/paired-remote-terminal-stall-recovery.spec.ts index 4406ebd2f18..2c81b76f077 100644 --- a/tests/e2e/paired-remote-terminal-stall-recovery.spec.ts +++ b/tests/e2e/paired-remote-terminal-stall-recovery.spec.ts @@ -1,3 +1,4 @@ +import { runProcess } from '../../src/shared/child-process/run-process' import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' import os from 'node:os' import path from 'node:path' @@ -101,11 +102,26 @@ async function minimizeHeadedHost(electronApp: ElectronApplication, page: Page): .poll(() => host.evaluate((window) => ({ backgroundThrottling: window.webContents.getBackgroundThrottling(), - minimized: window.isMinimized(), - visible: window.isVisible() + minimized: window.isMinimized() })) ) - .toEqual({ backgroundThrottling: true, minimized: true, visible: false }) + .toEqual({ backgroundThrottling: true, minimized: true }) + // Linux reports isVisible/document visibility differently; the window manager owns iconification. + if (process.platform === 'linux') { + const nativeId = await host.evaluate((window) => window.getNativeWindowHandle().readUInt32LE(0)) + await expect + .poll(async () => { + const result = await runProcess({ + program: 'xprop', + args: ['-id', String(nativeId), '_NET_WM_STATE'], + timeoutMs: 5_000 + }) + return result.stdout + }) + .toContain('_NET_WM_STATE_HIDDEN') + } else { + await expect.poll(() => page.evaluate(() => document.visibilityState)).toBe('hidden') + } } async function restoreHeadedHost(electronApp: ElectronApplication, page: Page): Promise { From fba90e017c81eff36697373a728e7e9029669738 Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Sat, 5 Sep 2026 21:11:28 -0700 Subject: [PATCH 38/69] fix(windows): copy the daemon host exe verbatim instead of renaming it (MDE T1036) (#17865) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * docs(windows): document the EDR signal surface Six Microsoft Defender for Endpoint incidents fired against Orca 1.4.192 in eight days on one enterprise Windows 11 / Intune tenant. All six were behavioural process-tree scoring, not signature hits; two escalated to multi-stage incidents mapped to ATT&CK Execution and Collection. Add a reference doc mapping each attack-technique-shaped behaviour to the code that produces it and to why it exists: the renamed daemon image (T1036), the per-process PEB read, encoded policy-bypassed PowerShell (T1049), caret-escaped cmd.exe lines, and computer-use screen capture plus runtime-compiled MSIL (T1113). Records that signing is not the gate -- reputation is signer plus hash-keyed prevalence -- and carries the two evidence gaps the report noted. Adds an engineer checklist, deployment guidance for admins (AV path exclusions do not suppress EDR behavioural alerts; an MDE alert suppression rule does), and an explicit pre-deployment warning about computer use. * docs(windows): correct the PowerShell flag inventory and admin paths Review corrections to the EDR posture doc. The "encoded, policy-bypassing PowerShell" list conflated three different shapes and was incomplete. Split it into the three tiers an EDR actually scores differently -- bypass plus encoding, encoding alone, and bypass alone -- and add the sites it missed, including windows-mobile-firewall.ts, which encodes a script and launches it elevated through Start-Process -Verb RunAs. system-fonts.ts (-Command) and desktop-script-provider-bridge.ts (-File) were listed as encoded and are not. Notes that a raw grep under-reports, because the hook sites reach -EncodedCommand through wrapWindowsPowerShellEncodedCommand. Attribute the in-payload Set-ExecutionPolicy move to #16576 rather than to #16003's measurement, which keyed on -WindowStyle Hidden + -EncodedCommand, and record that the launcher's own tradeoff is unverified on a real box. Admin guidance was missing two ways a suppression rule pinned to one full path misses real activity: the .staging- sibling that exists mid-update, which is when the update-cluster incidents fire, and the userData fallback when LOCALAPPDATA is unset. Also: state the measurement conditions on the process-table timings, note that Hermes has surface even though we have no telemetry for it, note that the uninstaller names are electron-builder-generated and in no repo file, drop a volatile line count, and mark the per-operation computer-use shape as being addressed by an unmerged change. Drops the duplicated AGENTS.md section, keeping the indexed bullet. * docs(windows): reconcile the EDR posture doc with the shipped remediation Three claims in this doc became false once the rest of the Windows EDR set landed, and two told engineers the opposite of what the release does. The process-table section still described one shared snapshot taken with `Memory | CommandLine | CreationTime`, argued that splitting the cache per field set "would restore exactly the fan-out it exists to prevent", and concluded the shape was unfixable because "the information is only in the PEB". The split shipped (identity opens no handle at all), `Memory` is retired, and the command line now comes from the kernel through `ProcessCommandLineInformation` -- `ReadProcessMemory` is absent from the compiled addon and a ratchet asserts it against the import table. An engineer reading the old text would have concluded both fixes were dead ends. The PowerShell site inventories were stale in three of four lists: the port scan went native, every `-ExecutionPolicy Bypass` + `-EncodedCommand` pair was dropped as a measured no-op, and of the unencoded-bypass list only `wsl-cli-scripts.ts` survives. Regenerated against the merged tree, including the sites that reach the flag through `wrapWindowsPowerShellEncodedCommand` and never spell it, which a raw `rg` misses. Incident-evidence sections are left alone: they record what the tenant observed on 1.4.192, not what the code does now. * fix(windows): copy the daemon host exe verbatim instead of renaming it Microsoft Defender for Endpoint flagged `orca-terminal-daemon.exe` as MITRE T1036 (Masquerading): Orca copied its own `Orca.exe` into %LOCALAPPDATA% under a different name, specifically so the NSIS updater's `taskkill /IM Orca.exe` could not match, then ran it detached. Because that process is what every other flagged action was attributed to, the name mismatch acted as a reputation multiplier on unrelated findings. The rename was never what made the daemon survive. In app-builder-lib 26.15.3 the installer's FIND_PROCESS/KILL_PROCESS select processes whose image path is under $INSTDIR; `taskkill /IM` is only the fallback for hosts where PowerShell is missing or blocked. Survival is a property of the path, and %LOCALAPPDATA%\Orca\daemon-host is outside $INSTDIR whatever the file is called. Derive the host exe name from process.execPath so the copy is byte-for-byte, name included — it keeps its Authenticode signature and carries no renamed-image signal. On the no-PowerShell fallback the daemon is now killed with the app and terminals cold-restore, which is the documented pre-relocation outcome the update harness already asserts, not a regression. The uninstall macro no longer needs a distinct name to find the daemon; it kills the app's own image name (plus the legacy name, for hosts left by older builds). Adds docs/reference/windows-daemon-host-relocation.md with the survival contract, the rejected alternatives and their measured costs, and the invariants to keep. * fix(windows): apply daemon-host relocation review corrections Scope the uninstall taskkill to the current user with `/FI "USERNAME eq %USERNAME%"` via cmd.exe, matching upstream's per-user KILL_PROCESS — without it an elevated machine-wide uninstall reaches another logged-on user's session, so the "no collateral" claim in the comment was overstated. Comment the rmSync-before-publish: Windows refuses to delete a running image, so a live daemon already hosted in this version's dir (same-version reinstall, or a dev channel reusing a version) throws and materialization fails open. Doc corrections: - The fallback selector is the full per-user `taskkill /F /IM ".exe" /FI "PID ne $pid" /FI "USERNAME eq %USERNAME%"`, not a bare `taskkill /IM`. - The probe reads `Get-ExecutionPolicy -Scope Process`, not the effective policy, and GPO writes MachinePolicy/UserPolicy — so GPO-managed hosts take the primary path-scoped branch. Narrow the fallback triggers accordingly. - Drop the Authenticode sentence: the old name was equally byte-identical and equally signed, so a filename has no bearing on signature validity. - Name the new update-abort path: the daemon now matches FIND_PROCESS, so on the fallback branch an unkillable host reaches the retry loop's MessageBox /SD IDCANCEL and Quits, aborting a silent update. - Correct the customCheckAppRunning rejection. It is ~6 lines, not a rewrite; it is wrong because forcing the PowerShell branch where PowerShell is absent makes FIND/KILL silently no-op and leaves the real app running with files in use. - Bound the win honestly: OriginalFilename is empty on the shipped binary, so the strongest T1036 indicator never fired, and the residual copy-and-run-detached shape still maps to T1036.005. Reconcile docs/reference/windows-edr-posture.md, which documents the rename as a live finding and would otherwise contradict this change. Content-only edit: markdown under docs/reference/ is not oxfmt-formatted as a matter of practice and nothing in CI gates it, so the file is left consistent with its neighbours. * fix(windows): expand USERNAME in NSIS instead of spawning cmd.exe The uninstall macro routed both taskkills through `"$SYSDIR\cmd.exe" /C` purely so `%USERNAME%` would expand — two extra interpreter spawns on the uninstall path, in a change whose whole point is not adding scored behaviour, and the exact `cmd.exe /c` shape the new AGENTS.md EDR bullet warns about. NSIS reads the variable itself with ReadEnvStr, so the spawns buy nothing. Verified on Windows 11 that the generated command line does what the filter is there for: a copy of cmd.exe running as orca-nonexistent-probe.exe (pid 34244) was terminated by `taskkill /F /IM "orca-nonexistent-probe.exe" /FI "USERNAME eq "` — SUCCESS, exit 0, process gone. Guarded on an empty USERNAME because the degenerate case is silent: taskkill rejects an empty filter value outright ("The search filter cannot be recognized") and kills nothing, which would leave exactly the orphaned daemon this macro exists to reap. `*` is rejected as a filter value too, so there is no branchless spelling. With no USERNAME to scope by it kills unfiltered, as the macro did before the filter was added. Stack stays balanced: three pushes, two nsExec pops, three restores. Also strike the last stale row in windows-edr-posture.md's remediation table. "Copying our own image under a different name" read as outstanding work; it is done by this change, so the row now points at the relocation doc. Same class of staleness as the section reconciled in the previous commit, and git would not have flagged it either. * fix(windows): port the daemon-host uninstall sweep into the live NSIS include The uninstall macro this branch rewrote lived in config/nsis/daemon-host-uninstall.nsh, which main no longer includes: #17906 consolidated every Windows installer hook into config/nsis/orca-installer-hooks.nsh because electron-builder accepts exactly one `nsis.include`. Merged as-is, the rewritten macro would have been dead code while the shipped uninstaller kept running main's stale sweep — `taskkill /F /IM orca-terminal-daemon.exe`, which matches nothing now that the relocated host is a verbatim Orca.exe copy. The RMDir that follows then cannot delete the running image, so a live orphaned daemon and its ~224 MB tree would survive every uninstall. Ported into the live include: the ${APP_EXECUTABLE_FILENAME} kill, the USERNAME filter that keeps an elevated machine-wide uninstall out of another logged-on user's session, and the register save/restore around both. The legacy orca-terminal-daemon.exe kill stays so hosts left by older builds are still reaped. The ratchet that was meant to catch exactly this pinned only the legacy image name, which main's stale macro already satisfied, so it passed both ways. It now asserts the app-exe kill and the USERNAME filter, against comment-stripped script — the prose above the macro names both image names, so a toContain over the raw file proves nothing. --------- Co-authored-by: Orca Worker --- .gitignore | 1 + AGENTS.md | 1 + config/nsis/orca-installer-hooks.nsh | 44 +++++-- ...ron-builder-markdown-associations.test.mjs | 20 ++- .../windows-daemon-host-relocation.md | 118 ++++++++++++++++++ docs/reference/windows-edr-posture.md | 37 ++++-- .../daemon/daemon-host-relocation.test.ts | 38 ++++-- src/main/daemon/daemon-host-relocation.ts | 24 +++- tests/tools/win-crash-survival-e2e/README.md | 4 +- .../win-crash-survival-e2e/crash-step.mjs | 2 +- tests/tools/win-crash-survival-e2e/run.mjs | 2 +- 11 files changed, 247 insertions(+), 44 deletions(-) create mode 100644 docs/reference/windows-daemon-host-relocation.md diff --git a/.gitignore b/.gitignore index 8be3fc5b6f4..08be52f0751 100644 --- a/.gitignore +++ b/.gitignore @@ -110,6 +110,7 @@ docs/** !docs/reference/macos-press-and-hold.md !docs/reference/orcad-operations.md !docs/reference/relay-grace-time-reconfiguration.md +!docs/reference/windows-daemon-host-relocation.md !docs/reference/windows-edr-posture.md !docs/reference/windows-process-enumeration.md !docs/reference/wsl-runner-verification.md diff --git a/AGENTS.md b/AGENTS.md index 306c9c8d5ed..491d270b815 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -55,6 +55,7 @@ Orca targets macOS, Linux, and Windows. Keep all platform-dependent behavior beh - **Windows setup scripts**: the setup/issue-command runner is a `.cmd` batch file unless the script starts with a `#!` line — never derive that from the user's terminal-shell preference, and never launch a `.cmd` runner with a bare `cmd.exe /c` from a Git Bash pane (MSYS rewrites the `/c`). See [`docs/reference/windows-setup-shell.md`](./docs/reference/windows-setup-shell.md). - **Windows child processes**: start them through `runProcess`/`spawnProcess` in `src/shared/child-process/` — never `child_process` directly. It pins `windowsHide`, refuses `shell: true`, and encodes `.cmd`/`.bat` arguments so neither `CommandLineToArgvW` nor `cmd.exe` mangles them. A ratchet test fails on any new direct import. - **Windows process enumeration**: read the table through `src/main/windows/windows-process-table.ts`, never by forking `powershell.exe`. See [`docs/reference/windows-process-enumeration.md`](./docs/reference/windows-process-enumeration.md). +- **Windows daemon-host relocation**: the terminal daemon runs from a copy of the app runtime under `%LOCALAPPDATA%`, which is what survives an auto-update. Before touching that copy, its exe name, or the NSIS uninstall macro, read [`docs/reference/windows-daemon-host-relocation.md`](./docs/reference/windows-daemon-host-relocation.md). - **Windows EDR signal**: don't add `-ExecutionPolicy Bypass`, `-EncodedCommand`, `cmd.exe /c` with escaped free text, per-operation interpreter spawning, or runtime `Add-Type` compilation without reading [`docs/reference/windows-edr-posture.md`](./docs/reference/windows-edr-posture.md) first — behavioural EDR scores each of those, and being signed does not clear them. - **WSL commands**: build argv with `buildWslExecArgs` (always `--exec` — under `--`, `wsl.exe` expands `$name` in every argument and silently rewrites the script), and fence anything whose stdout you parse with `buildWslCapturedLoginShellCommand`, because the interactive login shell prints the distro banner to stdout. See [`docs/reference/wsl-command-execution.md`](./docs/reference/wsl-command-execution.md). - **Linux native modules**: keep the glibc floor at Ubuntu 20.04 / glibc 2.31. A module compiled from source on a newer runner can reference symbol versions absent on the floor and crash the app on startup. See [`docs/reference/linux-glibc-compatibility.md`](./docs/reference/linux-glibc-compatibility.md); packaging fails if a bundled native binary needs newer glibc. diff --git a/config/nsis/orca-installer-hooks.nsh b/config/nsis/orca-installer-hooks.nsh index ca80c99fc6d..d89439073ab 100644 --- a/config/nsis/orca-installer-hooks.nsh +++ b/config/nsis/orca-installer-hooks.nsh @@ -49,22 +49,48 @@ ; --------------------------------------------------------------------------- ; Clean up the relocated terminal daemon on a REAL uninstall. ; -; Why: the daemon host is deliberately copied to a distinct image name -; (orca-terminal-daemon.exe) under %LOCALAPPDATA%\Orca\daemon-host so that app -; UPDATES cannot kill it — that relocation is what keeps terminals alive across -; updates. The same design means a normal uninstall's process sweep and file -; removal both miss it, leaving an orphaned daemon plus its runtime copy behind. +; Why: the daemon host is deliberately copied OUT of the install dir into +; %LOCALAPPDATA%\Orca\daemon-host so that app UPDATES cannot kill it — +; electron-builder's kill sweep selects processes whose image path is under +; $INSTDIR, and that relocation is what keeps terminals alive across updates. +; The same design means a normal uninstall's process sweep and file removal both +; miss it, leaving an orphaned daemon plus its runtime copy behind. ; ; The ${isUpdated} guard is essential: electron-builder runs this uninstaller as ; part of uninstallOldVersion on EVERY update, and killing the daemon there would ; defeat the whole feature. Only clean up on a genuine uninstall. ; -; The image name and the LOCALAPPDATA folder name must stay in sync with -; DAEMON_HOST_EXE_NAME and LOCAL_HOST_ROOT_NAME in -; src/main/daemon/daemon-host-relocation.ts. +; The LOCALAPPDATA folder name must stay in sync with LOCAL_HOST_ROOT_NAME in +; src/main/daemon/daemon-host-relocation.ts. See +; docs/reference/windows-daemon-host-relocation.md. !macro customUnInstall ${ifNot} ${isUpdated} - nsExec::Exec 'taskkill /F /IM orca-terminal-daemon.exe' + Push $0 + Push $1 + Push $2 + ; The host exe is a verbatim copy of the app exe, so the app's own image name + ; reaches it; the second name covers hosts left by builds that renamed the copy. + ; Filtered to the current user like upstream's per-user KILL_PROCESS, so an + ; elevated machine-wide uninstall cannot reach another logged-on user's session. + ; NSIS expands USERNAME itself: routing through cmd.exe only to get %USERNAME% + ; would add two interpreter spawns to the uninstall path for nothing. + ReadEnvStr $1 USERNAME + ${if} $1 == "" + ; Measured: taskkill rejects an empty filter value outright ("The search filter + ; cannot be recognized") and kills nothing, so with no USERNAME to scope by, + ; kill unfiltered rather than not at all. USERNAME is set in every session an + ; uninstaller runs in, so this is a backstop, not the expected path. + StrCpy $2 "" + ${else} + StrCpy $2 '/FI "USERNAME eq $1"' + ${endIf} + nsExec::Exec 'taskkill /F /IM "${APP_EXECUTABLE_FILENAME}" $2' + Pop $0 + nsExec::Exec 'taskkill /F /IM "orca-terminal-daemon.exe" $2' + Pop $0 + Pop $2 + Pop $1 + Pop $0 ; Give the OS a moment to release the image lock before removing the tree. Sleep 500 RMDir /r "$LOCALAPPDATA\Orca\daemon-host" diff --git a/config/scripts/electron-builder-markdown-associations.test.mjs b/config/scripts/electron-builder-markdown-associations.test.mjs index 7ae3b1c9428..58f6f8d8865 100644 --- a/config/scripts/electron-builder-markdown-associations.test.mjs +++ b/config/scripts/electron-builder-markdown-associations.test.mjs @@ -103,14 +103,24 @@ describe('electron-builder markdown file associations', () => { // Why: this include was renamed from daemon-host-uninstall.nsh to carry the markdown // hooks too. electron-builder allows only one include, so a merge that drops the daemon - // sweep would silently orphan a running orca-terminal-daemon.exe on every uninstall. + // sweep would silently orphan a running daemon host on every uninstall. + // + // Asserted against comment-stripped script, and on the app exe name first: the relocated + // host is a verbatim copy of the app exe (daemonHostExeName, daemon-host-relocation.ts), + // so a macro that kills only orca-terminal-daemon.exe matches no running process. The + // prose above the macro names both, so a toContain over the raw file proves nothing. it('keeps the daemon-host uninstall sweep across the include rename', async () => { - const hooks = await readInstallerHooks() + const script = stripNsisCommentLines(await readInstallerHooks()) - expect(hooks).toContain('orca-terminal-daemon.exe') - expect(hooks).toContain('$LOCALAPPDATA\\Orca\\daemon-host') + expect(script).toMatch(/taskkill[^\n]*\/IM\s+"?\$\{APP_EXECUTABLE_FILENAME\}"?/) + // Legacy name, so hosts left by builds that renamed the copy still get reaped. + expect(script).toMatch(/taskkill[^\n]*\/IM\s+"?orca-terminal-daemon\.exe"?/) + // Scopes both kills to the uninstalling user: an elevated machine-wide uninstall must + // not reach another logged-on user's session. + expect(script).toMatch(/\/FI\s+"USERNAME eq /) + expect(script).toContain('$LOCALAPPDATA\\Orca\\daemon-host') // Without this guard, uninstallOldVersion would kill the daemon on every update — // defeating the relocation that keeps terminals alive across updates. - expect(hooks).toMatch(/\$\{ifNot\}\s+\$\{isUpdated\}/) + expect(script).toMatch(/\$\{ifNot\}\s+\$\{isUpdated\}/) }) }) diff --git a/docs/reference/windows-daemon-host-relocation.md b/docs/reference/windows-daemon-host-relocation.md new file mode 100644 index 00000000000..f560597e25e --- /dev/null +++ b/docs/reference/windows-daemon-host-relocation.md @@ -0,0 +1,118 @@ +# Windows daemon-host relocation + +On Windows the terminal daemon does not run from the install directory. Before it forks the +daemon, Orca materializes a trimmed copy of its own runtime under +`%LOCALAPPDATA%\Orca\daemon-host\\` and forks the daemon from there +(`src/main/daemon/daemon-host-relocation.ts`). This is what keeps live terminals alive across an +auto-update and across a crash of the main process. + +Read this before changing the copy plan, the host exe name, the LOCALAPPDATA layout, or +`config/nsis/orca-installer-hooks.nsh`. + +## What the relocation actually escapes + +The killer is **electron-builder's process sweep, matched on image path** — not file deletion. +Windows will not delete a running image, so `RMDir /r "$INSTDIR"` cannot end the daemon on its own. + +In app-builder-lib's `allowOnlyOneInstallerInstance.nsh`, `FIND_PROCESS` / `KILL_PROCESS` have two +branches: + +| Branch | Condition | Selector | +| -------- | --------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Primary | `powershell.exe` runs, `Get-CimInstance` resolves, and `Get-ExecutionPolicy -Scope Process` is not `Restricted` | `Win32_Process` where `$_.Path.StartsWith('$INSTDIR', 'CurrentCultureIgnoreCase')` — **path-scoped** | +| Fallback | otherwise | per-user: `taskkill /F /IM ".exe" /FI "PID ne $pid" /FI "USERNAME eq %USERNAME%"`; per-machine: the same without the username filter — **image-name-scoped** | + +The probe reads the **process** scope, not the effective policy, and Group Policy writes +`MachinePolicy`/`UserPolicy` — so a GPO-managed host whose effective policy is `Restricted` still +exits 0 and takes the primary branch. The fallback is reached only when `powershell.exe` is absent, +`Get-CimInstance` does not resolve, PowerShell is blocked outright (WDAC/AppLocker, Server Core), or +an inherited `PSExecutionPolicyPreference=Restricted` is in the environment. + +So on essentially every machine the sweep is path-scoped, and a daemon whose image lives under +`%LOCALAPPDATA%` is out of range regardless of what the file is called. **Survival is a property of +the path.** The name only matters on the fallback branch. + +## Why the exe is copied verbatim (and not renamed) + +The host exe keeps the app exe's own file name (`daemonHostExeName()` returns +`basename(process.execPath)`), so the relocated image is a byte-for-byte copy of the app binary +under its original name. + +An earlier revision copied it as `orca-terminal-daemon.exe` specifically so the fallback +`taskkill /IM Orca.exe` could not match. That bought survival on the rare no-PowerShell host and +cost a textbook defence-evasion signature: _a process copies its own image into a user-writable +directory under a different name so a kill-by-image-name cannot match it, then runs detached and +survives the installer._ Microsoft Defender for Endpoint flagged it as MITRE **T1036 +(Masquerading)**, and — because it is the process every other flagged action is attributed to — it +acted as a reputation multiplier on unrelated findings. No VS Code fork does this. + +Trading the fallback branch for the name is the right trade: + +- On the primary branch nothing changes: the daemon still survives the update. +- On the fallback branch the daemon is killed with the app and terminals **cold-restore** on + relaunch. That is the documented pre-relocation behaviour, a first-class outcome the update + harness already asserts (`--expect cold-restore`), not a failure. +- Relocation is fail-open end to end anyway: any materialization failure returns `null` and the + caller forks the install-dir host. + +One new failure mode comes with it, on the fallback branch only. The daemon now matches +`FIND_PROCESS` under the app's image name, so it enters electron-builder's retry loop +(`allowOnlyOneInstallerInstance.nsh:136-141`). If the `taskkill` there fails to end it — an elevated +or otherwise unkillable host — the loop reaches `MessageBox ... /SD IDCANCEL` and `Quit`s, aborting a +silent update rather than completing it. Under the old distinct name the daemon was invisible to +that loop. Low probability (fallback branch _and_ an unkillable daemon), but it is a real new path. + +What this does **not** buy. Two things bound the win honestly: + +- The strongest T1036 indicator is a PE-resource-vs-disk-name mismatch, and it was **never firing**: + the shipped binary's `OriginalFilename` is empty (only `InternalName = Orca` is set), so there was + no embedded name for the old disk name to contradict. +- The remaining behaviour — a signed app copying its own ~225 MB image into user-writable + `%LOCALAPPDATA%` and running it detached under `ELECTRON_RUN_AS_NODE=1` — is still execution from + a non-standard user-writable location, which maps to **T1036.005** and is a standard heuristic on + its own. + +So this removes a real but partial signal. Expect the score to drop; do not expect the process to +stop being scored. + +## Options that were rejected + +| Option | Why not | +| ----------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Materialize the tree from the NSIS installer | The daemon host is ~246 MB. Writing it at install time doubles install footprint and lengthens the window in which the app is down during a silent update. Worse, on a per-machine install (`INSTALL_MODE_PER_ALL_USERS`) the installer runs as the installing admin, so `$LOCALAPPDATA` is the wrong user's — every other user still needs the runtime path, which means the runtime self-copy stays in the product and the signal is only made rarer. | +| Ship a second signed `orca-terminal-daemon.exe` in the installer | `Orca.exe` is 235,555,328 bytes (224.6 MiB). electron-builder's NSIS uses solid LZMA with a 64 MB dictionary, so a second copy 224 MB downstream does not dedupe; the compressed installer grows by roughly a whole compressed Electron binary, paid by every user on every update download. It also does not remove the runtime copy — the helper still has to reach `%LOCALAPPDATA%` to escape the sweep — so it buys the same signal reduction as the verbatim copy at a large download cost. | +| Override `customCheckAppRunning` to force a path-scoped kill on both branches | Cheap to write (~6 lines: `!include "getProcessInfo.nsh"`, `Var pid`, and a macro that pins `IsPowerShellAvailable`, reusing upstream's dialog, retry loop and elevated handling) — but wrong at any size. Forcing the PowerShell branch on a host where PowerShell is genuinely absent makes `FIND_PROCESS` and `KILL_PROCESS` silently no-op, so the installer proceeds with the **real app** still running and its files in use. That is a worse outcome than the cold restore it would prevent, so this is not worth doing ever, not merely not now. | +| Hardlink instead of copy | Avoids the 246 MB entirely and is not a "copy" at all, but is NTFS-and-same-volume-only and introduces fresh failure modes (link counts, AV interception, cross-volume installs). Worth revisiting deliberately, not as part of a signal fix. | + +## Invariants to preserve + +- The host exe name is **derived from `process.execPath`**, never a literal. A future + `executableName` or dev-channel rename must follow automatically; pinning a name of our own is + how the mismatch creeps back. +- The daemon is identified by **PID and command line**, never by image name — in the product + (`daemon-pid-file-parse`, `daemon-process-inspection`) and in the harness + (`tests/tools/win-update-e2e/daemon-processes.mjs`). Nothing may start matching on the exe name. +- `config/nsis/orca-installer-hooks.nsh` kills the daemon by image name. That now also matches the + app's own exe, which is correct on a genuine uninstall — the product is being removed — but its + `${isUpdated}` guard must stay: electron-builder runs the uninstaller during every update's + `uninstallOldVersion`, and killing the daemon there defeats the whole feature. The legacy + `orca-terminal-daemon.exe` name stays in the macro to reap hosts left by older builds. +- `LOCAL_HOST_ROOT_NAME` in `daemon-host-relocation.ts` and the path in the uninstall macro are the + same directory. Change both together. + +## Verifying a change + +Unit coverage lives in `src/main/daemon/daemon-host-relocation.test.ts` (copy plan, verbatim +naming, marker/atomic publish, fail-open, prune veto). Nothing in unit tests can prove survival, so +any change to this file or to the NSIS macro needs the packaged harnesses: + +- `.github/workflows/win-update-survival-e2e.yml` — builds an installer from the branch and updates + it over itself with `--expect survival`. The primary proof. +- `.github/workflows/win-crash-survival-e2e.yml` — proves the daemon survives a main-process crash. +- `.github/workflows/windows-terminal-restart-e2e.yml` — terminal restart behaviour. +- `.github/workflows/win-update-e2e.yml` — release-tag-to-release-tag update, both `survival` and + `cold-restore` profiles. + +All four are `workflow_dispatch`-only (the two update workflows also carry a push trigger pinned to +one historical feature branch), so they must be dispatched by hand against this branch before +merging a change here — which requires the workflow files to already exist on `main`. diff --git a/docs/reference/windows-edr-posture.md b/docs/reference/windows-edr-posture.md index 65287ac0459..68614932c14 100644 --- a/docs/reference/windows-edr-posture.md +++ b/docs/reference/windows-edr-posture.md @@ -50,25 +50,37 @@ and `orca-terminal-daemon.exe` report `Valid CN=SignPath Foundation`. ## The behaviours, and why each one exists -### The daemon runs from a renamed copy of our own image +### The daemon runs from a copy of our own image `src/main/daemon/daemon-host-relocation.ts` copies the Electron runtime into -`%LOCALAPPDATA%\Orca\daemon-host\\` and renames `Orca.exe` to -`orca-terminal-daemon.exe`. The comment on `DAEMON_HOST_EXE_NAME` states the -reason without varnish: _"so the NSIS updater's `taskkill /IM Orca.exe` can't -match it."_ +`%LOCALAPPDATA%\Orca\daemon-host\\` and forks the terminal daemon from +there. It exists because the NSIS installer deletes the old install directory and force- kills every process imaged under it. Without relocation, an auto-update kills the terminal daemon and every live terminal with it. The copy is a run-as-node `Orca.exe` rather than `node.exe` so there is no console flash and asar still -resolves; `config/nsis/daemon-host-uninstall.nsh` reaps it on a real uninstall +resolves; `config/nsis/orca-installer-hooks.nsh` reaps it on a real uninstall (guarded by `${isUpdated}` so an update's `uninstallOldVersion` never fires it). -**How an EDR reads it: MITRE T1036, masquerading.** A signed executable copied -out of the install directory into `%LOCALAPPDATA%` under a different name, which -then spawns shells, matches the textbook description closely enough that no -behavioural engine can be expected to score it low. +**At the time of these incidents the copy was also renamed** to +`orca-terminal-daemon.exe`, the image name every incident here reports, and +`DAEMON_HOST_EXE_NAME`'s comment stated the reason without varnish: _"so the NSIS +updater's `taskkill /IM Orca.exe` can't match it."_ The rename has since been +removed; the copy now keeps the app exe's own file name, because the updater's +kill sweep is path-scoped on every host that has PowerShell and the rename only +ever bought the no-PowerShell fallback. See +[`windows-daemon-host-relocation.md`](./windows-daemon-host-relocation.md). + +**How an EDR reads it: MITRE T1036, masquerading** — and, for what remains, +**T1036.005**. A signed executable copied out of the install directory into +`%LOCALAPPDATA%` under a different name, which then spawns shells, matches the +textbook description closely enough that no behavioural engine can be expected to +score it low. Dropping the rename removes that literal indicator but not the +underlying shape: execution from a non-standard user-writable location is scored +on its own. Note also that the strongest form of the T1036 signal was never +present here — the shipped binary's `OriginalFilename` is empty, so there was no +embedded name for the old disk name to contradict. ### Every process gets a handle, on a timer @@ -232,7 +244,8 @@ obfuscated-command-line detector is tuned on. ### The spawn tree itself -`Orca.exe` → `orca-terminal-daemon.exe` → a shell → an agent CLI is what a +`Orca.exe` → the relocated daemon host (`orca-terminal-daemon.exe` in the builds +these incidents cover, `Orca.exe` since) → a shell → an agent CLI is what a terminal multiplexer for coding agents *is*. `reg.exe` appears from `src/main/win32-utils.ts`, `src/main/agent-hooks/managed-hook-owner-identity.ts` and @@ -363,7 +376,7 @@ The checklist. On Windows, do not reach for: | Forking `powershell.exe` to read system state | The native reader — [`windows-process-enumeration.md`](./windows-process-enumeration.md) is the standing rule for the process table | | A process per operation in a loop | One long-lived helper with a request channel. A burst of short-lived interpreters under one parent is itself the signal | | `Add-Type -TypeDefinition` at runtime | A precompiled, signed assembly, or a native helper | -| Copying our own image under a different name | An installer or updater that does not need the rename. Where the rename is load-bearing, document it as such | +| Copying our own image under a different name | Copy it verbatim — [`windows-daemon-host-relocation.md`](./windows-daemon-host-relocation.md) (done for the daemon host) | | Deriving a script runner from a UI preference | [`windows-setup-shell.md`](./windows-setup-shell.md) — the script declares its own interpreter | Two framing rules that outlast the table: diff --git a/src/main/daemon/daemon-host-relocation.test.ts b/src/main/daemon/daemon-host-relocation.test.ts index 0d2323fd449..e899a67ed43 100644 --- a/src/main/daemon/daemon-host-relocation.test.ts +++ b/src/main/daemon/daemon-host-relocation.test.ts @@ -5,12 +5,13 @@ import { mkdtempSync, readFileSync, readdirSync, + renameSync, rmSync, utimesSync, writeFileSync } from 'node:fs' import os from 'node:os' -import { dirname, join } from 'node:path' +import { basename, dirname, join } from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { setAppEnvironment, type AppEnvironment } from '../../shared/app-environment' @@ -141,12 +142,11 @@ describe('buildDaemonHostManifest', () => { entryRelPath: 'resources/app.asar.unpacked/out/main/daemon-entry.js' }) const byDest = new Map(ops.map((op) => [op.destRel, op])) - // The host exe is renamed to a distinct image name (NOT the source basename) - // so the NSIS updater's name-based `taskkill /IM Orca.exe` can't kill it. - expect(byDest.get('orca-terminal-daemon.exe')?.kind).toBe('file') - expect(byDest.has('Orca.exe')).toBe(false) + // The host exe keeps the source basename: a verbatim, signature-preserving copy with no + // image-name mismatch. What escapes the updater's sweep is the path, not the name. + expect(byDest.get('Orca.exe')?.kind).toBe('file') const exeOp = ops.find((op) => op.sourcePath === 'C:\\app\\Orca.exe') - expect(exeOp?.destRel).not.toBe('Orca.exe') + expect(exeOp?.destRel).toBe('Orca.exe') // V8/ICU data blobs are read by the Electron bootstrap and kept. expect(byDest.has('icudtl.dat')).toBe(true) // GPU/graphics DLLs are never loaded by the windowless host, so not copied. @@ -170,7 +170,7 @@ describe('materializeRelocatedDaemonHost', () => { const result = materializeRelocatedDaemonHost() expect(result).not.toBeNull() const dest = join(localAppDataDir, 'Orca', 'daemon-host', '9.9.9') - expect(result?.execPath).toBe(join(dest, 'orca-terminal-daemon.exe')) + expect(result?.execPath).toBe(join(dest, 'Orca.exe')) expect(result?.entryPath).toBe( join(dest, 'resources', 'app.asar.unpacked', 'out', 'main', 'daemon-entry.js') ) @@ -203,6 +203,28 @@ describe('materializeRelocatedDaemonHost', () => { expect(marker.entryRelPath).toBe('resources/app.asar.unpacked/out/main/daemon-entry.js') }) + it('copies the exe verbatim: same file name and same bytes as the install-dir exe', () => { + const result = materializeRelocatedDaemonHost() + const sourceExe = join(installDir, 'Orca.exe') + // Byte-for-byte under the same name is what preserves the Authenticode signature and leaves + // no renamed-image signal for endpoint detection to read as masquerading. + expect(basename(result!.execPath)).toBe(basename(sourceExe)) + expect(readFileSync(result!.execPath)).toEqual(readFileSync(sourceExe)) + }) + + it('tracks a differently-named app exe rather than pinning an image name of its own', () => { + // A dev-channel or rebranded build ships a different executableName; the host copy must follow + // it, which is what keeps the copy verbatim instead of reintroducing a name mismatch. + renameSync(join(installDir, 'Orca.exe'), join(installDir, 'Orca Nightly.exe')) + setProcessProp('execPath', join(installDir, 'Orca Nightly.exe')) + const result = materializeRelocatedDaemonHost() + const dest = join(localAppDataDir, 'Orca', 'daemon-host', '9.9.9') + expect(result?.execPath).toBe(join(dest, 'Orca Nightly.exe')) + expect(existsSync(join(dest, 'orca-terminal-daemon.exe'))).toBe(false) + // Re-resolution must agree with materialization or the fork would target a missing exe. + expect(getRelocatedDaemonHost()?.execPath).toBe(join(dest, 'Orca Nightly.exe')) + }) + it('is idempotent: a valid marker short-circuits without recopying', () => { materializeRelocatedDaemonHost() const dest = join(localAppDataDir, 'Orca', 'daemon-host', '9.9.9') @@ -210,7 +232,7 @@ describe('materializeRelocatedDaemonHost', () => { const sentinel = join(dest, 'sentinel.txt') writeFileSync(sentinel, 'keep') const result = materializeRelocatedDaemonHost() - expect(result?.execPath).toBe(join(dest, 'orca-terminal-daemon.exe')) + expect(result?.execPath).toBe(join(dest, 'Orca.exe')) expect(existsSync(sentinel)).toBe(true) }) diff --git a/src/main/daemon/daemon-host-relocation.ts b/src/main/daemon/daemon-host-relocation.ts index 6d94bea06e4..13aab9fd8f9 100644 --- a/src/main/daemon/daemon-host-relocation.ts +++ b/src/main/daemon/daemon-host-relocation.ts @@ -22,6 +22,10 @@ import { inspectProcessLiveness, mergeProcessLivenessVerdict } from './daemon-pr * imaged under it, which would otherwise kill the daemon and its live terminals. The relocated exe is a * run-as-node Orca.exe copy (not node.exe) so there's no console flash and asar still resolves. Fail-open: * any failure returns null and the caller forks the install-dir host (pre-relocation behavior). + * + * What escapes the updater is the PATH, not the file name: electron-builder's kill sweep selects + * processes whose image path sits under $INSTDIR. See docs/reference/windows-daemon-host-relocation.md + * for the survival contract and why the exe is copied verbatim rather than renamed. */ export type RelocatedDaemonHost = { @@ -37,8 +41,14 @@ const MARKER_NAME = '.materialized.json' // LOCAL appData (not roaming) so OneDrive/roaming never syncs this ~260MB runtime. Shared with NSIS uninstall (config/nsis/orca-installer-hooks.nsh) — keep in sync. const LOCAL_HOST_ROOT_NAME = 'Orca' -// Copy of Orca.exe renamed to a distinct image name so the NSIS updater's `taskkill /IM Orca.exe` can't match it. -const DAEMON_HOST_EXE_NAME = 'orca-terminal-daemon.exe' +/** + * The host exe keeps the app exe's own file name, so the relocated image is a byte-for-byte, + * name-included copy of a signed binary — nothing for EDR to read as a renamed image (MITRE T1036). + * Survival comes from the path (see the module header). The one name-sensitive updater path is the + * no-PowerShell `taskkill /IM` fallback, where the daemon is killed and terminals cold-restore — + * the documented pre-relocation outcome, not a failure. + */ +const daemonHostExeName = (execPath: string): string => winPath.basename(execPath) // V8 snapshots + ICU data the Electron bootstrap reads even under ELECTRON_RUN_AS_NODE; siblings of Orca.exe. const RUNTIME_DATA_FILES = ['icudtl.dat', 'snapshot_blob.bin', 'v8_context_snapshot.bin'] @@ -146,8 +156,8 @@ export function buildDaemonHostManifest(sources: DaemonHostSources): CopyOp[] { const { appDir, execPath, resourcesPath, entrySourcePath, entryRelPath } = sources const ops: CopyOp[] = [] - // Host exe (renamed) + V8/ICU blobs at dest root. Top-level DLLs omitted: GPU/media libs a windowless run-as-node host never loads (~48MB saved). - ops.push({ sourcePath: execPath, destRel: DAEMON_HOST_EXE_NAME, kind: 'file' }) + // Host exe (verbatim name) + V8/ICU blobs at dest root. Top-level DLLs omitted: GPU/media libs a windowless run-as-node host never loads (~48MB saved). + ops.push({ sourcePath: execPath, destRel: daemonHostExeName(execPath), kind: 'file' }) for (const name of RUNTIME_DATA_FILES) { ops.push({ sourcePath: join(appDir, name), destRel: name, kind: 'file', optional: true }) } @@ -245,7 +255,7 @@ export function getRelocatedDaemonHost(): RelocatedDaemonHost | null { if (!marker || marker.version !== version) { return null } - const execPath = join(dest, DAEMON_HOST_EXE_NAME) + const execPath = join(dest, daemonHostExeName(sources.execPath)) const entryPath = destPath(dest, marker.entryRelPath) if (!existsSync(execPath) || !existsSync(entryPath)) { return null @@ -281,7 +291,9 @@ export function materializeRelocatedDaemonHost(): RelocatedDaemonHost | null { entryRelPath: sources.entryRelPath } writeFileSync(join(staging, MARKER_NAME), JSON.stringify(marker)) - // Replace any stale/partial dest, then publish the staging dir atomically. + // Replace any stale/partial dest, then publish atomically. Windows refuses to delete a running + // image, so a live daemon already hosted in THIS version's dir (same-version reinstall, or a dev + // channel reusing a version) throws here and materialization fails open to the install-dir host. rmSync(dest, { recursive: true, force: true }) renameSync(staging, dest) } catch { diff --git a/tests/tools/win-crash-survival-e2e/README.md b/tests/tools/win-crash-survival-e2e/README.md index beb56cf8c67..e0e773767c4 100644 --- a/tests/tools/win-crash-survival-e2e/README.md +++ b/tests/tools/win-crash-survival-e2e/README.md @@ -13,8 +13,8 @@ orphaned and PowerShell hard-crashed with a `0xE9` "No process is on the other end of the pipe" `FailFast`. Root cause: the terminal **daemon** (which hosts the ConPTYs) died together with the main process, severing the console pipe. -The fix re-architected the daemon into a standalone, relocated -`orca-terminal-daemon.exe` (see +The fix re-architected the daemon into a standalone daemon host relocated out of +the install dir (see [`src/main/daemon/daemon-host-relocation.ts`](../../src/main/daemon/daemon-host-relocation.ts)) that is spawned **detached** and **survives main-process death**. diff --git a/tests/tools/win-crash-survival-e2e/crash-step.mjs b/tests/tools/win-crash-survival-e2e/crash-step.mjs index 43dc18f04cd..3c7d80f51e1 100644 --- a/tests/tools/win-crash-survival-e2e/crash-step.mjs +++ b/tests/tools/win-crash-survival-e2e/crash-step.mjs @@ -4,7 +4,7 @@ // daemon (which hosts the ConPTYs) died with it, severing the console pipe, and // PowerShell hard-crashed with a 0xE9 "No process is on the other end of the // pipe" FailFast. The fix relocates the daemon into a standalone, detached -// orca-terminal-daemon.exe that SURVIVES main death (src/main/daemon/ +// host process outside the install dir that SURVIVES main death (src/main/daemon/ // daemon-host-relocation.ts). This module reproduces the crash and scans for the // pwsh FailFast that must no longer occur. diff --git a/tests/tools/win-crash-survival-e2e/run.mjs b/tests/tools/win-crash-survival-e2e/run.mjs index 4f9d8b242b0..68d8a7a3bd6 100644 --- a/tests/tools/win-crash-survival-e2e/run.mjs +++ b/tests/tools/win-crash-survival-e2e/run.mjs @@ -5,7 +5,7 @@ // process is on the other end of the pipe" FailFast, because the terminal daemon // (hosting the ConPTYs) died together with the main process and severed the // console pipe. The fix relocates the daemon into a standalone, detached -// orca-terminal-daemon.exe that survives main death (src/main/daemon/ +// host process outside the install dir that survives main death (src/main/daemon/ // daemon-host-relocation.ts). win-update-e2e proves the daemon survives a // Windows UPDATE; this harness proves it survives a CRASH of the main process. // From 687a22e1eee40af4ac25c4357cc7f7e8201c8bbb Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Sat, 5 Sep 2026 21:12:33 -0700 Subject: [PATCH 39/69] fix(computer-use): run the Windows runtime as one persistent helper (#17858) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(computer-use): run the Windows runtime as one persistent helper Microsoft Defender for Endpoint raised multi-stage Execution + Collection incidents against Orca on Windows ("Screenshots were taken unexpectedly on this device... Screen capture code was found in a script launched by powershell.exe", factor "Executes suspicious MSIL code"). The desktop script provider spawned a fresh powershell.exe per operation, so a single computer-use session produced a burst of short-lived PIDs and re-emitted runtime.ps1's inline Add-Type P/Invoke assembly on every click. runtime.ps1 gains a -Serve mode that loads its assemblies once and then reads NDJSON requests from stdin, and a new DesktopScriptRuntimeHost owns one long-lived child: lazy spawn, strict serialization, a 30s per-request timeout, restart on crash, a 120s idle shutdown, and dispose() on provider teardown. The one-shot -OperationPath path stays as the fallback, and Linux keeps its python3 bridge unchanged. Both Windows spawn sites now use -ExecutionPolicy RemoteSigned instead of Bypass, falling back once to Bypass (and logging) when a Restricted host refuses the unsigned script. * fix(computer-use): recover the runtime host instead of latching it off Review follow-up on the persistent Windows computer-use helper. A helper that died before producing a line set an unavailable flag nothing ever cleared, and the client then dropped the host for the life of the session. One transient bad spawn — a Defender scan, a locked CSC temp directory — silently restored the per-click powershell.exe burst and per-operation MSIL emission this work exists to remove, with computer use still working so nothing looked wrong. Start failures are now retried, then cool down for 60s, then re-probed; the client keeps the host so it can come back. Repeated post-answer crashes cool down too, and a single reply no longer clears the failure count. The one-shot bridge decided its execution-policy retry from a message that fell back to stdout, so a window title containing "SecurityError" could replay a non-idempotent operation — a double click, keystroke or paste — and stick the session on Bypass. The retry now requires empty stdout and a matching stderr. Serve-mode replies carry an echoed request id. Without one a single stray stdout line would make every later response answer the previous request, acting on stale element indexes with no error raised; a mismatch now kills the child. Non-JSON noise is ignored rather than counted as the helper having answered. Also: warnings reach the main process over the sidecar's IPC channel rather than its piped, unread stdio; the child is watched on close rather than exit; dispose latches so a queued request cannot respawn during teardown; and the host is split into a serve channel and an availability policy to stay under max-lines. * fix(computer-use): prove a helper never started before replaying its request The retry that replaced the permanent-latch bug could deliver unrequested input. send() re-sent the same request whenever the helper died without replying, but "no reply came back" is not "the operation did not run": runtime.ps1 synthesizes the click and only then builds the snapshot, which allocates a full-window bitmap and walks the UIA tree — a native GDI+/UIA fault there is uncatchable, and leaves the click already delivered. A deterministic fault meant three clicks from the host plus a fourth from the one-shot bridge, surfaced as a single failed operation. -Serve now writes one {"ready":true} line after its Add-Type work and before its first read, so "never started" is a fact rather than an inference. A request is replayed only when the helper died before announcing. A runtime.ps1 that predates the announcement — reachable through the provider path override — is covered by an observation-tool allowlist until a ready line proves otherwise. Host-detected aborts (timeout, desynchronised reply, oversized line) suppress the exit handler, so they were bypassing failure accounting entirely and a helper failing that way was respawned once per operation forever. They now count and are logged. Also stop charging twice for one outage: entering the cooldown resets the failure count, so the first death after recovery no longer re-enters a full cooldown and an interleaved workload cannot be stranded on the one-shot bridge. * fix(computer-use): ignore a stdin write callback from a torn-down helper stop() destroys stdin, so a write still queued at teardown calls back with ERR_STREAM_DESTROYED. The callback carried no channel or request identity and write() had no closed guard, so it ran abortChannel a second time: stopChannel no-opped but recordFailure and the warning did not, charging two failures for one operation and reaching the 3-strike cooldown at half the intended rate. That feeds the same accounting that keeps a persistently broken helper from respawning once per operation. The same root also allowed a late callback landing after a replacement channel existed to stop that channel and reject a different request with the previous one's error. Node fires the destroyed-stream callback on the next tick, well before a new request arrives, so the double-count is the reachable effect; binding the callback closes both. write() now drops payloads and error reports once closed, and the host ignores any report whose channel or request id is no longer current. * test(computer-use): pin each stale-write guard independently The channel's closed guard and the host's request-identity check are redundant by design, and the existing tests only failed when both were absent. Someone deleting one, believing the other was the covered one, would have got a green suite and a live regression — the same shape as a test that passes without the fix it was written for. Each is now pinned on its own. The channel's half is tested against the channel directly: after stop() it takes no writes and reports no error from one already queued, which the host cannot observe because it drops the channel at the same moment. The host's half is pinned by the case the channel cannot see — a live channel whose request was already answered, where backpressure delivers a write callback for a request that is no longer pending. Removing either guard alone now fails a test. Both carry a comment saying they are deliberately redundant and separately pinned, so the next reader does not have to rediscover this from the diff. * ci(windows): run the computer-use runtime host suite in CI The win32 suite only self-skips off Windows, so it passed vacuously in every lane. Register it the way the cmd-shim suite is registered. * fix(computer-use): time the runtime host cooldown on a monotonic clock The start-failure cooldown was a wall-clock deadline, so a backwards step — an NTP correction, a VM snapshot restore, a user changing the clock — left `remainingCooldown()` returning the cooldown plus the whole step. A one-hour step measured 3,660,000ms, and ten real minutes later still 3,060,000ms. Nothing shortens it from there. Only `recordSuccess()` clears the cooldown on a non-dispose path, and no request can reach a helper to succeed while it holds, so every `send()` throws `runtime_host_unavailable` first. The host is built with no `now` override and its lifecycle is a module-level singleton that shuts down at process exit, so the latch held for the sidecar's life — computer use kept working via the one-shot bridge while the per-click powershell.exe burst this host exists to remove came back silently. Store the instant the cooldown began and compare elapsed monotonic time, following the two fixes in #17884. The field is `number | null` rather than sentinel 0 because `performance.now()` legitimately returns 0. Both new tests leave `now` unset, because the bug was in the default the host picks and a test that injects a clock cannot see it. * fix(computer-use): give a queued request its own deadline The 30s request timeout was armed only in `sendOnce`, once a request reached a helper. A request behind N timing-out ones therefore waited roughly N times that with no deadline of its own: bounded, but the caller sees an `await` that looks hung for minutes and gets no error to act on. Move the serialization tail into its own class and arm a deadline at enqueue time. Only the wait is bounded — a request that reaches a helper still gets its full execution budget, so nothing that used to succeed now fails. An expired request is dropped rather than sent late: the caller has already been told it failed, and a click delivered after that is worse than no click. The tail keeps its never-rejecting shape and chains on the turn rather than on the raced promise, so a caller giving up early cannot release the next request while its predecessor is still in flight. * fix(computer-use): stop reading a locked file as an execution policy block `UnauthorizedAccess` is the FullyQualifiedErrorId PowerShell reports for a policy block, and it is also a strict prefix of `UnauthorizedAccessException`, which .NET raises for any ordinary locked or ACL-denied file. The predicate matched the token unanchored, so an AV scan holding runtime.ps1 or a locked CSC temp directory was read as a policy block. Two consequences, both bad. `escalateExecutionPolicy()` has no path back, so one false match spent the rest of the session on `-ExecutionPolicy Bypass` — the exact command line token this stack exists to stop emitting. And on the one-shot path `isPolicyBlockedStart` re-runs the operation: one-shot mode writes stdout only after the operation returns, so a crash partway through an action is indistinguishable from a helper that never started, and the click lands twice. Measured on Windows against all three records, which the test carries verbatim as fixtures: policy/Restricted FullyQualifiedErrorId: UnauthorizedAccess policy/RemoteSigned FullyQualifiedErrorId: UnauthorizedAccess genuine access denied FullyQualifiedErrorId: UnauthorizedAccessException `\b` is the whole discriminator: between `s` and `E` both sides are word characters, so no boundary exists there and the exception cannot match. Dropped two alternatives that measurement showed were wrong. `PSSecurityException` never appears — the record surfaces through a native-command wrapper and reports `ParentContainsErrorRecordException`. The prose is wrong three times over: it differs by policy, it is localized, and PowerShell hard-wraps it mid-sentence. Anchoring on the `FullyQualifiedErrorId:`/`CategoryInfo:` labels would be more precise again, but those labels are localized where the values are not, so it would lose a real block on a non-English host and strand it with no fallback. Matching the values with word boundaries keeps both directions; a fixture with translated labels pins it. The escalation stays sticky. With the predicate correct, it only fires on a machine that really does block, where re-probing the preferred policy would buy a guaranteed failed spawn per operation. * fix(computer-use): route a malformed request back to the request that caused it `ConvertFrom-Json` throws before `$requestId` is read, so the serve loop answered an unparseable request with an untagged error. On the client that is not an error at all: `deliver()` sees no matching id, calls `abortChannel`, kills the helper and charges a failure — and the helper's own message is discarded. A parse failure was reported as a stream desync with no trace of the real cause, and three of them walked into the 60s cooldown behind three misleading "did not match" messages. Recover the id from the raw line when the parse fails. No wire change: the response shape is untouched and `BridgeResponse.requestId` already documents this echo. It is the same shape the helper already returns for `not_a_tool`, where the id survives because it is read before the operation runs. Both mixed pairings degrade safely — a new script with an old client resolves the error normally, and an old script with a new client still aborts, but now reports what the helper said. When the line is mangled past recovering an id, the desync abort is the honest outcome, so keep it and carry the helper's text into it rather than replacing it. A line the helper could not tag is usually the only account of the cause. Proven against the real `runtime.ps1 -Serve`: the host can only write well-formed JSON, so the parse-failure branch is unreachable through it and the test drives the channel directly. * fix(computer-use): keep the Bypass escalation only when Bypass actually works AppLocker and WDAC constrained language mode raise PSSecurityException under the same SecurityError category a real execution-policy block uses, so the predicate matches them - correctly, on the evidence available. But those block the script at parse time, which `-ExecutionPolicy Bypass` cannot lift. The escalation was sticky unconditionally, so on a WDAC host we misdiagnosed, retried, failed again, and then latched: every later command line carried the most heavily weighted MDE token there is, on exactly the hardened, monitored enterprise machine that is watching for it. Treat the escalation as the diagnosis it is. A fallback that cannot start a helper either disproves it - the policy was not what stopped the first attempt - so revert to RemoteSigned instead of latching. When Bypass does start a helper the diagnosis is confirmed and it stays sticky exactly as before, so a genuinely Restricted machine still never pays a re-probe per operation. The revert lands inside the outage rather than only at its end, so a misdiagnosis costs one Bypass command line instead of one per attempt, and an escalation that never proved itself does not outlive the cooldown that ends the outage. Deliberately not a permanent "fallback is useless" flag: a Bypass attempt that failed for a transient reason would then disable the fallback for the session, which is the same latch in the other direction. Only `runtime_host_unavailable` proves no helper started, so only that reverts; a helper that started and then died proves Bypass works. That also makes the policy branch reachable on a final attempt for the first time, so it now rejects as unavailable rather than a generic error - that code is what routes the operation to the one-shot bridge, which carries its own policy fallback, and without it an all-blocked host would fail operations outright instead of degrading. The pre-existing "reports itself unavailable when Bypass is also refused" test pins that. --------- Co-authored-by: Orca Worker --- .github/workflows/pr.yml | 1 + config/scripts/pr-code-change-scope.mjs | 1 + native/computer-use-windows/runtime.ps1 | 67 +- .../computer-sidecar-diagnostics.test.ts | 45 + .../computer/computer-sidecar-diagnostics.ts | 38 + src/main/computer/desktop-script-action.ts | 14 + .../desktop-script-provider-bridge.ts | 109 ++- .../desktop-script-provider-client.ts | 45 +- ...ript-provider-runtime-host-routing.test.ts | 139 +++ .../desktop-script-provider-test-harness.ts | 20 +- .../computer/desktop-script-provider-types.ts | 4 + .../computer/desktop-script-request-queue.ts | 73 ++ .../desktop-script-runtime-availability.ts | 176 ++++ .../desktop-script-runtime-host.test.ts | 857 ++++++++++++++++++ .../computer/desktop-script-runtime-host.ts | 384 ++++++++ .../desktop-script-runtime-host.win32.test.ts | 130 +++ .../desktop-script-serve-channel.test.ts | 99 ++ .../computer/desktop-script-serve-channel.ts | 145 +++ src/main/computer/sidecar-client.ts | 6 + src/main/computer/sidecar-entry.ts | 5 + ...indows-powershell-execution-policy.test.ts | 92 ++ .../windows-powershell-execution-policy.ts | 59 ++ 22 files changed, 2475 insertions(+), 34 deletions(-) create mode 100644 src/main/computer/computer-sidecar-diagnostics.test.ts create mode 100644 src/main/computer/computer-sidecar-diagnostics.ts create mode 100644 src/main/computer/desktop-script-provider-runtime-host-routing.test.ts create mode 100644 src/main/computer/desktop-script-request-queue.ts create mode 100644 src/main/computer/desktop-script-runtime-availability.ts create mode 100644 src/main/computer/desktop-script-runtime-host.test.ts create mode 100644 src/main/computer/desktop-script-runtime-host.ts create mode 100644 src/main/computer/desktop-script-runtime-host.win32.test.ts create mode 100644 src/main/computer/desktop-script-serve-channel.test.ts create mode 100644 src/main/computer/desktop-script-serve-channel.ts create mode 100644 src/main/computer/windows-powershell-execution-policy.test.ts create mode 100644 src/main/computer/windows-powershell-execution-policy.ts diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index 0e2fa3f273c..0cd7960e0c7 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -856,6 +856,7 @@ jobs: src/main/wsl/wsl-w1-w3-contract.test.ts src/shared/source-scan/source-tree-scan.test.ts src/main/cli/wsl-cli-powershell-boundary.test.ts + src/main/computer/desktop-script-runtime-host.win32.test.ts src/main/cursor/hook-service.test.ts src/main/orca-profiles/profile-index-store.test.ts src/main/startup/windows-install-dir-acl-repair.win32.test.ts diff --git a/config/scripts/pr-code-change-scope.mjs b/config/scripts/pr-code-change-scope.mjs index fd36a803bb9..f5916d6a79c 100644 --- a/config/scripts/pr-code-change-scope.mjs +++ b/config/scripts/pr-code-change-scope.mjs @@ -228,6 +228,7 @@ const WINDOWS_PACKAGE_TESTS = [ 'src/main/wsl/wsl-w1-w3-contract.test.ts', 'src/shared/source-scan/source-tree-scan.test.ts', 'src/main/cli/wsl-cli-powershell-boundary.test.ts', + 'src/main/computer/desktop-script-runtime-host.win32.test.ts', 'src/main/cursor/hook-service.test.ts', 'src/main/orca-profiles/profile-index-store.test.ts', 'src/main/startup/windows-install-dir-acl-repair.win32.test.ts', diff --git a/native/computer-use-windows/runtime.ps1 b/native/computer-use-windows/runtime.ps1 index 4b68525c7c6..efd44e6cde7 100644 --- a/native/computer-use-windows/runtime.ps1 +++ b/native/computer-use-windows/runtime.ps1 @@ -1,9 +1,15 @@ param( - [Parameter(Mandatory = $true)] - [string]$OperationPath + [Parameter(Position = 0)] + [string]$OperationPath, + # Serve mode keeps one process alive so the Add-Type P/Invoke assembly below + # is emitted once per session instead of once per operation. + [switch]$Serve ) $ErrorActionPreference = "Stop" +# Progress records render to the host, which in serve mode is a pipe carrying +# one JSON response per line; a stray record would desynchronise the stream. +$ProgressPreference = "SilentlyContinue" $utf8NoBom = New-Object System.Text.UTF8Encoding $false [Console]::InputEncoding = $utf8NoBom [Console]::OutputEncoding = $utf8NoBom @@ -1313,9 +1319,56 @@ function Invoke-OrcaOperation($Operation) { [pscustomobject]@{ ok = $true; action = $action; snapshot = $snapshot } } -try { - $operation = Read-OrcaOperation $OperationPath - Write-OrcaJson (Invoke-OrcaOperation $operation) -} catch { - Write-OrcaJson ([pscustomobject]@{ ok = $false; error = [string]$_.Exception.Message }) +function Invoke-OrcaServeLoop { + # Announced before the first read, and after every Add-Type above: a caller + # that never sees this line knows the helper cannot have read a request, let + # alone synthesized a click, so replaying it is provably safe. Inferring that + # from a missing response instead would replay operations that did run. + [Console]::Out.WriteLine('{"ready":true}') + [Console]::Out.Flush() + # One NDJSON request per line in, one response per line out, until stdin closes. + # Responses carry base64 screenshots and routinely exceed a megabyte; ReadLine + # and the console writer are both length-bounded only by memory. + while ($true) { + $line = [Console]::In.ReadLine() + if ($null -eq $line) { break } + if ([string]::IsNullOrWhiteSpace($line)) { continue } + $requestId = $null + try { + $operation = $line | ConvertFrom-Json + $requestId = $operation.requestId + $response = Invoke-OrcaOperation $operation + } catch { + $response = [pscustomobject]@{ ok = $false; error = [string]$_.Exception.Message } + # ConvertFrom-Json throws before the id is read, so recover it from the + # raw line. An error the caller can match is delivered to the request + # that caused it; an unmatched one only trips the caller's desync + # guard, which kills this helper, charges a failure toward its cooldown + # and discards the message below - so a malformed request would be + # reported as a broken stream and its real cause never surface. + if ($null -eq $requestId -and $line -match '"requestId"\s*:\s*(\d+)') { + $requestId = [long]$Matches[1] + } + } + # Echoed so the caller can prove which request a line answers; a reply it + # cannot match is a desynchronised stream, not a usable response. + if ($null -ne $requestId) { + $response | Add-Member -NotePropertyName requestId -NotePropertyValue $requestId -Force + } + [Console]::Out.WriteLine((ConvertTo-Json $response -Depth 100 -Compress)) + [Console]::Out.Flush() + } +} + +if ($Serve) { + Invoke-OrcaServeLoop +} elseif ([string]::IsNullOrWhiteSpace($OperationPath)) { + Write-OrcaJson ([pscustomobject]@{ ok = $false; error = "runtime.ps1 requires an operation path or -Serve" }) +} else { + try { + $operation = Read-OrcaOperation $OperationPath + Write-OrcaJson (Invoke-OrcaOperation $operation) + } catch { + Write-OrcaJson ([pscustomobject]@{ ok = $false; error = [string]$_.Exception.Message }) + } } diff --git a/src/main/computer/computer-sidecar-diagnostics.test.ts b/src/main/computer/computer-sidecar-diagnostics.test.ts new file mode 100644 index 00000000000..06d1b1cfa02 --- /dev/null +++ b/src/main/computer/computer-sidecar-diagnostics.test.ts @@ -0,0 +1,45 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + isComputerSidecarDiagnostic, + reportComputerDiagnostic +} from './computer-sidecar-diagnostics' + +describe('computer sidecar diagnostics', () => { + const originalSend = process.send + + afterEach(() => { + process.send = originalSend + vi.restoreAllMocks() + }) + + it('sends over IPC when running inside the sidecar', () => { + const send = vi.fn((_message: unknown) => true) + process.send = send as unknown as typeof process.send + const console_ = vi.spyOn(console, 'warn').mockImplementation(() => {}) + + reportComputerDiagnostic('fell back to Bypass') + + // The sidecar's stdout is piped and never read, so this must not go there. + expect(console_).not.toHaveBeenCalled() + expect(send).toHaveBeenCalledWith({ + kind: 'computer-sidecar-diagnostic', + message: 'fell back to Bypass' + }) + expect(isComputerSidecarDiagnostic(send.mock.calls[0][0])).toBe(true) + }) + + it('logs directly when there is no IPC channel', () => { + process.send = undefined + const console_ = vi.spyOn(console, 'warn').mockImplementation(() => {}) + + reportComputerDiagnostic('fell back to Bypass') + + expect(console_).toHaveBeenCalledWith('[computer-use] fell back to Bypass') + }) + + it('does not mistake a sidecar response for a diagnostic', () => { + expect(isComputerSidecarDiagnostic({ id: 1, ok: true, result: {} })).toBe(false) + expect(isComputerSidecarDiagnostic({ kind: 'computer-sidecar-diagnostic' })).toBe(false) + expect(isComputerSidecarDiagnostic(null)).toBe(false) + }) +}) diff --git a/src/main/computer/computer-sidecar-diagnostics.ts b/src/main/computer/computer-sidecar-diagnostics.ts new file mode 100644 index 00000000000..2b23418940d --- /dev/null +++ b/src/main/computer/computer-sidecar-diagnostics.ts @@ -0,0 +1,38 @@ +/** + * Warnings from the computer-use provider, routed to somewhere a human sees. + * + * Why not `console.warn`: the provider runs inside the forked sidecar, which + * `sidecar-client.ts` starts with piped stdio that nothing ever reads. Anything + * written there is discarded — including the only signal that a machine has + * fallen back to `-ExecutionPolicy Bypass`, a state that persists for the + * session. The sidecar has an IPC channel already, so the warning takes it. + */ +export type ComputerSidecarDiagnostic = { + kind: 'computer-sidecar-diagnostic' + message: string +} + +const DIAGNOSTIC_KIND = 'computer-sidecar-diagnostic' + +export function isComputerSidecarDiagnostic( + message: unknown +): message is ComputerSidecarDiagnostic { + if (!message || typeof message !== 'object') { + return false + } + const record = message as Record + return record.kind === DIAGNOSTIC_KIND && typeof record.message === 'string' +} + +export function reportComputerDiagnostic(message: string): void { + if (process.send) { + process.send({ kind: DIAGNOSTIC_KIND, message } satisfies ComputerSidecarDiagnostic) + return + } + logComputerDiagnostic(message) +} + +/** The main-process end: how a sidecar's forwarded diagnostic is printed. */ +export function logComputerDiagnostic(message: string): void { + console.warn(`[computer-use] ${message}`) +} diff --git a/src/main/computer/desktop-script-action.ts b/src/main/computer/desktop-script-action.ts index 38dde4b56fa..7c2c21f1e8c 100644 --- a/src/main/computer/desktop-script-action.ts +++ b/src/main/computer/desktop-script-action.ts @@ -228,3 +228,17 @@ export function elementParam( } return element } + +/** + * Tools that only observe, and so may be safely re-sent to a fresh helper. + * + * Why an allowlist: a helper can die after running an operation but before + * writing its reply, so a replayed mutation is a second click, keystroke or + * paste. Only the observation tools are provably safe to repeat, and a tool + * added later has to opt in rather than inherit a replay by default. + */ +const OBSERVATION_TOOLS = new Set(['handshake', 'list_apps', 'list_windows', 'get_app_state']) + +export function isReplayableTool(tool: string): boolean { + return OBSERVATION_TOOLS.has(tool) +} diff --git a/src/main/computer/desktop-script-provider-bridge.ts b/src/main/computer/desktop-script-provider-bridge.ts index c3c21496d29..fecc4036df1 100644 --- a/src/main/computer/desktop-script-provider-bridge.ts +++ b/src/main/computer/desktop-script-provider-bridge.ts @@ -1,28 +1,98 @@ import { execFile } from 'node:child_process' +import { windowsPowerShellPath } from '../../shared/child-process/windows-system-binary' +import { reportComputerDiagnostic } from './computer-sidecar-diagnostics' import { RuntimeClientError } from './runtime-client-error' import type { DesktopScriptPlatform } from './desktop-script-provider-paths' +import { + FALLBACK_WINDOWS_EXECUTION_POLICY, + PREFERRED_WINDOWS_EXECUTION_POLICY, + isExecutionPolicyBlocked, + windowsPowerShellRuntimeArgs +} from './windows-powershell-execution-policy' const REQUEST_TIMEOUT_MS = 30_000 const FORCE_KILL_GRACE_MS = 1_000 -export function execBridge( +export async function execBridge( platform: DesktopScriptPlatform, scriptPath: string, operationPath: string ): Promise<{ stdout: string; stderr: string }> { - const command = platform === 'windows' ? 'powershell.exe' : 'python3' - const args = - platform === 'windows' - ? [ - '-NoProfile', - '-NonInteractive', - '-ExecutionPolicy', - 'Bypass', - '-File', - scriptPath, - operationPath - ] - : [scriptPath, operationPath] + if (platform !== 'windows') { + return await mapped(runBridgeProcess('python3', [scriptPath, operationPath])) + } + const command = windowsPowerShellPath() + try { + return await runBridgeProcess( + command, + windowsPowerShellRuntimeArgs(scriptPath, PREFERRED_WINDOWS_EXECUTION_POLICY, [operationPath]) + ) + } catch (error) { + if (!isPolicyBlockedStart(error)) { + throw error instanceof BridgeProcessFailure ? error.mapped : error + } + reportComputerDiagnostic( + `bridge start blocked at ${PREFERRED_WINDOWS_EXECUTION_POLICY}; retrying once with ${FALLBACK_WINDOWS_EXECUTION_POLICY}` + ) + return await mapped( + runBridgeProcess( + command, + windowsPowerShellRuntimeArgs(scriptPath, FALLBACK_WINDOWS_EXECUTION_POLICY, [operationPath]) + ) + ) + } +} + +/** Unwrap the raw-stream carrier back into the error callers expect. */ +async function mapped( + run: Promise<{ stdout: string; stderr: string }> +): Promise<{ stdout: string; stderr: string }> { + try { + return await run + } catch (error) { + throw error instanceof BridgeProcessFailure ? error.mapped : error + } +} + +/** + * Only a run that produced no stdout at all may be replayed. + * + * What the stdout guard covers: operations are not idempotent, and the response + * embeds window titles and element names, so a snapshot that merely contains + * the word "SecurityError" must not be read as a policy block and replayed as a + * second click, keystroke or paste. It closes that injection route only. + * + * What it does not cover: one-shot mode runs the operation to completion and + * writes stdout only afterwards, so stdout is empty for the whole action, not + * just before it starts. A crash after the click but before the write looks + * identical to a helper that never started. Nothing here can tell those apart — + * only a policy pattern that cannot match a non-policy failure keeps the replay + * off, which is why its `\b` is load-bearing rather than cosmetic. + */ +function isPolicyBlockedStart(error: unknown): error is BridgeProcessFailure { + return ( + error instanceof BridgeProcessFailure && + !error.stdout.trim() && + isExecutionPolicyBlocked(error.stderr) + ) +} + +/** Carries the raw streams so the retry decision does not read a mapped message. */ +class BridgeProcessFailure extends Error { + constructor( + readonly stdout: string, + readonly stderr: string, + readonly mapped: RuntimeClientError + ) { + super(mapped.message) + this.name = 'BridgeProcessFailure' + } +} + +function runBridgeProcess( + command: string, + args: readonly string[] +): Promise<{ stdout: string; stderr: string }> { return new Promise((resolve, reject) => { let child: ReturnType | null = null let settled = false @@ -76,7 +146,7 @@ export function execBridge( try { child = execFile( command, - args, + [...args], { env: process.env, maxBuffer: 20 * 1024 * 1024, @@ -86,11 +156,10 @@ export function execBridge( (error, stdout, stderr) => { if (error) { const message = stderr.trim() || stdout.trim() || error.message - finish( - error.killed - ? new RuntimeClientError('action_timeout', message) - : mapBridgeError(message) - ) + const mapped = error.killed + ? new RuntimeClientError('action_timeout', message) + : mapBridgeError(message) + finish(new BridgeProcessFailure(stdout, stderr, mapped)) return } finish(null, { stdout, stderr }) diff --git a/src/main/computer/desktop-script-provider-client.ts b/src/main/computer/desktop-script-provider-client.ts index 5e8e01e0d44..a655eeddb07 100644 --- a/src/main/computer/desktop-script-provider-client.ts +++ b/src/main/computer/desktop-script-provider-client.ts @@ -35,6 +35,7 @@ import type { BridgeResponse, NativeActionMethod } from './desktop-script-provider-types' +import { DesktopScriptRuntimeHost, isRuntimeHostUnavailable } from './desktop-script-runtime-host' import { DesktopScriptSnapshotStore } from './desktop-script-snapshot-store' import { normalizeBridgeApp, renderSnapshot } from './desktop-script-snapshot-rendering' import { normalizeComputerActionResult } from './computer-action-verification-normalization' @@ -51,12 +52,17 @@ export class DesktopScriptProviderClient { constructor( private readonly platform: DesktopScriptPlatform = requiredPlatform(), - private readonly scriptPath: string = requiredScriptPath() + private readonly scriptPath: string = requiredScriptPath(), + private readonly runtimeHost: DesktopScriptRuntimeHost | null = defaultRuntimeHost( + platform, + scriptPath + ) ) {} shutdown(): void { this.snapshotStore.clear() this.providerCapabilities = null + this.runtimeHost?.dispose() } async listApps(): Promise { @@ -203,6 +209,23 @@ export class DesktopScriptProviderClient { } private async callBridge(request: BridgeRequest): Promise { + const host = this.runtimeHost + if (host) { + try { + return checkedBridgeResponse(await host.request(request), '') + } catch (error) { + // Only a helper that cannot start falls back; operation errors surface. + // The host is kept: it re-probes after its cooldown, so a transient bad + // spawn cannot strand the session on one powershell.exe per operation. + if (!isRuntimeHostUnavailable(error)) { + throw error + } + } + } + return await this.callOneShotBridge(request) + } + + private async callOneShotBridge(request: BridgeRequest): Promise { const operationDirectory = await mkdtemp(join(tmpdir(), 'orca-computer-use-')) const operationPath = join(operationDirectory, 'operation.json') try { @@ -217,10 +240,7 @@ export class DesktopScriptProviderClient { `desktop provider returned invalid JSON: ${error instanceof Error ? error.message : String(error)}` ) } - if (!response.ok) { - throw mapBridgeError(response.error ?? stderr) - } - return response + return checkedBridgeResponse(response, stderr) } finally { await rm(operationDirectory, { force: true, recursive: true }) } @@ -255,6 +275,21 @@ export class DesktopScriptProviderClient { } } +function checkedBridgeResponse(response: BridgeResponse, stderr: string): BridgeResponse { + if (!response.ok) { + throw mapBridgeError(response.error ?? stderr) + } + return response +} + +// Why Windows only: the Linux provider is a python3 one-shot with no serve mode. +function defaultRuntimeHost( + platform: DesktopScriptPlatform, + scriptPath: string +): DesktopScriptRuntimeHost | null { + return platform === 'windows' ? new DesktopScriptRuntimeHost(scriptPath) : null +} + function requiredPlatform(): DesktopScriptPlatform { const platform = desktopScriptPlatform() if (!platform) { diff --git a/src/main/computer/desktop-script-provider-runtime-host-routing.test.ts b/src/main/computer/desktop-script-provider-runtime-host-routing.test.ts new file mode 100644 index 00000000000..e50a3878a11 --- /dev/null +++ b/src/main/computer/desktop-script-provider-runtime-host-routing.test.ts @@ -0,0 +1,139 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + bridgeProcessArgs, + createDesktopScriptProviderClient, + expectDesktopProviderSubprocessStartCount, + mockBridgeProcessFailure, + mockBridgeResponse, + resetDesktopScriptProviderTestHarness, + sampleCapabilities +} from './desktop-script-provider-test-harness' +import type { BridgeResponse } from './desktop-script-provider-types' +import type { DesktopScriptRuntimeHost } from './desktop-script-runtime-host' +import { RuntimeClientError } from './runtime-client-error' + +const POLICY_STDERR = + 'File runtime.ps1 cannot be loaded because running scripts is disabled on this system. + CategoryInfo : SecurityError' + +function fakeRuntimeHost(request: DesktopScriptRuntimeHost['request']) { + const dispose = vi.fn() + return { host: { request, dispose } as unknown as DesktopScriptRuntimeHost, dispose } +} + +describe('desktop script provider runtime host routing', () => { + afterEach(resetDesktopScriptProviderTestHarness) + + it('serves Windows operations from the runtime host without spawning a one-shot bridge', async () => { + const request = vi.fn( + async () => ({ ok: true, capabilities: sampleCapabilities() }) as BridgeResponse + ) + const { host } = fakeRuntimeHost(request) + + const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1', host) + + await expect(client.capabilities()).resolves.toMatchObject({ platform: 'linux' }) + expect(request).toHaveBeenCalledWith({ tool: 'handshake' }) + expectDesktopProviderSubprocessStartCount(0) + }) + + it('maps runtime host operation failures without falling back to the one-shot bridge', async () => { + const { host } = fakeRuntimeHost( + vi.fn(async () => ({ ok: false, error: 'appBlocked("1Password")' }) as BridgeResponse) + ) + + const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1', host) + + await expect(client.listApps()).rejects.toMatchObject({ code: 'app_blocked' }) + expectDesktopProviderSubprocessStartCount(0) + }) + + it('degrades to the one-shot bridge for the operations a host cannot serve', async () => { + const request = vi.fn(async () => { + throw new RuntimeClientError('runtime_host_unavailable', 'could not start') + }) + const { host, dispose } = fakeRuntimeHost(request as never) + mockBridgeResponse({ ok: true, apps: [{ name: 'Notepad', pid: 42 }] }) + mockBridgeResponse({ ok: true, apps: [{ name: 'Notepad', pid: 42 }] }) + + const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1', host) + + await expect(client.listApps()).resolves.toMatchObject({ apps: [{ pid: 42 }] }) + await client.listApps() + + // The host is kept and asked again: it owns its own cooldown, so one bad + // spawn must not stand the session down to a powershell.exe per click. + expect(request).toHaveBeenCalledTimes(2) + expect(dispose).not.toHaveBeenCalled() + expectDesktopProviderSubprocessStartCount(2) + }) + + it('returns to the runtime host once it recovers', async () => { + let healthy = false + const request = vi.fn(async () => { + if (!healthy) { + throw new RuntimeClientError('runtime_host_unavailable', 'could not start') + } + return { ok: true, apps: [] } as BridgeResponse + }) + const { host } = fakeRuntimeHost(request) + mockBridgeResponse({ ok: true, apps: [] }) + + const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1', host) + + await client.listApps() + expectDesktopProviderSubprocessStartCount(1) + + healthy = true + await expect(client.listApps()).resolves.toEqual({ apps: [] }) + expectDesktopProviderSubprocessStartCount(1) + }) + + it('runs the one-shot bridge under RemoteSigned and falls back to Bypass once', async () => { + mockBridgeProcessFailure(POLICY_STDERR) + mockBridgeResponse({ ok: true, apps: [] }) + + const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1') + + await expect(client.listApps()).resolves.toEqual({ apps: [] }) + expectDesktopProviderSubprocessStartCount(2) + expect(bridgeProcessArgs(0)).toContain('-NoLogo') + expect(bridgeProcessArgs(0)).toContain('RemoteSigned') + expect(bridgeProcessArgs(0)).not.toContain('Bypass') + expect(bridgeProcessArgs(1)).toContain('Bypass') + }) + + it('does not retry the one-shot bridge for a non-policy failure', async () => { + mockBridgeProcessFailure('No top-level UI Automation window is available for Notepad') + + const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1') + + await expect(client.listApps()).rejects.toMatchObject({ code: 'window_not_found' }) + expectDesktopProviderSubprocessStartCount(1) + }) + + it('never replays an operation whose own output merely mentions a policy error', async () => { + // Window titles and element names are user-controlled text that lands in + // stdout; matching them would double a click, a keystroke or a paste. + mockBridgeProcessFailure({ + stdout: JSON.stringify({ + ok: true, + snapshot: { windowTitle: 'SecurityError - UnauthorizedAccess.log - Notepad' } + }), + stderr: '' + }) + + const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1') + + await expect(client.listApps()).rejects.toBeInstanceOf(Error) + expectDesktopProviderSubprocessStartCount(1) + }) + + it('keeps Linux on the one-shot python bridge with no execution policy flags', async () => { + mockBridgeResponse({ ok: true, apps: [] }) + + const client = await createDesktopScriptProviderClient('linux', '/tmp/runtime.py') + + await expect(client.listApps()).resolves.toEqual({ apps: [] }) + expect(bridgeProcessArgs(0)).toEqual(['/tmp/runtime.py', expect.any(String)]) + }) +}) diff --git a/src/main/computer/desktop-script-provider-test-harness.ts b/src/main/computer/desktop-script-provider-test-harness.ts index 2cbd1e9a776..bcb0a4b0118 100644 --- a/src/main/computer/desktop-script-provider-test-harness.ts +++ b/src/main/computer/desktop-script-provider-test-harness.ts @@ -1,4 +1,5 @@ import { expect, vi } from 'vitest' +import type { DesktopScriptRuntimeHost } from './desktop-script-runtime-host' const { execFileMock, operationFiles, mkdtempMock, rmMock, writeFileMock } = vi.hoisted(() => { const files = new Map() @@ -23,12 +24,14 @@ vi.mock('fs/promises', () => ({ writeFile: writeFileMock })) +/** Builds a client on the one-shot bridge; pass a host to exercise serve mode. */ export async function createDesktopScriptProviderClient( platform: 'linux' | 'windows', - executablePath: string + executablePath: string, + runtimeHost: DesktopScriptRuntimeHost | null = null ) { const { DesktopScriptProviderClient } = await import('./desktop-script-provider-client') - return new DesktopScriptProviderClient(platform, executablePath) + return new DesktopScriptProviderClient(platform, executablePath, runtimeHost) } export function resetDesktopScriptProviderTestHarness(): void { @@ -77,6 +80,19 @@ export function mockBridgeResponse( }) } +export function mockBridgeProcessFailure(streams: string | { stdout?: string; stderr?: string }) { + const { stdout = '', stderr = '' } = typeof streams === 'string' ? { stderr: streams } : streams + execFileMock.mockImplementationOnce((_command, _args, _options, callback) => { + const done = callback as (error: Error | null, stdout: string, stderr: string) => void + done(new Error('Command failed'), stdout, stderr) + return null as never + }) +} + +export function bridgeProcessArgs(call: number): string[] { + return (execFileMock.mock.calls[call]?.[1] ?? []) as string[] +} + export function sampleBridgeSnapshot(name: string, value: string) { return { app: { name, bundleIdentifier: name, pid: 100 }, diff --git a/src/main/computer/desktop-script-provider-types.ts b/src/main/computer/desktop-script-provider-types.ts index 0ff48e80b4f..6e474fdcf29 100644 --- a/src/main/computer/desktop-script-provider-types.ts +++ b/src/main/computer/desktop-script-provider-types.ts @@ -101,6 +101,8 @@ export type BridgeWindow = { export type BridgeResponse = { ok: boolean + /** Echo of BridgeRequest.requestId; set only on the persistent serve path. */ + requestId?: number error?: string capabilities?: ComputerProviderCapabilities apps?: { @@ -122,6 +124,8 @@ export type BridgeResponse = { export type BridgeRequest = { tool: string + /** Correlates a serve-mode reply with its request; the one-shot path omits it. */ + requestId?: number app?: string element?: BridgeElement fromElement?: BridgeElement diff --git a/src/main/computer/desktop-script-request-queue.ts b/src/main/computer/desktop-script-request-queue.ts new file mode 100644 index 00000000000..8f32b5f6888 --- /dev/null +++ b/src/main/computer/desktop-script-request-queue.ts @@ -0,0 +1,73 @@ +import { RuntimeClientError } from './runtime-client-error' + +/** + * Serializes operations onto one helper and bounds how long one may wait its + * turn. + * + * Why the wait needs its own deadline: the in-flight timeout is armed only once + * a request reaches a helper, so a request behind N timing-out ones waited N + * times that timeout with no deadline of its own — bounded, but the caller sees + * an `await` that looks hung for minutes and gets no error to act on. + * + * Why only the wait: a request that reaches a helper still gets its full + * execution budget. A single deadline covering both would fail operations that + * queued briefly and would otherwise have succeeded. + */ +export class DesktopScriptRequestQueue { + /** + * Never rejects: downstream turns chain onto it, and a rejection here would + * be delivered to whichever request happened to queue behind the failure. + */ + private tail: Promise | null = null + + constructor( + private readonly waitTimeoutMs: number, + /** Called when the queue empties, so the host can arm its idle shutdown. */ + private readonly onDrained: () => void + ) {} + + enqueue(run: () => Promise): Promise { + const queued = this.tail + if (!queued) { + return this.track(run()) + } + let expiry: RuntimeClientError | null = null + let waitTimer: NodeJS.Timeout | undefined + const waited = new Promise((_resolve, reject) => { + waitTimer = setTimeout(() => { + expiry = new RuntimeClientError( + 'action_timeout', + `desktop provider timed out after ${this.waitTimeoutMs}ms waiting for earlier operations` + ) + reject(expiry) + }, this.waitTimeoutMs) + waitTimer.unref?.() + }) + // An abandoned request is never handed to a helper. The caller has already + // been told it failed, and a click delivered after that is worse than none. + const turn = (): Promise => { + clearTimeout(waitTimer) + return expiry ? Promise.reject(expiry) : run() + } + // The tail chains on the turn, not on the race: a caller giving up early + // must not release the next request while this one's predecessor is still + // in flight. + return Promise.race([waited, this.track(queued.then(turn, turn))]) + } + + private track(result: Promise): Promise { + const tail = result.then( + () => undefined, + () => undefined + ) + this.tail = tail + void tail.finally(() => { + if (this.tail !== tail) { + return + } + this.tail = null + this.onDrained() + }) + return result + } +} diff --git a/src/main/computer/desktop-script-runtime-availability.ts b/src/main/computer/desktop-script-runtime-availability.ts new file mode 100644 index 00000000000..58123f5eab7 --- /dev/null +++ b/src/main/computer/desktop-script-runtime-availability.ts @@ -0,0 +1,176 @@ +import { + FALLBACK_WINDOWS_EXECUTION_POLICY, + PREFERRED_WINDOWS_EXECUTION_POLICY, + type WindowsExecutionPolicy +} from './windows-powershell-execution-policy' + +/** + * Consecutive child failures before the helper is believed dead, and how long + * the one-shot bridge covers for it afterwards. + * + * Why not a latch: every plausible cause is transient — a Defender scan touching + * the script mid-launch, a locked CSC temp directory failing one `Add-Type`, + * momentary memory pressure. Giving up permanently silently restores the + * per-click process burst the host exists to remove, and computer use keeps + * working throughout, so nothing looks wrong while the MDE signature returns. + */ +export const MAX_START_ATTEMPTS = 3 +export const START_FAILURE_COOLDOWN_MS = 60_000 + +/** + * Why not `Date.now`: an NTP correction, a VM snapshot restore or a user changing + * the clock steps the wall clock backwards, which extended the cooldown by the + * size of the step. Nothing shortens it from there — only `recordSuccess` clears + * it, and no request can reach a helper to succeed while it holds — so a one-hour + * step disabled the persistent helper for the life of the sidecar, silently + * restoring the per-click process burst. Elapsed monotonic time cannot go + * backwards. + */ +const monotonicNowMs = (): number => performance.now() + +/** + * Whether the persistent helper is currently believed usable, and the execution + * policy it should be started under. + * + * Split from the host so the recovery rules are readable on their own: they are + * what stands between a transient bad spawn and a session that silently spends + * the rest of its life on one powershell.exe per click. + */ +export class RuntimeHostAvailability { + private policy: WindowsExecutionPolicy = PREFERRED_WINDOWS_EXECUTION_POLICY + private retryUnderFallbackPolicy = false + private consecutiveFailures = 0 + private consecutiveSuccesses = 0 + /** Null, not 0, for "no cooldown": `performance.now()` legitimately returns 0. */ + private cooldownStartedAtMs: number | null = null + /** + * Set while the escalated policy has yet to start a helper, so a wrong + * diagnosis can be taken back. + * + * Why it can be wrong: AppLocker and WDAC constrained language mode raise + * PSSecurityException under the same SecurityError category a policy block + * uses, but they refuse the script at parse time, which `Bypass` cannot lift. + * Latching there would spend the session putting the most heavily weighted + * MDE token on every command line, on exactly the hardened hosts watching + * for it. + */ + private fallbackPolicyUnproven = false + + constructor( + private readonly cooldownMs: number, + /** Public so the host can report its own start attempts to the same sink. */ + readonly warn: (message: string) => void, + /** Overridden only by tests; the default must stay monotonic. */ + private readonly now: () => number = monotonicNowMs + ) {} + + get executionPolicy(): WindowsExecutionPolicy { + return this.policy + } + + get policyRetryPending(): boolean { + return this.retryUnderFallbackPolicy + } + + get atPreferredPolicy(): boolean { + return this.policy === PREFERRED_WINDOWS_EXECUTION_POLICY + } + + /** Milliseconds left before the host may try a helper again; 0 when it may. */ + remainingCooldown(): number { + if (this.cooldownStartedAtMs === null) { + return 0 + } + // Elapsed since the cooldown began, never a stored deadline: a deadline is + // only as trustworthy as the clock it was computed against. + return Math.max(0, Math.ceil(this.cooldownMs - (this.now() - this.cooldownStartedAtMs))) + } + + requestPolicyRetry(): void { + this.retryUnderFallbackPolicy = true + } + + escalateExecutionPolicy(): void { + this.retryUnderFallbackPolicy = false + this.policy = FALLBACK_WINDOWS_EXECUTION_POLICY + this.fallbackPolicyUnproven = true + // Sticky once proven: a genuinely Restricted machine would otherwise pay a + // guaranteed failed spawn per operation. Only a helper that produced no + // output at all can reach here, so a snapshot cannot talk the host into it. + this.warn( + `runtime host start blocked at ${PREFERRED_WINDOWS_EXECUTION_POLICY}; trying ${FALLBACK_WINDOWS_EXECUTION_POLICY}` + ) + } + + /** A helper started under the current policy, so the policy is the right one. */ + confirmExecutionPolicy(): void { + this.fallbackPolicyUnproven = false + } + + /** + * Undo an escalation the fallback never justified. + * + * The escalation is a diagnosis, and a fallback that cannot start a helper + * either disproves it: the policy was not what stopped the first attempt. Go + * back rather than latch, so a re-probe can escalate again later if the real + * cause clears. Re-probing costs one spawn per outage, which the failure + * count and its cooldown already bound, and never latching is the whole point + * of this class. + */ + abandonUnprovenFallback(): void { + if (!this.fallbackPolicyUnproven) { + return + } + this.fallbackPolicyUnproven = false + this.policy = PREFERRED_WINDOWS_EXECUTION_POLICY + this.warn( + `${FALLBACK_WINDOWS_EXECUTION_POLICY} did not start a helper either, so the execution policy was not the cause; returning to ${PREFERRED_WINDOWS_EXECUTION_POLICY}` + ) + } + + recordFailure(): void { + this.consecutiveSuccesses = 0 + this.consecutiveFailures++ + } + + /** True once a helper has died often enough that respawning is just thrash. */ + get exhausted(): boolean { + return this.consecutiveFailures >= MAX_START_ATTEMPTS + } + + recordSuccess(): void { + this.consecutiveSuccesses++ + this.fallbackPolicyUnproven = false + // Why a clean run and not a single reply: a helper that answers one + // operation and dies on the next would otherwise reset the count forever, + // and respawn once per operation — the exact burst the host removes. + if (this.consecutiveSuccesses >= MAX_START_ATTEMPTS) { + this.consecutiveFailures = 0 + } + if (this.cooldownStartedAtMs === null) { + return + } + this.cooldownStartedAtMs = null + this.warn('runtime host recovered; operations are served by the persistent helper again') + } + + enterCooldown(): void { + // An escalation that never started a helper must not outlive the outage it + // was guessed from; the next one re-diagnoses from the preferred policy. + this.abandonUnprovenFallback() + const failures = this.consecutiveFailures + this.cooldownStartedAtMs = this.now() + // The wait is the penalty; leaving the count at the limit would charge twice + // and let the first death after recovery re-enter a full cooldown, so an + // interleaved workload would spend its life on the one-shot bridge. + this.consecutiveFailures = 0 + this.consecutiveSuccesses = 0 + this.warn( + `runtime host unavailable after ${failures} consecutive failures; falling back to one powershell.exe per operation for ${this.cooldownMs}ms` + ) + } + + clearCooldown(): void { + this.cooldownStartedAtMs = null + } +} diff --git a/src/main/computer/desktop-script-runtime-host.test.ts b/src/main/computer/desktop-script-runtime-host.test.ts new file mode 100644 index 00000000000..4d42a3c3e37 --- /dev/null +++ b/src/main/computer/desktop-script-runtime-host.test.ts @@ -0,0 +1,857 @@ +import { EventEmitter } from 'node:events' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { ProcessSpec } from '../../shared/child-process/process-spec' +import type { RuntimeChildProcess } from './desktop-script-serve-channel' +import { DesktopScriptRuntimeHost, isRuntimeHostUnavailable } from './desktop-script-runtime-host' + +const POLICY_ERROR = + 'File runtime.ps1 cannot be loaded because running scripts\nis disabled on this system.\n + CategoryInfo : SecurityError' + +class FakeRuntimeChild extends EventEmitter { + readonly stdout = new EventEmitter() + readonly stderr = new EventEmitter() + readonly writes: string[] = [] + killed = false + stdinEnded = false + /** Holds write callbacks so a late stdin failure can be fired deliberately. */ + deferWrites = false + private readonly pendingWrites: ((error?: Error | null) => void)[] = [] + + readonly stdin = { + write: (chunk: string, callback?: (error?: Error | null) => void): boolean => { + this.writes.push(chunk) + if (this.deferWrites) { + if (callback) { + this.pendingWrites.push(callback) + } + return true + } + callback?.(null) + return true + }, + end: (): void => { + this.stdinEnded = true + }, + on: (): void => {} + } + + kill(): boolean { + this.killed = true + return true + } + + /** What a destroyed stdin does to writes still queued at teardown. */ + failQueuedWrites(): void { + for (const callback of this.pendingWrites.splice(0)) { + callback(new Error('ERR_STREAM_DESTROYED')) + } + } + + /** Fail one queued write, leaving later ones outstanding. */ + failQueuedWrite(index: number): void { + this.pendingWrites.splice(index, 1)[0](new Error('EPIPE')) + } + + /** Requests written to this child, decoded. */ + requests(): Record[] { + return this.writes.map((line) => JSON.parse(line) as Record) + } + + /** The id the host is currently waiting on, so replies can echo it. */ + pendingId(): number { + return this.requests().at(-1)?.requestId as number + } + + /** The announcement the real serve loop writes before its first read. */ + ready(): void { + this.write('{"ready":true}\n') + } + + respond(response: Record, requestId = this.pendingId()): void { + this.write(`${JSON.stringify({ ...response, requestId })}\n`) + } + + write(raw: string): void { + this.stdout.emit('data', Buffer.from(raw, 'utf8')) + } + + exit(code: number | null, stderr = ''): void { + if (stderr) { + this.stderr.emit('data', Buffer.from(stderr, 'utf8')) + } + this.emit('close', code, null) + } +} + +function createHost( + options: { + idleShutdownMs?: number + requestTimeoutMs?: number + cooldownMs?: number + now?: () => number + deferWrites?: boolean + } = {} +) { + const children: FakeRuntimeChild[] = [] + const specs: ProcessSpec[] = [] + const warnings: string[] = [] + const host = new DesktopScriptRuntimeHost('C:\\orca\\runtime.ps1', { + ...options, + powerShellPath: () => 'C:\\Windows\\System32\\powershell.exe', + warn: (message) => warnings.push(message), + spawn: (spec) => { + specs.push(spec) + const child = new FakeRuntimeChild() + child.deferWrites = options.deferWrites === true + children.push(child) + return child as unknown as RuntimeChildProcess + } + }) + return { host, children, specs, warnings } +} + +/** Let the host's queue microtasks drain so the next request reaches its child. */ +async function settle(): Promise { + for (let index = 0; index < 6; index++) { + await Promise.resolve() + } +} + +/** The wait the host reported, read back out of its refusal message. */ +function remainingCooldownMs(error: Error | null): number { + const match = /retrying the runtime host in (\d+)ms/.exec(error?.message ?? '') + return match ? Number(match[1]) : Number.NaN +} + +/** Kill each helper the host starts, until it stops starting them. */ +async function failEveryStart(children: FakeRuntimeChild[], stderr: string): Promise { + for (let index = 0; index < 8; index++) { + if (index >= children.length) { + return + } + children[index].exit(1, stderr) + await settle() + } +} + +describe('DesktopScriptRuntimeHost', () => { + afterEach(() => { + vi.useRealTimers() + vi.restoreAllMocks() + }) + + it('starts one helper for many operations and never writes an operation file', async () => { + const { host, children, specs } = createHost() + + const first = host.request({ tool: 'handshake' }) + await settle() + children[0].respond({ ok: true, capabilities: {} }) + await expect(first).resolves.toMatchObject({ ok: true }) + + for (let index = 0; index < 5; index++) { + const next = host.request({ tool: 'click', app: 'Notepad' }) + await settle() + children[0].respond({ ok: true, action: { path: 'synthetic' } }) + await expect(next).resolves.toMatchObject({ ok: true }) + } + + expect(children).toHaveLength(1) + expect(children[0].requests()).toHaveLength(6) + expect(specs[0].args).toEqual([ + '-NoLogo', + '-NoProfile', + '-NonInteractive', + '-ExecutionPolicy', + 'RemoteSigned', + '-File', + 'C:\\orca\\runtime.ps1', + '-Serve' + ]) + host.dispose() + }) + + it('serializes requests so only one operation is ever in flight', async () => { + const { host, children } = createHost() + + const first = host.request({ tool: 'click', app: 'A' }) + const second = host.request({ tool: 'click', app: 'B' }) + await settle() + + expect(children[0].requests()).toEqual([{ tool: 'click', app: 'A', requestId: 1 }]) + + children[0].respond({ ok: true, action: { path: 'synthetic' } }) + await expect(first).resolves.toMatchObject({ ok: true }) + await settle() + + expect(children[0].requests()).toHaveLength(2) + children[0].respond({ ok: true, action: { path: 'accessibility' } }) + await expect(second).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + it('strips the echoed id from the response it hands back', async () => { + const { host, children } = createHost() + const promise = host.request({ tool: 'handshake' }) + await settle() + children[0].respond({ ok: true, capabilities: {} }) + + await expect(promise).resolves.toEqual({ ok: true, capabilities: {} }) + host.dispose() + }) + + it('reassembles a response split across chunks, including a split code point', async () => { + const { host, children } = createHost() + const promise = host.request({ tool: 'get_app_state', app: 'Editor' }) + await settle() + + const payload = Buffer.from( + `${JSON.stringify({ ok: true, snapshot: { app: 'né' }, requestId: 1 })}\r\n`, + 'utf8' + ) + const split = payload.indexOf(Buffer.from('é', 'utf8')) + 1 + children[0].stdout.emit('data', payload.subarray(0, split)) + children[0].stdout.emit('data', payload.subarray(split)) + + await expect(promise).resolves.toEqual({ ok: true, snapshot: { app: 'né' } }) + host.dispose() + }) + + it('kills the helper rather than answering a request with another reply', async () => { + const { host, children } = createHost() + + const first = host.request({ tool: 'handshake' }) + await settle() + // A stray line would otherwise shift every later response by one. + children[0].respond({ ok: true, capabilities: {} }, 999) + + await expect(first).rejects.toThrow(/did not match the pending request/) + expect(children[0].killed).toBe(true) + host.dispose() + }) + + it('kills the helper when an unsolicited line arrives with nothing pending', async () => { + const { host, children } = createHost() + + const first = host.request({ tool: 'handshake' }) + await settle() + children[0].respond({ ok: true, capabilities: {} }) + await first + + children[0].write(`${JSON.stringify({ ok: true, requestId: 77 })}\n`) + expect(children[0].killed).toBe(true) + host.dispose() + }) + + it('times out a wedged operation and starts a fresh helper for the next one', async () => { + vi.useFakeTimers() + const { host, children } = createHost({ requestTimeoutMs: 30_000 }) + + const promise = host.request({ tool: 'click', app: 'Frozen' }) + await settle() + await vi.advanceTimersByTimeAsync(30_001) + + await expect(promise).rejects.toMatchObject({ code: 'action_timeout' }) + expect(children[0].killed).toBe(true) + + const next = host.request({ tool: 'handshake' }) + await settle() + expect(children).toHaveLength(2) + children[1].respond({ ok: true, capabilities: {} }) + await expect(next).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + it('rejects the in-flight request when a working helper crashes, then restarts', async () => { + const { host, children } = createHost() + + const first = host.request({ tool: 'handshake' }) + await settle() + children[0].respond({ ok: true, capabilities: {} }) + await first + + const second = host.request({ tool: 'click', app: 'Notepad' }) + await settle() + children[0].exit(1, 'boom') + + await expect(second).rejects.toMatchObject({ code: 'accessibility_error' }) + await expect(second).rejects.toThrow(/runtime host exited/) + + const third = host.request({ tool: 'handshake' }) + await settle() + expect(children).toHaveLength(2) + children[1].respond({ ok: true, capabilities: {} }) + await expect(third).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + it('stops respawning a helper that dies on every second operation', async () => { + let clock = 1_000 + const { host, children } = createHost({ cooldownMs: 60_000, now: () => clock }) + + // One good answer per helper is exactly the pattern that used to respawn + // forever: the success reset the failure count before it could ever trip. + for (let round = 0; round < 3; round++) { + const good = host.request({ tool: 'handshake' }) + await settle() + children.at(-1)?.respond({ ok: true, capabilities: {} }) + await expect(good).resolves.toMatchObject({ ok: true }) + await settle() + + const crash = host.request({ tool: 'click', app: 'Crashy' }) + await settle() + children.at(-1)?.exit(1, 'boom') + await expect(crash).rejects.toThrow(/runtime host exited/) + await settle() + } + + const spawned = children.length + await expect(host.request({ tool: 'handshake' })).rejects.toSatisfy(isRuntimeHostUnavailable) + expect(children).toHaveLength(spawned) + host.dispose() + }) + + it('keeps serving a healthy helper after an isolated crash', async () => { + const { host, children } = createHost({ cooldownMs: 60_000 }) + + const crashed = host.request({ tool: 'handshake' }) + await settle() + children[0].respond({ ok: true, capabilities: {} }) + await crashed + const second = host.request({ tool: 'click', app: 'Notepad' }) + await settle() + children[0].exit(1, 'boom') + await expect(second).rejects.toThrow(/runtime host exited/) + + for (let index = 0; index < 4; index++) { + const next = host.request({ tool: 'handshake' }) + await settle() + children.at(-1)?.respond({ ok: true, capabilities: {} }) + await expect(next).resolves.toMatchObject({ ok: true }) + } + + // A clean run clears the count, so one bad helper cannot degrade a good one. + expect(children).toHaveLength(2) + host.dispose() + }) + + it('stops respawning a helper that keeps answering the wrong request', async () => { + let clock = 1_000 + const { host, children } = createHost({ cooldownMs: 60_000, now: () => clock }) + + // Desync is host-detected, so it bypassed the exit handler entirely: without + // its own accounting this respawned once per operation, forever. + for (let round = 0; round < 3; round++) { + const promise = host.request({ tool: 'handshake' }) + await settle() + const child = children.at(-1) + child?.respond({ ok: true, capabilities: {} }, child.pendingId() + 500) + await expect(promise).rejects.toThrow(/did not match the pending request/) + await settle() + } + + const spawned = children.length + await expect(host.request({ tool: 'handshake' })).rejects.toSatisfy(isRuntimeHostUnavailable) + expect(children).toHaveLength(spawned) + host.dispose() + }) + + it('stops respawning a helper that times out on every operation', async () => { + vi.useFakeTimers() + let clock = 1_000 + const { host, children } = createHost({ + requestTimeoutMs: 1_000, + cooldownMs: 60_000, + now: () => clock + }) + + for (let round = 0; round < 3; round++) { + const promise = host.request({ tool: 'get_app_state', app: 'Frozen' }) + await settle() + await vi.advanceTimersByTimeAsync(1_001) + await expect(promise).rejects.toMatchObject({ code: 'action_timeout' }) + await settle() + } + + const spawned = children.length + await expect(host.request({ tool: 'handshake' })).rejects.toSatisfy(isRuntimeHostUnavailable) + expect(children).toHaveLength(spawned) + host.dispose() + }) + + it('never re-sends a mutation to a fresh helper after a pre-answer death', async () => { + const { host, children } = createHost() + + const promise = host.request({ tool: 'click', app: 'Notepad', x: 10, y: 10 }) + await settle() + children[0].exit(1, 'Add-Type : Cannot access the temporary directory') + + // The click may already have landed inside the helper that died; replaying + // it would click twice. An observation in the same position is retried. + await expect(promise).rejects.toSatisfy(isRuntimeHostUnavailable) + expect(children).toHaveLength(1) + host.dispose() + }) + + it('never replays a mutation once the helper announced it was reading', async () => { + const { host, children } = createHost() + + const promise = host.request({ tool: 'click', app: 'Notepad', x: 10, y: 10 }) + await settle() + children[0].ready() + // Past the announcement the click may already have been synthesized: the + // snapshot that follows it is the fault-prone part, so a missing reply + // proves nothing about whether the input landed. + children[0].exit(1, 'faulting module gdiplus.dll') + + await expect(promise).rejects.toThrow(/runtime host exited/) + expect(children).toHaveLength(1) + host.dispose() + }) + + it('replays a mutation only for a helper that died before announcing readiness', async () => { + const { host, children } = createHost() + + const first = host.request({ tool: 'handshake' }) + await settle() + children[0].ready() + children[0].respond({ ok: true, capabilities: {} }) + await first + + const crashed = host.request({ tool: 'click', app: 'Notepad', x: 1, y: 1 }) + await settle() + children[0].exit(1, 'boom') + await expect(crashed).rejects.toThrow(/runtime host exited/) + + const retried = host.request({ tool: 'click', app: 'Notepad', x: 1, y: 1 }) + await settle() + // This helper never announced, so it cannot have read the click: replaying + // is a fact rather than a guess, and the caller never sees the stumble. + children[1].exit(1, 'Add-Type : Cannot access the temporary directory') + await settle() + + expect(children).toHaveLength(3) + children[2].ready() + children[2].respond({ ok: true, action: { path: 'synthetic' } }) + await expect(retried).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + it('does not treat the readiness announcement as an unmatched reply', async () => { + const { host, children } = createHost() + + const promise = host.request({ tool: 'handshake' }) + await settle() + children[0].ready() + + expect(children[0].killed).toBe(false) + children[0].respond({ ok: true, capabilities: {} }) + await expect(promise).resolves.toEqual({ ok: true, capabilities: {} }) + host.dispose() + }) + + it('charges one cooldown per outage, not one per later death', async () => { + let clock = 1_000 + const { host, children } = createHost({ cooldownMs: 60_000, now: () => clock }) + + const failed = host.request({ tool: 'handshake' }) + await settle() + await failEveryStart(children, 'The term is not recognized') + await expect(failed).rejects.toSatisfy(isRuntimeHostUnavailable) + + clock += 61_000 + const recovered = host.request({ tool: 'handshake' }) + await settle() + children.at(-1)?.respond({ ok: true, capabilities: {} }) + await recovered + + // One death after recovery must not re-enter a full cooldown; the previous + // outage was already paid for. + const crashed = host.request({ tool: 'handshake' }) + await settle() + children.at(-1)?.exit(1, 'boom') + await expect(crashed).rejects.toBeInstanceOf(Error) + + const next = host.request({ tool: 'handshake' }) + await settle() + children.at(-1)?.respond({ ok: true, capabilities: {} }) + await expect(next).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + it('charges one failure when a write fails after the helper was torn down', async () => { + const { host, children, warnings } = createHost({ deferWrites: true }) + + const promise = host.request({ tool: 'handshake' }) + await settle() + children[0].respond({ ok: true, capabilities: {} }, 999) + await expect(promise).rejects.toThrow(/did not match the pending request/) + + // stop() destroys stdin, so the queued write calls back with an error. That + // is the same operation failing, not a second one, and counting it twice + // would drive a 3-strike cooldown at half the intended rate. + children[0].failQueuedWrites() + + expect(warnings.filter((line) => /helper stopped/.test(line))).toHaveLength(1) + host.dispose() + }) + + it('never lets a stale write error stop a replacement helper', async () => { + const { host, children } = createHost({ deferWrites: true }) + + const first = host.request({ tool: 'handshake' }) + await settle() + children[0].respond({ ok: true, capabilities: {} }, 999) + await expect(first).rejects.toBeInstanceOf(Error) + + const second = host.request({ tool: 'handshake' }) + await settle() + expect(children).toHaveLength(2) + + // The late callback belongs to a channel and a request that are both gone. + children[0].failQueuedWrites() + + expect(children[1].killed).toBe(false) + children[1].respond({ ok: true, capabilities: {} }) + await expect(second).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + it('ignores a write error for a request that already finished', async () => { + const { host, children } = createHost({ deferWrites: true }) + + const first = host.request({ tool: 'handshake' }) + await settle() + children[0].respond({ ok: true, capabilities: {} }) + await first + + const second = host.request({ tool: 'handshake' }) + await settle() + + // Backpressure can hold a write callback past its own response. The channel + // is alive and was never stopped, so only the request id can tell that this + // report is stale — this is what pins the host-side guard on its own. + children[0].failQueuedWrite(0) + + expect(children[0].killed).toBe(false) + children[0].respond({ ok: true, capabilities: {} }) + await expect(second).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + it('shuts the helper down when idle and starts a new one on the next operation', async () => { + vi.useFakeTimers() + const { host, children } = createHost({ idleShutdownMs: 60_000 }) + + const first = host.request({ tool: 'handshake' }) + await settle() + children[0].respond({ ok: true, capabilities: {} }) + await first + await settle() + + expect(children[0].killed).toBe(false) + await vi.advanceTimersByTimeAsync(60_001) + expect(children[0].stdinEnded).toBe(true) + expect(children[0].killed).toBe(true) + + const next = host.request({ tool: 'handshake' }) + await settle() + expect(children).toHaveLength(2) + children[1].respond({ ok: true, capabilities: {} }) + await expect(next).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + it('disposes the helper and rejects the in-flight request', async () => { + const { host, children } = createHost() + const promise = host.request({ tool: 'click', app: 'Notepad' }) + await settle() + + host.dispose() + + expect(children[0].stdinEnded).toBe(true) + expect(children[0].killed).toBe(true) + await expect(promise).rejects.toThrow(/shut down/) + }) + + it('never respawns for a request queued behind dispose', async () => { + const { host, children } = createHost() + const first = host.request({ tool: 'handshake' }) + const queued = host.request({ tool: 'handshake' }) + await settle() + + host.dispose() + await expect(first).rejects.toBeInstanceOf(Error) + await expect(queued).rejects.toSatisfy(isRuntimeHostUnavailable) + await settle() + + expect(children).toHaveLength(1) + }) + + it('falls back to Bypass once when the execution policy blocks the start', async () => { + const { host, children, specs, warnings } = createHost() + + const promise = host.request({ tool: 'handshake' }) + await settle() + children[0].exit(1, POLICY_ERROR) + await settle() + + expect(children).toHaveLength(2) + expect(specs[1].args).toContain('Bypass') + children[1].respond({ ok: true, capabilities: {} }) + await expect(promise).resolves.toMatchObject({ ok: true }) + expect(warnings.some((line) => /trying Bypass/.test(line))).toBe(true) + + // A helper started under Bypass, so the diagnosis is proven and the fallback + // is remembered for the session rather than re-probed per call. + const next = host.request({ tool: 'handshake' }) + await settle() + expect(children).toHaveLength(2) + children[1].respond({ ok: true, capabilities: {} }) + await next + expect(warnings.some((line) => /returning to RemoteSigned/.test(line))).toBe(false) + host.dispose() + }) + + it('returns to RemoteSigned when Bypass does not start a helper either', async () => { + let clock = 1_000 + const { host, children, specs, warnings } = createHost({ cooldownMs: 60_000, now: () => clock }) + + // What AppLocker and WDAC constrained language mode look like: the same + // SecurityError category, but the block is at script load, so Bypass cannot + // lift it and the escalation was a misdiagnosis. + const promise = host.request({ tool: 'handshake' }) + await settle() + await failEveryStart(children, POLICY_ERROR) + await expect(promise).rejects.toSatisfy(isRuntimeHostUnavailable) + + expect(specs[1].args).toContain('Bypass') + expect(warnings.some((line) => /returning to RemoteSigned/.test(line))).toBe(true) + // The revert lands inside the outage, not just at its end: every attempt + // after the fallback is disproved is back on the preferred policy, so the + // misdiagnosis costs one Bypass command line rather than one per attempt. + expect(specs).toHaveLength(3) + expect(specs[2].args).not.toContain('Bypass') + + // Latching here would put the most heavily weighted MDE token on every + // later command line, on exactly the hardened host that is watching. + clock += 61_000 + const recovered = host.request({ tool: 'handshake' }) + await settle() + expect(specs.at(-1)?.args).not.toContain('Bypass') + children.at(-1)?.respond({ ok: true, capabilities: {} }) + await expect(recovered).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + it('reports itself unavailable when Bypass is also refused', async () => { + const { host, children } = createHost() + + const promise = host.request({ tool: 'handshake' }) + await settle() + await failEveryStart(children, POLICY_ERROR) + + await expect(promise).rejects.toSatisfy(isRuntimeHostUnavailable) + host.dispose() + }) + + it('reports itself unavailable when the helper cannot be spawned at all', async () => { + const host = new DesktopScriptRuntimeHost('C:\\orca\\runtime.ps1', { + powerShellPath: () => 'C:\\Windows\\System32\\powershell.exe', + warn: () => {}, + spawn: () => { + throw new Error('spawn ENOENT') + } + }) + + await expect(host.request({ tool: 'handshake' })).rejects.toSatisfy(isRuntimeHostUnavailable) + host.dispose() + }) + + it('retries a transient pre-answer death without the caller ever seeing it', async () => { + const { host, children } = createHost() + + const promise = host.request({ tool: 'handshake' }) + await settle() + children[0].exit(1, 'Add-Type : Cannot access the temporary directory') + await settle() + + expect(children).toHaveLength(2) + children[1].respond({ ok: true, capabilities: {} }) + + await expect(promise).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + it('gives up only after repeated start failures, then serves from the host again after the cooldown', async () => { + let clock = 1_000 + const { host, children, warnings } = createHost({ cooldownMs: 60_000, now: () => clock }) + + const failed = host.request({ tool: 'handshake' }) + await settle() + await failEveryStart(children, 'The term is not recognized') + await expect(failed).rejects.toSatisfy(isRuntimeHostUnavailable) + + const attempts = children.length + expect(attempts).toBe(3) + + // Inside the cooldown the host stays out of the way without respawning. + clock += 30_000 + await expect(host.request({ tool: 'handshake' })).rejects.toSatisfy(isRuntimeHostUnavailable) + expect(children).toHaveLength(attempts) + + // Past it, the next operation re-probes rather than staying degraded forever. + clock += 31_000 + const recovered = host.request({ tool: 'handshake' }) + await settle() + expect(children).toHaveLength(attempts + 1) + children[attempts].respond({ ok: true, capabilities: {} }) + await expect(recovered).resolves.toMatchObject({ ok: true }) + + expect(warnings.at(-1)).toMatch(/recovered/) + host.dispose() + }) + + it('keeps the helper account of a reply it could not tag', async () => { + const { host, children } = createHost() + + const promise = host.request({ tool: 'handshake' }) + await settle() + const child = children[0] + // What an old runtime.ps1 sends when a request will not parse: a real error, + // with no id to route it by. The desync is honest, but replacing its message + // reports a broken stream and loses the only account of the cause. + child.respond({ ok: false, error: 'Invalid object passed in' }, child.pendingId() + 500) + + await expect(promise).rejects.toThrow( + /did not match the pending request: Invalid object passed in/ + ) + host.dispose() + }) + + it('does not charge a cooldown for requests the helper rejects as malformed', async () => { + const { host, children } = createHost({ cooldownMs: 60_000 }) + + // A tagged error is the helper working, not failing. Three of them used to + // arrive untagged, and three desync aborts is exactly the cooldown. + for (let round = 0; round < 3; round++) { + const promise = host.request({ tool: 'handshake' }) + await settle() + children[0].respond({ ok: false, error: 'Invalid object passed in' }) + await expect(promise).resolves.toMatchObject({ ok: false }) + await settle() + } + + expect(children).toHaveLength(1) + const next = host.request({ tool: 'handshake' }) + await settle() + children[0].respond({ ok: true, capabilities: {} }) + await expect(next).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + it('fails a request that spends its whole timeout queued behind others', async () => { + vi.useFakeTimers() + const { host, children } = createHost({ requestTimeoutMs: 1_000 }) + + // Two ahead of it, because one puts the turn exactly on the deadline. + const first = host.request({ tool: 'get_app_state', app: 'Frozen' }) + const second = host.request({ tool: 'get_app_state', app: 'Frozen' }) + const queued = host.request({ tool: 'click', app: 'Notepad' }) + // Asserted before the clock moves: both reject while the test is still + // inside advanceTimersByTimeAsync. + const firstFailed = expect(first).rejects.toMatchObject({ code: 'action_timeout' }) + // Its own deadline, not the one it would inherit by reaching the head. + const queuedFailed = expect(queued).rejects.toMatchObject({ + code: 'action_timeout', + message: /waiting for earlier operations/ + }) + await settle() + expect(children[0].requests()).toHaveLength(1) + + await vi.advanceTimersByTimeAsync(1_001) + await firstFailed + await queuedFailed + + // Drain past the abandoned request: it is never handed to a helper, because + // a click the caller has been told failed must not still land. + children[1].respond({ ok: true, state: {} }) + await expect(second).resolves.toMatchObject({ ok: true }) + await settle() + expect(children.flatMap((child) => child.requests())).not.toContainEqual( + expect.objectContaining({ tool: 'click' }) + ) + + // The request that gave up does not poison the queue behind it. + const next = host.request({ tool: 'handshake' }) + await settle() + children[1].respond({ ok: true, capabilities: {} }) + await expect(next).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + it('gives a queued request its full timeout once it reaches the helper', async () => { + vi.useFakeTimers() + const { host, children } = createHost({ requestTimeoutMs: 1_000 }) + + const head = host.request({ tool: 'handshake' }) + const queued = host.request({ tool: 'get_app_state', app: 'Slow' }) + await settle() + + await vi.advanceTimersByTimeAsync(900) + children[0].respond({ ok: true, capabilities: {} }) + await expect(head).resolves.toMatchObject({ ok: true }) + await settle() + + // Past the point the enqueue deadline would have fired: waiting its turn + // must not eat the budget the operation itself is entitled to. + await vi.advanceTimersByTimeAsync(900) + children[0].respond({ ok: true, state: {} }) + await expect(queued).resolves.toMatchObject({ ok: true }) + host.dispose() + }) + + // Both of these deliberately leave `now` unset: the bug was in the default the + // host picks, so a test that injects a clock cannot see it. + it('does not stretch the cooldown when the wall clock steps backwards', async () => { + const wallClock = vi.spyOn(Date, 'now').mockReturnValue(2_000_000_000_000) + const { host, children } = createHost({ cooldownMs: 60_000 }) + + const failed = host.request({ tool: 'handshake' }) + await settle() + await failEveryStart(children, 'The term is not recognized') + await expect(failed).rejects.toSatisfy(isRuntimeHostUnavailable) + + // An NTP correction, a VM snapshot restore, a user changing the clock. + wallClock.mockReturnValue(2_000_000_000_000 - 3_600_000) + + const refused = await host.request({ tool: 'handshake' }).then( + () => null, + (error: Error) => error + ) + expect(refused?.message).toMatch(/retrying the runtime host in/) + expect(remainingCooldownMs(refused)).toBeLessThanOrEqual(60_000) + host.dispose() + }) + + it('serves from the persistent helper again after a backwards clock step', async () => { + vi.spyOn(Date, 'now').mockReturnValue(2_000_000_000_000) + const { host, children } = createHost({ cooldownMs: 25 }) + + const failed = host.request({ tool: 'handshake' }) + await settle() + await failEveryStart(children, 'The term is not recognized') + await expect(failed).rejects.toSatisfy(isRuntimeHostUnavailable) + const attempts = children.length + + vi.mocked(Date.now).mockReturnValue(2_000_000_000_000 - 3_600_000) + // Real elapsed time, because the clock under test is the real monotonic one. + await new Promise((resolve) => setTimeout(resolve, 60)) + + const recovered = host.request({ tool: 'handshake' }) + await settle() + expect(children).toHaveLength(attempts + 1) + children[attempts].respond({ ok: true, capabilities: {} }) + await expect(recovered).resolves.toMatchObject({ ok: true }) + host.dispose() + }) +}) diff --git a/src/main/computer/desktop-script-runtime-host.ts b/src/main/computer/desktop-script-runtime-host.ts new file mode 100644 index 00000000000..09aff5ec479 --- /dev/null +++ b/src/main/computer/desktop-script-runtime-host.ts @@ -0,0 +1,384 @@ +import { spawnProcess } from '../../shared/child-process/run-process' +import { windowsPowerShellPath } from '../../shared/child-process/windows-system-binary' +import { reportComputerDiagnostic } from './computer-sidecar-diagnostics' +import { isReplayableTool } from './desktop-script-action' +import type { BridgeRequest, BridgeResponse } from './desktop-script-provider-types' +import { DesktopScriptRequestQueue } from './desktop-script-request-queue' +import { + startServeChannel, + type DesktopScriptServeChannel, + type RuntimeProcessSpawn +} from './desktop-script-serve-channel' +import { + MAX_START_ATTEMPTS, + RuntimeHostAvailability, + START_FAILURE_COOLDOWN_MS +} from './desktop-script-runtime-availability' +import { RuntimeClientError } from './runtime-client-error' +import { + isExecutionPolicyBlocked, + windowsPowerShellRuntimeArgs +} from './windows-powershell-execution-policy' + +const REQUEST_TIMEOUT_MS = 30_000 +const IDLE_SHUTDOWN_MS = 120_000 + +/** Code the client keys on to serve this one operation from the one-shot bridge. */ +export const RUNTIME_HOST_UNAVAILABLE = 'runtime_host_unavailable' + +export type DesktopScriptRuntimeHostOptions = { + spawn?: RuntimeProcessSpawn + powerShellPath?: () => string + requestTimeoutMs?: number + idleShutdownMs?: number + cooldownMs?: number + now?: () => number + warn?: (message: string) => void +} + +type PendingRequest = { + id: number + resolve: (response: BridgeResponse) => void + reject: (error: Error) => void + timer: NodeJS.Timeout +} + +export function isRuntimeHostUnavailable(error: unknown): boolean { + return error instanceof RuntimeClientError && error.code === RUNTIME_HOST_UNAVAILABLE +} + +/** + * One long-lived `runtime.ps1 -Serve` process serving every computer-use + * operation over NDJSON on stdin/stdout. + * + * Why persistent: the one-shot bridge started a powershell.exe per click, and + * each one re-emitted the script's inline `Add-Type` P/Invoke assembly, which + * Defender for Endpoint reports as suspicious MSIL emission alongside the + * screen capture. Compiling once per session collapses a burst of short-lived + * PIDs into a single process. + * + * Requests are strictly serialized, and each carries an id the helper echoes. + * Serialization alone would leave a single stray line answering every later + * request with the previous response — silently acting on stale element + * indexes, with no error raised — so the id is checked and a mismatch is fatal + * to the child rather than merely logged. + */ +export class DesktopScriptRuntimeHost { + private channel: DesktopScriptServeChannel | null = null + private pending: PendingRequest | null = null + private idleTimer: NodeJS.Timeout | null = null + private childReady = false + private childAnswered = false + /** + * Set once any helper has announced itself, which proves the script on disk + * speaks the ready protocol. Until then a mutating request is not replayed + * even on a clean start failure, because ORCA_COMPUTER_DESKTOP_SCRIPT_PROVIDER_PATH + * can point at an older runtime.ps1 that simply never announces. + */ + private readyProtocolConfirmed = false + private disposed = false + private nextRequestId = 1 + private readonly availability: RuntimeHostAvailability + private readonly queue: DesktopScriptRequestQueue + private readonly requestTimeoutMs: number + private readonly idleShutdownMs: number + + constructor( + private readonly scriptPath: string, + private readonly options: DesktopScriptRuntimeHostOptions = {} + ) { + this.requestTimeoutMs = options.requestTimeoutMs ?? REQUEST_TIMEOUT_MS + this.idleShutdownMs = options.idleShutdownMs ?? IDLE_SHUTDOWN_MS + this.queue = new DesktopScriptRequestQueue(this.requestTimeoutMs, () => this.armIdleTimer()) + this.availability = new RuntimeHostAvailability( + options.cooldownMs ?? START_FAILURE_COOLDOWN_MS, + (message) => (options.warn ?? reportComputerDiagnostic)(message), + options.now + ) + } + + request(request: BridgeRequest): Promise { + return this.queue.enqueue(() => this.send(request)) + } + + /** Permanently stop this host. Callers build a new one for a new session. */ + dispose(): void { + this.disposed = true + this.clearIdleTimer() + this.availability.clearCooldown() + this.stopChannel() + this.rejectPending( + new RuntimeClientError('accessibility_error', 'desktop provider runtime host was shut down') + ) + } + + private async send(request: BridgeRequest): Promise { + this.clearIdleTimer() + // Why checked here and not only on entry: requests queue, and dispose can + // land while one waits its turn. Without this a teardown respawns a helper. + if (this.disposed) { + throw this.unavailableError('runtime host was disposed') + } + const cooldown = this.availability.remainingCooldown() + if (cooldown > 0) { + throw this.unavailableError(`retrying the runtime host in ${cooldown}ms`) + } + let lastError: unknown + for (let attempt = 1; attempt <= MAX_START_ATTEMPTS; attempt++) { + try { + const response = await this.sendOnce(request) + this.availability.recordSuccess() + return response + } catch (error) { + lastError = error + if (this.availability.policyRetryPending) { + this.availability.escalateExecutionPolicy() + continue + } + // Only this error proves no helper started, which is what disproves the + // escalation; a helper that started and then died proves the opposite. + if (isRuntimeHostUnavailable(error)) { + this.availability.abandonUnprovenFallback() + } + // A helper that answered and then died is a crash, not a bad start: the + // caller sees it and the next operation gets a fresh process — unless it + // keeps happening, which is thrash the one-shot bridge should absorb. + if (!isRuntimeHostUnavailable(error) || !this.mayReplay(request)) { + if (this.availability.exhausted) { + this.availability.enterCooldown() + } + throw error + } + this.availability.warn( + `runtime host failed to start (attempt ${attempt}/${MAX_START_ATTEMPTS}): ${errorText(error)}` + ) + } + } + this.availability.enterCooldown() + throw lastError + } + + private sendOnce(request: BridgeRequest): Promise { + let channel: DesktopScriptServeChannel + try { + channel = this.ensureChannel() + } catch (error) { + this.availability.recordFailure() + return Promise.reject(this.unavailableError(errorText(error))) + } + const id = this.nextRequestId++ + return new Promise((resolve, reject) => { + // Why kill rather than wait: a hung UI Automation call cannot be + // cancelled, so the process itself is the only thing left to reclaim. + const timer = setTimeout(() => { + this.abortChannel( + new RuntimeClientError( + 'action_timeout', + `desktop provider timed out after ${this.requestTimeoutMs}ms` + ) + ) + }, this.requestTimeoutMs) + timer.unref?.() + this.pending = { id, resolve, reject, timer } + channel.write(`${JSON.stringify({ ...request, requestId: id })}\n`, (error) => { + // Bind the report to what it was written for: a late callback must not + // charge a second failure for this operation, nor stop a replacement + // helper and reject a later request with this one's error. Deliberately + // redundant with the channel's own closed guard — keep both. This one + // also covers a live channel whose request has already been answered, + // which the channel cannot see; that case is what pins it. + // + // Redundant does not mean untested: removing either guard alone fails a + // test, so neither can be deleted as "the one the other covers". + if (this.channel !== channel || this.pending?.id !== id) { + return + } + this.abortChannel(new RuntimeClientError('accessibility_error', error.message)) + }) + }) + } + + private ensureChannel(): DesktopScriptServeChannel { + if (this.channel) { + return this.channel + } + this.childReady = false + this.childAnswered = false + const channel: DesktopScriptServeChannel = startServeChannel( + { + program: (this.options.powerShellPath ?? windowsPowerShellPath)(), + args: windowsPowerShellRuntimeArgs(this.scriptPath, this.availability.executionPolicy, [ + '-Serve' + ]), + env: process.env + }, + this.options.spawn ?? spawnProcess, + { + onLine: (line) => this.deliver(line), + // A replaced channel can still report; that must not fail the live one. + onGone: (detail) => { + if (this.channel === channel) { + this.handleGone(detail) + } + }, + onOverflow: () => + this.abortChannel( + new RuntimeClientError( + 'accessibility_error', + 'desktop provider response exceeded the runtime host buffer' + ) + ) + } + ) + this.channel = channel + return channel + } + + /** + * Whether the helper that just died can be proved not to have run the request. + * + * Why proof and not inference: "no reply came back" is not "nothing happened". + * runtime.ps1 synthesizes the input and only then builds the snapshot, which + * allocates a full-window bitmap and walks the UIA tree — a native fault there + * is uncatchable and would leave a click already delivered. Retrying on that + * inference turns one requested click into four. + */ + private mayReplay(request: BridgeRequest): boolean { + if (this.childReady || this.childAnswered) { + return false + } + return this.readyProtocolConfirmed || isReplayableTool(request.tool) + } + + private deliver(line: string): void { + let parsed: Record + try { + parsed = JSON.parse(line) as Record + } catch { + // Not a response at all — a PowerShell banner, a stray write. Dropping it + // is safe now that the id below is what decides which request is answered, + // and it keeps a chatty console from making the helper unusable. + return + } + // The readiness announcement carries no request id and answers nothing. + if (parsed.ready === true && parsed.requestId === undefined) { + this.childReady = true + this.readyProtocolConfirmed = true + this.availability.confirmExecutionPolicy() + return + } + const pending = this.pending + if (!pending || parsed.requestId !== pending.id) { + // One unmatched reply would otherwise shift every later response by one. + // Carry the helper's own message when it sent one: a line it could not tag + // with an id is usually the only account of what went wrong, and reporting + // a bare desync in its place loses the cause for good. + const reported = typeof parsed.error === 'string' ? `: ${parsed.error}` : '' + this.abortChannel( + new RuntimeClientError( + 'accessibility_error', + `desktop provider response did not match the pending request${reported}` + ) + ) + return + } + // Only a reply this host can prove is its own counts as the helper working. + this.childAnswered = true + this.pending = null + clearTimeout(pending.timer) + const { requestId: _echoed, ...response } = parsed + pending.resolve(response as BridgeResponse) + } + + private handleGone(detail: string): void { + const started = this.childReady || this.childAnswered + this.channel = null + this.availability.recordFailure() + if (!started && this.availability.atPreferredPolicy && isExecutionPolicyBlocked(detail)) { + this.availability.requestPolicyRetry() + // Unavailable rather than a generic error, because this can now be the + // final attempt: reverting an unproven escalation puts the host back on + // the preferred policy, so a later attempt can land here again. Only this + // code routes the operation to the one-shot bridge, which carries its own + // policy fallback; anything else fails the operation outright. + this.rejectPending(this.unavailableError(detail)) + return + } + if (!started) { + this.rejectPending(this.unavailableError(detail)) + return + } + this.rejectPending( + new RuntimeClientError( + 'accessibility_error', + `desktop provider runtime host exited: ${detail}` + ) + ) + } + + /** + * Stop a helper this host has judged unusable — a timeout, a desynchronised + * reply, an oversized line. + * + * Why it counts as a failure: stopping the channel suppresses the exit + * handler, so without this these paths bypassed the accounting entirely and a + * helper that failed this way on every operation was respawned once per + * operation forever — the burst this host exists to remove, restored through + * its own recovery path. + */ + private abortChannel(error: Error): void { + this.stopChannel() + this.availability.recordFailure() + this.availability.warn(`runtime host helper stopped: ${error.message}`) + this.rejectPending(error) + } + + private stopChannel(): void { + const channel = this.channel + this.channel = null + channel?.stop() + } + + private takePending(): PendingRequest | null { + const pending = this.pending + this.pending = null + if (pending) { + clearTimeout(pending.timer) + } + return pending + } + + private rejectPending(error: Error): void { + this.takePending()?.reject(error) + } + + private armIdleTimer(): void { + this.clearIdleTimer() + if (!this.channel) { + return + } + this.idleTimer = setTimeout(() => { + this.idleTimer = null + this.stopChannel() + }, this.idleShutdownMs) + this.idleTimer.unref?.() + } + + private clearIdleTimer(): void { + if (this.idleTimer) { + clearTimeout(this.idleTimer) + this.idleTimer = null + } + } + + private unavailableError(message: string): RuntimeClientError { + return new RuntimeClientError( + RUNTIME_HOST_UNAVAILABLE, + `desktop provider runtime host could not start: ${message}` + ) + } +} + +function errorText(error: unknown): string { + return error instanceof Error ? error.message : String(error) +} diff --git a/src/main/computer/desktop-script-runtime-host.win32.test.ts b/src/main/computer/desktop-script-runtime-host.win32.test.ts new file mode 100644 index 00000000000..76927a7dd65 --- /dev/null +++ b/src/main/computer/desktop-script-runtime-host.win32.test.ts @@ -0,0 +1,130 @@ +import { resolve } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import { spawnProcess } from '../../shared/child-process/run-process' +import { windowsPowerShellPath } from '../../shared/child-process/windows-system-binary' +import { DesktopScriptRuntimeHost } from './desktop-script-runtime-host' +import { startServeChannel } from './desktop-script-serve-channel' +import { + PREFERRED_WINDOWS_EXECUTION_POLICY, + windowsPowerShellRuntimeArgs +} from './windows-powershell-execution-policy' + +/** + * The other half of the serve-mode proof: the unit test drives a fake child, + * this one drives the real `runtime.ps1 -Serve` on a real Windows box. + * + * Both are needed. The framing that matters — one NDJSON line per response, + * megabyte-scale screenshot payloads, a console writer that actually flushes — + * only exists in PowerShell, and a fake child cannot disprove any of it. + * + * Runs only on win32; skipped elsewhere. + */ +const describeOnWindows = process.platform === 'win32' ? describe : describe.skip + +const SCRIPT_PATH = resolve(__dirname, '../../../native/computer-use-windows/runtime.ps1') + +describeOnWindows('runtime.ps1 serve mode', () => { + let host: DesktopScriptRuntimeHost | null = null + let spawns = 0 + + function startHost(): DesktopScriptRuntimeHost { + spawns = 0 + host = new DesktopScriptRuntimeHost(SCRIPT_PATH, { + warn: () => {}, + spawn: (spec) => { + spawns++ + return spawnProcess(spec) + } + }) + return host + } + + afterEach(() => { + host?.dispose() + host = null + }) + + it('answers repeated operations from a single PowerShell process', async () => { + const runtime = startHost() + + await expect(runtime.request({ tool: 'handshake' })).resolves.toMatchObject({ + ok: true, + capabilities: { protocolVersion: 1, provider: 'orca-computer-use-windows' } + }) + + const apps = await runtime.request({ tool: 'list_apps' }) + expect(apps.ok).toBe(true) + expect(Array.isArray(apps.apps)).toBe(true) + + await expect(runtime.request({ tool: 'handshake' })).resolves.toMatchObject({ ok: true }) + + expect(spawns).toBe(1) + }) + + it('returns a structured error for a bad request without killing the helper', async () => { + const runtime = startHost() + + await expect(runtime.request({ tool: 'not_a_tool' })).resolves.toMatchObject({ ok: false }) + await expect(runtime.request({ tool: 'handshake' })).resolves.toMatchObject({ ok: true }) + expect(spawns).toBe(1) + }) + + /** + * The host can only write well-formed JSON, so the parse-failure branch of the + * serve loop is unreachable through it. Driving the channel directly is the + * only way to prove what the real PowerShell answers. + */ + it('echoes the id it can recover when a request will not parse', async () => { + const answer = await answerRawLine('{"tool":"handshake","requestId":7') + + // Tagged, so the host resolves the waiting request with a failed operation + // instead of reading an untagged line as a desynchronised stream. + expect(answer).toMatchObject({ ok: false, requestId: 7 }) + expect(String(answer.error)).not.toBe('') + }) + + it('reports an error for a line with no recoverable id', async () => { + const answer = await answerRawLine('{"tool":"handshake"') + + expect(answer).toMatchObject({ ok: false }) + expect(answer.requestId).toBeUndefined() + expect(String(answer.error)).not.toBe('') + }) +}) + +/** One raw line into a real `runtime.ps1 -Serve`, and the line it writes back. */ +function answerRawLine(raw: string): Promise> { + return new Promise((settle, fail) => { + const channel = startServeChannel( + { + program: windowsPowerShellPath(), + args: windowsPowerShellRuntimeArgs(SCRIPT_PATH, PREFERRED_WINDOWS_EXECUTION_POLICY, [ + '-Serve' + ]), + env: process.env + }, + spawnProcess, + { + onLine: (line) => { + let parsed: Record + try { + parsed = JSON.parse(line) as Record + } catch { + return + } + if (parsed.ready === true) { + channel.write(`${raw}\n`, fail) + return + } + channel.stop() + settle(parsed) + }, + onGone: (detail) => fail(new Error(`helper exited before answering: ${detail}`)), + onOverflow: () => { + channel.stop() + fail(new Error('helper overflowed the response buffer')) + } + } + ) + }) +} diff --git a/src/main/computer/desktop-script-serve-channel.test.ts b/src/main/computer/desktop-script-serve-channel.test.ts new file mode 100644 index 00000000000..80a5dd491d3 --- /dev/null +++ b/src/main/computer/desktop-script-serve-channel.test.ts @@ -0,0 +1,99 @@ +import { EventEmitter } from 'node:events' +import { describe, expect, it, vi } from 'vitest' +import { DesktopScriptServeChannel, type RuntimeChildProcess } from './desktop-script-serve-channel' + +class FakeChild extends EventEmitter { + readonly stdout = new EventEmitter() + readonly stderr = new EventEmitter() + readonly writes: string[] = [] + killed = false + private readonly pendingWrites: ((error?: Error | null) => void)[] = [] + + readonly stdin = { + write: (chunk: string, callback?: (error?: Error | null) => void): boolean => { + this.writes.push(chunk) + if (callback) { + this.pendingWrites.push(callback) + } + return true + }, + end: (): void => {}, + on: (): void => {} + } + + kill(): boolean { + this.killed = true + return true + } + + /** What a destroyed stdin does to writes still queued at teardown. */ + failQueuedWrites(): void { + for (const callback of this.pendingWrites.splice(0)) { + callback(new Error('ERR_STREAM_DESTROYED')) + } + } +} + +function createChannel() { + const child = new FakeChild() + const handlers = { onLine: vi.fn(), onGone: vi.fn(), onOverflow: vi.fn() } + const channel = new DesktopScriptServeChannel(child as unknown as RuntimeChildProcess, handlers) + return { channel, child, handlers } +} + +describe('DesktopScriptServeChannel', () => { + it('splits responses into lines and tolerates a trailing carriage return', () => { + const { child, handlers } = createChannel() + + child.stdout.emit('data', Buffer.from('{"a":1}\r\n{"b":2}\n', 'utf8')) + + expect(handlers.onLine.mock.calls.map(([line]) => line)).toEqual(['{"a":1}', '{"b":2}']) + }) + + it('reports the exit reason with the stderr tail', () => { + const { child, handlers } = createChannel() + + child.stderr.emit('data', Buffer.from('it broke', 'utf8')) + child.emit('close', 1, null) + + expect(handlers.onGone).toHaveBeenCalledWith('code 1: it broke') + }) + + describe('once stopped', () => { + /** + * The channel's half of the stale-callback guard, pinned here rather than + * through the host: the host refuses a stale report too, so a host-level + * test passes with either guard alone and neither ends up covered. + */ + it('accepts no further writes', () => { + const { channel, child } = createChannel() + + channel.stop() + channel.write('{"tool":"click"}\n', vi.fn()) + + expect(child.writes).toEqual([]) + }) + + it('reports no error from a write that was already queued', () => { + const { channel, child } = createChannel() + const onError = vi.fn() + + channel.write('{"tool":"click"}\n', onError) + channel.stop() + child.failQueuedWrites() + + expect(onError).not.toHaveBeenCalled() + }) + + it('reports neither lines nor the exit it was asked to cause', () => { + const { channel, child, handlers } = createChannel() + + channel.stop() + child.stdout.emit('data', Buffer.from('{"a":1}\n', 'utf8')) + child.emit('close', 0, null) + + expect(handlers.onLine).not.toHaveBeenCalled() + expect(handlers.onGone).not.toHaveBeenCalled() + }) + }) +}) diff --git a/src/main/computer/desktop-script-serve-channel.ts b/src/main/computer/desktop-script-serve-channel.ts new file mode 100644 index 00000000000..afabb47962a --- /dev/null +++ b/src/main/computer/desktop-script-serve-channel.ts @@ -0,0 +1,145 @@ +import { StringDecoder } from 'node:string_decoder' +import type { ProcessSpec } from '../../shared/child-process/process-spec' +import type { spawnProcess } from '../../shared/child-process/run-process' + +/** The all-pipes child `spawnProcess` returns; avoids a node:child_process import. */ +export type RuntimeChildProcess = ReturnType + +export type RuntimeProcessSpawn = (spec: ProcessSpec) => RuntimeChildProcess + +/** UTF-16 units, not bytes — this bounds the buffer, it is not a payload contract. */ +const MAX_RESPONSE_CHARS = 20 * 1024 * 1024 +const MAX_STDERR_CHARS = 4096 + +export type ServeChannelHandlers = { + /** One complete line from the helper, without its terminator. */ + onLine: (line: string) => void + /** The helper is gone; detail carries the exit reason and its stderr tail. */ + onGone: (detail: string) => void + /** The helper produced more than one buffer's worth without a line break. */ + onOverflow: () => void +} + +/** + * One `runtime.ps1 -Serve` child, framed as NDJSON lines. + * + * Split from the host so the host reads as what it is — a queue, a retry policy + * and a correlation check — rather than that plus stream plumbing. Responses + * carry base64 screenshots and routinely exceed a megabyte, so lines are + * reassembled across chunks with a decoder that survives a code point split + * across a chunk boundary. + */ +export class DesktopScriptServeChannel { + private readonly decoder = new StringDecoder('utf8') + private buffer = '' + private stderrTail = '' + private detach: (() => void) | null = null + private closed = false + + constructor( + private readonly child: RuntimeChildProcess, + private readonly handlers: ServeChannelHandlers + ) { + const onStdout = (chunk: Buffer | string): void => this.readStdout(chunk) + const onStderr = (chunk: Buffer | string): void => { + this.stderrTail = `${this.stderrTail}${chunk.toString()}`.slice(-MAX_STDERR_CHARS) + } + // Why close and not exit: the caller classifies the failure from stderr, and + // only close guarantees the stdio streams were drained first. + const onClose = (code: number | null, signal: NodeJS.Signals | null): void => + this.reportGone(signal ? `signal ${signal}` : `code ${code ?? 'unknown'}`) + const onError = (error: Error): void => this.reportGone(error.message) + child.stdout.on('data', onStdout) + child.stderr.on('data', onStderr) + child.once('close', onClose) + child.once('error', onError) + // An unhandled stream error is an uncaught exception in the main process. + child.stdin.on('error', () => {}) + this.detach = (): void => { + child.stdout.off('data', onStdout) + child.stderr.off('data', onStderr) + child.off('close', onClose) + child.off('error', onError) + child.on('error', () => {}) + } + } + + write(payload: string, onError: (error: Error) => void): void { + if (this.closed) { + return + } + this.child.stdin.write(payload, (error) => { + // A destroyed stdin calls back after stop(); reporting then charges the + // caller a second failure for one operation. Deliberately redundant with + // the host's own staleness check — keep both, and note that each is + // pinned separately, this one by the "once stopped" tests here. + if (error && !this.closed) { + onError(error) + } + }) + } + + /** Stop the helper and go silent; handlers are not called afterwards. */ + stop(): void { + if (this.closed) { + return + } + this.closed = true + this.detach?.() + this.detach = null + this.buffer = '' + // Closing stdin ends the serve loop; the kill covers a wedged helper. + try { + this.child.stdin.end() + } catch { + /* already closed */ + } + this.child.kill() + } + + private reportGone(detail: string): void { + if (this.closed) { + return + } + const text = [detail, this.stderrTail.trim()].filter(Boolean).join(': ') + this.closed = true + this.detach?.() + this.detach = null + this.handlers.onGone(text) + } + + private readStdout(chunk: Buffer | string): void { + if (this.closed) { + return + } + this.buffer += typeof chunk === 'string' ? chunk : this.decoder.write(chunk) + if (this.buffer.length > MAX_RESPONSE_CHARS) { + this.buffer = '' + this.handlers.onOverflow() + return + } + for (let newline = this.buffer.indexOf('\n'); newline >= 0;) { + // Slice a trailing CR off by index; trimming copies the whole payload. + const end = newline > 0 && this.buffer.charCodeAt(newline - 1) === 13 ? newline - 1 : newline + const line = this.buffer.slice(0, end) + this.buffer = this.buffer.slice(newline + 1) + if (line.length > 0) { + this.handlers.onLine(line) + // A handler may have stopped this channel; stop reading its backlog. + if (this.closed) { + this.buffer = '' + return + } + } + newline = this.buffer.indexOf('\n') + } + } +} + +export function startServeChannel( + spec: ProcessSpec, + spawn: RuntimeProcessSpawn, + handlers: ServeChannelHandlers +): DesktopScriptServeChannel { + return new DesktopScriptServeChannel(spawn(spec), handlers) +} diff --git a/src/main/computer/sidecar-client.ts b/src/main/computer/sidecar-client.ts index 489c463dd93..23af1aa603b 100644 --- a/src/main/computer/sidecar-client.ts +++ b/src/main/computer/sidecar-client.ts @@ -9,6 +9,7 @@ import type { ComputerSnapshotResult } from '../../shared/runtime-types' import { normalizeComputerActionResult } from './computer-action-verification-normalization' +import { isComputerSidecarDiagnostic, logComputerDiagnostic } from './computer-sidecar-diagnostics' import { validateComputerSidecarPasteText } from './computer-sidecar-paste-validation' import { RuntimeClientError } from './runtime-client-error' @@ -245,6 +246,11 @@ class ComputerSidecarProcess { } private handleMessage(message: unknown): void { + // The sidecar's stdio is piped and unread, so its warnings arrive here. + if (isComputerSidecarDiagnostic(message)) { + logComputerDiagnostic(message.message) + return + } if (!isSidecarResponse(message)) { return } diff --git a/src/main/computer/sidecar-entry.ts b/src/main/computer/sidecar-entry.ts index 8489f71e7f3..961d2261ede 100644 --- a/src/main/computer/sidecar-entry.ts +++ b/src/main/computer/sidecar-entry.ts @@ -8,6 +8,11 @@ type SidecarRequest = { params?: Record } +// Why disconnect carries the weight on Windows: the parent stops the sidecar +// with kill('SIGTERM'), which is TerminateProcess there, so the SIGTERM handler +// below never runs and teardown rides on the IPC channel closing instead. A +// helper wedged inside a UI Automation call can still outlive that and deliver +// input after teardown; only a real signal would preempt it. process.once('disconnect', shutdownProviders) process.once('SIGTERM', () => { shutdownProviders() diff --git a/src/main/computer/windows-powershell-execution-policy.test.ts b/src/main/computer/windows-powershell-execution-policy.test.ts new file mode 100644 index 00000000000..033eec58fb5 --- /dev/null +++ b/src/main/computer/windows-powershell-execution-policy.test.ts @@ -0,0 +1,92 @@ +import { describe, expect, it } from 'vitest' +import { + FALLBACK_WINDOWS_EXECUTION_POLICY, + PREFERRED_WINDOWS_EXECUTION_POLICY, + isExecutionPolicyBlocked, + windowsPowerShellRuntimeArgs +} from './windows-powershell-execution-policy' + +/** + * Captured from powershell.exe on Windows, verbatim including the hard wrapping. + * + * The discriminator has to be pinned in both directions: a policy block must + * escalate once, and a plain access denial must not, because escalation is + * sticky for the session and lands on `-ExecutionPolicy Bypass`. + */ +const POLICY_BLOCKED_RESTRICTED = [ + 'File C:\\Temp\\runtime.ps1 cannot be loaded because running scripts is disabled on this system. For more ', + 'information, see about_Execution_Policies at https:/go.microsoft.com/fwlink/?LinkID=135170.', + ' + CategoryInfo : SecurityError: (:) [], ParentContainsErrorRecordException', + ' + FullyQualifiedErrorId : UnauthorizedAccess' +].join('\r\n') + +const POLICY_BLOCKED_REMOTE_SIGNED = [ + 'File C:\\Temp\\runtime.ps1 cannot be loaded. The file ', + 'C:\\Temp\\runtime.ps1 is not digitally signed. You cannot run this script on the current system. For more ', + 'information about running scripts and setting execution policy, see about_Execution_Policies at https:/go.microsoft.com/fwlink/?LinkID=135170.', + ' + CategoryInfo : SecurityError: (:) [], ParentContainsErrorRecordException', + ' + FullyQualifiedErrorId : UnauthorizedAccess' +].join('\r\n') + +/** No execution policy involved: .NET refusing a file the process may not read. */ +const GENUINE_ACCESS_DENIED = [ + 'Exception calling "ReadAllText" with "1" argument(s): "Access to the path \'C:\\Windows\\System32\\config\\SAM\' is denied."', + 'At C:\\Temp\\runtime.ps1:1 char:1', + '+ [System.IO.File]::ReadAllText("C:\\Windows\\System32\\config\\SAM")', + '+ ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~', + ' + CategoryInfo : NotSpecified: (:) [], MethodInvocationException', + ' + FullyQualifiedErrorId : UnauthorizedAccessException' +].join('\r\n') + +describe('isExecutionPolicyBlocked', () => { + it('recognises a policy block under either policy', () => { + expect(isExecutionPolicyBlocked(POLICY_BLOCKED_RESTRICTED)).toBe(true) + expect(isExecutionPolicyBlocked(POLICY_BLOCKED_REMOTE_SIGNED)).toBe(true) + }) + + it('does not read a plain access denial as a policy block', () => { + // UnauthorizedAccessException merely starts with the policy error id. Without + // the word boundary this matched, and one locked file downgraded the whole + // session to Bypass with no path back. + expect(isExecutionPolicyBlocked(GENUINE_ACCESS_DENIED)).toBe(false) + }) + + it('keeps recognising a block when the record labels are localized', () => { + // The labels are translated on a non-English host; the ids and the help + // topic are not, so the match must not depend on the labels. + const localized = POLICY_BLOCKED_RESTRICTED.replace('CategoryInfo', 'Categoria') + .replace('FullyQualifiedErrorId', 'IdErroreCompleto') + .replace( + 'cannot be loaded because running scripts is disabled on this system', + 'non puo essere caricato' + ) + expect(isExecutionPolicyBlocked(localized)).toBe(true) + }) + + it('ignores the failures the helper reports every day', () => { + expect(isExecutionPolicyBlocked('code 1: The term is not recognized')).toBe(false) + expect(isExecutionPolicyBlocked('Add-Type : Cannot access the temporary directory')).toBe(false) + expect(isExecutionPolicyBlocked('')).toBe(false) + }) +}) + +describe('windowsPowerShellRuntimeArgs', () => { + it('never emits Bypass unless the caller escalated to it', () => { + const preferred = windowsPowerShellRuntimeArgs( + 'C:\\orca\\runtime.ps1', + PREFERRED_WINDOWS_EXECUTION_POLICY, + ['-Serve'] + ) + expect(preferred).not.toContain(FALLBACK_WINDOWS_EXECUTION_POLICY) + expect(preferred).toEqual([ + '-NoLogo', + '-NoProfile', + '-NonInteractive', + '-ExecutionPolicy', + 'RemoteSigned', + '-File', + 'C:\\orca\\runtime.ps1', + '-Serve' + ]) + }) +}) diff --git a/src/main/computer/windows-powershell-execution-policy.ts b/src/main/computer/windows-powershell-execution-policy.ts new file mode 100644 index 00000000000..204f5b02b06 --- /dev/null +++ b/src/main/computer/windows-powershell-execution-policy.ts @@ -0,0 +1,59 @@ +/** + * Execution-policy handling for the Windows computer-use runtime script. + * + * Why not `Bypass` outright: it is the highest-weighted token on a + * powershell.exe command line for Defender for Endpoint, and the shipped + * runtime.ps1 does not need it — NSIS extraction writes no Zone.Identifier, so + * an unsigned local script runs under `RemoteSigned`. `Restricted` is still the + * Windows client default though, so a policy-blocked start must fall back once + * rather than leaving computer use broken. + */ +export type WindowsExecutionPolicy = 'RemoteSigned' | 'Bypass' + +export const PREFERRED_WINDOWS_EXECUTION_POLICY: WindowsExecutionPolicy = 'RemoteSigned' +export const FALLBACK_WINDOWS_EXECUTION_POLICY: WindowsExecutionPolicy = 'Bypass' + +/** + * Matches the SecurityError PowerShell emits for `-File` under a blocking policy. + * + * Every alternative is a PowerShell or .NET identifier, never prose. The prose + * differs by policy ("running scripts is disabled" under Restricted, "is not + * digitally signed" under RemoteSigned), is localized, and PowerShell hard-wraps + * it mid-sentence at the console width, so it can anchor nothing. + * + * The `\b` after UnauthorizedAccess is the whole discriminator and must not be + * dropped. `UnauthorizedAccess` is the FullyQualifiedErrorId of a policy block, + * but it is also a strict prefix of `UnauthorizedAccessException`, which .NET + * raises for an ordinary locked or ACL-denied file: an AV scan holding + * runtime.ps1, a locked CSC temp directory, a roaming-profile hiccup. Matching + * that escalates to `Bypass` for the rest of the session — the exact command + * line token this stack exists to stop emitting — and on the one-shot path + * replays an operation that already ran. + * + * Anchoring on the `FullyQualifiedErrorId:`/`CategoryInfo:` labels would be more + * precise still, but the labels are localized where these values are not, so a + * non-English host would stop recognising a real block and lose the fallback. + */ +const EXECUTION_POLICY_BLOCKED = /\bUnauthorizedAccess\b|\bSecurityError\b|about_Execution_Policies/ + +export function isExecutionPolicyBlocked(text: string): boolean { + return EXECUTION_POLICY_BLOCKED.test(text) +} + +export function windowsPowerShellRuntimeArgs( + scriptPath: string, + policy: WindowsExecutionPolicy, + scriptArgs: readonly string[] = [] +): string[] { + return [ + // -NoLogo: a banner on stdout would be read as a malformed response line. + '-NoLogo', + '-NoProfile', + '-NonInteractive', + '-ExecutionPolicy', + policy, + '-File', + scriptPath, + ...scriptArgs + ] +} From cff202c16a79bbcd3d24cb7ab62abf2898449a23 Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Sat, 5 Sep 2026 21:12:40 -0700 Subject: [PATCH 40/69] fix(windows): drop EDR-flagged -ExecutionPolicy Bypass from encoded PowerShell (#17880) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(windows): drop EDR-flagged -ExecutionPolicy Bypass from encoded PowerShell MDE flags `-ExecutionPolicy Bypass` paired with base64 `-EncodedCommand` as a behavioural signal. Measured on Windows 11: neither `-Command` nor `-EncodedCommand` is execution-policy gated (both run under an explicit `-ExecutionPolicy Restricted` and `AllSigned`; only `-File` fails), so the switch was a pure no-op on every one of these command lines. Removes the switch from all four sites that spelled it, and de-encodes the one site whose payload never passes through a re-parsing shell: - ssh-remote-powershell: one chokepoint for ~40 remote-Windows call sites. Base64 kept — the remote sshd DefaultShell re-parses this string. - setup-agent-sequencing / windows-cmd-runner-delayed-launch: base64 kept — these strings are typed into a terminal pane. - windows-interactive-login-spawn: base64 kept — `cmd.exe /c start` re-parses, and the cmd-safe-token guard rejects the `&` and `"` in the raw relay script. - windows-mobile-firewall local runner: `-EncodedCommand` -> `-Command`, since execFile reaches CreateProcess with no shell in between. The setup startup gate keeps execution-policy relief in-payload (process scope), because it evals a user-authored startup command that may invoke a `.ps1`, and a `.ps1` IS gated. Caught by the real-process suite; mirrors the agent-hooks launcher's trade. The elevated firewall child deliberately stays encoded: `Start-Process -ArgumentList` joins its array into one ShellExecuteEx string without quoting and PowerShell re-splits on whitespace, measured to collapse `C:\My App\...` to `C:\My App\...` — a firewall rule for the wrong program. * test(ssh): enforce the no-script-file invariant remote payloads rely on Dropping `-ExecutionPolicy Bypass` from `powerShellCommand` is a no-op only while no remote payload loads a PowerShell script file — execution policy has never gated anything else. That invariant held by inspection and was guarded by nothing, so a future payload that dot-sourced, used `-File`, or imported a `.psm1` would break only on a remote host with a Restricted/AllSigned LocalMachine policy and no GPO: a failure on someone else's machine. States the invariant at the wrapper, and adds a ratchet that scans every module importing it for `.ps1`/`.psm1`, `Import-Module`, `-File`, and dot-sourcing. The scan discovers importers itself (13 today) so new ones are covered, and asserts it found some, so an emptied list cannot pass vacuously. Mutation-checked: injecting each construct into a real importer fails the matching case and names the file. The first dot-source pattern passed a `;`-prefixed sample but missed `powerShellCommand(". '$x'")` — the likelier shape — so the pattern now accepts a string-literal start and the self-test samples carry their surrounding quotes. * test(ssh): close two blind spots in the remote-payload ratchet Both found by independent mutation testing of the ratchet itself, and both let a real violation pass while the guard reported green. `-File` was matched case-sensitively, so `-file $scriptVar` slipped through — PowerShell switches are case-insensitive, and with a variable path the `.ps1` pattern does not cover for it, so that shape escaped both nets. The naive fix is wrong: bare /-File\b/i matches `--credential-file`, `--log-file` and `--body-file`, which occur in three of these importers. Anchoring to a token boundary catches the lowercase, odd-spacing and argv-element forms with zero offenders across all 14. Comment stripping paired a `/*` appearing inside a string (a glob such as 'src/*.ts') with any later comment close and deleted everything between, hiding violations in the gap. Anchoring the block strip to line start, as the `//` strip already was, fixes it — verified by injecting an `Import-Module` after a glob string: the unanchored form misses it, the anchored form catches it. Extends the same case-insensitivity to `.ps1`/`.psm1` and `Import-Module`, which had the identical flaw (`import-module`, `DEPLOY.PS1` are legitimate spellings); measured to add no false positive. Each construct now carries the fixtures it must catch AND the near-misses it must not, so a future tightening cannot quietly trade one for the other — the negative fixtures are what would have caught the naive `-File` fix. Non-vacuity bound tightened to >10 against 14 importers. * docs(ssh): state what the remote-payload ratchet cannot see The scan matches source text, so a script file reached only through a variable (`& $scriptPath`) never appears in source and no pattern can catch it. The ratchet narrows the hole; the invariant note on `powerShellCommand` covers the remainder. Recorded because a guard that reads as complete coverage when it is not is worse than one that states its edge: the next author trusts it further than it deserves, and should learn this limit from the test rather than an incident. * test(ssh): scan remote payloads with the shared source walk The ratchet had its own tree walk and comment stripper. The walk skipped neither node_modules/dist/.git nor dot-directories and excluded tests by `.test.ts` alone, so its importer count -- the guard's own goalpost -- could be wrong about what it scanned. The stripper was anchored to line start to dodge a `/*` inside a glob string, which silently skipped trailing comments; `stripComments` tracks quote state and handles both. Importer set re-derived against the shared walk: 15, floor unchanged at 10. * fix(setup): report a failed execution-policy relief instead of swallowing it The in-payload Set-ExecutionPolicy carried -ErrorAction SilentlyContinue and an empty catch, so any failure vanished. A Windows PowerShell 5.1 install with duplicate extended type data fails every cmdlet in Microsoft.PowerShell.Security -- autoload, not policy -- and the user then saw only their own .ps1 being refused, with no trace that the relief had been attempted or why. -ErrorAction Stop is what routes a non-terminating failure into the catch at all; the catch reports the FullyQualifiedErrorId to stderr and deliberately does not rethrow, so a broken policy cmdlet cannot take down the startup this gate exists to run. Success path is unchanged and stays stderr-clean. Verified by execution on a clean child environment: success -> policy=Bypass, stderr empty; shadowed failing cmdlet -> diagnostic on stderr and the gate still continues; the old empty catch -> silent. --------- Co-authored-by: Orca Worker --- .../runtime/windows-mobile-firewall.test.ts | 54 ++++++ src/main/runtime/windows-mobile-firewall.ts | 9 +- src/main/ssh/ssh-remote-powershell.test.ts | 164 ++++++++++++++++++ src/main/ssh/ssh-remote-powershell.ts | 23 ++- src/shared/setup-agent-sequencing.test.ts | 31 +++- src/shared/setup-agent-sequencing.ts | 26 ++- src/shared/setup-runner-command.test.ts | 4 +- .../windows-cmd-runner-delayed-launch.test.ts | 36 ++++ .../windows-cmd-runner-delayed-launch.ts | 5 +- .../windows-interactive-login-spawn.test.ts | 26 +-- src/shared/windows-interactive-login-spawn.ts | 6 +- 11 files changed, 363 insertions(+), 21 deletions(-) create mode 100644 src/main/ssh/ssh-remote-powershell.test.ts create mode 100644 src/shared/windows-cmd-runner-delayed-launch.test.ts diff --git a/src/main/runtime/windows-mobile-firewall.test.ts b/src/main/runtime/windows-mobile-firewall.test.ts index 19109a444b8..d561878a538 100644 --- a/src/main/runtime/windows-mobile-firewall.test.ts +++ b/src/main/runtime/windows-mobile-firewall.test.ts @@ -1,4 +1,5 @@ import { describe, expect, it, vi } from 'vitest' +import { execFile } from 'node:child_process' import { getWebSocketPort, inspectWindowsMobileFirewall, @@ -6,6 +7,9 @@ import { type WindowsMobileFirewallEnvironment } from './windows-mobile-firewall' +// Why: every other case injects `runPowerShell`, so only the argv case below reaches execFile. +vi.mock('node:child_process', () => ({ execFile: vi.fn() })) + function environment( runPowerShell: WindowsMobileFirewallEnvironment['runPowerShell'], overrides: Partial = {} @@ -179,6 +183,56 @@ describe('windows mobile firewall', () => { expect(repairScript).toContain('-EdgeTraversalPolicy Block') }) + it('keeps the elevated child encoded because Start-Process re-splits its ArgumentList', async () => { + // Why: `Start-Process -ArgumentList` joins the array into one ShellExecuteEx parameter + // string without quoting and PowerShell re-splits it on whitespace, which collapses runs + // of spaces. Measured on Windows 11: a `-Command` payload turned `C:\My App\Orca.exe` + // into `C:\My App\Orca.exe`, i.e. a firewall rule for the wrong program. Base64 is the + // only form that survives that hop, so this site must not follow the local runner. + const runPowerShell = vi.fn().mockResolvedValue('{"launched":true,"exitCode":0}') + await repairWindowsMobileFirewall( + 6769, + environment(runPowerShell, { executablePath: 'C:\\My App\\Orca.exe' }) + ) + + const outerScript = runPowerShell.mock.calls[0]![0] as string + expect(outerScript).toContain("'-EncodedCommand'") + expect(outerScript).not.toContain("'-Command'") + expect(outerScript).toContain('-Verb RunAs') + + const encoded = outerScript.match(/'-EncodedCommand', '([^']+)'/)?.[1] + const repairScript = Buffer.from(encoded!, 'base64').toString('utf16le') + expect(repairScript).toContain("-Program 'C:\\My App\\Orca.exe'") + }) + + it('runs the local PowerShell over argv with a plain -Command script', async () => { + // Why: execFile reaches CreateProcess with no shell in between, so the script needs no + // base64 armouring, and argv preserves runs of spaces that the elevated hop cannot. + // `-EncodedCommand` here was pure EDR signal. + const execFileMock = vi.mocked(execFile) + execFileMock.mockImplementation(((_file, _args, _options, callback) => { + callback(null, '{"privateFirewallEnabled":true,"networkCategory":"Private"}', '') + return {} + }) as unknown as typeof execFile) + + await inspectWindowsMobileFirewall(6768, undefined, { + platform: 'win32', + isPackaged: true, + executablePath: 'C:\\My App\\Orca.exe', + systemRoot: 'C:\\Windows' + }) + + const [file, args] = execFileMock.mock.calls[0]! + expect(file).toMatch(/WindowsPowerShell\\v1\.0\\powershell\.exe$/i) + expect(args!.slice(0, 3)).toEqual(['-NoProfile', '-NonInteractive', '-Command']) + expect(args).not.toContain('-EncodedCommand') + expect(args).not.toContain('-ExecutionPolicy') + // The script travels as ONE argv element, so its spaces and newlines survive verbatim. + expect(args).toHaveLength(4) + expect(args![3]).toContain("-Program 'C:\\My App\\Orca.exe'") + expect(args![3]).toContain('\n') + }) + it('distinguishes a cancelled UAC prompt from repair failure', async () => { await expect( repairWindowsMobileFirewall( diff --git a/src/main/runtime/windows-mobile-firewall.ts b/src/main/runtime/windows-mobile-firewall.ts index 88b885ab5f0..c90a025cb14 100644 --- a/src/main/runtime/windows-mobile-firewall.ts +++ b/src/main/runtime/windows-mobile-firewall.ts @@ -229,6 +229,11 @@ Get-NetFirewallRule -Name ${quotePowerShell(FIREWALL_RULE_NAME)} -ErrorAction Si New-NetFirewallRule -Name ${quotePowerShell(FIREWALL_RULE_NAME)} -DisplayName ${quotePowerShell(FIREWALL_RULE_DISPLAY_NAME)} -Description 'Allows Orca Mobile to connect to this Orca desktop on private networks.' -Direction Inbound -Action Allow -Enabled True -Profile Private -Protocol TCP -LocalPort ${port} -Program ${quotePowerShell(executablePath)} -EdgeTraversalPolicy Block | Out-Null` } +// Why the elevated child keeps `-EncodedCommand` while the local runner does not: `Start-Process +// -ArgumentList` joins its array into one ShellExecuteEx parameter string without quoting, and +// PowerShell then re-splits it on whitespace — measured to collapse `C:\My App\...` to +// `C:\My App\...`, which would silently write the firewall rule for the wrong program. Node's +// argv path (createPowerShellRunner) preserves runs of spaces, so only this hop needs base64. function buildElevationScript(powershellPath: string, encodedRepairScript: string): string { return `$ErrorActionPreference = 'Stop' try { @@ -257,7 +262,9 @@ function createPowerShellRunner(systemRoot?: string): PowerShellRunner { new Promise((resolve, reject) => { execFile( powershellPath, - ['-NoProfile', '-NonInteractive', '-EncodedCommand', encodePowerShell(script)], + // Why: argv reaches CreateProcess with no shell in between, so the script needs no base64 + // armouring — and plain `-Command` keeps this off EDR's encoded-PowerShell heuristics. + ['-NoProfile', '-NonInteractive', '-Command', script], { encoding: 'utf8', timeout: timeoutMs, windowsHide: true, maxBuffer: 1024 * 1024 }, (error, stdout) => { if (error) { diff --git a/src/main/ssh/ssh-remote-powershell.test.ts b/src/main/ssh/ssh-remote-powershell.test.ts new file mode 100644 index 00000000000..f6ba6136378 --- /dev/null +++ b/src/main/ssh/ssh-remote-powershell.test.ts @@ -0,0 +1,164 @@ +import { describe, expect, it } from 'vitest' +import { join } from 'node:path' +import { scanSourceTree, stripComments } from '../../shared/source-scan/source-tree-scan' +import { powerShellCommand } from './ssh-remote-powershell' + +function decodePayload(command: string): string { + const encoded = command.match(/ -EncodedCommand (\S+)$/)?.[1] + if (!encoded) { + throw new Error(`no -EncodedCommand payload in: ${command}`) + } + return Buffer.from(encoded, 'base64').toString('utf16le') +} + +/** + * This one helper builds the command line for every remote-Windows SSH call site + * (relay deploy, install locks, upload staging, GC claim, browse, CLI launch), so + * its switches are worth pinning. + */ +describe('powerShellCommand', () => { + it('spells no -ExecutionPolicy switch', () => { + const command = powerShellCommand('exit 0') + const switches = command.replace(/ -EncodedCommand \S+$/, '') + + // Why: `-EncodedCommand` is not execution-policy gated — only `-File` is — so the switch + // was a no-op, and `-ExecutionPolicy Bypass` beside base64 is among the most heavily + // EDR-flagged PowerShell command lines there is. + expect(switches).not.toMatch(/-ExecutionPolicy/i) + expect(switches).not.toMatch(/Bypass/i) + expect(switches).toBe('powershell.exe -NoProfile -NonInteractive') + }) + + it('keeps the base64 payload the remote shell cannot rewrite', () => { + // Why: this string is re-parsed by the remote host's sshd DefaultShell, which is + // cmd.exe on a stock Windows OpenSSH install. Base64 is load-bearing here. + const command = powerShellCommand("Write-Output 'a & b' | Out-String") + + expect(command).toMatch( + /^powershell\.exe -NoProfile -NonInteractive -EncodedCommand [A-Za-z0-9+/=]+$/ + ) + expect(decodePayload(command)).toBe("Write-Output 'a & b' | Out-String") + }) +}) + +const MAIN_DIR = join(import.meta.dirname, '..') + +// Why: execution policy gates loading script FILES and nothing else, so dropping +// `-ExecutionPolicy Bypass` is a no-op exactly while no remote payload loads one. That +// invariant is what makes the switch safe to omit, and it was previously guarded by nothing: +// a future payload that dot-sourced or used `-File` would fail only on a remote host whose +// LocalMachine policy is Restricted/AllSigned. See the invariant note on `powerShellCommand`. +// Every pattern is case-insensitive: PowerShell switches and cmdlet names are, and Windows +// paths are, so `-file`, `import-module` and `DEPLOY.PS1` are all legitimate spellings that a +// case-sensitive pattern would wave through. Verified to add no false positive across the real +// importers. Each entry carries the fixtures it must catch AND the near-misses it must not, so +// a future tightening cannot quietly trade one for the other. +// +// Limit: this is a source-text scan, so a script file reached only through a variable +// (`& $scriptPath`) never appears in source and no pattern here can catch it — this narrows +// the hole rather than sealing it. The invariant note on `powerShellCommand` covers the rest. +// +// Matched against `stripComments`, the shared quote-tracking stripper, so a construct named in +// prose is not counted as code. A line-anchored regex pair cannot do this job: it either eats +// live code by pairing a `/*` inside a glob string with a later comment close, or — anchoring +// to avoid that — skips every trailing comment. Quote state is the only fix. +const POLICY_GATED_CONSTRUCTS = [ + { + label: 'a PowerShell script file (.ps1/.psm1)', + pattern: /\.psm?1\b/i, + catches: [ + `powerShellCommand("$script = 'C:\\tools\\deploy.ps1'")`, + `powerShellCommand("Import-Module '$dir\\orca.psm1'")`, + `powerShellCommand("& '$root\\DEPLOY.PS1'")` + ], + ignores: [`const build = 'artifact.ps10'`] + }, + { + label: 'Import-Module', + pattern: /\bImport-Module\b/i, + catches: [ + `powerShellCommand("Import-Module 'NetSecurity'")`, + `powerShellCommand("import-module $modulePath")` + ], + ignores: [`const name = 'Import-ModuleList'`] + }, + { + // Anchored to a token boundary: a bare /-File\b/i also matches `--credential-file`, + // `--log-file` and `--body-file`, which are real arguments in three of these importers. + label: 'the -File switch', + pattern: /(^|[\s'"`([{,])-File\b/i, + catches: [ + `runRemote("powershell.exe -NoProfile -File 'C:\\x.ps1'")`, + `runRemote("powershell.exe -file $scriptVar")`, + `runRemote(["-NoProfile", "-File", scriptVar])` + ], + ignores: [`fetchWith("--credential-file", path)`, `run("--log-file $p --body-file $b")`] + }, + { + // The quote/backtick prefixes matter: a dot-source in a generated payload usually sits at + // the very start of a TS string literal — `powerShellCommand(". '$x'")` — not after a `;`. + label: 'dot-sourcing', + pattern: /(^|[;{'"`]|\n)[ \t]*\.[ \t]+['"$]/, + catches: [ + `powerShellCommand(". '$profileScript'")`, + `powerShellCommand("$ErrorActionPreference = 'Stop'; . '$profile'")`, + `powerShellCommand(". $profileScript")` + ], + ignores: [ + `cp -a $sourcePath/. $destinationPath/`, + `Host key verification failed for $displayHost. $detail` + ] + } +] as const + +describe('remote PowerShell payload invariant', () => { + // `scanSourceTree` is the shared walk: it skips node_modules/dist/out/build/.git, + // dot-directories and `__fixtures__`, and excludes tests by the shared `isTestFile` (which + // also covers `.spec.ts`, `__tests__/` and `-test-harness.ts`). A hand-rolled walk that got + // any of those wrong would move the floor below, which is this guard's own goalpost. + const importers = scanSourceTree(MAIN_DIR).filter((file) => + file.source.includes('ssh-remote-powershell') + ) + + it('finds the modules that build remote payloads', () => { + // Guards the scan itself: a resolution change that emptied this list would make every + // assertion below vacuously pass. 15 importers today, re-derived against the shared walk. + expect(importers.length).toBeGreaterThan(10) + }) + + it.each(POLICY_GATED_CONSTRUCTS)('loads no remote payload through $label', ({ + label, + pattern + }) => { + const offenders = importers + .filter((file) => pattern.test(stripComments(file.source))) + .map((file) => file.relativePath) + + expect( + offenders, + `${offenders.join(', ')} uses ${label}, which IS execution-policy gated on the remote ` + + 'host. Do not restore `-ExecutionPolicy Bypass` to the command line (a GPO scope ' + + 'beats it). Set the policy in-payload at process scope instead — see the note on ' + + 'powerShellCommand.' + ).toEqual([]) + }) + + // Why: these patterns only earn trust if they fire on a real violation spelled the way a + // generated payload spells it — inside a TS string literal — and stay quiet on the near + // misses. Both halves are load-bearing: an earlier dot-source pattern passed a `;`-prefixed + // sample but missed `powerShellCommand(". '$x'")`, and the obvious case-insensitive fix for + // `-File` matches `--credential-file` in three real importers. A fixture written from the + // pattern confirms the pattern; these are written from the requirement. + it.each(POLICY_GATED_CONSTRUCTS)('detects $label wherever it is spelled', ({ + pattern, + catches, + ignores + }) => { + for (const sample of catches) { + expect(pattern.test(sample), `should catch: ${sample}`).toBe(true) + } + for (const sample of ignores) { + expect(pattern.test(sample), `should ignore: ${sample}`).toBe(false) + } + }) +}) diff --git a/src/main/ssh/ssh-remote-powershell.ts b/src/main/ssh/ssh-remote-powershell.ts index 420223ced29..31587bcc668 100644 --- a/src/main/ssh/ssh-remote-powershell.ts +++ b/src/main/ssh/ssh-remote-powershell.ts @@ -18,6 +18,27 @@ const WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS = 8_000 */ export type WindowsPowerShellExecutable = 'powershell.exe' | 'pwsh.exe' +// Why: `-EncodedCommand` is not execution-policy gated (only `-File` is), so `-ExecutionPolicy +// Bypass` was a no-op here — and it is one of the most heavily EDR-flagged PowerShell tokens. +// The base64 stays: this string is re-parsed by the remote host's default SSH shell, which may +// be cmd.exe, PowerShell, or bash. +// +// INVARIANT — no remote payload may load a PowerShell *script file*. +// +// Execution policy has only ever gated loading script files (2.0 through 7.x). Inline +// statements, `& some.exe` and `Add-Type -TypeDefinition` are never gated, which is what makes +// dropping the switch a no-op for every payload we send today — the compressed path below stays +// inline too, since `Invoke-Expression` on a decompressed string loads no file. Loading a script +// file is the one thing the dropped switch actually covered, so a payload that dot-sources, runs +// `& '.ps1'`, calls `Import-Module '.psm1'`, or passes `-File` would silently fail on a +// remote host whose LocalMachine policy is Restricted/AllSigned with no GPO — a break that +// surfaces on someone else's machine, not ours. +// +// If you ever need one, do NOT restore the command-line switch (it loses to a GPO scope anyway, +// so it never covered the locked-down case): set the policy in-payload at process scope, the way +// `buildWindowsStartupCommand` in src/shared/setup-agent-sequencing.ts does. +// +// Enforced by the ratchet in ssh-remote-powershell.test.ts, which scans every importer. export function powerShellCommand( script: string, executable: WindowsPowerShellExecutable = 'powershell.exe' @@ -38,7 +59,7 @@ export function powerShellCommand( } function encodedPowerShellCommand(script: string, executable: WindowsPowerShellExecutable): string { - return `${executable} -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}` + return `${executable} -NoProfile -NonInteractive -EncodedCommand ${encodePowerShellCommand(script)}` } /** Orca-prefixed names so the payload can never shadow the bootstrap's own state. */ diff --git a/src/shared/setup-agent-sequencing.test.ts b/src/shared/setup-agent-sequencing.test.ts index fd567c145b0..1b8d69f0aa5 100644 --- a/src/shared/setup-agent-sequencing.test.ts +++ b/src/shared/setup-agent-sequencing.test.ts @@ -271,13 +271,13 @@ describe('createSequencedSetupAgentCommands', () => { const startupPowerShell = decodePowerShellScript(result.startupCommand) expect(result.setupCommand).toContain( - 'powershell.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand' + 'powershell.exe -NoProfile -NonInteractive -EncodedCommand' ) expect(setupPowerShell).toContain("$runner = 'C:\\repo\\.git\\orca\\setup-runner.cmd'") expect(setupPowerShell).toContain('$nonce + ":" + $setupStatus') expect(result.startupCommand.match(/powershell\.exe/g)).toHaveLength(1) expect(result.startupCommand).toContain( - 'powershell.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand' + 'powershell.exe -NoProfile -NonInteractive -EncodedCommand' ) expect(startupPowerShell).toContain('AddSeconds(3)') expect(startupPowerShell).toContain('Missing setup marker path.') @@ -292,6 +292,31 @@ describe('createSequencedSetupAgentCommands', () => { expect(result.startupEnv).toEqual({ [SETUP_AGENT_SEQUENCE_STARTUP_COMMAND_ENV]: "codex --model gpt-5 'fix !PATH! & test'" }) + // Why: `-EncodedCommand` is not execution-policy gated — only `-File` is — so the switch + // was a no-op, and base64 beside `-ExecutionPolicy Bypass` is a heavily EDR-flagged shape. + // The base64 itself must stay: these strings are typed into a terminal pane. + expect(result.setupCommand).not.toMatch(/-ExecutionPolicy/i) + expect(result.startupCommand).not.toMatch(/-ExecutionPolicy/i) + // Why: dropping the switch alone would break a user startup command that invokes a + // `.ps1` — a `.ps1` IS policy gated even though `-EncodedCommand` is not. The relief + // moves into the payload, where it is not part of the flagged command-line shape. + expect(startupPowerShell).toContain( + 'Set-ExecutionPolicy -Scope Process -ExecutionPolicy Bypass -Force -ErrorAction Stop' + ) + // Why `-ErrorAction Stop` and a reporting catch: autoload can fail for reasons that are + // not about policy at all (a 5.1 install with duplicate extended type data fails every + // cmdlet in Microsoft.PowerShell.Security), and the old SilentlyContinue plus `catch {}` + // hid that -- the user saw only their own script being refused. The catch must report and + // must NOT rethrow, or a broken policy cmdlet would take the whole startup with it. + expect(startupPowerShell).not.toContain('catch {}') + expect(startupPowerShell).toMatch(/catch \{ \[Console\]::Error\.WriteLine\(/) + expect(startupPowerShell).toContain('$_.FullyQualifiedErrorId') + expect(startupPowerShell).not.toMatch(/catch \{[^}]*throw/) + // Why: the autoloaded module's progress record would otherwise corrupt this gate's stderr. + expect(startupPowerShell).toContain("$ProgressPreference = 'SilentlyContinue'") + expect(startupPowerShell).toContain('$ProgressPreference = $orcaProgress') + // The setup gate only ever launches a .cmd/.bat runner, so it needs no relief. + expect(setupPowerShell).not.toMatch(/Set-ExecutionPolicy/i) }) it('launches a batch runner through the cmd launcher inside a Git Bash gate', () => { @@ -307,7 +332,7 @@ describe('createSequencedSetupAgentCommands', () => { }) expect(result.setupCommand).toContain( - 'powershell.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand' + 'powershell.exe -NoProfile -NonInteractive -EncodedCommand' ) expect(result.setupCommand).not.toMatch(/bash\s+\S*setup-runner/) expect(decodePowerShellScript(result.setupCommand)).toContain( diff --git a/src/shared/setup-agent-sequencing.ts b/src/shared/setup-agent-sequencing.ts index 7108360f645..367be99c212 100644 --- a/src/shared/setup-agent-sequencing.ts +++ b/src/shared/setup-agent-sequencing.ts @@ -227,6 +227,27 @@ function buildWindowsStartupCommand( // Why: native Windows setup runners launch through cmd.exe, but PowerShell // gives us safe bounded file polling/parsing without a fragile batch label loop. const script = [ + // Why: the startup command is user-authored and may invoke a `.ps1`, which IS + // execution-policy gated even though `-EncodedCommand` is not. This is the in-payload + // stand-in for the `-ExecutionPolicy Bypass` switch dropped from the command line + // (same trade as the agent-hooks launcher). Progress must be silenced first and + // restored after: Set-ExecutionPolicy autoloads a module whose "Preparing modules for + // first use." record would otherwise land on the stderr this gate writes to. + // + // The failure is reported rather than swallowed. Autoload can fail for reasons that + // have nothing to do with policy -- a 5.1 install with duplicate extended type data + // fails every cmdlet in Microsoft.PowerShell.Security -- and the old + // `-ErrorAction SilentlyContinue` plus empty `catch` hid that completely, leaving the + // user with an execution-policy refusal from their own script and no trace that the + // relief had been attempted. `-ErrorAction Stop` is what routes a non-terminating + // failure into the catch at all. Still never throws: a diagnostic is worth a line of + // stderr, but not the startup this gate exists to run. + "$orcaProgress = $ProgressPreference; $ProgressPreference = 'SilentlyContinue'", + 'try { Set-ExecutionPolicy -Scope Process -ExecutionPolicy Bypass -Force -ErrorAction Stop } ' + + 'catch { [Console]::Error.WriteLine("Orca: could not relax the execution policy for this " + ' + + '"session (" + $_.FullyQualifiedErrorId + "). A startup command that runs a .ps1 " + ' + + '"may be blocked.") }', + '$ProgressPreference = $orcaProgress', `$marker = ${quotePowerShellString(markerPath)}`, 'if ([string]::IsNullOrWhiteSpace($marker)) {', ' [Console]::Error.WriteLine("Missing setup marker path.")', @@ -269,8 +290,11 @@ function buildWindowsStartupCommand( return encodePowerShellInvocation(script) } +// Why: `-EncodedCommand` is not execution-policy gated (only `-File` is), so `-ExecutionPolicy +// Bypass` was a no-op — and it is one of the most heavily EDR-flagged PowerShell tokens. The +// base64 stays: these strings are typed into a terminal pane and re-parsed by its shell. function encodePowerShellInvocation(script: string): string { - return `powershell.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}` + return `powershell.exe -NoProfile -NonInteractive -EncodedCommand ${encodePowerShellCommand(script)}` } function quotePosixArg(value: string): string { diff --git a/src/shared/setup-runner-command.test.ts b/src/shared/setup-runner-command.test.ts index 069130b1313..4280ed4f6e0 100644 --- a/src/shared/setup-runner-command.test.ts +++ b/src/shared/setup-runner-command.test.ts @@ -82,7 +82,7 @@ describe('buildSetupRunnerCommand', () => { expect(command).not.toContain('cmd.exe /c') expect(command).toMatch( - /^powershell\.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand [A-Za-z0-9+/=]+$/ + /^powershell\.exe -NoProfile -NonInteractive -EncodedCommand [A-Za-z0-9+/=]+$/ ) }) @@ -129,7 +129,7 @@ describe('buildSetupRunnerCommand cmd metacharacter guard', () => { }) expect(command).toMatch( - /^powershell\.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand [A-Za-z0-9+/=]+$/ + /^powershell\.exe -NoProfile -NonInteractive -EncodedCommand [A-Za-z0-9+/=]+$/ ) } ) diff --git a/src/shared/windows-cmd-runner-delayed-launch.test.ts b/src/shared/windows-cmd-runner-delayed-launch.test.ts new file mode 100644 index 00000000000..21c5cc8befe --- /dev/null +++ b/src/shared/windows-cmd-runner-delayed-launch.test.ts @@ -0,0 +1,36 @@ +import { describe, expect, it } from 'vitest' +import { buildWindowsCmdRunnerDelayedLaunchCommand } from './windows-cmd-runner-delayed-launch' + +function decodePayload(command: string): string { + const encoded = command.match(/ -EncodedCommand (\S+)$/)?.[1] + if (!encoded) { + throw new Error(`no -EncodedCommand payload in: ${command}`) + } + return Buffer.from(encoded, 'base64').toString('utf16le') +} + +describe('buildWindowsCmdRunnerDelayedLaunchCommand', () => { + it('spells no -ExecutionPolicy switch', () => { + const command = buildWindowsCmdRunnerDelayedLaunchCommand('C:\\work\\setup.cmd') + const switches = command.replace(/ -EncodedCommand \S+$/, '') + + // Why: `-EncodedCommand` is not execution-policy gated — only `-File` is — so the switch + // was a no-op next to a heavily EDR-flagged base64 command line. + expect(switches).not.toMatch(/-ExecutionPolicy/i) + expect(switches).not.toMatch(/Bypass/i) + expect(switches).toBe('powershell.exe -NoProfile -NonInteractive') + }) + + it('keeps the base64 that shields the runner path from the pane shell', () => { + // Why: this whole module exists because the path carries cmd metacharacters; the + // command is typed into a terminal pane, so the base64 must stay. + const command = buildWindowsCmdRunnerDelayedLaunchCommand('C:\\work (x86)\\se&tup.cmd') + + expect(command).toMatch( + /^powershell\.exe -NoProfile -NonInteractive -EncodedCommand [A-Za-z0-9+/=]+$/ + ) + const script = decodePayload(command) + expect(script).toContain("$runner = 'C:\\work (x86)\\se&tup.cmd'") + expect(script).toContain('/d /s /v:on /c ""!ORCA_SETUP_RUNNER!""') + }) +}) diff --git a/src/shared/windows-cmd-runner-delayed-launch.ts b/src/shared/windows-cmd-runner-delayed-launch.ts index 50ed131dc99..cdc0af3a48a 100644 --- a/src/shared/windows-cmd-runner-delayed-launch.ts +++ b/src/shared/windows-cmd-runner-delayed-launch.ts @@ -34,7 +34,10 @@ export function buildWindowsCmdRunnerDelayedLaunchCommand(runnerScriptPath: stri 'exit $process.ExitCode' ].join('; ') - return `powershell.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}` + // Why: `-EncodedCommand` is not execution-policy gated (only `-File` is), so `-ExecutionPolicy + // Bypass` was a no-op — and it is one of the most heavily EDR-flagged PowerShell tokens. The + // base64 stays: this string is typed into a shell, which is the whole point of the guard above. + return `powershell.exe -NoProfile -NonInteractive -EncodedCommand ${encodePowerShellCommand(script)}` } function quotePowerShellString(value: string): string { diff --git a/src/shared/windows-interactive-login-spawn.test.ts b/src/shared/windows-interactive-login-spawn.test.ts index 07cde964095..ef99f8edabe 100644 --- a/src/shared/windows-interactive-login-spawn.test.ts +++ b/src/shared/windows-interactive-login-spawn.test.ts @@ -17,8 +17,14 @@ function encodedValue(value: string): string { return `Read-OrcaValue '${Buffer.from(value).toString('base64')}'` } +/** Positional-independent so the argv shape can change without silently reading the wrong slot. */ +function decodedScript(args: string[]): string { + const payload = args[args.indexOf('-EncodedCommand') + 1] ?? '' + return Buffer.from(payload, 'base64').toString('utf16le') +} + function pidFilePathFromSpawnArgs(args: string[]): string { - const script = Buffer.from(args[11] ?? '', 'base64').toString('utf16le') + const script = decodedScript(args) const encodedPath = script.match( /WriteAllText\(\(Read-OrcaValue '([^']+)'\), \[string\]\$PID\)/ )?.[1] @@ -40,15 +46,15 @@ describe('buildWindowsHostInteractiveLoginSpawn', () => { expect(spawn.command).toBe(getCmdExePath()) expect(spawn.args.slice(0, 5)).toEqual(['/d', '/c', 'start', '', '/wait']) expect(spawn.args[5]).toMatch(/WindowsPowerShell\\v1\.0\\powershell\.exe$/i) - expect(spawn.args.slice(6, 11)).toEqual([ - '-NoLogo', - '-NoProfile', - '-ExecutionPolicy', - 'Bypass', - '-EncodedCommand' - ]) + // Why: `-ExecutionPolicy Bypass` is a no-op next to `-EncodedCommand` (only `-File` is + // policy gated) and is a heavily EDR-flagged token, so it must not come back. The base64 + // must stay — `start` re-parses this through cmd.exe, whose safe-token guard rejects the + // `&` and `"` in the raw relay script. + expect(spawn.args.slice(6, 9)).toEqual(['-NoLogo', '-NoProfile', '-EncodedCommand']) + expect(spawn.args).not.toContain('-ExecutionPolicy') + expect(spawn.args).not.toContain('Bypass') - const script = Buffer.from(spawn.args[11] ?? '', 'base64').toString('utf16le') + const script = decodedScript(spawn.args) expect(script).toContain('[string]$PID') expect(script).toContain(encodedValue(getCmdExePath())) expect(script).toContain(encodedValue('C:\\Tools\\claude.cmd')) @@ -68,7 +74,7 @@ describe('buildWindowsHostInteractiveLoginSpawn', () => { const spawn = withWindows(() => buildWindowsHostInteractiveLoginSpawn('C:\\Tools\\codex.exe', ['login']) ) - const script = Buffer.from(spawn.args[11] ?? '', 'base64').toString('utf16le') + const script = decodedScript(spawn.args) expect(script).toContain(encodedValue('C:\\Tools\\codex.exe')) expect(script).toContain(encodedValue('login')) spawn.cleanup() diff --git a/src/shared/windows-interactive-login-spawn.ts b/src/shared/windows-interactive-login-spawn.ts index 1068607e38f..a38ba1ae7a8 100644 --- a/src/shared/windows-interactive-login-spawn.ts +++ b/src/shared/windows-interactive-login-spawn.ts @@ -78,11 +78,13 @@ export function buildWindowsHostInteractiveLoginSpawn( 'powershell.exe' ) const script = buildPidRelayScript(spawnCmd, spawnArgs, pidFilePath) + // Why: `-EncodedCommand` is not execution-policy gated (only `-File` is), so `-ExecutionPolicy + // Bypass` was a no-op — and it is one of the most heavily EDR-flagged PowerShell tokens. The + // base64 stays: `wrapWindowsStartWait` sends this through `cmd.exe /c start`, whose + // `assertWindowsCmdSafeTokens` guard rejects the `&` and `"` the raw relay script contains. const wrapped = wrapWindowsStartWait(powershell, [ '-NoLogo', '-NoProfile', - '-ExecutionPolicy', - 'Bypass', '-EncodedCommand', Buffer.from(script, 'utf16le').toString('base64') ]) From bfc6a262a7489b5f41ada02d4b133901933fc636 Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Sat, 5 Sep 2026 21:12:47 -0700 Subject: [PATCH 41/69] fix(windows): read command lines from the kernel, not each process's PEB (#17886) * fix(windows): read command lines from the kernel, not each process's PEB MDE incident D scored Orca for suspicious memory activity: the vendored `@vscode/windows-process-tree` recovered every process's command line by opening it with `PROCESS_QUERY_INFORMATION | PROCESS_VM_READ` and chaining three `ReadProcessMemory` calls through the PEB and `RTL_USER_PROCESS_PARAMETERS`. On a 750ms/2s cadence over the whole table that is the credential-dumping primitive, whatever the intent. Windows 8.1 added `NtQueryInformationProcess`'s `ProcessCommandLineInformation` class (60), which returns the same string as a kernel-built `UNICODE_STRING` under `PROCESS_QUERY_LIMITED_INFORMATION` alone. Electron's floor is Windows 10, so every supported OS has it. The PEB reader stays behind a process-wide latch that only `STATUS_INVALID_INFO_CLASS`/`NOT_SUPPORTED`/`NOT_IMPLEMENTED` can set; a pid that merely denied a handle does not re-arm it, because `PROCESS_QUERY_INFORMATION` implicitly grants the limited right and so cannot be obtained where the weaker open already failed. The same hunk drops `PROCESS_VM_READ` from `GetProcessMemoryUsage` and `GetCpuUsage`, which acquired it and never read an address space. Measured on Windows 11 (514 processes), counted in-process by swapping the addon's import table entries for counting stubs, per CommandLine scan: `ReadProcessMemory` 1128 -> 0, desired access 0x0410 -> 0x1000, p50 12.7ms -> 9.3ms. Command lines were byte-identical on every process both readers recovered (376/376, 379/379 across runs), including a 24,068-character argv with quotes, non-ASCII and trailing whitespace, and a WOW64 target. Three processes that refused the old rights granted the new one; none went the other way. * chore(deps): refresh the windows-process-tree patch hash in the lockfile * fix(windows): drop the PEB fallback and detect the unpatched prebuilt Review of #17886 found three ways the reader could still perform, or silently resume, the primitive it exists to remove. The class-missing latch was a permanent, process-wide, one-way downgrade back to the PEB read, and any single target returning STATUS_INVALID_INFO_CLASS / NOT_SUPPORTED / NOT_IMPLEMENTED could trip it. On an EDR-hooked ntdll -- the entire premise of this change -- a hook that does not recognise class 60 would have restored PROCESS_VM_READ plus three ReadProcessMemory per pid per scan for the life of the process, unobservably, on precisely the machines this was written for. The fallback is deleted rather than guarded: GetProcessCommandLine now returns false and leaves the command line empty, which callers already handle, so the addon imports no ReadProcessMemory at all. That absence is what makes the property checkable on the artifact. The published 0.8.0 tarball ships a loadable prebuilt built from unpatched source; it is node-addon-api, so a bare require() accepts it, allowBuilds is false and CI installs with --ignore-scripts, and a rebuild that soft-exits on a Windows file lock leaves it in place. Source-text guards could never see it. windowsProcessTreeAddonReadsProcessMemory() checks the compiled binary instead, and is wired into the install check, the rebuild, and the relay build. The repair itself never worked: `git apply` run inside a work tree prefixes patch paths with the cwd-relative prefix, skips what does not match, and exits 0, so the branch always fell through to its own post-check throw. The package dir is always under the project root, while the fixture that covered it was in %TEMP%, outside any repo. Blinding git with GIT_DIR fixes it, and the test now runs inside a real work tree. Also from review: bounds-check the returned UNICODE_STRING against the allocation (not the size the second query clobbers) and cap the probe so a bogus length cannot bad_alloc a whole scan; test NT_SUCCESS explicitly; value- initialize ProcessInfo, which left `memory` as stack garbage -- measured, 82 processes reported the same bogus working set; and correct a comment in windows-process-table.ts that still described the command line as a PEB read. Re-measured on Windows 11 (543 processes): ReadProcessMemory 1128 -> 0, with the symbol absent from the import table so the IAT hook finds no slot to count; desired access 0x0410 -> 0x1000 on all 543 opens; p50 13.5 -> 12.3ms; 405/405 command lines byte-identical including a 24,087-character quoted non-ASCII argv and a WOW64 target; 3 processes recovered only by the new path, 0 only by the old. * chore(deps): refresh the windows-process-tree patch hash in the lockfile * test(scripts): stage a script's local imports into the native-runtime fixture ensure-native-runtime.mjs gained an import of windows-process-tree-gyp-rebuild.mjs, but the fixture copied only the script itself, so every case in the suite died with ERR_MODULE_NOT_FOUND before reaching its own assertions. copyScriptWithLocalModules already walks a script's co-located imports for exactly this reason -- its own doc comment names this failure -- so use it rather than listing files by hand. The two Windows cases still fail here, on a missing node-pty ConPTY runtime that also fails on main; this only stops a resolution error from standing in front of whatever they were meant to catch. * fix(windows): route a locked stale addon to the Windows file-lock message `pnpm install` with Orca running aborted with a raw EPERM stack. The stale-binary guard -- which deletes an addon that still imports ReadProcessMemory so a skipped rebuild cannot use it -- ran outside the try whose catch classifies Windows file locks, and whose message is literally "Close running Orca/Electron/dev processes for this worktree": exactly this situation. Measured rather than assumed: rmSync against a loaded (memory-mapped) addon throws EPERM, and `force: true` does not help, since it only swallows ENOENT. Cold copies of the same file delete fine. So the delete threw a page before the handler that knows what it means. Moving the guard inside the try is the whole fix; the classifier already matches the EPERM text. The new case runs the real script against a temp project whose stale addon is held open by a live child process, and fails against the old placement with the raw `syscall: 'rm'` stack the report described. * feat(windows): warn once when command-line recovery is refused host-wide Removing the PEB fallback removed a total-defeat vector, but it left a cliff: if NtQueryInformationProcess(ProcessCommandLineInformation) is refused -- a hooked ntdll that does not know class 60 -- every command line comes back empty and agent identity matching silently degrades to image names. The addon still loads and still enumerates, so every health check the app has stays green. A cliff nobody can see is the failure mode this area keeps producing. The querying process is the unambiguous probe. A process can always open itself with PROCESS_QUERY_LIMITED_INFORMATION, so its own command line coming back empty means the query is refused for every process -- not that some target denied a handle, which is normal for roughly a quarter of the table. Keying on our own row rather than a fraction means no threshold to tune and no false positive on a hardened box where most processes deny. One warning per session, gated on the CommandLine flag actually being requested so a future identity-only reader cannot trip it. The suite's own SELF fixture gains a command line for the same reason: a self row without one is the alarm, not a detail. * fix(windows): check the relay's staged addon at load, and answer tri-state Two gaps in the ReadProcessMemory check, both about what it does not see. It only ever looked at node_modules/@vscode/windows-process-tree. A relay host has no node_modules of ours: it loads ./windows-process-tree.node staged beside the bundle. The relay build asserts the symbol on the artifact it produces, but a bundle and the addon beside it redeploy independently, so a host that has not taken a new bundle keeps whatever binary is already there -- and the published prebuilt is node-addon-api, so it binds cleanly and then walks every process's address space. loadWindowsProcessTree now checks that file too and refuses it, falling back to the CIM scan: slower, but not the thing an EDR quarantines a host for. The predicate is duplicated rather than imported, because the config-script copy is install-time tooling that drags in node-gyp and child_process, and this module is bundled into the app and the relay. And it returned false for a binary that is not there. All three callers happened to be safe, but the name read as a safety predicate, so a future caller would take a missing binary as verified. inspectWindowsProcessTreeAddon() now answers clean/unpatched/missing over an explicit binary path -- which is also what lets the relay's staged addon be checked at all -- and each caller states which state it acts on. Both are covered by cases that fail against the old code: without the load-time check the unpatched staged addon is bound and the CIM fallback never runs, and with 'missing' folded back into 'clean' the absence case fails outright. * test(windows): load the addon in beforeAll, not at collection time loadAddon() ran while the file was being collected, so on a Windows checkout with no built addon the require threw before any case existed and took the seven patch-text cases down with it -- cases that read only the patch file and need no binary at all. Verified both ways against a deliberately unresolvable addon path: at collection time vitest reports "no tests" for the file; from beforeAll the seven text cases pass and only the three addon cases go. * fix(deps): normalize the windows-process-tree patch to LF and let pnpm own its hash `pnpm install --frozen-lockfile` failed on this branch on every platform with ERR_PNPM_LOCKFILE_CONFIG_MISMATCH, which breaks CI and the release build. Two coupled defects. The patch file was committed with CRLF -- 174 CR bytes, against zero on main -- and `.gitattributes` pins `/config/patches/*.patch -text` precisely so checkout cannot convert it, so those bytes reached every runner. And pnpm hashes a patch **LF-normalized**, so the raw sha256 of a CRLF file is a value pnpm never computes: raw sha256 322965470c05f63d8527f7d8e892ee26ee444136b66b57fd64c362a9f2ff05d1 LF-normalized f8ea245391c94da5770045aeea01fa6de466c2199c6ef46b5b769b398aa9823e The lockfile carried the raw one, at all three sites. It is the only one of the seven patches where the two digests differ, which is why the other six passed. Normalized the patch to LF and took pnpm's own value from `pnpm install --no-frozen-lockfile`; nothing here is hand-computed. With the file LF-only the two interpretations coincide, so the lockfile, the contract test's no-CR assertion and its hash assertion all agree at one number -- and `config/scripts/windows-process-tree-patch-contract.test.mjs`, which was red on this branch for the same reason, is green again. The lockfile diff is exactly the three hash lines. The regression check is the installer, not a digest. Two separate reviews "verified" the shipped hash by recomputing sha256(patchBytes) and matching the lockfile; both were wrong, because both repeated the same wrong assumption about which bytes pnpm hashes. A check that reproduces the original mistake is not independent. So the new case runs `pnpm install --frozen-lockfile --lockfile-only --ignore-scripts` against a copy of the manifest, lockfile and patches, and asserts exit 0 -- verified by deletion: restoring the shipped hash fails it with the exact ERR_PNPM_LOCKFILE_CONFIG_MISMATCH from the branch's package (windows) job. Also corrected the `.gitattributes` comment claiming pnpm hashes patches byte-for-byte. The `-text` setting is right -- `git apply` needs the exact bytes -- but that sentence is the claim that produced the wrong hash twice. * ci(windows): run the process-tree patch suites in CI Both suites only self-skip off Windows, so the binary-level check that the addon carries no ReadProcessMemory passed vacuously in every lane. * fix(windows): force core.autocrlf=input for the patch repair My LF normalization of the windows-process-tree patch broke the `git apply` repair path introduced in this PR. The two are coupled and I checked only one. Those 174 CR bytes were not editor noise. They sat on exactly the pre-image lines and nowhere else -- 107/107 in src/process.cc, 67/67 in src/process_commandline.cc, 0 on every added or context line -- because @vscode/windows-process-tree@0.8.0 ships those two sources as CRLF. Normalizing the patch made its pre-image stop matching the file it is applied against. Measured, reconstructing the true CRLF pre-image from the pre-normalization blob and applying the current LF patch: core.autocrlf plain -c core.autocrlf=input true exit 0 exit 0 input exit 0 exit 0 false exit 1 exit 0 `false` is Git's own built-in default and what "checkout as-is" selects in the Git for Windows installer -- on this box the `true` that hides it comes from the installer's system gitconfig, not from anything in the repo. There the repair throws, ensureWindowsProcessTreeCommandLinePatch reports "still reads the PEB, and repairing it ... failed", isWindowsNativeLockError does not match that text, and `pnpm install` dies with no path forward. Forcing the mode rather than `--ignore-whitespace`: both fix every cell and both leave the applied file fully LF, but `input` relaxes line endings only, so a hunk whose real content drifted is still rejected. The repair rewrites a security-relevant source file; it should stay strict about everything except the thing that is legitimately ambiguous. Not reverting the patch to CRLF: windows-process-tree-patch-contract.test.mjs (pre-existing on main) forbids CR bytes in it, and pnpm computes the same hash either way. LF plus the forced mode is the end state. The suite could not have caught this. The fixture built its pre-image from the patch itself and joined with '\n', so fixture and patch agreed by construction on any encoding -- once again a test that passes without its fix. It now emits the CRLF the real package ships, and the case runs under both autocrlf modes pinned through a temp HOME gitconfig, because the repair blinds git to the repo and so reads global config. Verified by deletion in both directions: with the flag removed the autocrlf=false case fails with the exact "still reads the PEB" dead end while autocrlf=true still passes, and with the fixture back on LF all eight cases pass with no fix present at all. Also corrected the .gitattributes comment I added last commit. It said `git apply` needs the bytes the patch was written against, which is now false -- the pinned bytes are LF and the bytes it was written against are CRLF. That is the same class of confident-and-wrong claim that produced the bad hash twice. * fix(windows): assert the rebuilt addon, and install the patch for real in tests Three follow-ups from review. **The packaged binary had no check.** The relay build asserts its own artifact and ensure-native-runtime asserts what it loads, but nothing looked at the addon copied into the packaged app -- so a rebuild that silently produced the upstream reader shipped. `rebuild-native-deps.mjs` now asserts `clean` on it after `rebuild()`. This is also the caller D4's tri-state was missing: every existing site branches on `=== 'unpatched'`, so `missing` still behaved exactly like `clean` everywhere, which was the thing making it a state rather than a boolean. Here both non-clean states fail, and they fail differently: after a rebuild that reported success, an absent binary is a broken build, not an absence to shrug at. The fake `rebuild()` had to start producing a binary for that to mean anything, so it now emits stand-in bytes and takes `addon: 'clean' | 'unpatched' | 'none'`. Verified by deletion: with the assertion removed both new cases pass. **The frozen-install case could not see a patch at all.** `--lockfile-only` resolves and never applies one, so its coverage stops at hash consistency. Added a case that installs `@vscode/windows-process-tree@0.8.0` for real with the patch and asserts the materialized `src/process_commandline.cc` carries the marker and no longer carries `ReadProcessMemory` -- about 1.5s for the pair. Correcting the brief on that one: it does **not** catch the `git apply` breakage from the previous commit. Measured -- with `-c core.autocrlf=input` removed it passes cleanly, because `pnpm install` uses pnpm's own patch applier and never runs our repair script. What it does catch is a patch pnpm can no longer apply: corrupting one pre-image line fails both cases. The repair path stays covered by the CRLF fixture in rebuild-native-deps-node-pty.test.mjs. Worth recording, since it decides whether the LF normalization was safe at all: pnpm applies the LF patch to the CRLF tarball sources without complaint, and materializes them as LF with the marker present and `ReadProcessMemory` absent. The primary install path was never affected -- only the `git apply` fallback was. **Dead timeout.** The frozen-install case passed `timeoutMs: 300_000` to the spawn while vitest capped the case itself at 30s, so on a cold runner vitest would have killed it first. Both cases now declare the budget they use. * test(windows): route the frozen-install check through the pnpm invocation owner The new patched-dependencies check hand-rolled a PATH walk naming 'pnpm.cmd', which the windows batch shim spawn boundary ratchet rejects: pnpm-cli-invocation already owns that decision for every other script, and its allowlist only shrinks. Reuse resolvePnpmCliInvocation for the command and prefixArgs, and the shared resolveCliCommand for the presence check, so no shim name is spelled here. Its `shell` flag is dropped because runProcessSync refuses it and already drives a shim through the interpreter itself. --------- Co-authored-by: Orca Worker Co-authored-by: Neil <4138956+nwparker@users.noreply.github.com> --- .gitattributes | 9 +- .github/workflows/pr.yml | 4 + .../@vscode__windows-process-tree@0.8.0.patch | 425 +++++++++++++++++- ...build-windows-process-tree-relay-addon.mjs | 27 +- config/scripts/ensure-native-runtime.mjs | 29 +- config/scripts/ensure-native-runtime.test.mjs | 10 +- ...tched-dependencies-frozen-install.test.mjs | 155 +++++++ config/scripts/pr-code-change-scope.mjs | 2 + .../rebuild-native-deps-node-pty.test.mjs | 123 ++++- .../rebuild-native-deps-test-fixtures.mjs | 123 ++++- ...-native-deps-windows-process-tree.test.mjs | 103 +++++ config/scripts/rebuild-native-deps.mjs | 64 ++- .../windows-process-tree-gyp-rebuild.mjs | 126 +++++- .../windows-process-tree-gyp-rebuild.test.mjs | 40 +- docs/reference/windows-edr-posture.md | 56 ++- docs/reference/windows-process-enumeration.md | 118 ++++- pnpm-lock.yaml | 6 +- ...ndows-command-line-recovery-health.test.ts | 59 +++ .../windows-command-line-recovery-health.ts | 47 ++ .../windows/windows-process-table.test.ts | 129 +++++- src/main/windows/windows-process-table.ts | 94 +++- ...ws-process-tree-command-line-patch.test.ts | 188 ++++++++ 22 files changed, 1853 insertions(+), 84 deletions(-) create mode 100644 config/scripts/patched-dependencies-frozen-install.test.mjs create mode 100644 config/scripts/rebuild-native-deps-windows-process-tree.test.mjs create mode 100644 src/main/windows/windows-command-line-recovery-health.test.ts create mode 100644 src/main/windows/windows-command-line-recovery-health.ts create mode 100644 src/main/windows/windows-process-tree-command-line-patch.test.ts diff --git a/.gitattributes b/.gitattributes index 2a99890023b..1ce5b29ee45 100644 --- a/.gitattributes +++ b/.gitattributes @@ -8,7 +8,14 @@ /src/cli/bundled-skill-guides.ts text eol=lf # Bundled plugin trees are byte-hashed; CRLF checkout would break the pinned hash. /resources/plugins/** text eol=lf -# pnpm hashes every patch byte-for-byte, so a CRLF checkout breaks the install. +# Pin the bytes so a patch reads and diffs identically on every host. It is NOT +# what makes the hash right: pnpm hashes a patch LF-normalized, so a CRLF checkout +# cannot change it. Believing otherwise put a hand-computed raw digest in the +# lockfile twice and broke every install (#17886). +# These files are stored LF, which is not always the encoding they were written +# against -- @vscode/windows-process-tree ships CRLF sources -- so any code that +# runs `git apply` on one must force `-c core.autocrlf=input` rather than trust +# the host's setting. See config/scripts/windows-process-tree-gyp-rebuild.mjs. /config/patches/*.patch -text # The xterm bundle hunks also make a diff nobody can read; review the hand-written # source patch under xterm-src/ instead. The sibling patches stay diffable. diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index 0cd7960e0c7..536b5c0f273 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -832,10 +832,13 @@ jobs: node_modules/.pnpm/@vscode+windows-process-tree@*/node_modules/@vscode/windows-process-tree/build key: native-modules-${{ runner.os }}-${{ steps.deps.outputs.native-cache-scope }}-${{ runner.arch }}-node-node${{ steps.deps.outputs.node-version }}-${{ hashFiles('pnpm-lock.yaml', '.github/actions/install-node-dependencies/action.yml', 'config/scripts/ensure-native-runtime.mjs', 'config/scripts/rebuild-native-deps.mjs', 'config/patches/node-pty@1.1.0.patch', 'config/patches/@vscode__windows-process-tree@0.8.0.patch') }} + # vitest runs here directly rather than through `pnpm test`, so the addon + # assertions only hold once install-node-dependencies has rebuilt natives. - name: Test Windows-specific boundaries run: >- pnpm exec vitest run --config config/vitest.config.ts config/scripts/rebuild-native-deps.test.mjs + config/scripts/rebuild-native-deps-windows-process-tree.test.mjs src/main/browser/browser-client-page-renderer-lifecycle.electron.test.ts src/main/browser/browser-route-tcp-egress.electron.test.ts src/main/browser/browser-route-webrtc-egress.electron.test.ts @@ -848,6 +851,7 @@ jobs: src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts src/main/windows/windows-pty-job.win32.test.ts src/main/windows/windows-host-job.win32.test.ts + src/main/windows/windows-process-tree-command-line-patch.test.ts src/main/windows-live-tree-kill.win32.test.ts src/main/wsl/wsl-runner.test.ts src/main/wsl/wsl-guest-environment.test.ts diff --git a/config/patches/@vscode__windows-process-tree@0.8.0.patch b/config/patches/@vscode__windows-process-tree@0.8.0.patch index 10780f5288a..fe5e4be44b1 100644 --- a/config/patches/@vscode__windows-process-tree@0.8.0.patch +++ b/config/patches/@vscode__windows-process-tree@0.8.0.patch @@ -27,15 +27,424 @@ index 855bd4b86f0a3c18c7594212c0e42b6e35bc4001..33774e7ae296f0de39dd94156673c9e7 "/guard:cf", "/sdl", diff --git a/src/process.cc b/src/process.cc -index 3eea92077c4d1d433119361d5c432881859131e9..1998f4addd4d7e9aba946ea6f7f7a4a5d13291bc 100644 +index 3eea92077c4d1d433119361d5c432881859131e9..738775f6fcdfb676054386fe34c0380327ed1863 100644 --- a/src/process.cc +++ b/src/process.cc -@@ -37,7 +37,7 @@ uint32_t GetRawProcessList(std::vector& process_info, - process_info.push_back(std::move(pinfo)); - process_count++; - } +@@ -1,108 +1,112 @@ +-/*--------------------------------------------------------------------------------------------- +- * Copyright (c) Microsoft Corporation. All rights reserved. +- * Licensed under the MIT License. See License.txt in the project root for license information. +- *--------------------------------------------------------------------------------------------*/ +- +-#include "process.h" +-#include "process_commandline.h" +- +-#include +-#include +-#include +- +-uint32_t GetRawProcessList(std::vector& process_info, +- DWORD process_data_flags) { +- // Fetch the PID and PPIDs +- PROCESSENTRY32 process_entry = { 0 }; +- DWORD parent_pid = 0; +- uint32_t process_count = 0; +- HANDLE snapshot_handle = CreateToolhelp32Snapshot(TH32CS_SNAPPROCESS, 0); +- process_entry.dwSize = sizeof(PROCESSENTRY32); +- if (Process32First(snapshot_handle, &process_entry)) { +- do { +- if (process_entry.th32ProcessID != 0) { +- ProcessInfo pinfo; +- pinfo.pid = process_entry.th32ProcessID; +- pinfo.ppid = process_entry.th32ParentProcessID; +- +- if (MEMORY & process_data_flags) { +- GetProcessMemoryUsage(pinfo); +- } +- +- if (COMMANDLINE & process_data_flags) { +- GetProcessCommandLine(pinfo); +- } +- +- strcpy(pinfo.name, process_entry.szExeFile); +- process_info.push_back(std::move(pinfo)); +- process_count++; +- } - } while (process_count < 1024 && Process32Next(snapshot_handle, &process_entry)); +- } +- +- CloseHandle(snapshot_handle); +- return process_count; +-} +- +-void GetProcessMemoryUsage(ProcessInfo& process_info) { +- DWORD pid = process_info.pid; +- HANDLE hProcess; +- PROCESS_MEMORY_COUNTERS pmc; +- +- hProcess = OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, false, pid); +- +- if (hProcess == NULL) { +- return; +- } +- +- if (GetProcessMemoryInfo(hProcess, &pmc, sizeof(pmc))) { +- process_info.memory = (DWORD)pmc.WorkingSetSize; +- } +- +- CloseHandle(hProcess); +-} +- +-// Per documentation, it is not recommended to add or subtract values from the FILETIME +-// structure, or to cast it to ULARGE_INTEGER as this can cause alignment faults on 64-bit Windows. +-// Copy the high and low part to a ULARGE_INTEGER and peform arithmetic on that instead. +-// See https://msdn.microsoft.com/en-us/library/windows/desktop/ms724284(v=vs.85).aspx +-ULONGLONG GetTotalTime(const FILETIME* kernelTime, const FILETIME* userTime) { +- ULARGE_INTEGER kt, ut; +- kt.LowPart = (*kernelTime).dwLowDateTime; +- kt.HighPart = (*kernelTime).dwHighDateTime; +- +- ut.LowPart = (*userTime).dwLowDateTime; +- ut.HighPart = (*userTime).dwHighDateTime; +- +- return kt.QuadPart + ut.QuadPart; +-} +- +-void GetCpuUsage(Cpu& cpu_info, bool first_pass) { +- DWORD pid = cpu_info.pid; +- HANDLE hProcess; +- +- hProcess = OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, false, pid); +- +- if (hProcess == NULL) { +- return; +- } +- +- FILETIME creationTime, exitTime, kernelTime, userTime; +- FILETIME sysIdleTime, sysKernelTime, sysUserTime; +- if (GetProcessTimes(hProcess, &creationTime, &exitTime, &kernelTime, &userTime) +- && GetSystemTimes(&sysIdleTime, &sysKernelTime, &sysUserTime)) { +- if (first_pass) { +- cpu_info.initialProcRunTime = GetTotalTime(&kernelTime, &userTime); +- cpu_info.initialSystemTime = GetTotalTime(&sysKernelTime, &sysUserTime); +- } else { +- ULONGLONG endProcTime = GetTotalTime(&kernelTime, &userTime); +- ULONGLONG endSysTime = GetTotalTime(&sysKernelTime, &sysUserTime); +- +- cpu_info.cpu = 100.0 * (endProcTime - cpu_info.initialProcRunTime) / (endSysTime - cpu_info.initialSystemTime); +- } +- } else { +- cpu_info.cpu = std::numeric_limits::quiet_NaN(); +- } +- +- CloseHandle(hProcess); ++/*--------------------------------------------------------------------------------------------- ++ * Copyright (c) Microsoft Corporation. All rights reserved. ++ * Licensed under the MIT License. See License.txt in the project root for license information. ++ *--------------------------------------------------------------------------------------------*/ ++ ++#include "process.h" ++#include "process_commandline.h" ++ ++#include ++#include ++#include ++ ++uint32_t GetRawProcessList(std::vector& process_info, ++ DWORD process_data_flags) { ++ // Fetch the PID and PPIDs ++ PROCESSENTRY32 process_entry = { 0 }; ++ DWORD parent_pid = 0; ++ uint32_t process_count = 0; ++ HANDLE snapshot_handle = CreateToolhelp32Snapshot(TH32CS_SNAPPROCESS, 0); ++ process_entry.dwSize = sizeof(PROCESSENTRY32); ++ if (Process32First(snapshot_handle, &process_entry)) { ++ do { ++ if (process_entry.th32ProcessID != 0) { ++ // Value-initialize: `memory` is otherwise stack garbage when the flag is unset. ++ ProcessInfo pinfo{}; ++ pinfo.pid = process_entry.th32ProcessID; ++ pinfo.ppid = process_entry.th32ParentProcessID; ++ ++ if (MEMORY & process_data_flags) { ++ GetProcessMemoryUsage(pinfo); ++ } ++ ++ if (COMMANDLINE & process_data_flags) { ++ GetProcessCommandLine(pinfo); ++ } ++ ++ strcpy(pinfo.name, process_entry.szExeFile); ++ process_info.push_back(std::move(pinfo)); ++ process_count++; ++ } + } while (Process32Next(snapshot_handle, &process_entry)); - } - - CloseHandle(snapshot_handle); ++ } ++ ++ CloseHandle(snapshot_handle); ++ return process_count; ++} ++ ++void GetProcessMemoryUsage(ProcessInfo& process_info) { ++ DWORD pid = process_info.pid; ++ HANDLE hProcess; ++ PROCESS_MEMORY_COUNTERS pmc; ++ ++ // PROCESS_VM_READ is never used here -- GetProcessMemoryInfo reads counters the ++ // kernel keeps, not the address space -- and acquiring it is what EDR scores. ++ hProcess = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, false, pid); ++ ++ if (hProcess == NULL) { ++ return; ++ } ++ ++ if (GetProcessMemoryInfo(hProcess, &pmc, sizeof(pmc))) { ++ process_info.memory = (DWORD)pmc.WorkingSetSize; ++ } ++ ++ CloseHandle(hProcess); ++} ++ ++// Per documentation, it is not recommended to add or subtract values from the FILETIME ++// structure, or to cast it to ULARGE_INTEGER as this can cause alignment faults on 64-bit Windows. ++// Copy the high and low part to a ULARGE_INTEGER and peform arithmetic on that instead. ++// See https://msdn.microsoft.com/en-us/library/windows/desktop/ms724284(v=vs.85).aspx ++ULONGLONG GetTotalTime(const FILETIME* kernelTime, const FILETIME* userTime) { ++ ULARGE_INTEGER kt, ut; ++ kt.LowPart = (*kernelTime).dwLowDateTime; ++ kt.HighPart = (*kernelTime).dwHighDateTime; ++ ++ ut.LowPart = (*userTime).dwLowDateTime; ++ ut.HighPart = (*userTime).dwHighDateTime; ++ ++ return kt.QuadPart + ut.QuadPart; ++} ++ ++void GetCpuUsage(Cpu& cpu_info, bool first_pass) { ++ DWORD pid = cpu_info.pid; ++ HANDLE hProcess; ++ ++ // GetProcessTimes needs no more than PROCESS_QUERY_LIMITED_INFORMATION. ++ hProcess = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, false, pid); ++ ++ if (hProcess == NULL) { ++ return; ++ } ++ ++ FILETIME creationTime, exitTime, kernelTime, userTime; ++ FILETIME sysIdleTime, sysKernelTime, sysUserTime; ++ if (GetProcessTimes(hProcess, &creationTime, &exitTime, &kernelTime, &userTime) ++ && GetSystemTimes(&sysIdleTime, &sysKernelTime, &sysUserTime)) { ++ if (first_pass) { ++ cpu_info.initialProcRunTime = GetTotalTime(&kernelTime, &userTime); ++ cpu_info.initialSystemTime = GetTotalTime(&sysKernelTime, &sysUserTime); ++ } else { ++ ULONGLONG endProcTime = GetTotalTime(&kernelTime, &userTime); ++ ULONGLONG endSysTime = GetTotalTime(&sysKernelTime, &sysUserTime); ++ ++ cpu_info.cpu = 100.0 * (endProcTime - cpu_info.initialProcRunTime) / (endSysTime - cpu_info.initialSystemTime); ++ } ++ } else { ++ cpu_info.cpu = std::numeric_limits::quiet_NaN(); ++ } ++ ++ CloseHandle(hProcess); + } +\ No newline at end of file +diff --git a/src/process_commandline.cc b/src/process_commandline.cc +index ea822b120e8038a4803e34647042f08f4aaf5ca1..25907c0bf542bed6c72b1b462b19bcf3210c3cfd 100644 +--- a/src/process_commandline.cc ++++ b/src/process_commandline.cc +@@ -1,67 +1,125 @@ +-/*--------------------------------------------------------------------------------------------- +- * Copyright (c) Microsoft Corporation. All rights reserved. +- * Licensed under the MIT License. See License.txt in the project root for license information. +- *--------------------------------------------------------------------------------------------*/ +- +-#include "process.h" +-#include "process_commandline.h" +-#include +-#include +-#include +- +-bool GetProcessCommandLine(ProcessInfo& process_info) { +- HINSTANCE ntdll = GetModuleHandleW(L"ntdll.dll"); +- if (!ntdll) { +- return false; +- } +- +- decltype(NtQueryInformationProcess)* nt_query_information_process = +- reinterpret_cast( +- GetProcAddress(ntdll, "NtQueryInformationProcess")); +- +- if (!nt_query_information_process) { +- return false; +- } +- +- PROCESS_BASIC_INFORMATION pbi{}; +- PEB peb = {NULL}; +- RTL_USER_PROCESS_PARAMETERS process_parameters = {NULL}; +- +- // Get process handle +- DWORD pid = process_info.pid; +- HANDLE hProcess = OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, FALSE, pid); +- if (hProcess == INVALID_HANDLE_VALUE) { +- return false; +- } +- +- // Get Process Environment Block (PEB) +- NTSTATUS status = nt_query_information_process(hProcess, ProcessBasicInformation, &pbi, sizeof(pbi), nullptr); +- if (NT_SUCCESS(status) && pbi.PebBaseAddress) { +- // Read PEB +- if (ReadProcessMemory(hProcess, pbi.PebBaseAddress, &peb, sizeof(peb), nullptr)) { +- // Read the processs parameters +- if (ReadProcessMemory(hProcess, peb.ProcessParameters, &process_parameters, sizeof(RTL_USER_PROCESS_PARAMETERS), nullptr)) { +- if (process_parameters.CommandLine.Length > 0) { +- std::wstring buffer; +- buffer.resize(process_parameters.CommandLine.Length / sizeof(wchar_t)); +- if (ReadProcessMemory(hProcess, process_parameters.CommandLine.Buffer, &buffer[0], process_parameters.CommandLine.Length, nullptr)) { +- int wide_length = static_cast(buffer.length()); +- int charcount = WideCharToMultiByte(CP_UTF8, 0, buffer.data(), wide_length, +- NULL, 0, NULL, NULL); +- if (charcount) { +- process_info.commandLine.resize(static_cast(charcount)); +- WideCharToMultiByte(CP_UTF8, 0, buffer.data(), wide_length, +- &process_info.commandLine[0], charcount, +- NULL, NULL); +- } +- CloseHandle(hProcess); +- return true; +- } +- } +- } +- } +- } +- +- CloseHandle(hProcess); +- return false; +-} ++/*--------------------------------------------------------------------------------------------- ++ * Copyright (c) Microsoft Corporation. All rights reserved. ++ * Licensed under the MIT License. See License.txt in the project root for license information. ++ *--------------------------------------------------------------------------------------------*/ ++ ++#include "process.h" ++#include "process_commandline.h" ++#include ++#include ++#include ++ ++namespace { ++ ++// Windows 8.1 and later hand back a process's command line as a UNICODE_STRING ++// the kernel builds, needing only PROCESS_QUERY_LIMITED_INFORMATION. ++// ++// There is deliberately no PEB fallback. Reading the command line out of the ++// target's address space -- opening it for VM reads and then chaining ++// memory reads across every pid on a timer -- is the credential-dumping ++// primitive this reader exists to not perform, so it is absent from the binary ++// rather than one anomalous NTSTATUS away. Electron's floor is Windows 10, so ++// every OS Orca supports has this class; if a hooked ntdll refuses it anyway, ++// the command line comes back empty, which callers already handle, instead of ++// silently reinstating the primitive on exactly the instrumented machines this ++// reader was written for. ++const ULONG kProcessCommandLineInformation = 60; ++ ++const NTSTATUS kStatusInfoLengthMismatch = static_cast(0xC0000004L); ++const NTSTATUS kStatusBufferTooSmall = static_cast(0xC0000023L); ++ ++// A command line is a UNICODE_STRING, whose Length is a USHORT, so the kernel ++// can never need more than the header plus 64 KiB. Refusing anything larger ++// keeps a bogus size from throwing bad_alloc out of a scan that has already ++// walked most of the table. ++const ULONG kMaxCommandLineBytes = sizeof(UNICODE_STRING) + 0xFFFF + sizeof(wchar_t); ++ ++// winternl.h's PROCESSINFOCLASS does not name class 60 and its enumerator range ++// stops far short of it, so the class travels as a ULONG rather than a cast enum. ++typedef NTSTATUS(NTAPI* NtQueryInformationProcessFn)(HANDLE, ULONG, PVOID, ULONG, PULONG); ++ ++// ntdll ships no import library for this entry point; it has to be resolved. ++NtQueryInformationProcessFn ResolveNtQueryInformationProcess() { ++ HMODULE ntdll = GetModuleHandleW(L"ntdll.dll"); ++ if (!ntdll) { ++ return nullptr; ++ } ++ return reinterpret_cast( ++ GetProcAddress(ntdll, "NtQueryInformationProcess")); ++} ++ ++NtQueryInformationProcessFn NtQueryInformationProcessEntry() { ++ static NtQueryInformationProcessFn entry = ResolveNtQueryInformationProcess(); ++ return entry; ++} ++ ++bool StoreCommandLineUtf8(ProcessInfo& process_info, const wchar_t* data, size_t wide_length) { ++ if (wide_length == 0) { ++ return false; ++ } ++ int length = static_cast(wide_length); ++ int charcount = WideCharToMultiByte(CP_UTF8, 0, data, length, NULL, 0, NULL, NULL); ++ if (!charcount) { ++ return false; ++ } ++ process_info.commandLine.resize(static_cast(charcount)); ++ WideCharToMultiByte(CP_UTF8, 0, data, length, &process_info.commandLine[0], charcount, NULL, ++ NULL); ++ return true; ++} ++ ++} // namespace ++ ++bool GetProcessCommandLine(ProcessInfo& process_info) { ++ NtQueryInformationProcessFn query = NtQueryInformationProcessEntry(); ++ if (!query) { ++ return false; ++ } ++ ++ HANDLE process = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, FALSE, process_info.pid); ++ if (process == NULL) { ++ return false; ++ } ++ ++ ULONG size = 0; ++ NTSTATUS status = query(process, kProcessCommandLineInformation, nullptr, 0, &size); ++ if (NT_SUCCESS(status)) { ++ // Nothing was written, so there is no command line to read. ++ CloseHandle(process); ++ return false; ++ } ++ if (status != kStatusInfoLengthMismatch && status != kStatusBufferTooSmall) { ++ CloseHandle(process); ++ return false; ++ } ++ if (size < sizeof(UNICODE_STRING) || size > kMaxCommandLineBytes) { ++ CloseHandle(process); ++ return false; ++ } ++ ++ std::vector buffer(size); ++ status = query(process, kProcessCommandLineInformation, &buffer[0], size, &size); ++ CloseHandle(process); ++ if (!NT_SUCCESS(status)) { ++ return false; ++ } ++ ++ // Header and characters arrive in one allocation, but treat the header as ++ // untrusted: a hooked ntdll is the case this reader is written for, and an ++ // unchecked Buffer/Length here would be an over-read encoded straight into JS. ++ // Bound against buffer.size(), never `size` -- the second query overwrote it. ++ const UNICODE_STRING* command_line = reinterpret_cast(&buffer[0]); ++ const unsigned char* begin = &buffer[0]; ++ const unsigned char* end = begin + buffer.size(); ++ const unsigned char* chars = reinterpret_cast(command_line->Buffer); ++ if (chars == nullptr || chars < begin + sizeof(UNICODE_STRING) || chars > end || ++ command_line->Length > static_cast(end - chars)) { ++ return false; ++ } ++ ++ // True only when a command line was actually stored, so "empty" and "not ++ // recovered" stay the same answer they were before this reader replaced the ++ // PEB read. `src/process.cc` discards the result either way. ++ return StoreCommandLineUtf8(process_info, command_line->Buffer, ++ command_line->Length / sizeof(wchar_t)); ++} diff --git a/config/scripts/build-windows-process-tree-relay-addon.mjs b/config/scripts/build-windows-process-tree-relay-addon.mjs index d3b9db939cd..9243f5a5b78 100644 --- a/config/scripts/build-windows-process-tree-relay-addon.mjs +++ b/config/scripts/build-windows-process-tree-relay-addon.mjs @@ -32,6 +32,8 @@ import { import { join, resolve } from 'node:path' import { RELAY_WINDOWS_PROCESS_TREE_FILENAME } from '../../src/shared/relay-artifacts.ts' import { + ensureWindowsProcessTreeCommandLinePatch, + inspectWindowsProcessTreeAddon, nodeGypRebuildInvocation, stageWindowsProcessTreeNodeAddonApiHeaders, WINDOWS_PROCESS_TREE_PACKAGE_DIR as PACKAGE_DIR @@ -89,6 +91,13 @@ function assertPatchApplied() { 'config/patches/@vscode__windows-process-tree@0.8.0.patch; run pnpm install.' ) } + if (processCc.includes('OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ')) { + throw new Error( + 'src/process.cc still takes PROCESS_VM_READ for memory or CPU counters it never reads ' + + 'from the address space. pnpm did not apply ' + + 'config/patches/@vscode__windows-process-tree@0.8.0.patch; run pnpm install.' + ) + } } // pnpm can materialize this CRLF package without applying its patch. Repair the @@ -123,6 +132,13 @@ function applyWindowsProcessTreeBuildFixes() { '' ) processCc = processCc.replace(/process_count < 1024 && /, '') + // The memory and CPU readers only ever call GetProcessMemoryInfo/GetProcessTimes, + // which need no more than PROCESS_QUERY_LIMITED_INFORMATION; taking VM_READ is + // what EDR scores. + processCc = processCc.replaceAll( + 'OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, false, pid)', + 'OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, false, pid)' + ) if (bindingGyp !== originalBinding) { writeFileSync(bindingPath, bindingGyp) @@ -131,7 +147,8 @@ function applyWindowsProcessTreeBuildFixes() { writeFileSync(processPath, processCc) } stageWindowsProcessTreeNodeAddonApiHeaders(PACKAGE_DIR) - if (bindingGyp !== originalBinding || processCc !== originalProcess) { + const repairedCommandLine = ensureWindowsProcessTreeCommandLinePatch(PACKAGE_DIR) + if (bindingGyp !== originalBinding || processCc !== originalProcess || repairedCommandLine) { console.warn('[windows-process-tree] Repaired un-applied pnpm patch hunks before build.') } } @@ -173,6 +190,14 @@ function main() { if (!existsSync(built)) { throw new Error(`node-gyp reported success but ${built} is missing.`) } + // Why check the artifact and not only the source: the source checks above run + // before node-gyp, and a stale build directory can outlive them. + if (inspectWindowsProcessTreeAddon(built) === 'unpatched') { + throw new Error( + 'The built addon still calls ReadProcessMemory, so it did not come from the patched ' + + 'command-line reader. A relay would get the primitive MDE scores as credential dumping.' + ) + } const machine = readPeMachine(built) if (machine !== PE_MACHINE[arch]) { throw new Error( diff --git a/config/scripts/ensure-native-runtime.mjs b/config/scripts/ensure-native-runtime.mjs index a4cc6db8843..b2a47b99d5b 100644 --- a/config/scripts/ensure-native-runtime.mjs +++ b/config/scripts/ensure-native-runtime.mjs @@ -5,6 +5,12 @@ import { createRequire } from 'node:module' import { existsSync, readFileSync } from 'node:fs' import { release } from 'node:os' import { basename, dirname, resolve } from 'node:path' +import { + ensureWindowsProcessTreeCommandLinePatch, + inspectWindowsProcessTreeAddon, + stageWindowsProcessTreeNodeAddonApiHeaders, + windowsProcessTreeAddonPath +} from './windows-process-tree-gyp-rebuild.mjs' const require = createRequire(import.meta.url) const { assertNodePtyJobOwnership } = require('./node-pty-job-ownership.cjs') @@ -253,11 +259,18 @@ function collectNativeModuleFailures() { function loadNativeModule(moduleName) { if (moduleName === '@vscode/windows-process-tree') { - // A bare require already loads the .node addon on win32, so it catches an - // ABI mismatch on its own. What it cannot catch is a snapshot that comes - // back empty -- the shape a blocked CreateToolhelp32Snapshot produces -- - // so check the addon actually enumerates before calling the runtime healthy. + // A bare require loads the .node addon on win32, so it catches an ABI + // mismatch on its own. What it cannot catch is *which* addon loaded: the + // published tarball ships a prebuilt built from unpatched source that is + // node-addon-api, so it requires cleanly and then reads every process's + // command line out of its address space. Check the binary, not the load. require(moduleName) + if (inspectWindowsProcessTreeAddon(windowsProcessTreeAddonPath()) === 'unpatched') { + throw new Error( + 'the loaded addon still calls ReadProcessMemory, so it was not built from the patched ' + + 'source. Rebuild it (pnpm run rebuild:electron) rather than using the published prebuild.' + ) + } return } if (moduleName === 'windows-native-registry') { @@ -368,6 +381,14 @@ function getWindowsBuildNumber() { function rebuildNodeRuntimeModules(moduleNames) { for (const moduleName of moduleNames) { const moduleDir = dirname(require.resolve(`${moduleName}/package.json`)) + if (moduleName === '@vscode/windows-process-tree') { + // Why before node-gyp: this module is rebuilt precisely because the + // binary was the unpatched one, and pnpm materializes it unpatched often + // enough that compiling the source as-is would just rebuild the same + // reader and fail the verify pass. + ensureWindowsProcessTreeCommandLinePatch(moduleDir) + stageWindowsProcessTreeNodeAddonApiHeaders(moduleDir) + } console.warn(`[native-runtime] Rebuilding ${moduleName} with node-gyp.`) runPnpm(['exec', 'node-gyp', 'rebuild'], { cwd: moduleDir }) if (moduleName === 'node-pty' && process.platform === 'win32') { diff --git a/config/scripts/ensure-native-runtime.test.mjs b/config/scripts/ensure-native-runtime.test.mjs index ea6e876e619..973e2f6852d 100644 --- a/config/scripts/ensure-native-runtime.test.mjs +++ b/config/scripts/ensure-native-runtime.test.mjs @@ -12,6 +12,7 @@ import { tmpdir } from 'node:os' import { delimiter, join } from 'node:path' import { fileURLToPath } from 'node:url' import { describe, expect, it } from 'vitest' +import { copyScriptWithLocalModules } from './script-module-dependencies.mjs' const sourceScriptPath = fileURLToPath(new URL('./ensure-native-runtime.mjs', import.meta.url)) const sourceNodePtyJobOwnershipPath = fileURLToPath( @@ -27,7 +28,6 @@ describe('ensure-native-runtime', () => { const logPath = join(projectDir, 'native-runtime.log') const markerPath = join(projectDir, 'rebuilt.marker') const binDir = join(projectDir, 'bin') - copyFileSync(sourceScriptPath, scriptPath) writeFakeNativeModules(projectDir) writeNodePtyPatchFile(projectDir) writeFakePnpm(binDir) @@ -67,7 +67,6 @@ describe('ensure-native-runtime', () => { const logPath = join(projectDir, 'native-runtime.log') const markerPath = join(projectDir, 'rebuilt.marker') const binDir = join(projectDir, 'bin') - copyFileSync(sourceScriptPath, scriptPath) writeFakeNativeModules(projectDir, { windowsRegistryRequiresMarker: true }) writeNodePtyPatchFile(projectDir) writeFakePnpm(binDir) @@ -102,7 +101,6 @@ describe('ensure-native-runtime', () => { const logPath = join(projectDir, 'native-runtime.log') const markerPath = join(projectDir, 'rebuilt.marker') const binDir = join(projectDir, 'bin') - copyFileSync(sourceScriptPath, scriptPath) writeLoadableNativeModules(projectDir) writeNodePtyPatchFile(projectDir) writeFakePnpm(binDir) @@ -137,7 +135,6 @@ describe('ensure-native-runtime', () => { const logPath = join(projectDir, 'native-runtime.log') const markerPath = join(projectDir, 'rebuilt.marker') const binDir = join(projectDir, 'bin') - copyFileSync(sourceScriptPath, scriptPath) writeLoadableNativeModules(projectDir) writeNodePtyPatchFile(projectDir) writePatchedNodePtyBuildArtifacts(projectDir) @@ -171,7 +168,6 @@ describe('ensure-native-runtime', () => { const logPath = join(projectDir, 'native-runtime.log') const markerPath = join(projectDir, 'rebuilt.marker') const binDir = join(projectDir, 'bin') - copyFileSync(sourceScriptPath, scriptPath) writeLoadableNativeModules(projectDir, { nativeDir: '../build/Release/' }) writeNodePtyPatchFile(projectDir) writePatchedNodePtyBuildArtifacts(projectDir) @@ -198,7 +194,9 @@ describe('ensure-native-runtime', () => { function mkTempProject() { const projectDir = mkdtempSync(join(tmpdir(), 'orca-native-runtime-')) - mkdirSync(join(projectDir, 'config', 'scripts'), { recursive: true }) + // Walked, not listed: the script imports windows-process-tree-gyp-rebuild.mjs, and a fixture + // missing it fails every case with a module-resolution error instead of the defect under test. + copyScriptWithLocalModules(sourceScriptPath, join(projectDir, 'config', 'scripts')) copyFileSync( sourceNodePtyJobOwnershipPath, join(projectDir, 'config', 'scripts', 'node-pty-job-ownership.cjs') diff --git a/config/scripts/patched-dependencies-frozen-install.test.mjs b/config/scripts/patched-dependencies-frozen-install.test.mjs new file mode 100644 index 00000000000..89f98ef074b --- /dev/null +++ b/config/scripts/patched-dependencies-frozen-install.test.mjs @@ -0,0 +1,155 @@ +import { + cpSync, + copyFileSync, + existsSync, + mkdirSync, + mkdtempSync, + readFileSync, + writeFileSync +} from 'node:fs' +import { tmpdir } from 'node:os' +import { isAbsolute, join, parse, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' +import { runProcessSync } from '../../src/shared/child-process/run-process.ts' +import { resolveCliCommand } from '../../src/shared/node-cli-command-resolution.ts' +import { removeTreeSync } from '../../src/shared/windows-transient-lock-removal.ts' +import { resolvePnpmCliInvocation } from './pnpm-cli-invocation.mjs' + +/** + * Run the command that actually consumes the patch hashes. + * + * A hash comparison is not this check. `@vscode/windows-process-tree@0.8.0` shipped + * twice with a hand-computed `sha256(patchBytes)` in the lockfile, and two separate + * reviews "verified" it by recomputing the same number the same wrong way. pnpm + * hashes the **LF-normalized** content, so a CRLF patch makes the raw digest a value + * pnpm will never produce, and `--frozen-lockfile` dies with + * ERR_PNPM_LOCKFILE_CONFIG_MISMATCH on every runner. An independent check that + * repeats the original assumption is not independent; only the installer is. + * + * `--lockfile-only --ignore-scripts` keeps it to the resolution pnpm rejects on, + * with no node_modules and no native builds. + */ +const PROJECT_DIR = resolve(import.meta.dirname, '../..') +const WINDOWS_PROCESS_TREE_PATCH = '@vscode__windows-process-tree@0.8.0.patch' + +/** + * Which pnpm to run belongs to pnpm-cli-invocation.mjs, not to this file: naming + * the Windows shim here is what windows-cmd-shim-spawn-boundary.test.mjs rejects. + * Its `shell` is dropped on purpose -- runProcessSync refuses that flag and + * already drives a shim through the interpreter itself. + */ +function resolvePnpmInvocation() { + const { command, prefixArgs } = resolvePnpmCliInvocation() + if (isAbsolute(command)) { + return existsSync(command) ? { program: command, prefixArgs } : null + } + // Bare name only when npm_execpath is unset (bare `vitest`, not `pnpm test`). + // Drop the extension so the shared resolver tries every executable form of it. + const resolved = resolveCliCommand(parse(command).name) + return isAbsolute(resolved) ? { program: resolved, prefixArgs } : null +} + +describe('patched dependencies', () => { + it('installs with --frozen-lockfile, which is what validates every patch hash', () => { + const pnpm = resolvePnpmInvocation() + expect(pnpm, 'pnpm must be installed; it is the only thing that can check this').not.toBeNull() + + // A copy, because a --frozen-lockfile run still rewrites parts of the + // lockfile this repo does not track, and the real one must not move. + const scratch = mkdtempSync(join(tmpdir(), 'orca-frozen-install-')) + try { + for (const file of ['package.json', 'pnpm-lock.yaml', 'pnpm-workspace.yaml']) { + copyFileSync(join(PROJECT_DIR, file), join(scratch, file)) + } + mkdirSync(join(scratch, 'config'), { recursive: true }) + cpSync(join(PROJECT_DIR, 'config', 'patches'), join(scratch, 'config', 'patches'), { + recursive: true + }) + + const result = runProcessSync({ + program: pnpm.program, + args: [ + ...pnpm.prefixArgs, + 'install', + '--frozen-lockfile', + '--lockfile-only', + '--ignore-scripts' + ], + cwd: scratch, + timeoutMs: 300_000 + }) + + expect(result.code, `${result.stdout}\n${result.stderr}`).toBe(0) + } finally { + removeTreeSync(scratch) + } + // The 300s spawn budget is only reachable if the case is allowed to take it; + // config/vitest.config.ts caps every case at 30s by default. + }, 300_000) + + /** + * `--lockfile-only` resolves; it never applies a patch. So the case above is + * bounded to hash consistency, and the actual question -- can pnpm still put + * the patched reader on disk? -- had nothing covering it. + * + * One package, patch applied for real, assert the marker landed. Scoped to the + * single dependency so it stays a ~2s check rather than a full install. + */ + it('materializes the patched command-line reader on a real install', () => { + const pnpm = resolvePnpmInvocation() + expect(pnpm, 'pnpm must be installed; it is the only thing that can check this').not.toBeNull() + + const scratch = mkdtempSync(join(tmpdir(), 'orca-patch-apply-')) + try { + mkdirSync(join(scratch, 'config', 'patches'), { recursive: true }) + copyFileSync( + join(PROJECT_DIR, 'config', 'patches', WINDOWS_PROCESS_TREE_PATCH), + join(scratch, 'config', 'patches', WINDOWS_PROCESS_TREE_PATCH) + ) + writeFileSync( + join(scratch, 'package.json'), + `${JSON.stringify( + { + name: 'orca-patch-apply-probe', + version: '1.0.0', + dependencies: { '@vscode/windows-process-tree': '0.8.0' } + }, + null, + 2 + )}\n` + ) + writeFileSync( + join(scratch, 'pnpm-workspace.yaml'), + 'packages: []\n' + + 'patchedDependencies:\n' + + ` '@vscode/windows-process-tree@0.8.0': config/patches/${WINDOWS_PROCESS_TREE_PATCH}\n` + ) + + const result = runProcessSync({ + program: pnpm.program, + args: [...pnpm.prefixArgs, 'install', '--no-frozen-lockfile', '--ignore-scripts'], + cwd: scratch, + timeoutMs: 300_000 + }) + expect(result.code, `${result.stdout}\n${result.stderr}`).toBe(0) + + const materialized = readFileSync( + join( + scratch, + 'node_modules', + '@vscode', + 'windows-process-tree', + 'src', + 'process_commandline.cc' + ), + 'utf8' + ) + expect(materialized).toContain('kProcessCommandLineInformation') + // The whole point of the patch: the upstream reader is gone, not merely + // supplemented. + expect(materialized).not.toContain('ReadProcessMemory') + } finally { + removeTreeSync(scratch) + } + }, 300_000) +}) diff --git a/config/scripts/pr-code-change-scope.mjs b/config/scripts/pr-code-change-scope.mjs index f5916d6a79c..c1d5c731cd4 100644 --- a/config/scripts/pr-code-change-scope.mjs +++ b/config/scripts/pr-code-change-scope.mjs @@ -213,6 +213,7 @@ const LINUX_PACKAGE_TESTS = [ const WINDOWS_PACKAGE_TESTS = [ ...LINUX_PACKAGE_TESTS, 'config/scripts/rebuild-native-deps.test.mjs', + 'config/scripts/rebuild-native-deps-windows-process-tree.test.mjs', 'src/main/providers/windows-conpty-wide-char-duplication.node-pty.test.ts', 'src/main/providers/pty-repaint-wide-char-buffer.node-pty.test.ts', 'src/shared/child-process/windows-command-line.win32.test.ts', @@ -220,6 +221,7 @@ const WINDOWS_PACKAGE_TESTS = [ 'src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts', 'src/main/windows/windows-pty-job.win32.test.ts', 'src/main/windows/windows-host-job.win32.test.ts', + 'src/main/windows/windows-process-tree-command-line-patch.test.ts', 'src/main/windows-live-tree-kill.win32.test.ts', 'src/main/wsl/wsl-runner.test.ts', 'src/main/wsl/wsl-guest-environment.test.ts', diff --git a/config/scripts/rebuild-native-deps-node-pty.test.mjs b/config/scripts/rebuild-native-deps-node-pty.test.mjs index 09e38853371..871732dd53d 100644 --- a/config/scripts/rebuild-native-deps-node-pty.test.mjs +++ b/config/scripts/rebuild-native-deps-node-pty.test.mjs @@ -4,6 +4,8 @@ import { join } from 'node:path' import { describe, expect, it } from 'vitest' import { + gitLineEndingEnv, + initGitWorkTree, mkTempProject, runRebuildScript, writeFakeElectronRebuild, @@ -14,7 +16,8 @@ import { writeFakeWindowsProcessTreeWithNodeAddonApi, writeFakeWindowsRegistry, writeNodePtyPatchFile, - writePatchedNodePtyBuildArtifacts + writePatchedNodePtyBuildArtifacts, + writeWindowsProcessTreePatchFile } from './rebuild-native-deps-test-fixtures.mjs' describe('rebuild-native-deps patched node-pty rebuild', () => { @@ -85,6 +88,91 @@ describe('rebuild-native-deps patched node-pty rebuild', () => { } }) + const commandLineSourcePath = (projectDir) => + join( + projectDir, + 'node_modules', + '@vscode', + 'windows-process-tree', + 'src', + 'process_commandline.cc' + ) + + // Why inside a git work tree: `git apply` run under one prefixes patch paths + // with the cwd-relative prefix, silently skips what does not match, and still + // exits 0. The package dir is always under the project root in production, so + // a fixture in %TEMP% alone would pass while the real repair did nothing. + // + // Why both line-ending modes: the patch is stored LF while upstream ships this + // source CRLF, so whether the pre-image matches depends on `core.autocrlf` -- + // and under `false`, Git's own built-in default, it did not. The repair blinds + // git to the repo, so that value comes from global config, i.e. from whichever + // option the developer's installer wrote. Pinning both makes the case cover the + // host that breaks rather than the host that happens to run it. + for (const autocrlf of ['false', 'true']) { + it(`repairs an un-applied command-line patch in a work tree (autocrlf=${autocrlf})`, () => { + const projectDir = mkTempProject() + + try { + initGitWorkTree(projectDir) + writeFakeUsableElectronPackage(projectDir, { platform: 'win32' }) + writeFakeElectronRebuild(projectDir) + writeFakeNodePtyConptyPayload(projectDir, 'x64') + writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir, { + commandLinePatchApplied: false + }) + writeWindowsProcessTreePatchFile(projectDir) + + const result = runRebuildScript( + projectDir, + { + npm_config_platform: 'win32', + npm_config_arch: 'x64', + ...gitLineEndingEnv(autocrlf) + }, + ['--platform=win32', '--arch=x64', '--force'] + ) + + expect(result.status, result.stderr).toBe(0) + expect(readFileSync(commandLineSourcePath(projectDir), 'utf8')).toContain( + 'kProcessCommandLineInformation' + ) + } finally { + removeTreeSync(projectDir) + } + }) + } + + // Why fail rather than build: an unpatched command-line reader compiles fine + // and then opens every process with PROCESS_VM_READ to walk its PEB, which is + // the primitive the patch exists to remove. + it('refuses a Windows rebuild when the command-line patch cannot be applied', () => { + const projectDir = mkTempProject() + + try { + initGitWorkTree(projectDir) + writeFakeUsableElectronPackage(projectDir, { platform: 'win32' }) + writeFakeElectronRebuild(projectDir) + writeFakeNodePtyConptyPayload(projectDir, 'x64') + writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir, { commandLinePatchApplied: false }) + // No patch file, so the repair has nothing to apply. + + const result = runRebuildScript( + projectDir, + { npm_config_platform: 'win32', npm_config_arch: 'x64' }, + ['--platform=win32', '--arch=x64', '--force'] + ) + + expect(result.status).not.toBe(0) + expect(result.stderr).toContain('process_commandline.cc') + expect(readFileSync(commandLineSourcePath(projectDir), 'utf8')).not.toContain( + 'kProcessCommandLineInformation' + ) + } finally { + removeTreeSync(projectDir) + } + }) + it('restores the ConPTY runtime payload after a Windows Electron rebuild', () => { const projectDir = mkTempProject() @@ -256,4 +344,37 @@ describe('rebuild-native-deps patched node-pty rebuild', () => { } } ) + + // The binary this step produces is the one copied into the packaged app. The + // relay build checks its own artifact and ensure-native-runtime checks what it + // loads; nothing checked this one, so a rebuild that quietly emitted the + // upstream reader shipped. Both non-clean states have to fail, which is the + // caller the tri-state was missing: after a rebuild that reported success, an + // absent binary is a broken build, not an absence to shrug at. + for (const [addon, expected] of [ + ['unpatched', 'still imports ReadProcessMemory'], + ['none', 'is not there'] + ]) { + it(`fails a Windows rebuild that leaves ${addon} windows-process-tree bytes`, () => { + const projectDir = mkTempProject() + + try { + writeFakeUsableElectronPackage(projectDir, { platform: 'win32' }) + writeFakeElectronRebuild(projectDir, { addon }) + writeFakeNodePtyConptyPayload(projectDir, 'x64') + writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir) + + const result = runRebuildScript( + projectDir, + { npm_config_platform: 'win32', npm_config_arch: 'x64' }, + ['--platform=win32', '--arch=x64', '--force'] + ) + + expect(result.status).not.toBe(0) + expect(result.stderr).toContain(expected) + } finally { + removeTreeSync(projectDir) + } + }) + } }) diff --git a/config/scripts/rebuild-native-deps-test-fixtures.mjs b/config/scripts/rebuild-native-deps-test-fixtures.mjs index 585e7a58ef2..2cb7d8ba8b4 100644 --- a/config/scripts/rebuild-native-deps-test-fixtures.mjs +++ b/config/scripts/rebuild-native-deps-test-fixtures.mjs @@ -1,5 +1,12 @@ import { spawnSync } from 'node:child_process' -import { chmodSync, copyFileSync, mkdirSync, mkdtempSync, writeFileSync } from 'node:fs' +import { + chmodSync, + copyFileSync, + mkdirSync, + mkdtempSync, + readFileSync, + writeFileSync +} from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' import { fileURLToPath } from 'node:url' @@ -15,6 +22,68 @@ const sourceNodePtyJobOwnershipPath = fileURLToPath( const sourceWindowsProcessTreeGypRebuildPath = fileURLToPath( new URL('./windows-process-tree-gyp-rebuild.mjs', import.meta.url) ) +const sourceWindowsProcessTreePatchPath = fileURLToPath( + new URL('../patches/@vscode__windows-process-tree@0.8.0.patch', import.meta.url) +) + +/** + * The command-line reader as it is *before* the patch, taken from the patch's + * own pre-image so no upstream copy has to be vendored. + * + * Written back as **CRLF**, which is what `@vscode/windows-process-tree@0.8.0` + * actually ships: all 67 pre-image lines of this file carried a CR before the + * patch was normalized to LF. Rebuilding it with the patch's current newline + * instead would make fixture and patch agree by construction, on any encoding — + * which is exactly how a repair that cannot apply to the real package passed + * this suite. + */ +function unpatchedWindowsProcessTreeCommandLineSource() { + const lines = readFileSync(sourceWindowsProcessTreePatchPath, 'utf8').split('\n') + const start = lines.findIndex((line) => + line.startsWith('diff --git a/src/process_commandline.cc ') + ) + const rest = lines.slice(start + 1) + const end = rest.findIndex((line) => line.startsWith('diff --git ')) + const preImage = (end === -1 ? rest : rest.slice(0, end)) + .filter((line) => line.startsWith(' ') || line.startsWith('-')) + .filter((line) => !line.startsWith('---')) + .map((line) => line.slice(1).replace(/\r$/, '')) + .join('\r\n') + // Splitting drops the file's own trailing newline as an empty element, and + // `git apply` needs the bytes exact. + return `${preImage}\r\n` +} + +/** + * Pin `core.autocrlf` for a spawned repair, whatever the host is set to. + * + * The repair blinds git to the surrounding repo with `GIT_DIR`, so the value it + * sees comes from global/system config — on a Git for Windows box that is + * whichever line-ending option the installer wrote, and `false` (Git's built-in + * default, "checkout as-is") is the one the repair used to fail under. A global + * config in a temp HOME outranks the system file, so this is deterministic + * rather than whatever the developer happens to have. + */ +export function gitLineEndingEnv(autocrlf) { + const home = mkdtempSync(join(tmpdir(), `orca-git-home-${autocrlf}-`)) + writeFileSync(join(home, '.gitconfig'), `[core]\n\tautocrlf = ${autocrlf}\n`) + return { HOME: home, USERPROFILE: home } +} + +/** Production always runs the repair from inside a work tree; `git apply` behaves differently there. */ +export function initGitWorkTree(projectDir) { + for (const args of [['init'], ['config', 'user.email', 'a@b.c'], ['config', 'user.name', 't']]) { + spawnSync('git', args, { cwd: projectDir, encoding: 'utf8' }) + } +} + +export function writeWindowsProcessTreePatchFile(projectDir) { + mkdirSync(join(projectDir, 'config', 'patches'), { recursive: true }) + copyFileSync( + sourceWindowsProcessTreePatchPath, + join(projectDir, 'config', 'patches', '@vscode__windows-process-tree@0.8.0.patch') + ) +} export function mkTempProject() { const projectDir = mkdtempSync(join(tmpdir(), 'orca-rebuild-native-deps-')) @@ -143,17 +212,46 @@ if (${JSON.stringify(createExecutable)}) { ) } -export function writeFakeElectronRebuild(projectDir, { logPathEnv = null } = {}) { +/** Bytes that stand in for a compiled addon's import table. */ +const FAKE_ADDON_BYTES = { + clean: 'MZ\0ntdll.dll\0NtQueryInformationProcess\0', + unpatched: 'MZ\0KERNEL32.dll\0ReadProcessMemory\0' +} + +/** + * A rebuild that produces nothing leaves no addon to inspect, and the script now + * asserts the binary it just built is a patched one. Emit a stand-in so the + * fixture models a rebuild that actually succeeded. `addon` picks which kind, + * because "produced the upstream reader" and "produced nothing" are both real + * outcomes that assertion has to tell apart. + */ +export function writeFakeElectronRebuild(projectDir, { logPathEnv = null, addon = 'clean' } = {}) { const rebuildDir = join(projectDir, 'node_modules', '@electron', 'rebuild') mkdirSync(rebuildDir, { recursive: true }) writeFileSync(join(rebuildDir, 'package.json'), JSON.stringify({ type: 'module' })) + const emitAddon = + addon === 'none' + ? '' + : ` + const packageDir = join('node_modules', '@vscode', 'windows-process-tree') + if (existsSync(join(packageDir, 'package.json'))) { + mkdirSync(join(packageDir, 'build', 'Release'), { recursive: true }) + writeFileSync( + join(packageDir, 'build', 'Release', 'windows_process_tree.node'), + ${JSON.stringify(FAKE_ADDON_BYTES[addon])} + ) + }` + const emitImports = + addon === 'none' + ? '' + : "import { existsSync, mkdirSync, writeFileSync } from 'node:fs'\nimport { join } from 'node:path'\n" writeFileSync( join(rebuildDir, 'index.js'), logPathEnv ? ` import { appendFileSync } from 'node:fs' - -export async function rebuild(options) { +${emitImports} +export async function rebuild(options) {${emitAddon} const logPath = process.env[${JSON.stringify(logPathEnv)}] if (!logPath) { return @@ -171,7 +269,10 @@ export async function rebuild(options) { ) } ` - : 'export async function rebuild() {}\n' + : `${emitImports} +export async function rebuild() {${emitAddon} +} +` ) } @@ -271,12 +372,22 @@ export function writeFakeWindowsProcessTree(projectDir) { writeFileSync(join(processTreeDir, 'index.js'), 'module.exports = {}\n') } -export function writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir) { +export function writeFakeWindowsProcessTreeWithNodeAddonApi( + projectDir, + { commandLinePatchApplied = true } = {} +) { const processTreeDir = join(projectDir, 'node_modules', '@vscode', 'windows-process-tree') const nodeAddonApiDir = join(processTreeDir, 'node_modules', 'node-addon-api') mkdirSync(nodeAddonApiDir, { recursive: true }) writeFileSync(join(processTreeDir, 'package.json'), '{"dependencies":{"node-addon-api":"*"}}\n') writeFileSync(join(processTreeDir, 'index.js'), 'module.exports = {}\n') + mkdirSync(join(processTreeDir, 'src'), { recursive: true }) + writeFileSync( + join(processTreeDir, 'src', 'process_commandline.cc'), + commandLinePatchApplied + ? '// kProcessCommandLineInformation = 60\n' + : unpatchedWindowsProcessTreeCommandLineSource() + ) writeFileSync(join(nodeAddonApiDir, 'package.json'), '{"name":"node-addon-api"}\n') writeFileSync(join(nodeAddonApiDir, 'napi.h'), '// napi.h\n') writeFileSync(join(nodeAddonApiDir, 'napi-inl.h'), '// napi-inl.h\n') diff --git a/config/scripts/rebuild-native-deps-windows-process-tree.test.mjs b/config/scripts/rebuild-native-deps-windows-process-tree.test.mjs new file mode 100644 index 00000000000..4f98c1b092d --- /dev/null +++ b/config/scripts/rebuild-native-deps-windows-process-tree.test.mjs @@ -0,0 +1,103 @@ +import { spawn } from 'node:child_process' +import { appendFileSync, copyFileSync, existsSync, mkdirSync } from 'node:fs' +import { createRequire } from 'node:module' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { removeTreeSync } from '../../src/shared/windows-transient-lock-removal.ts' + +import { + mkTempProject, + runRebuildScript, + writeFakeElectronRebuild, + writeFakeNodePtyConptyPayload, + writeFakeUsableElectronPackage, + writeFakeWindowsProcessTreeWithNodeAddonApi +} from './rebuild-native-deps-test-fixtures.mjs' + +const require = createRequire(import.meta.url) + +/** A real loadable addon, so the OS holds the same lock a running Orca holds. */ +function repoAddonPath() { + try { + const entry = require.resolve('@vscode/windows-process-tree') + const built = join(entry, '..', '..', 'build', 'Release', 'windows_process_tree.node') + return existsSync(built) ? built : null + } catch { + return null + } +} + +/** + * Stage a stale addon and keep it loaded, exactly as a running Orca does. + * + * The bytes are the repo's own patched build with the flagged import appended, + * because the guard keys on that symbol and the patched binary does not carry + * it. Trailing bytes are PE overlay, so the file still loads. + */ +async function stageLoadedStaleAddon(projectDir) { + const source = repoAddonPath() + const releaseDir = join( + projectDir, + 'node_modules', + '@vscode', + 'windows-process-tree', + 'build', + 'Release' + ) + mkdirSync(releaseDir, { recursive: true }) + const stale = join(releaseDir, 'windows_process_tree.node') + copyFileSync(source, stale) + appendFileSync(stale, 'ReadProcessMemory') + + const holder = spawn( + process.execPath, + ['-e', 'require(process.argv[1]); process.send("held"); setInterval(() => {}, 1000)', stale], + { stdio: ['ignore', 'ignore', 'ignore', 'ipc'] } + ) + await new Promise((resolve, reject) => { + holder.once('message', resolve) + holder.once('exit', () => reject(new Error('the addon holder exited before loading'))) + }) + return holder +} + +// Why an end-to-end run: the defect was purely one of placement. The guard threw +// a real EPERM, and the classifier that turns that into "close running Orca" +// already existed -- the throw simply happened before the try that reaches it. +// Only the whole script exercises that. +describe.runIf(process.platform === 'win32')('rebuild-native-deps stale addon under lock', () => { + it.skipIf(!repoAddonPath())( + 'reports a locked stale addon as a Windows file lock instead of an EPERM stack', + async () => { + const projectDir = mkTempProject() + let holder + + try { + writeFakeUsableElectronPackage(projectDir, { platform: 'win32' }) + writeFakeElectronRebuild(projectDir) + writeFakeNodePtyConptyPayload(projectDir, process.arch) + writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir) + holder = await stageLoadedStaleAddon(projectDir) + + const result = runRebuildScript( + projectDir, + { + npm_lifecycle_event: 'postinstall', + npm_config_platform: 'win32', + npm_config_arch: process.arch + }, + ['--platform=win32', `--arch=${process.arch}`, '--force'] + ) + + expect(result.stderr).toContain( + 'Close running Orca/Electron/dev processes for this worktree' + ) + // Non-strict postinstall soft-exits on a lock; the next dev/start re-checks. + expect(result.status, result.stderr).toBe(0) + } finally { + holder?.kill() + removeTreeSync(projectDir) + } + } + ) +}) diff --git a/config/scripts/rebuild-native-deps.mjs b/config/scripts/rebuild-native-deps.mjs index 3b17683e831..863aac850a1 100644 --- a/config/scripts/rebuild-native-deps.mjs +++ b/config/scripts/rebuild-native-deps.mjs @@ -20,7 +20,12 @@ import { rebuild } from '@electron/rebuild' import { execFileSync, spawnSync } from 'node:child_process' -import { stageWindowsProcessTreeNodeAddonApiHeaders } from './windows-process-tree-gyp-rebuild.mjs' +import { + ensureWindowsProcessTreeCommandLinePatch, + inspectWindowsProcessTreeAddon, + stageWindowsProcessTreeNodeAddonApiHeaders, + windowsProcessTreeAddonPath +} from './windows-process-tree-gyp-rebuild.mjs' import { copyFileSync, existsSync, @@ -141,15 +146,21 @@ if (!ignoreModules.includes('cpu-features')) { } } -if ( - rebuildPlatform === 'win32' && - modulesToRebuild.includes('@vscode/windows-process-tree') && - existsSync(join(projectDir, 'node_modules', '@vscode', 'windows-process-tree', 'package.json')) -) { - stageWindowsProcessTreeNodeAddonApiHeaders() -} - try { + // Why inside the try: the patch guard deletes a stale addon binary, and that + // delete fails EPERM when the addon is loaded -- exactly the running-Orca case + // the catch below is written for. Outside, it aborted `pnpm install` with a + // raw stack instead of the "close running Orca/Electron processes" message. + if ( + rebuildPlatform === 'win32' && + modulesToRebuild.includes('@vscode/windows-process-tree') && + existsSync(join(projectDir, 'node_modules', '@vscode', 'windows-process-tree', 'package.json')) + ) { + stageWindowsProcessTreeNodeAddonApiHeaders() + if (ensureWindowsProcessTreeCommandLinePatch()) { + console.warn('[rebuild] Repaired the un-applied windows-process-tree command-line patch.') + } + } await rebuild({ buildPath: projectDir, electronVersion, @@ -165,6 +176,7 @@ try { force: true }) restoreNodePtyWindowsConptyRuntime() + assertWindowsProcessTreeAddonIsPatched() } catch (/** @type {any} */ err) { console.error('[rebuild] Native module rebuild failed:', err?.message ?? err) if (isWindowsNativeLockError(err)) { @@ -184,6 +196,40 @@ try { process.exit(1) } +/** + * The binary this rebuild just produced is the one the packaged app ships. + * + * The relay build asserts its own artifact and `ensure-native-runtime.mjs` + * asserts what it loads, but nothing checked the addon that gets copied into the + * packaged `node_modules` -- so a rebuild that silently produced the upstream + * reader would reach users. Anything but `clean` fails: after a rebuild that + * reported success the binary must exist, so `missing` is a broken build, not an + * absence to shrug at. This is the caller that needs the state to be a state and + * not a boolean. + */ +function assertWindowsProcessTreeAddonIsPatched() { + if ( + rebuildPlatform !== 'win32' || + !modulesToRebuild.includes('@vscode/windows-process-tree') || + !existsSync(join(projectDir, 'node_modules', '@vscode', 'windows-process-tree', 'package.json')) + ) { + return + } + const addonPath = windowsProcessTreeAddonPath() + const state = inspectWindowsProcessTreeAddon(addonPath) + if (state === 'clean') { + return + } + throw new Error( + state === 'missing' + ? `the rebuild reported success but ${addonPath} is not there, so the packaged app would ` + + 'ship no windows-process-tree addon at all.' + : `${addonPath} still imports ReadProcessMemory, so it was not built from the patched ` + + 'command-line reader. The packaged app would carry the primitive MDE scores as ' + + 'credential dumping.' + ) +} + function restoreNodePtyWindowsConptyRuntime() { if (rebuildPlatform !== 'win32' || !onlyModules.includes('node-pty')) { return diff --git a/config/scripts/windows-process-tree-gyp-rebuild.mjs b/config/scripts/windows-process-tree-gyp-rebuild.mjs index c815407d6d0..20d91e55497 100644 --- a/config/scripts/windows-process-tree-gyp-rebuild.mjs +++ b/config/scripts/windows-process-tree-gyp-rebuild.mjs @@ -9,7 +9,8 @@ * hop escapes the store and configure fails with "node_addon_api.gyp not * found" (run 32999886072). */ -import { copyFileSync, mkdirSync, realpathSync } from 'node:fs' +import { execFileSync } from 'node:child_process' +import { copyFileSync, existsSync, mkdirSync, readFileSync, realpathSync, rmSync } from 'node:fs' import { createRequire } from 'node:module' import { dirname, join, resolve } from 'node:path' @@ -22,6 +23,16 @@ export const WINDOWS_PROCESS_TREE_PACKAGE_DIR = join( 'windows-process-tree' ) +export const WINDOWS_PROCESS_TREE_PATCH_PATH = join( + ROOT, + 'config', + 'patches', + '@vscode__windows-process-tree@0.8.0.patch' +) + +/** Only the patched reader defines this; the upstream one walks the PEB. */ +const COMMAND_LINE_PATCH_MARKER = 'kProcessCommandLineInformation' + export const WINDOWS_PROCESS_TREE_NODE_ADDON_API_HEADERS = [ 'napi.h', 'napi-inl.h', @@ -39,6 +50,119 @@ export function nodeGypRebuildInvocation(arch, packageDir = WINDOWS_PROCESS_TREE } } +/** The binary the addon actually loads. */ +export function windowsProcessTreeAddonPath(packageDir = WINDOWS_PROCESS_TREE_PACKAGE_DIR) { + return join(packageDir, 'build', 'Release', 'windows_process_tree.node') +} + +/** The import whose absence tells the patched binary from the published prebuilt. */ +const FLAGGED_IMPORT = 'ReadProcessMemory' + +/** + * Does this compiled addon still carry the flagged primitive? + * + * The patched reader never calls `ReadProcessMemory`, so the symbol is absent + * from its import table; the upstream build imports it. That makes this a + * property of the binary rather than of the source next to it, which matters + * because the published tarball ships a *loadable* prebuilt built from + * unpatched source: it is node-addon-api, so it satisfies a bare `require()` + * under both Node and Electron, and a skipped rebuild would use it. + * + * Tri-state, not a predicate: a binary that is not there has not been cleared, + * and a boolean makes "absent" indistinguishable from "verified clean" at every + * call site. Takes the binary path so the relay's staged addon -- which sits + * beside the bundle, with no package around it -- gets the same check. + * + * @param {string} addonPath + * @returns {'clean' | 'unpatched' | 'missing'} + */ +export function inspectWindowsProcessTreeAddon(addonPath) { + if (!existsSync(addonPath)) { + return 'missing' + } + return readFileSync(addonPath).includes(FLAGGED_IMPORT) ? 'unpatched' : 'clean' +} + +/** + * Refuse to compile or load the upstream command-line reader. + * + * Unpatched, it opens every process with `PROCESS_VM_READ` and walks the PEB to + * recover the command line -- the primitive MDE scores as credential dumping, + * and the reason this package is patched at all. pnpm has been seen + * materializing this CRLF package with its patch missing, so repair the source + * from the patch file, and drop any binary that predates the repair. + */ +export function ensureWindowsProcessTreeCommandLinePatch( + packageDir = WINDOWS_PROCESS_TREE_PACKAGE_DIR +) { + const source = join(packageDir, 'src', 'process_commandline.cc') + if (!existsSync(source)) { + throw new Error( + `${source} is missing, so the command-line patch cannot be verified. Run pnpm install.` + ) + } + let repaired = false + + if (!readFileSync(source, 'utf8').includes(COMMAND_LINE_PATCH_MARKER)) { + try { + execFileSync( + 'git', + [ + // Why force the line-ending mode: the patch is stored LF (a contract + // test forbids CR bytes in it), but upstream ships this source CRLF, + // so its pre-image lines and the file's differ by a CR. Under + // `core.autocrlf=false` -- Git's own built-in default, and what + // "checkout as-is" selects in the Git for Windows installer -- git + // compares them literally, the hunk does not match, and the repair + // throws. `input` normalizes line endings for that comparison and + // nothing else, so a hunk whose real content drifted is still + // rejected. Measured: without it, apply exits 1 at autocrlf=false and + // 0 at true/input; with it, 0 for CRLF and LF sources under all three. + '-c', + 'core.autocrlf=input', + 'apply', + '--include=src/process_commandline.cc', + WINDOWS_PROCESS_TREE_PATCH_PATH + ], + { + cwd: realpathSync(packageDir), + stdio: 'pipe', + // Why blind git to the repo: run inside a work tree, `git apply` + // prefixes patch paths with the cwd-relative prefix, silently skips + // everything that does not match -- and still exits 0. The package + // dir is always under the project root, so without this the repair + // reports success and changes nothing. + env: { ...process.env, GIT_DIR: join(packageDir, '.orca-no-such-git-dir') } + } + ) + } catch (error) { + throw new Error( + 'src/process_commandline.cc still reads the PEB, and repairing it from ' + + `${WINDOWS_PROCESS_TREE_PATCH_PATH} failed: ${error?.message ?? error}. Run pnpm install.` + ) + } + if (!readFileSync(source, 'utf8').includes(COMMAND_LINE_PATCH_MARKER)) { + throw new Error( + 'src/process_commandline.cc still reads the PEB after repair, so the patch did not ' + + 'apply. Run pnpm install.' + ) + } + repaired = true + } + + // A binary from before the repair -- or the tarball's own prebuilt -- would + // otherwise survive a skipped rebuild and load the flagged reader anyway. + // Deleting it can fail EPERM against a loaded (memory-mapped) addon, which + // `force: true` does not cover -- it only swallows ENOENT. That throw is the + // caller's to classify as a Windows file lock, so it must not be swallowed. + if (inspectWindowsProcessTreeAddon(windowsProcessTreeAddonPath(packageDir)) === 'unpatched') { + rmSync(windowsProcessTreeAddonPath(packageDir), { force: true }) + repaired = true + } + + return repaired +} + // Patched binding.gyp includes deps/node-addon-api; the tarball does not ship those headers. export function stageWindowsProcessTreeNodeAddonApiHeaders( packageDir = WINDOWS_PROCESS_TREE_PACKAGE_DIR diff --git a/config/scripts/windows-process-tree-gyp-rebuild.test.mjs b/config/scripts/windows-process-tree-gyp-rebuild.test.mjs index f4820e9430a..f2939b71179 100644 --- a/config/scripts/windows-process-tree-gyp-rebuild.test.mjs +++ b/config/scripts/windows-process-tree-gyp-rebuild.test.mjs @@ -10,8 +10,9 @@ import { } from 'node:fs' import { tmpdir } from 'node:os' import { join, resolve } from 'node:path' -import { describe, expect, it } from 'vitest' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' import { + inspectWindowsProcessTreeAddon, nodeGypRebuildInvocation, stageWindowsProcessTreeNodeAddonApiHeaders, WINDOWS_PROCESS_TREE_NODE_ADDON_API_HEADERS, @@ -59,3 +60,40 @@ describe('windows-process-tree node-gyp rebuild', () => { } }) }) + +describe('inspecting a compiled windows-process-tree addon', () => { + let dir + + beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), 'orca-windows-process-tree-addon-')) + }) + afterEach(() => { + rmSync(dir, { recursive: true, force: true }) + }) + + it('reports a binary that still imports ReadProcessMemory as unpatched', () => { + const addonPath = join(dir, 'windows_process_tree.node') + writeFileSync(addonPath, Buffer.from('MZ\0\0KERNEL32.dll\0ReadProcessMemory\0', 'binary')) + expect(inspectWindowsProcessTreeAddon(addonPath)).toBe('unpatched') + }) + + it('reports a binary without the import as clean', () => { + const addonPath = join(dir, 'windows_process_tree.node') + writeFileSync(addonPath, Buffer.from('MZ\0\0ntdll.dll\0NtQueryInformationProcess\0', 'binary')) + expect(inspectWindowsProcessTreeAddon(addonPath)).toBe('clean') + }) + + // The whole point of the tri-state: absence is not evidence of safety, and a + // boolean made "there is no binary" indistinguishable from "checked, clean". + it('reports an absent binary as missing rather than clean', () => { + expect(inspectWindowsProcessTreeAddon(join(dir, 'windows_process_tree.node'))).toBe('missing') + }) + + it('inspects whatever path it is handed, including a relay-staged addon', () => { + // The relay loads `./windows-process-tree.node` beside its bundle, which is + // nowhere near a node_modules package directory. + const staged = join(dir, 'windows-process-tree.node') + writeFileSync(staged, Buffer.from('MZ\0\0ReadProcessMemory\0', 'binary')) + expect(inspectWindowsProcessTreeAddon(staged)).toBe('unpatched') + }) +}) diff --git a/docs/reference/windows-edr-posture.md b/docs/reference/windows-edr-posture.md index 68614932c14..ff3e6d49cc6 100644 --- a/docs/reference/windows-edr-posture.md +++ b/docs/reference/windows-edr-posture.md @@ -13,7 +13,7 @@ two escalated to multi-stage incidents carrying ATT&CK tactic mappings The framing this document keeps throughout, because both halves matter: > **Defender is not malfunctioning. It is describing the code accurately.** Orca -> really does copy its own signed image under a different name, really does read +> really does copy its own signed image under a different name, really did read > every process's memory on a timer, really does run base64-encoded PowerShell > with the execution policy bypassed, and really does take screenshots and > synthesise input from a runtime-compiled assembly. Each of those is a @@ -88,7 +88,9 @@ embedded name for the old disk name to contradict. **one** flag set, `CommandLine | CreationTime`, shared by every caller. pid, ppid and name come out of the snapshot itself and open nothing. `CommandLine` is what opens a handle: the addon calls `GetProcessCommandLine` per process, which opens -`PROCESS_QUERY_INFORMATION | PROCESS_VM_READ` and walks the PEB with three +`PROCESS_QUERY_LIMITED_INFORMATION` — the same right Task Manager takes — and +asks the kernel for the string. Upstream it opened +`PROCESS_QUERY_INFORMATION | PROCESS_VM_READ` and walked the PEB with three `ReadProcessMemory` calls (`src/process_commandline.cc:32,41-47` in the vendored `@vscode/windows-process-tree` 0.8.0 source that `config/patches/` patches). @@ -96,8 +98,9 @@ opens a handle: the addon calls `GetProcessCommandLine` per process, which opens `GetProcessMemoryUsage` open a **second** `PROCESS_QUERY_INFORMATION | PROCESS_VM_READ` handle per process for a `GetProcessMemoryInfo` call whose result no caller read (`src/process.cc:47-63`). Dropping it halves the handles -opened per snapshot. It does not remove the remote memory read, because the -command line still performs one. +opened per snapshot. On its own it removed no memory read — both handles carried +`PROCESS_VM_READ` at the time — so it composes with the patch below rather than +substituting for it. It exists because seven independent readers used to fork `powershell.exe` for a `Get-CimInstance Win32_Process` scan. That cost, measured: a PowerShell @@ -121,30 +124,37 @@ processes. That design is not in the tree and those numbers describe no code path here; the figures that do apply are the module's own, in [`windows-process-enumeration.md`](./windows-process-enumeration.md). -**How an EDR reads it:** a cross-process handle plus a remote memory read against +**How an EDR read it:** a cross-process handle plus a remote memory read against every process on the box, repeating on a cadence, is the read half of the telemetry that credential dumping and process injection produce. MDE surfaced it as "suspicious memory activity". -**That signal is still present.** An earlier revision of this file claimed the -command line "now comes from the kernel" through `NtQueryInformationProcess`'s -`ProcessCommandLineInformation` class, needing only -`PROCESS_QUERY_LIMITED_INFORMATION`, and that `ReadProcessMemory` was absent from -the compiled addon. None of that is true of the code we ship. -`process_commandline.cc` calls `NtQueryInformationProcess` with -`ProcessBasicInformation` only — to locate the PEB — and then issues three -`ReadProcessMemory` calls against a `PROCESS_VM_READ` handle to read the PEB, the -`RTL_USER_PROCESS_PARAMETERS`, and the command-line buffer. Nothing asserts an -import table, and no such assertion would pass. +**The memory read is gone.** A fourth hunk in +`config/patches/@vscode__windows-process-tree@0.8.0.patch` has +`GetProcessCommandLine` call `NtQueryInformationProcess` with +`ProcessCommandLineInformation` (class 60, Windows 8.1+; Electron's floor is +Windows 10), which returns a `UNICODE_STRING` the kernel builds and needs only +`PROCESS_QUERY_LIMITED_INFORMATION`. Measured on ~540 processes, per detailed +scan: `ReadProcessMemory` 1128 → **0**, desired access `0x0410` → `0x1000`, with +byte-identical command lines on every process both readers recovered. There is no +PEB fallback to reinstate it — a hooked `ntdll` answering +`STATUS_INVALID_INFO_CLASS` for one target would have flipped a process-wide, +one-way switch back to `PROCESS_VM_READ` on exactly the machines this exists for. -What this change did remove is the `Memory` flag's second handle and its -`GetProcessMemoryInfo` call, so the per-process handle count per snapshot halves. -What remains to declare to administrators is unchanged in kind: one -`PROCESS_QUERY_INFORMATION | PROCESS_VM_READ` handle and a PEB read against every -process on the box, at the shared snapshot's cadence. Moving to -`ProcessCommandLineInformation` (Windows 8.1+, `PROCESS_QUERY_LIMITED_INFORMATION` -only) would genuinely retire the remote read, but it is an addon patch nobody has -written; treat it as unclaimed work, not as shipped. +Because the property is the *absence* of an import, it is checkable on the +artifact rather than the source: `inspectWindowsProcessTreeAddon()` answers +`clean` / `unpatched` / `missing`, and the rebuild, `ensure-native-runtime.mjs`, +the relay build and `loadWindowsProcessTree()` all key on it. That check is load- +bearing because the published tarball ships a *loadable* prebuilt built from +unpatched source, so "it required cleanly" is not evidence. + +What to declare to administrators is now one +`PROCESS_QUERY_LIMITED_INFORMATION` handle per process on a detailed snapshot and +no remote memory access at all. What this does not narrow is _which_ processes +are asked — a detailed scan still queries every pid, including `lsass.exe`. +Restricting the command-line pass to Orca's own subtree needs job-object +membership as its source of truth (a ppid-derived allowlist would miss the +detached, reparented descendants of #9045 and #10475), and remains unclaimed work. ### Encoded, policy-bypassing PowerShell diff --git a/docs/reference/windows-process-enumeration.md b/docs/reference/windows-process-enumeration.md index 87ac2a97fb1..fb58030be6e 100644 --- a/docs/reference/windows-process-enumeration.md +++ b/docs/reference/windows-process-enumeration.md @@ -189,7 +189,7 @@ on any other OS keeps using the scan. ## Why the package is patched -`config/patches/@vscode__windows-process-tree@0.8.0.patch` carries three hunks. +`config/patches/@vscode__windows-process-tree@0.8.0.patch` carries four hunks. 1. **Spectre mitigation.** The upstream `binding.gyp` requires Spectre-mitigated libraries, which Orca's Windows build agents do not install. `node-pty` is @@ -204,10 +204,122 @@ on any other OS keeps using the scan. realpath, then loads the relative path from the `node_modules` symlink, so `node_addon_api.gyp` resolves outside the repo and hourly Windows builds die at configure. `node-pty` is patched the same way for the same reason. +4. **No PEB reads, no `PROCESS_VM_READ`.** See below. The typings claim `commandLine` is truncated at 512 characters. Measured, it is not: the longest observed on a real host was 26,059. +### The command line comes from the kernel, not the target's memory + +Upstream, `GetProcessCommandLine` opens every process with +`PROCESS_QUERY_INFORMATION | PROCESS_VM_READ` and issues three chained +`ReadProcessMemory` calls — PEB, `RTL_USER_PROCESS_PARAMETERS`, then the string +— to recover the command line. Walking another process's address space for +credentials-adjacent data on a repeating timer is what a credential dumper does, +so Defender for Endpoint scores it as such regardless of intent. Nothing about +the flag sets above changes that; only removing the read does. + +Windows 8.1 added `NtQueryInformationProcess`'s `ProcessCommandLineInformation` +class (60), which returns the same string as a `UNICODE_STRING` the kernel +builds, needing only `PROCESS_QUERY_LIMITED_INFORMATION`. Electron's floor is +Windows 10, so every OS Orca supports has it. The entry point is resolved with +`GetProcAddress` on `ntdll.dll` — it has no import library — and the size is +probed with a null-buffer call that answers `STATUS_INFO_LENGTH_MISMATCH`. + +The same hunk drops `PROCESS_VM_READ` from `GetProcessMemoryUsage` and +`GetCpuUsage`, which acquired it and never read an address space: +`GetProcessMemoryInfo` and `GetProcessTimes` are satisfied by +`PROCESS_QUERY_LIMITED_INFORMATION`. Measured, both return identical values +under the weaker right on every process that opens at all. + +Measured on Windows 11, ~540 processes, counted in-process by replacing the +addon's import table entries with counting stubs: + +| per `CommandLine` scan | before | after | +| ---------------------- | ----------------------------------------- | -------------------------------------- | +| `OpenProcess` calls | 543 | 543 | +| desired access | `0x0410` (`VM_READ \| QUERY_INFORMATION`) | `0x1000` (`QUERY_LIMITED_INFORMATION`) | +| `ReadProcessMemory` | 1128 | **0** | +| p50 / p95 | 13.5 / 14.5 ms | 12.3 / 13.5 ms | + +Command lines were byte-identical on every process both readers recovered +(405/405, and 399/399 and 376/376 on other runs), including a 24,087-character +argv with embedded quotes, non-ASCII characters and trailing whitespace, and a +WOW64 target. The weaker right is also a strict superset in reach: three +processes that refused `PROCESS_QUERY_INFORMATION | PROCESS_VM_READ` granted +`PROCESS_QUERY_LIMITED_INFORMATION`, and none went the other way. + +### There is no PEB fallback, deliberately + +An earlier revision kept the PEB reader for a kernel without class 60, behind a +latch. That was wrong, and the reason is worth recording: `ClassifyQueryFailure` +mapped `STATUS_INVALID_INFO_CLASS` / `NOT_SUPPORTED` / `NOT_IMPLEMENTED` from +**any single target** onto a process-wide, one-way switch back to +`PROCESS_VM_READ` plus three `ReadProcessMemory` per pid per scan, for the life +of the process, with nothing observable from JS. + +The environment this reader exists for is one where an EDR hooks `ntdll`. A hook +that returns `STATUS_INVALID_INFO_CLASS` for a class it does not recognise would +have silently reinstated the exact primitive the patch removes, on precisely the +machines it was written for — and one stray status from one process was enough. +The same applies under Wine or any instrumented `ntdll`. + +So the fallback is gone rather than guarded. `GetProcessCommandLine` returns +false and leaves the command line empty, which is already a normal outcome +(`WindowsProcessRow.command` is documented as empty when a process denies a +query handle, and callers fall back to the image name). Degrading to no command +line is recoverable; silently resuming address-space reads is not. + +This also makes the property checkable on the artifact rather than the source: +the patched reader never calls `ReadProcessMemory`, so the symbol is absent from +the compiled addon's import table. `inspectWindowsProcessTreeAddon()` in +`config/scripts/windows-process-tree-gyp-rebuild.mjs` is that check, and it is +the only way to tell the two binaries apart — see below. It answers +`clean` / `unpatched` / `missing` rather than a boolean, because a binary that is +not there has not been cleared, and a caller reading `false` as “verified” would +pass exactly the thing the check exists to catch. + +Because the returned `UNICODE_STRING` comes from that same hookable boundary, +its `Buffer` and `Length` are bounds-checked against the allocation before the +characters are encoded, and the probed size is capped at the header plus 64 KiB +(`Length` is a `USHORT`) so a bogus size cannot turn into a `bad_alloc` that +fails an entire scan instead of one process. + +### The published tarball ships a loadable unpatched prebuilt + +`@vscode/windows-process-tree@0.8.0` publishes +`build/Release/windows_process_tree.node` in the tarball. It is node-addon-api, +so it is ABI-stable and loads cleanly under both Node and Electron — and it was +built from unpatched source, so it performs 1179 `ReadProcessMemory` calls and +opens every process at `0x0410` per scan. + +That matters because `allowBuilds` is `false` for this package and CI installs +with `--ignore-scripts`, so nothing compiles it at install time. A `require()` +health check cannot tell the two binaries apart, and a rebuild that is skipped — +`rebuild-native-deps.mjs` soft-exits 0 on a Windows file lock during postinstall +— leaves the upstream prebuilt in place and cached. + +Four checks close that, all keyed on the absent `ReadProcessMemory` import: + +- `ensureWindowsProcessTreeCommandLinePatch()` deletes a binary that still has + it, so a skipped rebuild fails loudly instead of using the prebuilt; +- `ensure-native-runtime.mjs` treats such a binary as a load failure, which is + what triggers the rebuild; +- the relay build asserts it on the artifact it just produced; +- `loadWindowsProcessTree()` asserts it again on the addon staged beside a relay + bundle and refuses to bind one that still imports the symbol, falling back to + the CIM scan. The build-time assertion is not enough on its own: a bundle and + the addon beside it redeploy independently, so a host that has not taken a new + bundle keeps whatever `.node` is already there. + +What none of this does is narrow _which_ processes are asked. A detailed scan +still queries every pid, including `lsass.exe`; it now asks with the same right +Task Manager uses instead of `PROCESS_VM_READ`. Restricting the command-line +pass to Orca's own subtree is the complementary change, and it belongs with the +identity/detailed reader split rather than here — a ppid-derived allowlist would +miss exactly the detached, reparented descendants the trackers exist to find +(#9045, #10475), so it needs the job-object membership as its source of truth. + ## Packaging The addon is Windows-only, so it follows the same contract as @@ -221,6 +333,10 @@ The addon is Windows-only, so it follows the same contract as `ensure-native-runtime.mjs`; - copied into the packaged `node_modules` for win32 only. +The relay's copy is a separate artifact staged beside the bundle, so a relay host +only picks up a rebuilt addon on redeploy. Until then it keeps whatever binary it +already has, which is why the addon is checked again at load. + ## What the snapshot does not provide `CreationDate` (process start time) has no equivalent. Anything using a start diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 6b59d23e026..a69e47f89b3 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -109,7 +109,7 @@ overrides: monaco-editor>dompurify: 3.4.13 patchedDependencies: - '@vscode/windows-process-tree@0.8.0': 9217ef36c01ed74127fef5512b0c92089cdbf820fd6c109dd671137eebdc7585 + '@vscode/windows-process-tree@0.8.0': f8ea245391c94da5770045aeea01fa6de466c2199c6ef46b5b769b398aa9823e '@xterm/addon-ligatures@0.11.0-beta.300': 47405b9994b5acf1b4e90b49250358c1ca03649854d59560e7732b72fe336920 '@xterm/addon-search@0.17.0-beta.300': eee5338dd2621ece46e79c61ec06766cd7fadaf79ffdb24e2a8ab68e97ef31f0 '@xterm/addon-serialize@0.15.0-beta.300': 851eac3d75e6d8c013b9f4c053e61d824b23965cb19ecc28e335e05059f3a294 @@ -510,7 +510,7 @@ importers: optionalDependencies: '@vscode/windows-process-tree': specifier: 0.8.0 - version: 0.8.0(patch_hash=9217ef36c01ed74127fef5512b0c92089cdbf820fd6c109dd671137eebdc7585) + version: 0.8.0(patch_hash=f8ea245391c94da5770045aeea01fa6de466c2199c6ef46b5b769b398aa9823e) sherpa-onnx-darwin-arm64: specifier: 1.12.37 version: 1.12.37 @@ -9821,7 +9821,7 @@ snapshots: convert-source-map: 2.0.0 tinyrainbow: 3.1.0 - '@vscode/windows-process-tree@0.8.0(patch_hash=9217ef36c01ed74127fef5512b0c92089cdbf820fd6c109dd671137eebdc7585)': + '@vscode/windows-process-tree@0.8.0(patch_hash=f8ea245391c94da5770045aeea01fa6de466c2199c6ef46b5b769b398aa9823e)': dependencies: node-addon-api: 7.1.0 optional: true diff --git a/src/main/windows/windows-command-line-recovery-health.test.ts b/src/main/windows/windows-command-line-recovery-health.test.ts new file mode 100644 index 00000000000..3791acf6c93 --- /dev/null +++ b/src/main/windows/windows-command-line-recovery-health.test.ts @@ -0,0 +1,59 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + reportWindowsCommandLineRecoveryHealth, + resetWindowsCommandLineRecoveryHealthForTests +} from './windows-command-line-recovery-health' + +describe('windows command line recovery health', () => { + let warn: ReturnType + + beforeEach(() => { + resetWindowsCommandLineRecoveryHealthForTests() + warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + }) + afterEach(() => { + warn.mockRestore() + }) + + const selfRow = (commandLine: string): { pid: number; commandLine: string } => ({ + pid: process.pid, + commandLine + }) + + it('warns when the querying process has no command line of its own', () => { + // We can always open ourselves with PROCESS_QUERY_LIMITED_INFORMATION, so + // an empty self command line means the query is refused host-wide. + reportWindowsCommandLineRecoveryHealth([selfRow(''), { pid: 4, commandLine: '' }]) + + expect(warn).toHaveBeenCalledTimes(1) + expect(warn.mock.calls[0][0]).toContain('ProcessCommandLineInformation') + expect(warn.mock.calls[0][1]).toEqual({ processes: 2, withCommandLine: 0 }) + }) + + it('warns once per session, not once per scan', () => { + for (let i = 0; i < 5; i++) { + reportWindowsCommandLineRecoveryHealth([selfRow('')]) + } + expect(warn).toHaveBeenCalledTimes(1) + }) + + it('stays quiet when only other processes denied a handle', () => { + // Roughly a quarter of a real table denies access; that is not a fault. + const denied = Array.from({ length: 40 }, (_, index) => ({ + pid: index + 1, + commandLine: '' + })) + reportWindowsCommandLineRecoveryHealth([selfRow('node.exe --run'), ...denied]) + expect(warn).not.toHaveBeenCalled() + }) + + it('stays quiet when our own row is absent, which the caller rejects separately', () => { + reportWindowsCommandLineRecoveryHealth([{ pid: process.pid + 1, commandLine: '' }]) + expect(warn).not.toHaveBeenCalled() + }) + + it('treats a missing commandLine field the same as an empty one', () => { + reportWindowsCommandLineRecoveryHealth([{ pid: process.pid }]) + expect(warn).toHaveBeenCalledTimes(1) + }) +}) diff --git a/src/main/windows/windows-command-line-recovery-health.ts b/src/main/windows/windows-command-line-recovery-health.ts new file mode 100644 index 00000000000..173e8f5a969 --- /dev/null +++ b/src/main/windows/windows-command-line-recovery-health.ts @@ -0,0 +1,47 @@ +/** + * One warning, once per session, when command-line recovery has stopped working. + * + * The reader has no PEB fallback by design: falling back was a total-defeat + * vector, because any single anomalous NTSTATUS reinstated address-space reads + * for the life of the process. The cost of removing it is a cliff -- if + * `NtQueryInformationProcess(ProcessCommandLineInformation)` is refused, every + * command line comes back empty and agent identity matching silently degrades + * to image names, while the addon still loads and still enumerates, so every + * health check stays green. A cliff nobody can see is the failure mode this + * area keeps producing, so it gets a signal. + * + * The querying process is the unambiguous probe. A process can always open + * itself with `PROCESS_QUERY_LIMITED_INFORMATION`, so its own command line + * coming back empty means the query is refused host-wide -- not that some + * target denied a handle, which is normal for roughly a quarter of the table. + * That is why this keys on our own row rather than a fraction: no threshold to + * tune, and no false positive on a hardened box where most processes deny. + */ +type CommandLineRow = { pid: number; commandLine?: string } + +let warned = false + +export function reportWindowsCommandLineRecoveryHealth(rows: CommandLineRow[]): void { + if (warned) { + return + } + const self = rows.find((row) => row.pid === process.pid) + // No self row is a different failure, and the caller's own guard rejects it. + if (!self || (self.commandLine ?? '') !== '') { + return + } + warned = true + const recovered = rows.filter((row) => (row.commandLine ?? '') !== '').length + console.warn( + '[windows-process-table] command-line recovery is refused on this host: the querying ' + + 'process has no command line of its own, so NtQueryInformationProcess' + + '(ProcessCommandLineInformation) is failing for every process. Agent identity matching ' + + 'falls back to image names. A hooked ntdll that does not know class 60 is the usual cause.', + { processes: rows.length, withCommandLine: recovered } + ) +} + +/** Test-only: the warning is once per session, so cases must not inherit it. */ +export function resetWindowsCommandLineRecoveryHealthForTests(): void { + warned = false +} diff --git a/src/main/windows/windows-process-table.test.ts b/src/main/windows/windows-process-table.test.ts index 360dae4ec14..96da6fcb4ef 100644 --- a/src/main/windows/windows-process-table.test.ts +++ b/src/main/windows/windows-process-table.test.ts @@ -1,3 +1,6 @@ +import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { __setWindowsProcessTableCimScanForTests, @@ -9,13 +12,16 @@ import { readWindowsProcessTableFresh, resetWindowsProcessTableForTests } from './windows-process-table' +import { resetWindowsCommandLineRecoveryHealthForTests } from './windows-command-line-recovery-health' const getAllProcesses = vi.fn() // A real snapshot always contains the querying process; the reader rejects a // table without it, because that is what a blocked CreateToolhelp32Snapshot -// returns -- an empty list rather than an error. -const SELF = { pid: process.pid, ppid: 0, name: 'vitest.exe' } +// returns -- an empty list rather than an error. It also always carries our own +// command line, since a process can always open itself -- an empty one there is +// the host-wide-refusal signal, not a fixture detail. +const SELF = { pid: process.pid, ppid: 0, name: 'vitest.exe', commandLine: 'vitest.exe --run' } const NATIVE = [ SELF, { @@ -52,7 +58,7 @@ describe('windows process table', () => { it('maps native rows, defaulting an unreadable command line to empty', async () => { const rows = await readWindowsProcessTableFresh() expect(rows).toEqual([ - { pid: process.pid, ppid: 0, name: 'vitest.exe', command: '' }, + { pid: process.pid, ppid: 0, name: 'vitest.exe', command: 'vitest.exe --run' }, { pid: 100, ppid: 4, @@ -360,6 +366,7 @@ describe('resolving the native reader', () => { let platform: PropertyDescriptor | undefined const PACKAGE_SPECIFIER = '@vscode/windows-process-tree' const ADDON_SPECIFIER = './windows-process-tree.node' + const stagedAddonDirs: string[] = [] beforeEach(() => { platform = Object.getOwnPropertyDescriptor(process, 'platform') @@ -369,6 +376,9 @@ describe('resolving the native reader', () => { afterEach(() => { __setWindowsProcessTreeRequireForTests() __setWindowsProcessTableCimScanForTests() + for (const dir of stagedAddonDirs.splice(0)) { + rmSync(dir, { recursive: true, force: true }) + } if (platform) { Object.defineProperty(process, 'platform', platform) } @@ -403,7 +413,7 @@ describe('resolving the native reader', () => { }) const rows = await readWindowsProcessTableFresh() expect(rows).toEqual([ - { pid: process.pid, ppid: 0, name: 'vitest.exe', command: '' }, + { pid: process.pid, ppid: 0, name: 'vitest.exe', command: 'vitest.exe --run' }, { pid: 100, ppid: 4, @@ -464,6 +474,64 @@ describe('resolving the native reader', () => { expect(cimScan).toHaveBeenCalledTimes(1) }) + // A relay bundle and the addon staged beside it redeploy independently, so a + // host that never took a new bundle can still be loading the published + // prebuilt -- which binds fine and then walks every process's address space. + // The relay build asserts the symbol is absent; nothing did at load. + function withStagedAddonBinary( + bytes: string, + addon: unknown + ): ((specifier: string) => unknown) & { resolve: (specifier: string) => string } { + const dir = mkdtempSync(join(tmpdir(), 'orca-relay-addon-')) + const addonPath = join(dir, 'windows-process-tree.node') + writeFileSync(addonPath, bytes) + stagedAddonDirs.push(dir) + const resolve = (specifier: string): unknown => { + if (specifier === ADDON_SPECIFIER) { + return addon + } + throw new Error('MODULE_NOT_FOUND') + } + resolve.resolve = (specifier: string): string => { + if (specifier === ADDON_SPECIFIER) { + return addonPath + } + throw new Error('MODULE_NOT_FOUND') + } + return resolve + } + + it('refuses a staged relay addon still built from unpatched source', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const cimScan = vi + .fn() + .mockResolvedValue([ + { pid: process.pid, ppid: 0, name: 'node.exe', command: 'node relay.js' } + ]) + __setWindowsProcessTableCimScanForTests(cimScan) + const addon = addonReturning(NATIVE) + __setWindowsProcessTreeRequireForTests( + withStagedAddonBinary('MZ\0KERNEL32.dll\0ReadProcessMemory\0', addon) + ) + + await expect(readWindowsProcessTableFresh()).resolves.toHaveLength(1) + expect(addon.getProcessList).not.toHaveBeenCalled() + expect(cimScan).toHaveBeenCalledTimes(1) + expect(isWindowsProcessTableAvailable()).toBe(false) + expect(warn.mock.calls[0]?.[0]).toContain('ReadProcessMemory') + warn.mockRestore() + }) + + it('binds a staged relay addon whose binary carries no such import', async () => { + const addon = addonReturning(NATIVE) + __setWindowsProcessTreeRequireForTests( + withStagedAddonBinary('MZ\0ntdll.dll\0NtQueryInformationProcess\0', addon) + ) + + await expect(readWindowsProcessTableFresh()).resolves.toHaveLength(2) + expect(addon.getProcessList).toHaveBeenCalledTimes(1) + }) + it('never probes either specifier off Windows', async () => { Object.defineProperty(process, 'platform', { configurable: true, value: 'darwin' }) const resolve = vi.fn() @@ -472,3 +540,56 @@ describe('resolving the native reader', () => { expect(resolve).not.toHaveBeenCalled() }) }) + +// The cliff the removed PEB fallback leaves behind: a hooked ntdll that refuses +// class 60 empties every command line, and the addon still loads and still +// enumerates, so every health check the app has stays green. +describe('warning when command-line recovery is refused host-wide', () => { + let platform: PropertyDescriptor | undefined + let warn: ReturnType + + beforeEach(() => { + platform = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + resetWindowsCommandLineRecoveryHealthForTests() + warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + }) + + afterEach(() => { + __setWindowsProcessTreeLoaderForTests() + warn.mockRestore() + if (platform) { + Object.defineProperty(process, 'platform', platform) + } + }) + + type NativeRow = { pid: number; ppid: number; name: string; commandLine?: string } + + function loaderReturning(rows: NativeRow[], commandLineFlag: number): void { + __setWindowsProcessTreeLoaderForTests(() => ({ + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: commandLineFlag, CreationTime: 4 }, + getAllProcesses: (cb: (r: NativeRow[] | undefined) => void) => cb(rows) + })) + } + + it('warns once when our own row comes back with no command line', async () => { + loaderReturning([{ pid: process.pid, ppid: 0, name: 'vitest.exe' }], 2) + await readWindowsProcessTableFresh() + await readWindowsProcessTableFresh() + expect(warn).toHaveBeenCalledTimes(1) + expect(warn.mock.calls[0][0]).toContain('ProcessCommandLineInformation') + }) + + it('stays quiet when our own command line came back', async () => { + loaderReturning(NATIVE, 2) + await readWindowsProcessTableFresh() + expect(warn).not.toHaveBeenCalled() + }) + + it('stays quiet when the read never asked for a command line', async () => { + // A reader that requests identity fields only must not read as a refusal. + loaderReturning([{ pid: process.pid, ppid: 0, name: 'vitest.exe' }], 0) + await readWindowsProcessTableFresh() + expect(warn).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/windows/windows-process-table.ts b/src/main/windows/windows-process-table.ts index 2bd63ccce9c..8d00f408132 100644 --- a/src/main/windows/windows-process-table.ts +++ b/src/main/windows/windows-process-table.ts @@ -1,5 +1,7 @@ +import { readFileSync } from 'node:fs' import { createRequire } from 'node:module' import { createProcessTableSnapshotReader } from '../../shared/process-table-snapshot-reader' +import { reportWindowsCommandLineRecoveryHealth } from './windows-command-line-recovery-health' import { readWindowsProcessRowsWithCim } from './windows-process-table-cim-scan' /** @@ -61,10 +63,15 @@ type WindowsProcessTreeModule = { const requireFromMain = createRequire(__filename) +/** `resolve` is optional so a test can inject a bare function for the require alone. */ +type NativeRequire = ((specifier: string) => unknown) & { + resolve?: (specifier: string) => string +} + // Why injectable: `createRequire` bypasses the module mocker, and the two // resolution steps below are the exact thing #15749 shipped untested -- the // relay suites replaced the loader wholesale, so nothing exercised the require. -let requireNative: (specifier: string) => unknown = requireFromMain +let requireNative: NativeRequire = requireFromMain /** * The bare addon a relay host receives, with no npm package around it. @@ -91,6 +98,42 @@ const PROCESS_DATA_FLAG = { None: 0, Memory: 1, CommandLine: 2 } as const /** Staged beside the relay bundle by build-relay; see RELAY_ARTIFACTS. */ const RELAY_ADDON_FILENAME = './windows-process-tree.node' +/** The import whose absence tells the patched binary from the published prebuilt. */ +const FLAGGED_ADDON_IMPORT = 'ReadProcessMemory' + +/** + * Refuse a staged relay addon built from unpatched source. + * + * The build asserts this on the artifact it produces, but a relay bundle and the + * addon beside it are redeployed independently: a host that has not taken a new + * bundle keeps whatever `.node` is already there, and the published prebuilt is + * node-addon-api, so it binds cleanly and then opens every process with + * `PROCESS_VM_READ` to walk its PEB -- the primitive MDE scores as credential + * dumping. Nothing checked that at load until here. + * + * Same predicate as `inspectWindowsProcessTreeAddon` in + * `config/scripts/windows-process-tree-gyp-rebuild.mjs`, which cannot be + * imported here: it is install-time tooling that pulls in node-gyp and + * `child_process`, and this module is bundled into the app and the relay. + * + * Falling back to the CIM scan is the correct loss: it is slower, and it is not + * the thing an EDR quarantines the host for. + */ +function stagedRelayAddonIsUnpatched(): boolean { + // No resolver means an injected test double, so there is no file to inspect. + // Production always has one, and a require that just succeeded proves the + // path is readable -- "cannot tell" here is never a real deployment. + const addonPath = requireNative.resolve?.(RELAY_ADDON_FILENAME) + if (!addonPath) { + return false + } + try { + return readFileSync(addonPath).includes(FLAGGED_ADDON_IMPORT) + } catch { + return false + } +} + let cachedModule: WindowsProcessTreeModule | null | undefined let moduleLoader: () => WindowsProcessTreeModule | null = loadWindowsProcessTree let cimScan: () => Promise = readWindowsProcessRowsWithCim @@ -135,8 +178,22 @@ function loadWindowsProcessTree(): WindowsProcessTreeModule | null { // Why check the shape: a truncated upload or an addon built for another // arch can load and still not answer. Binding to it would then reject every // read forever, where falling through reaches a scan that works. - cachedModule = - typeof addon?.getProcessList === 'function' ? adaptAddon(addon) : /* v8 ignore next */ null + if (typeof addon?.getProcessList !== 'function') { + /* v8 ignore next 2 */ + cachedModule = null + return cachedModule + } + if (stagedRelayAddonIsUnpatched()) { + console.warn( + `[windows-process-table] the addon staged beside the relay bundle still imports ` + + `${FLAGGED_ADDON_IMPORT}, so it was built from unpatched source and reads every ` + + 'process address space. Refusing it and falling back to the CIM scan; redeploy the ' + + 'relay so the staged addon is rebuilt.' + ) + cachedModule = null + return cachedModule + } + cachedModule = adaptAddon(addon) } catch { cachedModule = null } @@ -194,14 +251,14 @@ function readNativeRows(): Promise { const readId = ++readSequence const readerEpoch = nativeReaderEpoch // Why CommandLine but not Memory: each flag costs one OpenProcess per process - // inside the addon (process.cc), and every caller of this table matches on - // `command`, while nothing reads a working set off it -- the Resource Manager - // runs its own CIM sweep because it needs commit and CPU time in one pass, and - // `process.cc` truncates the working set into a DWORD anyway. Dropping Memory - // halves the per-snapshot handle count; the remaining flags stay in ONE flag - // set because every read shares one snapshot, so a 32-wide teardown collapses - // into a single scan. Splitting the cache per field set would restore exactly - // the fan-out it exists to prevent. + // inside the addon (CommandLine's is a kernel query, not a memory read), and + // every caller of this table matches on `command`, while nothing reads a + // working set off it -- the Resource Manager runs its own CIM sweep because it + // needs commit and CPU time in one pass, and `process.cc` truncates the working + // set into a DWORD anyway. Dropping Memory halves the per-snapshot handle + // count; the remaining flags stay in ONE flag set because every read shares one + // snapshot, so a 32-wide teardown collapses into a single scan. Splitting the + // cache per field set would restore exactly the fan-out it exists to prevent. const flags = native.ProcessDataFlag.CommandLine | (native.ProcessDataFlag.CreationTime ?? 0) return new Promise((resolve, reject) => { // Hoisted so a synchronous throw from getAllProcesses can clear it. An @@ -238,6 +295,10 @@ function readNativeRows(): Promise { reject(new Error('windows process table is unreadable')) return } + // Only meaningful when a command line was actually asked for. + if ((flags & native.ProcessDataFlag.CommandLine) !== 0) { + reportWindowsCommandLineRecoveryHealth(processes) + } resolve( processes.map((row) => ({ pid: row.pid, @@ -327,10 +388,13 @@ export function __setWindowsProcessTreeLoaderForTests( snapshotReader.reset() } -/** Test-only: substitute the require that resolves the package and the addon. */ -export function __setWindowsProcessTreeRequireForTests( - resolve?: (specifier: string) => unknown -): void { +/** + * Test-only: substitute the require that resolves the package and the addon. + * + * Attach a `resolve` to the injected function to also exercise the staged-addon + * binary check; without one, the loader has no path to inspect. + */ +export function __setWindowsProcessTreeRequireForTests(resolve?: NativeRequire): void { requireNative = resolve ?? requireFromMain moduleLoader = loadWindowsProcessTree cachedModule = undefined diff --git a/src/main/windows/windows-process-tree-command-line-patch.test.ts b/src/main/windows/windows-process-tree-command-line-patch.test.ts new file mode 100644 index 00000000000..1eaa4459c9e --- /dev/null +++ b/src/main/windows/windows-process-tree-command-line-patch.test.ts @@ -0,0 +1,188 @@ +import { spawn } from 'node:child_process' +import { readFileSync } from 'node:fs' +import { createRequire } from 'node:module' +import { join, resolve } from 'node:path' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' + +/** + * The command-line reader is a patch, not repo source, so its contract is + * asserted against the patch's post-image. MDE scored the addon for + * `OpenProcess(PROCESS_VM_READ)` + `ReadProcessMemory` over the whole process + * table on a timer; these cases exist so a patch refresh cannot quietly restore + * that primitive. + */ +const PATCH_PATH = resolve( + import.meta.dirname, + '../../../config/patches/@vscode__windows-process-tree@0.8.0.patch' +) + +/** Reconstruct a file as the patch leaves it: context plus added lines. */ +function patchedFile(patch: string, path: string): string { + const lines = patch.split('\n') + const start = lines.findIndex((line) => line.startsWith(`diff --git a/${path} `)) + if (start === -1) { + throw new Error(`${path} is not in the patch`) + } + const rest = lines.slice(start + 1) + const end = rest.findIndex((line) => line.startsWith('diff --git ')) + return ( + (end === -1 ? rest : rest.slice(0, end)) + // A context line for an empty source line is a bare space, and unified + // diffs may drop even that, so an empty string is context too. + .filter((line) => line === '' || line.startsWith(' ') || line.startsWith('+')) + .filter((line) => !line.startsWith('+++') && !line.startsWith('@@')) + .map((line) => line.slice(1)) + .join('\n') + ) +} + +const patch = readFileSync(PATCH_PATH, 'utf8') +const commandLineSource = patchedFile(patch, 'src/process_commandline.cc') +const processSource = patchedFile(patch, 'src/process.cc') + +describe('windows-process-tree command line patch', () => { + it('reads the command line through ProcessCommandLineInformation', () => { + // Class 60 is Windows 8.1+; Electron's floor is Windows 10, so every OS + // Orca supports has it. + expect(commandLineSource).toContain('kProcessCommandLineInformation = 60') + expect(commandLineSource).toContain('OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION') + }) + + it('resolves NtQueryInformationProcess dynamically rather than linking it', () => { + expect(commandLineSource).toContain('GetModuleHandleW(L"ntdll.dll")') + expect(commandLineSource).toContain('GetProcAddress(ntdll, "NtQueryInformationProcess")') + }) + + it('probes the buffer size before allocating, and caps it', () => { + // STATUS_INFO_LENGTH_MISMATCH / STATUS_BUFFER_TOO_SMALL carry the size. + expect(commandLineSource).toContain( + 'kStatusInfoLengthMismatch = static_cast(0xC0000004L)' + ) + expect(commandLineSource).toContain( + 'kStatusBufferTooSmall = static_cast(0xC0000023L)' + ) + expect(commandLineSource).toMatch( + /query\(process, kProcessCommandLineInformation, nullptr, 0, &size\)/ + ) + // UNICODE_STRING::Length is a USHORT, so a bogus size must not become a + // bad_alloc that fails the whole scan. + expect(commandLineSource).toContain('size > kMaxCommandLineBytes') + }) + + it('treats the returned UNICODE_STRING as untrusted', () => { + // A hooked ntdll is the environment this reader targets, so an unchecked + // Buffer/Length would be an over-read encoded straight into JS. The bound + // must be buffer.size(), not `size`, which the second query overwrites. + expect(commandLineSource).toContain('const unsigned char* end = begin + buffer.size()') + expect(commandLineSource).toMatch(/chars == nullptr \|\|/) + expect(commandLineSource).toMatch(/command_line->Length > static_cast\(end - chars\)/) + }) + + it('has no PEB fallback and no latch that could reinstate one', () => { + // The fallback used to be reachable from any single anomalous NTSTATUS, + // which on an EDR-hooked ntdll is the realistic case -- one stray status + // would have silently restored the primitive for the process lifetime. + expect(commandLineSource).not.toContain('ReadCommandLineFromPeb') + expect(commandLineSource).not.toContain('PROCESS_BASIC_INFORMATION') + expect(commandLineSource).not.toContain('InterlockedExchange') + expect(commandLineSource).not.toMatch(/ReadProcessMemory\(/) + }) + + it('acquires PROCESS_VM_READ nowhere in the addon', () => { + for (const source of [commandLineSource, processSource]) { + expect(source).not.toMatch(/OpenProcess\([^)]*PROCESS_VM_READ/) + expect(source).not.toMatch(/ReadProcessMemory\(/) + } + // Memory and CPU counters kept VM_READ and never read an address space. + expect(processSource.match(/OpenProcess\(PROCESS_QUERY_LIMITED_INFORMATION/g)).toHaveLength(2) + }) + + it('value-initializes ProcessInfo so memory is not stack garbage', () => { + // Measured before: 82 processes reported the same bogus working set. + expect(processSource).toContain('ProcessInfo pinfo{};') + }) +}) + +type Addon = { + getProcessList: ( + callback: (rows: { pid: number; commandLine?: string }[] | undefined) => void, + flags: number + ) => void +} + +const addonRequire = createRequire(import.meta.url) + +function loadAddon(): Addon { + const packageEntry = addonRequire.resolve('@vscode/windows-process-tree') + return addonRequire( + join(packageEntry, '..', '..', 'build', 'Release', 'windows_process_tree.node') + ) as Addon +} + +// Why fail rather than skip on win32: the published tarball ships a loadable +// prebuilt built from unpatched source, and both readers emit byte-identical +// strings, so a skipping suite would pass against the very binary this patch +// exists to keep out. On win32 the addon must be present, and it must be ours. +describe.runIf(process.platform === 'win32')('windows-process-tree command line addon', () => { + // Why not at collection time: a require that throws there fails the whole + // file, and the patch-text cases above need no binary at all -- a Windows + // checkout without a built addon would lose them to an unrelated failure. + let addon: Addon + beforeAll(() => { + addon = loadAddon() + }) + const children: { kill: () => void }[] = [] + afterAll(() => { + for (const child of children) { + try { + child.kill() + } catch { + // already gone + } + } + }) + + const scan = async (): Promise> => + new Promise((resolveScan) => { + addon.getProcessList((rows) => { + resolveScan(new Map((rows ?? []).map((row) => [row.pid, row.commandLine ?? '']))) + }, 2 /* ProcessDataFlag.CommandLine */) + }) + + it('was built from the patched source, not the published prebuild', () => { + // The patched reader never calls ReadProcessMemory, so the symbol is + // absent from its import table. This is the only check that tells the two + // binaries apart -- a bare require() cannot. + const packageEntry = addonRequire.resolve('@vscode/windows-process-tree') + const binary = readFileSync( + join(packageEntry, '..', '..', 'build', 'Release', 'windows_process_tree.node') + ) + expect(binary.includes('ReadProcessMemory')).toBe(false) + }) + + it('recovers command lines byte-for-byte, quoting and trailing spaces included', async () => { + const marker = `orca-cmdline-${Date.now()}` + // Quotes and trailing whitespace are exactly what a re-quoting bug eats. + const child = spawn(process.execPath, ['-e', 'setTimeout(() => {}, 20000)', `"${marker}" `], { + windowsHide: true, + stdio: 'ignore' + }) + children.push(child) + await new Promise((r) => setTimeout(r, 400)) + + const rows = await scan() + const command = rows.get(child.pid!) + expect(command).toBeDefined() + expect(command).toContain(marker) + expect(command!.endsWith(' "') || command!.endsWith(' ')).toBe(true) + }) + + it('reports the querying process and most of the table', async () => { + const rows = await scan() + expect(rows.has(process.pid)).toBe(true) + const recovered = [...rows.values()].filter((command) => command.length > 0) + // Protected and cross-session processes legitimately deny a handle; a + // wholesale regression would show up as almost nothing recovered. + expect(recovered.length).toBeGreaterThan(rows.size * 0.25) + }) +}) From 975bbdedcc2b3765f6611c5c4b20e59e6896a82d Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Sat, 5 Sep 2026 21:12:59 -0700 Subject: [PATCH 42/69] fix(windows): scan ports natively instead of encoded PowerShell (#17861) * fix(windows): scan ports natively instead of encoded PowerShell Microsoft Defender for Endpoint scored the relay's Windows port scan as suspicious PowerShell plus network discovery (T1049). The command line was `-ExecutionPolicy Bypass -EncodedCommand ` around a Get-NetTCPConnection/Get-Process join -- base64 next to a policy override is the highest-weighted token pair on a PowerShell command line, and netstat only ever ran as its fallback. Invert the chain. `netstat.exe -ano` is now the primary reader and the owning process name comes from the shared native process table, which exists to keep PID lookups off PowerShell. The payload survives only as a last resort, and without the override: execution policy gates script files, never `-Command`, so nothing needed it (verified: `-ExecutionPolicy Restricted -Command` runs). Drop `-p tcp` while inverting: on Windows that protocol name means IPv4 only, so as a primary reader it would have hidden every `[::]` listener the payload used to report. Names arrive as `sshd.exe` from the table and are published as `sshd`, keeping the sshd filter and old clients' rendering intact. Routes both spawns through runProcess, removing the file from the child_process and windowsHide ratchets. * fix(windows): read netstat state by shape and refuse a truncated table Review of the port-scan inversion found two ways the new primary path could be silently wrong, both of which would have kept the flagged PowerShell payload running on exactly the hosts this change targets. `LISTENING` is not in netstat.exe. It lives in System32\\netstat.exe.mui and MUI selection follows the UI language, so the pinned-locale env in relay-command-env.ts cannot reach it -- a German host prints `ABHOEREN` and the word test parsed zero rows. The zero-listeners guard then read that as a blocked reader and ran `Get-NetTCPConnection` every 12-30s forever, or returned nothing at all where PowerShell is also restricted. Keep the word as the fast path and, when it finds nothing over output that did contain TCP rows, re-read by shape: only a listening socket has no peer. Measured on this host across all four states present (LISTENING 47, ESTABLISHED 49, CLOSE_WAIT 29, TIME_WAIT 213): zero non-listening rows with a zero peer, zero listening rows without one, and the same 47 rows parse after substituting the German state words. Shape stays the fallback because `BOUND` also prints a zero peer. Truncation was invisible: createOutputSink discards overflow, ProcessResult carries no flag, so a capped read still exits 0 and its head still parses. netstat orders IPv4 TCP, then IPv6 TCP, then UDP, so a host with tens of thousands of TIME_WAIT rows would have lost every `[::]` listener -- the exact loss dropping `-p tcp` exists to prevent, and one the zero-listeners guard cannot see. Refuse the read instead. A `truncated` flag on the shared sink would be cleaner and is left as a follow-up rather than widened into this PR. Also: decline to wait on the shared process table once the request is aborted (it takes no signal and must not be cancelled for other callers); note the name lookup as best-effort, since a TTL-cached snapshot can hand a recycled PID its previous owner name; log once on either fall-through, because both are permanent and invisible when wrong; and drop a stderr assertion that any PowerShell autoload banner would redden. Correcting the cost claim in the previous commit: the aggregate win holds with the native addon (netstat 21ms vs the retired payload 860ms at 532 processes), not without it. The addon is optional, the snapshot TTL is 500ms and the scan cadence is 12-30s, so a relay with no active agent pane never warms its own cache and pays ~1.4s cold on the CIM path -- slower than what it replaced. * fix(windows): log the port-scan fall-through on the relay diagnostic stream Checked where this code actually runs before trusting the log. `console.warn` did reach a file, but relayLogLine is the right call and the reasoning is worth recording. `scanWindowsListeningPorts` runs only in the detached relay daemon: relay.ts returns early for --connect and --orca-cli, so PortScanHandler is reached only through runRelayDaemon, and both launchers start it detached with a log file (POSIX `> relay.log 2>&1`, Windows `1>relay.log 2>relay.err.log` via Win32_Process.Create). installRelayLogRotation then wraps both streams into relay.log, which is the file the documented diagnostics tail reads. Verified by installing the real rotation over a temp path and reading the file back. So the line surfaced -- but untimestamped, in a log whose format exists so reconnect flaps can be correlated with the events around them (#7773). relayLogLine is that format and the relay idiom in 41 other places, and "since when has this host been stuck on PowerShell" is most of what this line is for. The test spies on process.stderr to pin the stream and the ISO stamp rather than just asserting something was called, since a fall-through logged somewhere unread is the failure being guarded against. Also fixes a comment that ended its own block early: `relay-*/relay.log` in a doc comment contains `*/`. * fix(windows): keep the dominant zero-peer state when reading a localized netstat Shape alone promoted any zero-peer TCP row, not just listeners. `BOUND` and `CLOSED` print a zero peer too, and on a localized host their state words are exactly as unreadable as the listening one -- so a German host with listeners plus one BOUND socket published a phantom listener. Reachable on an English host too: with zero listeners a lone BOUND row is promoted AND, because the result is then non-empty, it suppresses the blocked-reader fall-through. Group the zero-peer rows by state word and keep only the largest group. A transient BOUND or CLOSED socket cannot outnumber the listeners (51 against 0 on this host), so this removes the class rather than special-casing the words, which would just be the localization bug again. An exact tie keeps every tied group rather than guessing -- no worse than reading shape alone. Verified against real netstat output: injecting a BOUND row into the localized capture leaves the result identical to the English answer (47 rows, no phantom 65001). The new test has teeth -- reverting the grouping fails it and nothing else. Corrects two claims that were slightly wrong: the docblock said shape was the fallback because BOUND prints a zero peer, which described the hazard without saying it was unhandled; and a test comment said an English host "never sees a bound socket", true only when it has at least one readable LISTENING row. Also gates the fall-through log per reason instead of per module, so a host that parses nothing today and truncates tomorrow reports both faults. Same one-shot cost, and the vocabulary is two fixed strings so the set cannot grow. That guard matters more than it looks: --log-file rotates stdout only, so the file stderr can land in is unrotated. * docs(windows): note the direction the zero-peer majority rule can fail in The docblock described the tie case and stopped there, which reads as a complete account of the limits when it is not: a majority rule inverts if the majority is wrong, and enough transient zero-peer sockets would publish the phantoms and drop the real listeners. Someone would reasonably have concluded the rule was safe in both directions. Trigger numbers and the repro stay in the PR discussion; the code only needs the reader to know the rule has a direction, and the hatch (defer to the PowerShell reader, which reads the state word instead of inferring it) since that is the part a future editor would otherwise re-derive. * ci(windows): run the real-netstat port scan suite in CI The win32 suite only self-skips off Windows, so it passed vacuously in every lane. Register it the way the cmd-shim suite is registered. * test(windows): lower both child-process ratchets to the ground this PR took Migrating the port scan off `node:child_process` onto `runProcess` drops `src/relay/windows-port-scan.ts` from both allowlists, so both offender counts fall by one. Each ratchet pins the count from below as well as above, so a pin left above reality fails and re-opens room for the next direct import to land for free. * docs(windows): qualify the no-PowerShell claim on the netstat scan The scan starts no PowerShell of its own, but no released relay carries the optional `windows-process-tree.node` addon (only dev-channel-win-build.yml builds it), so the shared process-table read falls back to a CIM scan that forks one `powershell.exe`. The EDR win is the removal of the `-EncodedCommand` / `-ExecutionPolicy Bypass` shape, not the elimination of PowerShell. Comment-only. * docs(windows): record the identity-reader follow-up and the perf table's addon attachWindowsProcessNames reads only `name`, so it should move to `readWindowsProcessIdentityTable` once #17866 lands -- on that PR's detailed reader it would open per-process handles for a field it discards. The reader does not exist on this branch, so the call stays as-is with the follow-up recorded rather than pulling #17866 in. The process-table perf table's two Toolhelp32 rows assume the optional `windows-process-tree.node` addon. The desktop bundles it; no released relay does, so on an SSH host the CIM row is the operative number. Comment-only. * docs(windows): state the CIM scan as the relay's normal path, not a fallback No released relay carries the optional `windows-process-tree.node` addon -- release-cut.yml has zero references to it and only dev-channel-win-build.yml builds it -- so the PowerShell CIM scan is what every SSH host runs. The call-site docstring read as a conditional fallback standalone. Comment-only. --------- Co-authored-by: Orca Worker --- .github/workflows/pr.yml | 1 + config/scripts/pr-code-change-scope.mjs | 3 +- src/main/windows/windows-process-table.ts | 4 + src/relay/windows-port-scan.test.ts | 413 +++++++++++++++--- src/relay/windows-port-scan.ts | 345 ++++++++++++--- src/relay/windows-port-scan.win32.test.ts | 52 +++ .../child-process-import-allowlist.txt | 1 - .../windows-console-visibility-allowlist.txt | 1 - .../child-process-import-boundary.test.ts | 2 +- .../windows-console-visibility.test.ts | 2 +- 10 files changed, 694 insertions(+), 130 deletions(-) create mode 100644 src/relay/windows-port-scan.win32.test.ts diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index 536b5c0f273..38daf8a71de 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -869,6 +869,7 @@ jobs: src/shared/secure-file-fsync-flags.test.ts src/main/ipc/pty-codex-account-attribution.test.ts src/main/ipc/pty-spawn-env-codex-resume-provenance.test.ts + src/relay/windows-port-scan.win32.test.ts # Why the :parallel variant: identical to build:release except the three # electron-vite targets overlap instead of running back to back. The Linux package diff --git a/config/scripts/pr-code-change-scope.mjs b/config/scripts/pr-code-change-scope.mjs index c1d5c731cd4..f83c1508c26 100644 --- a/config/scripts/pr-code-change-scope.mjs +++ b/config/scripts/pr-code-change-scope.mjs @@ -238,7 +238,8 @@ const WINDOWS_PACKAGE_TESTS = [ 'src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts', 'src/shared/secure-file-fsync-flags.test.ts', 'src/main/ipc/pty-codex-account-attribution.test.ts', - 'src/main/ipc/pty-spawn-env-codex-resume-provenance.test.ts' + 'src/main/ipc/pty-spawn-env-codex-resume-provenance.test.ts', + 'src/relay/windows-port-scan.win32.test.ts' ] const DESKTOP_IRRELEVANT_PREFIXES = [ diff --git a/src/main/windows/windows-process-table.ts b/src/main/windows/windows-process-table.ts index 8d00f408132..6683770435d 100644 --- a/src/main/windows/windows-process-table.ts +++ b/src/main/windows/windows-process-table.ts @@ -29,6 +29,10 @@ import { readWindowsProcessRowsWithCim } from './windows-process-table-cim-scan' * Those are the module's published figures for both extra fields together; the * only flag set this module asks for is `CommandLine` (+ `CreationTime`, free), * which sits between the two rows and has not been separately measured. + * + * Both Toolhelp32 rows assume the optional `windows-process-tree.node` addon. + * The desktop bundles it; no released relay carries it, so on an SSH host the + * CIM row is the operative number and the child process is not avoided at all. */ export type WindowsProcessRow = { diff --git a/src/relay/windows-port-scan.test.ts b/src/relay/windows-port-scan.test.ts index 139d3ede6be..9fb074180b9 100644 --- a/src/relay/windows-port-scan.test.ts +++ b/src/relay/windows-port-scan.test.ts @@ -1,22 +1,30 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -const { execFileAsyncMock, execFileMock, promisifyCustom } = vi.hoisted(() => ({ - execFileAsyncMock: vi.fn(), - execFileMock: vi.fn(), - promisifyCustom: Symbol.for('nodejs.util.promisify.custom') -})) - -vi.mock('child_process', () => ({ - execFile: Object.assign(execFileMock, { - [promisifyCustom]: execFileAsyncMock - }) +const runProcessMock = vi.fn() +vi.mock('../shared/child-process/run-process', () => ({ + runProcess: (spec: unknown) => runProcessMock(spec) })) vi.mock('./relay-command-env', () => ({ buildRelayCommandEnv: () => ({ PATH: 'C:\\Windows\\System32' }) })) -const { scanWindowsListeningPorts } = await import('./windows-port-scan') +import { + __setWindowsProcessTableCimScanForTests, + __setWindowsProcessTreeLoaderForTests, + resetWindowsProcessTableForTests +} from '../main/windows/windows-process-table' +import { + resetWindowsPortScanDiagnosticsForTests, + scanWindowsListeningPorts +} from './windows-port-scan' + +type Spec = { + program: string + args?: readonly string[] + timeoutMs?: number | null + signal?: AbortSignal +} // The scanner drops any row whose pid is the relay process or its parent, so a fixture pid // that happens to match the vitest worker's own pid silently empties the result and the @@ -32,78 +40,365 @@ function pidUnlikeSelf(seed: number): number { return pid } -const POWERSHELL_PID = pidUnlikeSelf(1234) const NETSTAT_PID = pidUnlikeSelf(2468) +const SSHD_PID = pidUnlikeSelf(4321) +const POWERSHELL_PID = pidUnlikeSelf(1234) + +const NETSTAT_STDOUT = [ + ' Proto Local Address Foreign Address State PID', + ` TCP 0.0.0.0:3000 0.0.0.0:0 LISTENING ${NETSTAT_PID}`, + ` TCP [::]:3000 [::]:0 LISTENING ${NETSTAT_PID}`, + ` TCP 0.0.0.0:4000 93.184.216.34:443 ESTABLISHED ${NETSTAT_PID}`, + ` UDP 0.0.0.0:5353 *:* ${NETSTAT_PID}`, + ` TCP 0.0.0.0:2222 0.0.0.0:0 LISTENING ${SSHD_PID}` +].join('\r\n') + +function ok(stdout: string): { + code: number + signal: null + stdout: string + stderr: string + timedOut: boolean +} { + return { code: 0, signal: null, stdout, stderr: '', timedOut: false } +} + +type NativeRow = { + pid: number + ppid: number + name: string + memory?: number + commandLine?: string + creationTimeMs?: number +} + +/** A native snapshot must contain the reader's own pid or the table rejects. */ +function nativeTable(rows: NativeRow[]) { + return () => ({ + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2 }, + getAllProcesses: (callback: (processes: NativeRow[] | undefined) => void) => + callback([{ pid: process.pid, ppid: 0, name: 'vitest.exe' }, ...rows]) + }) +} + +function specs(): Spec[] { + return runProcessMock.mock.calls.map((call) => call[0] as Spec) +} describe('scanWindowsListeningPorts', () => { beforeEach(() => { - execFileAsyncMock.mockReset() + runProcessMock.mockReset() + resetWindowsPortScanDiagnosticsForTests() + resetWindowsProcessTableForTests() + __setWindowsProcessTreeLoaderForTests( + nativeTable([ + { pid: NETSTAT_PID, ppid: 4, name: 'node.exe' }, + { pid: SSHD_PID, ppid: 4, name: 'sshd.exe' } + ]) + ) }) - it('bounds the PowerShell scan with the caller abort signal and timeout', async () => { + afterEach(() => { + __setWindowsProcessTreeLoaderForTests() + __setWindowsProcessTableCimScanForTests() + resetWindowsProcessTableForTests() + }) + + it('reads netstat first and never starts PowerShell', async () => { const controller = new AbortController() - execFileAsyncMock.mockResolvedValueOnce({ - stdout: JSON.stringify({ + runProcessMock.mockResolvedValueOnce(ok(NETSTAT_STDOUT)) + + await expect(scanWindowsListeningPorts(controller.signal)).resolves.toEqual([ + { host: '::', port: 3000, pid: NETSTAT_PID, processName: 'node' }, + { host: '0.0.0.0', port: 3000, pid: NETSTAT_PID, processName: 'node' } + ]) + + expect(specs()).toHaveLength(1) + expect(specs()[0].program).toMatch(/netstat\.exe$/) + // `-p tcp` is absent on purpose: on Windows it means IPv4-only and would + // hide every `[::]` listener. + expect(specs()[0].args).toEqual(['-ano']) + expect(specs()[0].signal).toBe(controller.signal) + expect(specs()[0].timeoutMs).toBe(5000) + }) + + it('keeps the sshd and self-pid filters working off native process names', async () => { + runProcessMock.mockResolvedValueOnce( + ok( + [ + NETSTAT_STDOUT, + ` TCP 0.0.0.0:9999 0.0.0.0:0 LISTENING ${process.pid}` + ].join('\r\n') + ) + ) + + const ports = await scanWindowsListeningPorts() + + // sshd.exe is matched despite the table's `.exe` spelling, and the relay's + // own listener never reaches a client. + expect(ports.map((port) => `${port.host}:${port.port}`)).toEqual([':::3000', '0.0.0.0:3000']) + }) + + it('still reports host/port/pid when no process table is readable', async () => { + runProcessMock.mockResolvedValueOnce(ok(NETSTAT_STDOUT)) + __setWindowsProcessTreeLoaderForTests(() => null) + __setWindowsProcessTableCimScanForTests(() => + Promise.reject(new Error('windows process table unavailable')) + ) + resetWindowsProcessTableForTests() + + await expect(scanWindowsListeningPorts()).resolves.toEqual([ + { host: '0.0.0.0', port: 2222, pid: SSHD_PID }, + { host: '::', port: 3000, pid: NETSTAT_PID }, + { host: '0.0.0.0', port: 3000, pid: NETSTAT_PID } + ]) + // Names were unavailable, so nothing else was spawned to go get them. + expect(specs()).toHaveLength(1) + }) + + it('falls back to PowerShell without an execution-policy override', async () => { + const controller = new AbortController() + runProcessMock + .mockResolvedValueOnce({ + code: 1, + signal: null, + stdout: '', + stderr: 'blocked', + timedOut: false + }) + .mockResolvedValueOnce( + ok( + JSON.stringify({ + host: '127.0.0.1', + port: 5173, + pid: POWERSHELL_PID, + processName: 'node' + }) + ) + ) + + await expect(scanWindowsListeningPorts(controller.signal)).resolves.toEqual([ + { host: '127.0.0.1', port: 5173, pid: POWERSHELL_PID, processName: 'node' - }), - stderr: '' - }) - - await expect(scanWindowsListeningPorts(controller.signal)).resolves.toEqual([ - { host: '127.0.0.1', port: 5173, pid: POWERSHELL_PID, processName: 'node' } + } ]) - expect(execFileAsyncMock).toHaveBeenCalledWith( - 'powershell.exe', - expect.arrayContaining(['-EncodedCommand', expect.any(String)]), - expect.objectContaining({ - signal: controller.signal, - timeout: 5000, - windowsHide: true - }) - ) + const powershell = specs()[1] + expect(powershell.program).toMatch(/powershell\.exe$/i) + expect(powershell.args?.slice(0, 3)).toEqual(['-NoProfile', '-NonInteractive', '-Command']) + expect(powershell.args).not.toContain('-ExecutionPolicy') + expect(powershell.args).not.toContain('-EncodedCommand') + expect(powershell.args).toHaveLength(4) + expect(powershell.args?.[3]).toContain('Get-NetTCPConnection') + expect(powershell.signal).toBe(controller.signal) + expect(powershell.timeoutMs).toBe(5000) }) - it('bounds the netstat fallback with the same abort signal and timeout', async () => { - const controller = new AbortController() - execFileAsyncMock + it('tries pwsh when Windows PowerShell cannot answer, then gives up empty', async () => { + runProcessMock + .mockResolvedValueOnce(ok('')) .mockRejectedValueOnce(new Error('powershell unavailable')) .mockRejectedValueOnce(new Error('pwsh unavailable')) - .mockResolvedValueOnce({ - stdout: [ - ' Proto Local Address Foreign Address State PID', - ` TCP 0.0.0.0:3000 0.0.0.0:0 LISTENING ${NETSTAT_PID}` - ].join('\r\n'), - stderr: '' - }) - await expect(scanWindowsListeningPorts(controller.signal)).resolves.toEqual([ - { host: '0.0.0.0', port: 3000, pid: NETSTAT_PID } + await expect(scanWindowsListeningPorts()).resolves.toEqual([]) + + expect(specs().map((spec) => spec.program)).toEqual([ + expect.stringMatching(/netstat\.exe$/), + expect.stringMatching(/powershell\.exe$/i), + 'pwsh.exe' ]) - - expect(execFileAsyncMock).toHaveBeenLastCalledWith( - 'netstat.exe', - ['-ano', '-p', 'tcp'], - expect.objectContaining({ - signal: controller.signal, - timeout: 5000, - windowsHide: true - }) - ) }) - it('does not start the netstat fallback after the scan is cancelled', async () => { + it('gives up rather than falling back once the scan is cancelled', async () => { const controller = new AbortController() controller.abort() - execFileAsyncMock.mockRejectedValueOnce( - Object.assign(new Error('cancelled'), { name: 'AbortError' }) - ) + runProcessMock.mockResolvedValueOnce({ + code: null, + signal: null, + stdout: '', + stderr: '', + timedOut: false + }) await expect(scanWindowsListeningPorts(controller.signal)).resolves.toEqual([]) - expect(execFileAsyncMock).toHaveBeenCalledTimes(1) + expect(runProcessMock).toHaveBeenCalledTimes(1) + }) + + // `LISTENING` ships in netstat.exe.mui, picked by UI language, so no env can + // pin it. Without the shape-based re-read a German host parses zero rows, + // reads that as a blocked reader, and runs the flagged payload every 12-30s. + it('reads a localized host by socket shape rather than the state word', async () => { + runProcessMock.mockResolvedValueOnce( + ok( + [ + 'Aktive Verbindungen', + '', + ' Proto Lokale Adresse Remoteadresse Status PID', + ' TCP 0.0.0.0:135 0.0.0.0:0 ABHÖREN 1116', + ` TCP 0.0.0.0:3000 0.0.0.0:0 ABHÖREN ${NETSTAT_PID}`, + ` TCP [::]:3000 [::]:0 ABHÖREN ${NETSTAT_PID}`, + ` TCP 192.168.0.5:52000 93.184.216.34:443 HERGESTELLT ${NETSTAT_PID}` + ].join('\r\n') + ) + ) + + await expect(scanWindowsListeningPorts()).resolves.toEqual([ + { host: '0.0.0.0', port: 135, pid: 1116 }, + { host: '::', port: 3000, pid: NETSTAT_PID, processName: 'node' }, + { host: '0.0.0.0', port: 3000, pid: NETSTAT_PID, processName: 'node' } + ]) + expect(specs()).toHaveLength(1) + }) + + // The gap the state-word cases cannot cover: on a localized host BOUND is as + // unreadable as ABHÖREN, so shape alone would publish 8080 as a listener. + // Listeners dominate, and that is what separates them. + it('drops a BOUND socket that shape alone would promote on a localized host', async () => { + runProcessMock.mockResolvedValueOnce( + ok( + [ + ` TCP 0.0.0.0:3000 0.0.0.0:0 ABHÖREN ${NETSTAT_PID}`, + ` TCP [::]:3000 [::]:0 ABHÖREN ${NETSTAT_PID}`, + ` TCP 0.0.0.0:8080 0.0.0.0:0 GEBUNDEN ${NETSTAT_PID}`, + ` TCP 192.168.0.5:52000 93.184.216.34:443 HERGESTELLT ${NETSTAT_PID}` + ].join('\r\n') + ) + ) + + await expect(scanWindowsListeningPorts()).resolves.toEqual([ + { host: '::', port: 3000, pid: NETSTAT_PID, processName: 'node' }, + { host: '0.0.0.0', port: 3000, pid: NETSTAT_PID, processName: 'node' } + ]) + }) + + // Only on an exact tie does shape have nothing left to go on, and then it + // keeps both rather than guessing — no worse than reading shape alone. + it('keeps every tied zero-peer state when none dominates', async () => { + runProcessMock.mockResolvedValueOnce( + ok( + [ + ` TCP 0.0.0.0:3000 0.0.0.0:0 ABHÖREN ${NETSTAT_PID}`, + ` TCP 0.0.0.0:8080 0.0.0.0:0 GEBUNDEN ${NETSTAT_PID}` + ].join('\r\n') + ) + ) + + await expect(scanWindowsListeningPorts()).resolves.toEqual([ + { host: '0.0.0.0', port: 3000, pid: NETSTAT_PID, processName: 'node' }, + { host: '0.0.0.0', port: 8080, pid: NETSTAT_PID, processName: 'node' } + ]) + }) + + // Windows prints BOUND with a zero peer too, so the shape test must stay the + // fallback: a host with at least one readable LISTENING row never reaches it. + it('does not promote a BOUND socket on a host whose state word parsed', async () => { + runProcessMock.mockResolvedValueOnce( + ok( + [ + ` TCP 0.0.0.0:3000 0.0.0.0:0 LISTENING ${NETSTAT_PID}`, + ` TCP 0.0.0.0:8080 0.0.0.0:0 BOUND ${NETSTAT_PID}` + ].join('\r\n') + ) + ) + + await expect(scanWindowsListeningPorts()).resolves.toEqual([ + { host: '0.0.0.0', port: 3000, pid: NETSTAT_PID, processName: 'node' } + ]) + }) + + // A capped read exits 0 and its head parses, and netstat orders IPv4 TCP + // before IPv6 TCP, so publishing the head would drop every `[::]` listener. + it('refuses a netstat table that hit the capture cap', async () => { + const filler = Array.from( + { length: 60_000 }, + (_, index) => + ` TCP 10.0.0.1:${1000 + (index % 5000)} 10.0.0.2:443 TIME_WAIT 4` + ).join('\r\n') + runProcessMock + .mockResolvedValueOnce(ok(`${NETSTAT_STDOUT}\r\n${filler}`.slice(0, 4 * 1024 * 1024))) + .mockResolvedValueOnce(ok('[]')) + + await expect(scanWindowsListeningPorts()).resolves.toEqual([]) + + // Fell through instead of publishing the IPv4 head it could still parse. + expect(specs()).toHaveLength(2) + expect(specs()[1].args).toContain('-Command') + }) + + it('does not wait on the shared process table once the scan is cancelled', async () => { + const controller = new AbortController() + const getAllProcesses = vi.fn() + __setWindowsProcessTreeLoaderForTests(() => ({ + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2 }, + getAllProcesses + })) + resetWindowsProcessTableForTests() + // netstat answered, then the request was abandoned before names were needed. + runProcessMock.mockImplementationOnce(() => { + controller.abort() + return Promise.resolve(ok(NETSTAT_STDOUT)) + }) + + await expect(scanWindowsListeningPorts(controller.signal)).resolves.toEqual([ + // Unnamed, so the sshd row survives its own filter — the cost of not + // waiting, and strictly better than blocking an abandoned request. + { host: '0.0.0.0', port: 2222, pid: SSHD_PID }, + { host: '::', port: 3000, pid: NETSTAT_PID }, + { host: '0.0.0.0', port: 3000, pid: NETSTAT_PID } + ]) + expect(getAllProcesses).not.toHaveBeenCalled() + }) + + // The relay daemon's stderr is what installRelayLogRotation routes into + // relay.log, so a fall-through logged anywhere else is a fall-through nobody + // can diagnose. Pin the stream, not just the fact that something was called. + it('reports leaving the native path on the relay diagnostic stream, once', async () => { + const lines: string[] = [] + const stderr = vi + .spyOn(process.stderr, 'write') + .mockImplementation((chunk: string | Uint8Array) => { + lines.push(String(chunk)) + return true + }) + try { + runProcessMock.mockResolvedValue(ok('')) + await scanWindowsListeningPorts() + await scanWindowsListeningPorts() + // A second, different fault on the same host must still be heard: one + // flag for the whole module would have swallowed it. + runProcessMock.mockResolvedValue(ok('x'.repeat(4 * 1024 * 1024))) + await scanWindowsListeningPorts() + await scanWindowsListeningPorts() + } finally { + stderr.mockRestore() + } + + const reported = lines.filter((line) => line.includes('[ports] netstat unusable')) + expect(reported).toHaveLength(2) + expect(reported[0]).toContain('no listening row parsed') + expect(reported[1]).toContain('truncated') + // relayLogLine's ISO stamp: an unplaceable line cannot be read against the + // reconnect flaps around it. + expect(reported[0]).toMatch(/^\d{4}-\d{2}-\d{2}T[\d:.]+Z /) + }) + + it('treats a netstat timeout as unanswered and falls through', async () => { + runProcessMock + .mockResolvedValueOnce({ + code: null, + signal: 'SIGKILL', + stdout: '', + stderr: '', + timedOut: true + }) + .mockResolvedValueOnce(ok('[]')) + + await expect(scanWindowsListeningPorts()).resolves.toEqual([]) + + expect(specs()).toHaveLength(2) }) }) diff --git a/src/relay/windows-port-scan.ts b/src/relay/windows-port-scan.ts index f26a2bb172a..f99fce4a087 100644 --- a/src/relay/windows-port-scan.ts +++ b/src/relay/windows-port-scan.ts @@ -1,84 +1,231 @@ -import { execFile } from 'node:child_process' -import { promisify } from 'node:util' +import { readWindowsProcessTable } from '../main/windows/windows-process-table' +import { runProcess } from '../shared/child-process/run-process' +import { + windowsPowerShellPath, + windowsSystem32Binary +} from '../shared/child-process/windows-system-binary' import { getProcessOutputFields } from '../shared/process-output-field-scanner' -import { encodePowerShellCommand } from '../shared/powershell-command-encoding' import type { DetectedPort } from './port-scan-handler' import { buildRelayCommandEnv } from './relay-command-env' +import { relayLogLine } from './relay-diagnostic-log' const SYSTEM_PORTS_TO_EXCLUDE = new Set([22]) const MAX_DETECTED_PORTS = 50 const WINDOWS_PORT_SCAN_TIMEOUT_MS = 5_000 -const execFileAsync = promisify(execFile) +// Wide enough for `netstat -ano` on a busy host: it prints every connection, not +// just the listeners, and a truncated table silently drops the tail. +const WINDOWS_PORT_SCAN_MAX_OUTPUT_BYTES = 4 * 1024 * 1024 +/** + * Listening TCP ports, attributed to their owning process. + * + * `netstat.exe -ano` answers all of it except the process name, which comes + * from the shared process table -- so this scan starts no PowerShell of its + * own. One still runs on a released relay: without the optional + * `windows-process-tree.node` addon (built only by dev-channel-win-build.yml, + * so no release carries it) that table falls back to a CIM scan that forks one + * `powershell.exe`. That scan is TTL-shared with pane naming, so a relay with a + * live pane pays nothing extra for it. + * + * The EDR win is therefore the shape, not the absence of PowerShell. The + * retired payload ran `-ExecutionPolicy Bypass -EncodedCommand ` + * wrapping `Get-NetTCPConnection` joined to `Get-Process`: base64 beside a + * policy override is the highest-weighted token pair Defender for Endpoint + * scores on a PowerShell command line, and listing listeners with their owners + * reads as network discovery (T1049) on top of it. The shared CIM scan carries + * neither token. That payload survives only as the last resort below, without + * the override. + */ export async function scanWindowsListeningPorts(signal?: AbortSignal): Promise { + const netstatPorts = await readWindowsNetstatPorts(signal) + if (netstatPorts) { + return normalizeWindowsDetectedPorts(await attachWindowsProcessNames(netstatPorts, signal)) + } + if (signal?.aborted) { + return [] + } try { const json = await runWindowsPortScanPowerShell(signal) return normalizeWindowsDetectedPorts(parseWindowsPowerShellPortRows(json)) } catch { - if (signal?.aborted) { - return [] - } - try { - const { stdout } = await execFileAsync('netstat.exe', ['-ano', '-p', 'tcp'], { - env: buildRelayCommandEnv(), - encoding: 'utf-8', - signal, - timeout: WINDOWS_PORT_SCAN_TIMEOUT_MS, - windowsHide: true - }) - return normalizeWindowsDetectedPorts(parseWindowsNetstatOutput(stdout)) - } catch { - return [] - } + return [] } } -async function runWindowsPortScanPowerShell(signal?: AbortSignal): Promise { - const script = [ - "$ErrorActionPreference = 'Stop'", - '$connections = Get-NetTCPConnection -State Listen -ErrorAction Stop', - '$items = foreach ($connection in $connections) {', - ' $name = $null', - ' try {', - ' $process = Get-Process -Id $connection.OwningProcess -ErrorAction Stop', - ' $name = $process.ProcessName', - ' } catch {}', - ' [pscustomobject]@{', - ' host = [string]$connection.LocalAddress', - ' port = [int]$connection.LocalPort', - ' pid = [int]$connection.OwningProcess', - ' processName = $name', - ' }', - '}', - '$items | ConvertTo-Json -Compress -Depth 3' - ].join('\n') - const encoded = encodePowerShellCommand(script) - const lastError: unknown[] = [] +/** Rows, or null when netstat could not answer and the fallback should run. */ +async function readWindowsNetstatPorts(signal?: AbortSignal): Promise { + let stdout: string + try { + const result = await runProcess({ + program: windowsSystem32Binary('netstat.exe'), + // No `-p tcp`: on Windows that protocol name means TCP over IPv4 only, so + // it hides every `[::]` listener the retired PowerShell payload reported. + args: ['-ano'], + env: buildRelayCommandEnv(), + timeoutMs: WINDOWS_PORT_SCAN_TIMEOUT_MS, + maxOutputBytes: WINDOWS_PORT_SCAN_MAX_OUTPUT_BYTES, + signal + }) + if (result.timedOut || result.code !== 0) { + return null + } + // A capped read still exits 0 and its head still parses, so nothing + // downstream can tell a partial table from a whole one. netstat prints IPv4 + // TCP, then IPv6 TCP, then UDP, so the rows lost first are exactly the + // `[::]` listeners that dropping `-p tcp` above exists to keep. Refuse the + // whole read rather than publish its head. + if (Buffer.byteLength(result.stdout) >= WINDOWS_PORT_SCAN_MAX_OUTPUT_BYTES) { + reportWindowsNetstatUnusable('output hit the capture cap and was truncated') + return null + } + stdout = result.stdout + } catch { + return null + } + const ports = parseWindowsNetstatOutput(stdout) + // Windows always has a listener (RPC endpoint mapper, SMB), so an exit-0 scan + // that parses to nothing is a reader that was blocked, not an idle host. + if (ports.length === 0) { + reportWindowsNetstatUnusable('exited 0 but no listening row parsed') + return null + } + return ports +} - for (const binary of ['powershell.exe', 'pwsh.exe']) { - try { - const { stdout } = await execFileAsync( - binary, - ['-NoProfile', '-NonInteractive', '-ExecutionPolicy', 'Bypass', '-EncodedCommand', encoded], - { - env: buildRelayCommandEnv(), - encoding: 'utf-8', - maxBuffer: 1024 * 1024, - signal, - timeout: WINDOWS_PORT_SCAN_TIMEOUT_MS, - windowsHide: true - } +/** Reasons already reported. A fixed two-value vocabulary, so it cannot grow. */ +const reportedNetstatFailures = new Set() + +/** + * Say once why the scan left the native path. + * + * Both fall-throughs are permanent when they are wrong — the host stays on the + * PowerShell payload, or on nothing, for the life of the relay — and the scan + * repeats every 12-30s, so this logs one line rather than a stream. + * + * Through relayLogLine, not console.warn: this only ever runs in the detached + * daemon, whose stderr installRelayLogRotation routes into the relay.log that + * the remote-diagnostics tail reads. An untimestamped line in that file cannot + * be placed against the reconnect flaps around it (#7773), and "since when" is + * most of what this line is for. + */ +function reportWindowsNetstatUnusable(reason: string): void { + // Per reason, not per module: a host that parses nothing today and truncates + // tomorrow has two different faults, and one flag would hide the second. + if (reportedNetstatFailures.has(reason)) { + return + } + reportedNetstatFailures.add(reason) + relayLogLine(`[ports] netstat unusable on this host (${reason}); falling back to PowerShell`) +} + +/** Test-only: re-arm the one-shot so each case can observe its own line. */ +export function resetWindowsPortScanDiagnosticsForTests(): void { + reportedNetstatFailures.clear() +} + +/** + * Fill in owning-process names from the shared process-table snapshot. + * + * Names are optional data — the panel renders host/port/pid without them — so a + * host that cannot read the table keeps its rows. This shares whatever scan the + * table already runs rather than avoiding one, and on a relay that scan is a + * `powershell.exe` CIM query -- no released relay carries the native addon, so + * that is the path every SSH host takes, not a fallback. + * See docs/reference/windows-process-enumeration.md. + * + * Only `name` is read here, so this wants `readWindowsProcessIdentityTable` + * once #17866 lands -- on the detailed reader it would pay per-process handles + * for a field it discards. + * + * Best-effort by design: the snapshot is shared and TTL-cached, so it can + * predate netstat and hand a recycled PID its previous owner's name. Only + * labels read this field, and a fresh read would cost every caller a scan. + */ +async function attachWindowsProcessNames( + ports: DetectedPort[], + signal?: AbortSignal +): Promise { + const pids = new Set(ports.flatMap((port) => (port.pid == null ? [] : [port.pid]))) + // The shared snapshot takes no signal and must not be cancelled on one + // caller's behalf, so an abandoned scan declines to wait for it instead. + if (pids.size === 0 || signal?.aborted) { + return ports + } + let names: Map + try { + const rows = await readWindowsProcessTable() + names = new Map( + rows.flatMap((row) => + pids.has(row.pid) && row.name ? [[row.pid, stripExecutableSuffix(row.name)] as const] : [] ) - return stdout + ) + } catch { + return ports + } + return ports.map((port) => { + const processName = port.pid == null ? undefined : names.get(port.pid) + return processName ? { ...port, processName } : port + }) +} + +// The process table reports `sshd.exe`; the retired `Get-Process` payload +// reported `sshd`. The sshd filter below and every client that already renders +// these rows read the bare name, so keep publishing that spelling. +function stripExecutableSuffix(name: string): string { + return name.replace(/\.exe$/i, '') +} + +/** + * Single line so it survives as one argv element regardless of how the + * shell-less spawn hands it to PowerShell's `-Command` parser. Exported so + * windows-port-scan.win32.test.ts can run it: a missing `;` between statements + * is a parse error the mocked tests cannot see. + */ +export const WINDOWS_PORT_SCAN_SCRIPT = [ + "$ErrorActionPreference = 'Stop';", + 'Get-NetTCPConnection -State Listen | ForEach-Object {', + '$connection = $_; $name = $null;', + 'try { $name = (Get-Process -Id $connection.OwningProcess -ErrorAction Stop).ProcessName } catch { };', + '[pscustomobject]@{ host = [string]$connection.LocalAddress; port = [int]$connection.LocalPort;', + 'pid = [int]$connection.OwningProcess; processName = $name }', + '} | ConvertTo-Json -Compress -Depth 3' +].join(' ') + +async function runWindowsPortScanPowerShell(signal?: AbortSignal): Promise { + let lastError: unknown + + for (const program of [windowsPowerShellPath(), 'pwsh.exe']) { + try { + const result = await runProcess({ + program, + // No `-ExecutionPolicy` override: the policy gates script *files*, never + // `-Command`. Verified on Windows 11 — `-ExecutionPolicy Restricted + // -Command` still runs, while `-File` against an unsigned .ps1 does not. + args: ['-NoProfile', '-NonInteractive', '-Command', WINDOWS_PORT_SCAN_SCRIPT], + env: buildRelayCommandEnv(), + timeoutMs: WINDOWS_PORT_SCAN_TIMEOUT_MS, + maxOutputBytes: WINDOWS_PORT_SCAN_MAX_OUTPUT_BYTES, + signal + }) + if (signal?.aborted) { + throw new Error('windows port scan aborted') + } + if (result.timedOut || result.code !== 0) { + lastError ??= new Error( + `windows port scan PowerShell failed (code=${result.code} timedOut=${result.timedOut})` + ) + continue + } + return result.stdout } catch (error) { if (signal?.aborted) { throw error } - lastError.push(error) + lastError ??= error } } - throw lastError[0] ?? new Error('PowerShell unavailable') + throw lastError ?? new Error('PowerShell unavailable') } export function parseWindowsPowerShellPortRows(json: string): DetectedPort[] { @@ -98,26 +245,85 @@ export function parseWindowsPowerShellPortRows(json: string): DetectedPort[] { return rows.flatMap((row) => parseWindowsPortRow(row)) } +/** + * Listening rows, on a host in any UI language. + * + * `LISTENING` is not in `netstat.exe` — it lives in + * `System32\\netstat.exe.mui` beside `ESTABLISHED` and `Proto`, and MUI + * selection follows the UI language, so the pinned-locale env in + * relay-command-env.ts cannot reach it. A German host prints `ABHÖREN` and the + * word test finds nothing at all. + * + * The shape is language-independent: a listening socket has no peer, so its + * foreign address is `0.0.0.0:0` / `[::]:0`, and on every state Windows prints + * with a real peer that port is non-zero. Shape stays the fallback because the + * converse does not hold — `BOUND` and `CLOSED` print a zero peer too, and on a + * localized host their words are just as unreadable as the listening one. + */ export function parseWindowsNetstatOutput(output: string): DetectedPort[] { - const rows: DetectedPort[] = [] + const { rows, tcpRows } = scanWindowsNetstatTcpRows(output) + const byStateWord = rows.filter((row) => row.state === 'LISTENING') + if (byStateWord.length > 0 || tcpRows === 0) { + return byStateWord.map((row) => row.port) + } + return readDominantZeroPeerState(rows) +} + +/** + * Of the zero-peer states, keep only the one that dominates. + * + * Shape alone would publish a phantom listener: one `BOUND` socket among real + * listeners looks identical to them once the state word is unreadable. But it + * cannot dominate — listeners outnumber those transients by roughly 50:1 on a + * real host (51 against 0 here), so the largest zero-peer group is the + * listening one. An exact tie keeps every tied group rather than guessing, + * which is no worse than reading shape alone. + * + * A majority rule inverts if the majority is wrong: enough transient zero-peer + * sockets and the phantoms win, publishing those and dropping the real + * listeners. The hatch is to return [] here and defer to the PowerShell reader, + * which reads the state word instead of inferring it. + */ +function readDominantZeroPeerState(rows: NetstatTcpRow[]): DetectedPort[] { + const countByState = new Map() + for (const row of rows) { + if (row.zeroPeer) { + countByState.set(row.state, (countByState.get(row.state) ?? 0) + 1) + } + } + const largest = Math.max(0, ...countByState.values()) + const dominant = new Set( + [...countByState].filter(([, count]) => count === largest).map(([state]) => state) + ) + return rows.flatMap((row) => (row.zeroPeer && dominant.has(row.state) ? [row.port] : [])) +} + +type NetstatTcpRow = { state: string; zeroPeer: boolean; port: DetectedPort } + +/** `tcpRows` separates a localized host from one with genuinely no TCP output. */ +function scanWindowsNetstatTcpRows(output: string): { rows: NetstatTcpRow[]; tcpRows: number } { + const rows: NetstatTcpRow[] = [] + let tcpRows = 0 for (const line of output.split(/\r?\n/)) { const fields = getProcessOutputFields(line, 5) if (fields.length < 5 || fields[0].toUpperCase() !== 'TCP') { continue } - if (fields[3].toUpperCase() !== 'LISTENING') { - continue - } + tcpRows += 1 const hostPort = parseWindowsNetstatAddress(fields[1]) const pid = Number.parseInt(fields[4], 10) if (!hostPort || !Number.isSafeInteger(pid) || pid <= 0) { continue } - rows.push({ ...hostPort, pid }) + rows.push({ + state: fields[3].toUpperCase(), + zeroPeer: readWindowsNetstatPort(fields[2]) === 0, + port: { ...hostPort, pid } + }) } - return rows + return { rows, tcpRows } } function parseWindowsPortRow(row: unknown): DetectedPort[] { @@ -165,13 +371,20 @@ function readInteger(value: unknown): number | undefined { return Number.isSafeInteger(parsed) ? parsed : undefined } -function parseWindowsNetstatAddress(value: string): { host: string; port: number } | null { - const ipv6Match = /^\[(.*)\]:(\d+)$/.exec(value) - const portText = ipv6Match?.[2] ?? value.slice(value.lastIndexOf(':') + 1) +/** Port alone, keeping 0 — the foreign-address test above turns on that value. */ +function readWindowsNetstatPort(value: string): number | null { + const ipv6Match = /^\[.*\]:(\d+)$/.exec(value) + const portText = ipv6Match?.[1] ?? value.slice(value.lastIndexOf(':') + 1) const port = Number.parseInt(portText, 10) - if (!Number.isSafeInteger(port) || port <= 0) { + return Number.isSafeInteger(port) ? port : null +} + +function parseWindowsNetstatAddress(value: string): { host: string; port: number } | null { + const port = readWindowsNetstatPort(value) + if (port == null || port <= 0) { return null } + const ipv6Match = /^\[(.*)\]:\d+$/.exec(value) if (ipv6Match) { return { host: ipv6Match[1], port } } diff --git a/src/relay/windows-port-scan.win32.test.ts b/src/relay/windows-port-scan.win32.test.ts new file mode 100644 index 00000000000..976f477cd9c --- /dev/null +++ b/src/relay/windows-port-scan.win32.test.ts @@ -0,0 +1,52 @@ +import { describe, expect, it } from 'vitest' +import { runProcess } from '../shared/child-process/run-process' +import { windowsPowerShellPath } from '../shared/child-process/windows-system-binary' +import { + WINDOWS_PORT_SCAN_SCRIPT, + parseWindowsPowerShellPortRows, + scanWindowsListeningPorts +} from './windows-port-scan' + +/** + * The mocked suite pins the argv; this pins that the argv works. + * + * Both halves are things a mock cannot see: netstat's real column layout (a + * `-p tcp` here silently drops every `[::]` listener), and whether the joined + * one-line PowerShell script even parses — a missing `;` between statements is + * a ParserError, and the fallback would then be dead on the day it is needed. + * + * Runs only on win32; skipped elsewhere. + */ +const describeOnWindows = process.platform === 'win32' ? describe : describe.skip + +describeOnWindows('windows port scan against the real host', () => { + it('finds listeners over both address families through netstat', async () => { + const ports = await scanWindowsListeningPorts() + + expect(ports.length).toBeGreaterThan(0) + for (const port of ports) { + expect(port.port).toBeGreaterThan(0) + expect(port.host.length).toBeGreaterThan(0) + } + // Windows binds RPC/SMB dual-stack, so both families must be represented. + expect(ports.some((port) => port.host.includes(':'))).toBe(true) + expect(ports.some((port) => !port.host.includes(':'))).toBe(true) + // Names come from the shared process table, never from a shell of our own. + expect(ports.some((port) => port.processName)).toBe(true) + // The table spells them `svchost.exe`; clients have always seen `svchost`. + expect(ports.every((port) => !port.processName?.endsWith('.exe'))).toBe(true) + }, 30_000) + + it('runs the de-escalated PowerShell fallback command line', async () => { + const result = await runProcess({ + program: windowsPowerShellPath(), + args: ['-NoProfile', '-NonInteractive', '-Command', WINDOWS_PORT_SCAN_SCRIPT], + timeoutMs: 20_000 + }) + + // No stderr assertion: an autoload or first-run banner writes there without + // the scan having failed. + expect(result.code).toBe(0) + expect(parseWindowsPowerShellPortRows(result.stdout).length).toBeGreaterThan(0) + }, 30_000) +}) diff --git a/src/shared/child-process/__fixtures__/child-process-import-allowlist.txt b/src/shared/child-process/__fixtures__/child-process-import-allowlist.txt index 929cc4f10ad..b2b82fdff69 100644 --- a/src/shared/child-process/__fixtures__/child-process-import-allowlist.txt +++ b/src/shared/child-process/__fixtures__/child-process-import-allowlist.txt @@ -176,7 +176,6 @@ src/relay/git-stdout-stream.ts src/relay/preflight-handler.ts src/relay/pty-shell-utils.ts src/relay/subprocess-tree-termination.ts -src/relay/windows-port-scan.ts src/relay/workspace-space-scan.ts src/shared/ephemeral-vm-recipe-process.ts src/shared/ephemeral-vm-recipe-runner.ts diff --git a/src/shared/child-process/__fixtures__/windows-console-visibility-allowlist.txt b/src/shared/child-process/__fixtures__/windows-console-visibility-allowlist.txt index a657002a8ff..068f9a8a96f 100644 --- a/src/shared/child-process/__fixtures__/windows-console-visibility-allowlist.txt +++ b/src/shared/child-process/__fixtures__/windows-console-visibility-allowlist.txt @@ -60,7 +60,6 @@ relay/fs-list-files-fallback-chain.ts relay/git-handler.ts relay/pty-shell-utils.ts relay/subprocess-tree-termination.ts -relay/windows-port-scan.ts relay/workspace-space-scan.ts shared/fish-binary-requirement.ts shared/process-table-snapshot-reader.ts diff --git a/src/shared/child-process/child-process-import-boundary.test.ts b/src/shared/child-process/child-process-import-boundary.test.ts index 32c6c9d890e..678bc53c4d0 100644 --- a/src/shared/child-process/child-process-import-boundary.test.ts +++ b/src/shared/child-process/child-process-import-boundary.test.ts @@ -29,7 +29,7 @@ const CHILD_PROCESS_IMPORT_ALLOWLIST: readonly string[] = readFileSync( * May only ever be DECREASED, and only by migrating a file off * `node:child_process`. Raising it is never the fix. */ -const DIRECT_IMPORTER_PIN = 158 +const DIRECT_IMPORTER_PIN = 157 const IMPORT_PATTERN = /(?:from\s+['"]node:child_process['"]|from\s+['"]child_process['"]|require\(\s*['"]node:child_process['"]|require\(\s*['"]child_process['"])/ diff --git a/src/shared/child-process/windows-console-visibility.test.ts b/src/shared/child-process/windows-console-visibility.test.ts index 596728fa245..f135fc10555 100644 --- a/src/shared/child-process/windows-console-visibility.test.ts +++ b/src/shared/child-process/windows-console-visibility.test.ts @@ -34,7 +34,7 @@ const ALLOWLIST: readonly string[] = readAllowlist( * the allowlist does not bound this: a swap (one file fixed and delisted, one * new file added with its entry) satisfies both membership assertions. */ -const UNHIDDEN_SPAWNER_PIN = 66 +const UNHIDDEN_SPAWNER_PIN = 65 const CHILD_PROCESS_IMPORT = /from\s+['"](?:node:)?child_process['"]|require\(\s*['"](?:node:)?child_process['"]/ From 0cbb01ef4b5cd931bfa81961eb147a0c67c408ce Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Sat, 5 Sep 2026 21:13:06 -0700 Subject: [PATCH 43/69] fix(security): apply the Windows path-hardening ACL that never ran (#17884) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(security): apply the Windows path-hardening ACL that never ran `buildWindowsRestrictAclArgs` invoked the hardening script as `powershell.exe -Command - - - `) - }) - await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve)) - const port = (server.address() as AddressInfo).port - return { - sourceUrl: `http://127.0.0.1:${port}/source`, - close: () => closeServer(server) - } -} - async function startBrowserWindowCloseServer(): Promise<{ url: string sourceUrl: string @@ -281,8 +204,8 @@ async function clickBrowserLink( browserTabId: string, selector: string, options: { - modifiers?: ('meta' | 'control')[] - button?: 'left' | 'middle' + modifiers?: ('meta' | 'control' | 'shift')[] + button?: 'left' | 'middle' | 'right' frameSelector?: string } = {} ): Promise { @@ -317,21 +240,31 @@ async function clickBrowserLink( if (!point) { throw new Error(`Missing browser link ${targetSelector}`) } - await webview.sendInputEvent({ type: 'mouseMove', modifiers: inputModifiers, ...point }) - await webview.sendInputEvent({ - type: 'mouseDown', - button, - clickCount: 1, - modifiers: inputModifiers, - ...point - }) - await webview.sendInputEvent({ - type: 'mouseUp', - button, - clickCount: 1, - modifiers: inputModifiers, - ...point - }) + const holdShift = inputModifiers.includes('shift') + if (holdShift) { + await webview.sendInputEvent({ type: 'keyDown', keyCode: 'Shift', modifiers: ['shift'] }) + } + try { + await webview.sendInputEvent({ type: 'mouseMove', modifiers: inputModifiers, ...point }) + await webview.sendInputEvent({ + type: 'mouseDown', + button, + clickCount: 1, + modifiers: inputModifiers, + ...point + }) + await webview.sendInputEvent({ + type: 'mouseUp', + button, + clickCount: 1, + modifiers: inputModifiers, + ...point + }) + } finally { + if (holdShift) { + await webview.sendInputEvent({ type: 'keyUp', keyCode: 'Shift' }) + } + } }, { targetBrowserTabId: browserTabId, @@ -343,21 +276,43 @@ async function clickBrowserLink( ) } -async function expectBrowserTabActive( +async function waitForTabIdByExactTitle( page: Parameters[0], title: string -): Promise { +): Promise { const resolveTabId = (): Promise => page.locator('[data-tab-id]').evaluateAll((tabs, exactTitle) => { const tab = tabs.find((candidate) => candidate.textContent?.trim() === exactTitle) return tab?.getAttribute('data-tab-id') ?? null }, title) await expect.poll(resolveTabId, { timeout: 10_000 }).not.toBeNull() - const tabId = await resolveTabId() - expect(tabId).toBeTruthy() + return (await resolveTabId()) as string +} + +async function expectBrowserTabActive( + page: Parameters[0], + title: string +): Promise { + const tabId = await waitForTabIdByExactTitle(page, title) await expect(page.locator(`[data-browser-overlay-tab-id="${tabId}"]`)).toHaveCSS('opacity', '1') } +async function expectBrowserTabOpenedInBackground( + page: Parameters[0], + sourceTabId: string, + title: string +): Promise { + const openedTabId = await waitForTabIdByExactTitle(page, title) + await expect(page.locator(`[data-browser-overlay-tab-id="${sourceTabId}"]`)).toHaveCSS( + 'opacity', + '1' + ) + await expect(page.locator(`[data-browser-overlay-tab-id="${openedTabId}"]`)).toHaveCSS( + 'opacity', + '0' + ) +} + async function readBrowserInputValue( page: Parameters[0], browserTabId: string @@ -680,7 +635,7 @@ test.describe('Browser Tab', () => { } }) - test('every new-tab link gesture activates an Orca tab and never a native window', async ({ + test('new-tab link gestures follow Chrome foreground and background behavior', async ({ electronApp, orcaPage }) => { @@ -698,38 +653,51 @@ test.describe('Browser Tab', () => { const baseWindowCount = await electronApp.evaluate( ({ BaseWindow }) => BaseWindow.getAllWindows().length ) - // A plain target=_blank click is a new-tab request, in the main frame and in an iframe; - // the source tab must stay put rather than navigate away under it. + // A plain main-frame target=_blank click must not navigate the source tab away. const sourceTabLocator = orcaPage.locator(`[data-tab-id="${sourceTab!.id}"]`) - await clickBrowserLink(orcaPage, sourceTab!.id, '#external-link') - await expectBrowserTabActive(orcaPage, 'Linked destination') + await clickBrowserLink(orcaPage, sourceTab!.id, '#blank-link') + await expectBrowserTabActive(orcaPage, 'Blank target destination') await expect(sourceTabLocator).toContainText('Source page') await switchToBrowserTab(orcaPage, worktreeId, sourceTab!.id) + // Context-menu links keep the source visible until the new tab is selected. + await clickBrowserLink(orcaPage, sourceTab!.id, '#external-link', { button: 'right' }) + await orcaPage + .getByRole('menuitem', { name: 'Open Link In Orca Browser', exact: true }) + .click() + await expectBrowserTabOpenedInBackground(orcaPage, sourceTab!.id, 'Linked destination') await clickBrowserLink(orcaPage, sourceTab!.id, '#frame-link', { frameSelector: '#link-frame' }) await expectBrowserTabActive(orcaPage, 'Frame destination') - await expect(sourceTabLocator).toContainText('Source page') await switchToBrowserTab(orcaPage, worktreeId, sourceTab!.id) await clickBrowserLink(orcaPage, sourceTab!.id, '#frame-modifier-link', { frameSelector: '#link-frame', modifiers: process.platform === 'darwin' ? ['meta'] : ['control'] }) - await expectBrowserTabActive(orcaPage, 'Frame modifier destination') - await switchToBrowserTab(orcaPage, worktreeId, sourceTab!.id) + await expectBrowserTabOpenedInBackground( + orcaPage, + sourceTab!.id, + 'Frame modifier destination' + ) await clickBrowserLink(orcaPage, sourceTab!.id, '#frame-middle-link', { button: 'middle', frameSelector: '#link-frame' }) - await expectBrowserTabActive(orcaPage, 'Frame middle destination') - await switchToBrowserTab(orcaPage, worktreeId, sourceTab!.id) + await expectBrowserTabOpenedInBackground(orcaPage, sourceTab!.id, 'Frame middle destination') await clickBrowserLink(orcaPage, sourceTab!.id, '#modifier-link', { modifiers: process.platform === 'darwin' ? ['meta'] : ['control'] }) - await expectBrowserTabActive(orcaPage, 'Modifier destination') + await expectBrowserTabOpenedInBackground(orcaPage, sourceTab!.id, 'Modifier destination') + + await clickBrowserLink(orcaPage, sourceTab!.id, '#frame-shift-middle-link', { + button: 'middle', + modifiers: ['shift'], + frameSelector: '#link-frame' + }) + await expectBrowserTabActive(orcaPage, 'Frame shift middle destination') await switchToBrowserTab(orcaPage, worktreeId, sourceTab!.id) const tabCountBeforeCancelledClick = await orcaPage.locator('[data-tab-id]').count() @@ -740,7 +708,7 @@ test.describe('Browser Tab', () => { await expect(orcaPage.locator('[data-tab-id]')).toHaveCount(tabCountBeforeCancelledClick) await clickBrowserLink(orcaPage, sourceTab!.id, '#middle-link', { button: 'middle' }) - await expectBrowserTabActive(orcaPage, 'Middle-click destination') + await expectBrowserTabOpenedInBackground(orcaPage, sourceTab!.id, 'Middle-click destination') await expect .poll(() => electronApp.evaluate(({ BaseWindow }) => BaseWindow.getAllWindows().length), { timeout: 5_000 diff --git a/tests/e2e/helpers/browser-link-server.ts b/tests/e2e/helpers/browser-link-server.ts new file mode 100644 index 00000000000..81debdc5859 --- /dev/null +++ b/tests/e2e/helpers/browser-link-server.ts @@ -0,0 +1,100 @@ +import { createServer, type Server } from 'node:http' +import type { AddressInfo } from 'node:net' + +async function closeServer(server: Server): Promise { + await new Promise((resolve, reject) => + server.close((error) => { + if (error) { + reject(error) + return + } + resolve() + }) + ) +} + +export async function startBrowserLinkServer(): Promise<{ + sourceUrl: string + close: () => Promise +}> { + const server = createServer((request, response) => { + const origin = `http://127.0.0.1:${(server.address() as AddressInfo).port}` + const pathname = new URL(request.url ?? '/', origin).pathname + response.writeHead(200, { 'Content-Type': 'text/html; charset=utf-8' }) + if (pathname === '/destination') { + response.end( + `Linked destinationDestination Return` + ) + return + } + if (pathname === '/blank-destination') { + response.end( + 'Blank target destinationBlank target destination' + ) + return + } + if (pathname === '/frame-destination') { + response.end( + `Frame destinationFrame destination Return` + ) + return + } + if (pathname === '/frame-modifier-destination') { + response.end( + 'Frame modifier destinationFrame modifier destination' + ) + return + } + if (pathname === '/frame-middle-destination') { + response.end( + 'Frame middle destinationFrame middle destination' + ) + return + } + if (pathname === '/frame') { + response.end( + `${request.url?.includes('shift-middle') ? 'Frame shift middle destination' : ''}Open frame destinationOpen frame modifier destinationOpen frame middle destinationOpen foreground frame tab` + ) + return + } + if (pathname === '/modifier-destination') { + response.end( + 'Modifier destinationModifier destination' + ) + return + } + if (pathname === '/middle-destination') { + response.end( + 'Middle-click destinationMiddle-click destination' + ) + return + } + response.end(` + + + ${request.url?.includes('shift-middle') ? 'Shift middle destination' : 'Source page'} + + Open destination + Open blank target destination + Open with modifier + Open with middle click + Open foreground tab + Handle in page + + + + + `) + }) + await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve)) + const port = (server.address() as AddressInfo).port + return { + sourceUrl: `http://127.0.0.1:${port}/source`, + close: () => closeServer(server) + } +} From fc5fa168705a94348d1d06d6ce15e709c7959cab Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Sat, 5 Sep 2026 21:42:34 -0700 Subject: [PATCH 50/69] perf(windows): split the process table into two flag sets (#17866) * perf(windows): split the process table into two flag sets MDE flags "suspicious memory activity" on the process-table reader: it opened a handle into every process on the box and read each one's PEB on a repeating cadence. Two changes narrow that. Drop `Memory` outright. It cost a second OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ) plus GetProcessMemoryInfo per process, and nothing reads a working set off this table -- the Resource Manager runs its own sweep, and the addon stores WorkingSetSize into a DWORD so anything above 4 GB wraps. Split the rest in two. `readWindowsProcessIdentityTable[Fresh]` is a bare Toolhelp32 walk with zero per-process handles, and returns `WindowsProcessIdentityRow`, which has no `command` to read. `readWindowsProcessTable[Fresh]` keeps the command line for the callers that match on it. PTY root identity and the owner start-time probe move to the cheap reader; agent recognition, port attribution, codex turn processes and structured-TUI matching all genuinely need the command line and stay. Two independently single-flighted caches, never one per caller: the fan-out this module prevents is one scan per caller, and each reader still serves every caller wanting its flag set. The wedge gate and the 3s deadline stay shared, because both readers call the same addon and one wedged read latches its one `requestInProgress`. With no binding there is only the 1.4s PowerShell scan to run, so the identity view rides the detailed snapshot rather than forking a second one. Measured on Windows 11, 492 processes (p50/p95): identity 6.3/7.0 ms, detailed 12.3/13.4 ms, previous memory+commandLine 13.1/14.1 ms. * fix(windows): serialize native process-table reads across flag sets The two flag-set readers could both be in flight at once, and the vendored wrapper does not tolerate that. `getRawProcessList` pushes the callback onto one list and calls the addon only when no request is in progress, so a second concurrent caller's `flags` are DISCARDED and it is handed the first caller's rows. Measured against the real addon: identity issued first, both callers got the same array, 0 of 541 rows with a command line. A detailed read overlapping an identity read therefore returned a table with every command line empty, which agent recognition reads as "no agent" -- silently, and only under concurrency. Nothing already here excluded that. Each snapshot cache single-flights only within itself, and the wedge set latches only after a read misses its 3s deadline, so through the healthy ~12ms of a scan neither reader excluded the other. Overlap is the normal state: panes poll detailed at 750ms while a teardown takes identity snapshots. `nativeReadGate` admits one native read at a time across both flag sets. It also fixes the relay path, where `adaptAddon` has no queue at all and two simultaneous CreateToolhelp32Snapshot calls are the crash the vendor's queue exists to prevent. Every link settles, so a wedged read never strands a waiter; the waiter re-checks the wedge and rejects. With one call outstanding, retention stays bounded at one callback rather than one per reader. Also from review: - The CIM fallback now belongs to the detailed flag set alone, and the identity view projects that snapshot through `toIdentityRow`, so an identity row carries no command line on a no-binding host either. - The concurrency test modelled the wrapper's coalescing queue, which the previous synchronous mock could not express; verified failing without the gate and passing with it. - `agent-session-process-identity-probe` early-returns when the creation-time flag is unavailable, which no shipped addon build provides, instead of scanning the table to produce null. - Corrected the cost framing: Memory took an OpenProcess(...|VM_READ) it never read through, so dropping it halves per-process handle opens and leaves the PEB/ReadProcessMemory telemetry unchanged. * test(windows): keep read exclusion across resets and flag each field Two review follow-ups, both about tests passing for the wrong reason. `resetNativeReaderState` replaced the read gate with a resolved promise, so waiters still holding the old chain ran beside reads queued on the new one. Reachable only from the `__set*ForTests` hooks, which is what makes it worth fixing: it hands a suite two concurrent calls into its own mock addon -- the exact condition the concurrency tests exist to detect. Chain onto the gate instead; every link settles within the deadline, so the bounded wait that costs is the right trade. The coalescing mock shaped every field off the CommandLine bit, so an identity read that did request CreationTime got `creationTimeMs` stripped. The identity-side assertion was then only `!('command' in row)`, which a correctly flagged read and a coalesced one satisfy equally: a future regression losing identity flags under concurrency would have kept the case green. Gate each field on its own bit and assert `creationTimeMs` positively, inside the helper both orderings share. Concurrency assertions move to a new bare-addon mock. The coalescing mock's own latch means it can never report more than one call in flight, so measuring exclusion there proved nothing; the bare addon has no queue -- like `adaptAddon` on a relay, where re-entering CreateToolhelp32Snapshot is a real crash -- and makes re-entry visible. Verified by deletion: restoring `nativeReadGate = Promise.resolve()` fails the reset case with `expected 2 to be 1`, and restoring the single-bit mock fails both overlap orderings on `creationTimeMs`. * docs(windows): count the third test defect in the list that names them The section opened "Two defects have now shipped", numbered two, then described the third in its closing paragraph -- a list that reads as a complete account while quietly omitting one, which is the exact failure the section exists to warn about. Say three and number it, and note that the third arrived inside the fix for the first two. Also record why the creationTimeMs and flags-array assertions are not redundant, in the doc and beside the assertions: the flags array catches a read served another flag set's rows, the positional creationTimeMs check catches field shaping (identity dropping CreationTime, or toIdentityRow not forwarding it). Neither sees the other's failure. * docs(windows): stop describing a PEB read this release removed Every comment here that justified the flag split in terms of PEB reads became false when the command-line reader moved to the kernel. Left alone, the enumeration doc contradicted itself inside one file: the flag-set section described three chained `ReadProcessMemory` calls per process while the sections below it explained that the addon contains no such primitive and has no PEB fallback. The measurement is now attributed rather than merged. Dropping `Memory` halved the per-process handle opens and nothing else -- both handles carried `PROCESS_VM_READ` at the time -- and it was replacing the PEB walk that took `PROCESS_VM_READ` and `ReadProcessMemory` out of the addon. Neither change substitutes for the other, which is worth keeping straight: the split's remaining value is the handle itself, not the memory access. Also adds `relay/windows-port-scan.ts` to the caller table, the one caller this effort introduced, and records that it reads only pid/name through the detailed reader -- free while a pane is polling, not free on a headless relay. * test(windows): pin the fresh links path against the identity TTL cache The identity and detailed tables are separate snapshot readers with independent TTLs, so the detailed path's existing freshness guard says nothing about the ancestry walk's. Cover the identity reader on its own. --------- Co-authored-by: Orca Worker --- docs/reference/windows-edr-posture.md | 39 ++- docs/reference/windows-process-enumeration.md | 200 ++++++++++-- .../windows-foreground-process-rows.test.ts | 27 +- .../windows-foreground-process-rows.ts | 12 + .../agent-session-process-identity-probe.ts | 16 +- src/main/windows-pty-root-identity.ts | 4 +- .../windows/windows-process-table-cim-scan.ts | 2 + .../windows/windows-process-table.test.ts | 307 +++++++++++++++++- src/main/windows/windows-process-table.ts | 228 ++++++++++--- 9 files changed, 727 insertions(+), 108 deletions(-) diff --git a/docs/reference/windows-edr-posture.md b/docs/reference/windows-edr-posture.md index ff3e6d49cc6..24854890fc7 100644 --- a/docs/reference/windows-edr-posture.md +++ b/docs/reference/windows-edr-posture.md @@ -84,10 +84,12 @@ embedded name for the old disk name to contradict. ### Every process gets a handle, on a timer -`src/main/windows/windows-process-table.ts` takes a Toolhelp32 snapshot under -**one** flag set, `CommandLine | CreationTime`, shared by every caller. pid, ppid -and name come out of the snapshot itself and open nothing. `CommandLine` is what -opens a handle: the addon calls `GetProcessCommandLine` per process, which opens +`src/main/windows/windows-process-table.ts` takes a Toolhelp32 snapshot under one +of **two** flag sets: identity (`None | CreationTime`) for callers that read only +pid, ppid and name, and detailed (`+ CommandLine`) for callers that match on a +command line. pid, ppid and name come out of the snapshot itself and open +nothing, so an identity scan opens nothing at all. `CommandLine` is what opens a +handle: the addon calls `GetProcessCommandLine` per process, which opens `PROCESS_QUERY_LIMITED_INFORMATION` — the same right Task Manager takes — and asks the kernel for the string. Upstream it opened `PROCESS_QUERY_INFORMATION | PROCESS_VM_READ` and walked the PEB with three @@ -113,15 +115,15 @@ panes multiplied it (#15036). The native snapshot answers the same question in See [`windows-process-enumeration.md`](./windows-process-enumeration.md). -Asking for fewer fields is cheaper, and the module now asks for the smallest set -that still answers every caller. There is **no** per-flag-set cache split: one -TTL-cached snapshot serves everyone, deliberately, because a split would restore -the per-pane fan-out the cache exists to remove — a 32-wide teardown has to -collapse into one scan. So the cheap identity-only read is not something any -caller can select; every read pays for `CommandLine`. An earlier revision of this -file described a two-cache design with 6.3 ms / 12.3 ms p50 figures at 492 -processes. That design is not in the tree and those numbers describe no code -path here; the figures that do apply are the module's own, in +Asking for fewer fields is cheaper, and each caller now asks for the smallest set +that answers it. There are exactly **two** TTL-cached snapshots, one per flag +set, never one per caller: the fan-out the cache exists to remove is one scan per +_caller_, and each reader still serves every caller wanting its flag set, so a +32-wide teardown still collapses into one scan of each. Teardown identity and the +owner probe select the identity set and therefore open no handles; the per-pane +foreground tracker genuinely needs a command line and still pays for one. A third +cache would need a third flag set, not a third caller. Measured at 492 processes, +p50: identity 6.3 ms, detailed 12.3 ms — see [`windows-process-enumeration.md`](./windows-process-enumeration.md). **How an EDR read it:** a cross-process handle plus a remote memory read against @@ -150,11 +152,12 @@ unpatched source, so "it required cleanly" is not evidence. What to declare to administrators is now one `PROCESS_QUERY_LIMITED_INFORMATION` handle per process on a detailed snapshot and -no remote memory access at all. What this does not narrow is _which_ processes -are asked — a detailed scan still queries every pid, including `lsass.exe`. -Restricting the command-line pass to Orca's own subtree needs job-object -membership as its source of truth (a ppid-derived allowlist would miss the -detached, reparented descendants of #9045 and #10475), and remains unclaimed work. +no remote memory access at all; an identity snapshot opens nothing. What this +does not narrow is _which_ processes are asked — a detailed scan still queries +every pid, including `lsass.exe`. Restricting the command-line pass to Orca's own +subtree needs job-object membership as its source of truth (a ppid-derived +allowlist would miss the detached, reparented descendants of #9045 and #10475), +and remains unclaimed work. ### Encoded, policy-bypassing PowerShell diff --git a/docs/reference/windows-process-enumeration.md b/docs/reference/windows-process-enumeration.md index fb58030be6e..34afb56c8e6 100644 --- a/docs/reference/windows-process-enumeration.md +++ b/docs/reference/windows-process-enumeration.md @@ -16,38 +16,193 @@ table. It wraps a Toolhelp32 snapshot from `@vscode/windows-process-tree`. ```ts import { + readWindowsProcessIdentityTable, + readWindowsProcessIdentityTableFresh, readWindowsProcessTable, readWindowsProcessTableFresh } from '../windows/windows-process-table' ``` -- `readWindowsProcessTable()` — shared TTL cache. Use for anything periodic. -- `readWindowsProcessTableFresh()` — a snapshot that starts after the call. Use - for teardown identity, where a cached row can predate the exit it is being - asked about. +Each pair is a shared TTL cache plus a `Fresh` variant that starts its scan +after the call. Use `Fresh` for teardown identity, where a cached row can +predate the exit it is being asked about, and the cached one for anything +periodic. -Both **reject** when the table cannot be read. Do not convert that into an empty -array. An empty table is a claim that nothing is running, and callers act on -that claim by declaring a tree dead or a shell childless. "Unavailable" has to -stay distinguishable from "empty" — collapsing the two is how a PTY tree +All four **reject** when the table cannot be read. Do not convert that into an +empty array. An empty table is a claim that nothing is running, and callers act +on that claim by declaring a tree dead or a shell childless. "Unavailable" has +to stay distinguishable from "empty" — collapsing the two is how a PTY tree survived its own teardown (#9045). -Measured on Windows 11 with 1050 processes (p50 / p95): +## Two flag sets: ask for a command line only if you read one + +Neither flag is a wider column on the same query. Each is a separate +per-process syscall sequence, and they are not equally expensive to the EDR +watching: + +- `CommandLine` (`process_commandline.cc`) — + `OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION)`, then + `NtQueryInformationProcess(ProcessCommandLineInformation)` twice: once to size + the buffer, once to fill it. The kernel builds the string, so no address space + is opened or read. It used to walk the target's PEB with three chained + `ReadProcessMemory` calls; the patched addon no longer contains that primitive. +- `Memory` (`process.cc`) — retired. It took a **second** `OpenProcess`, and that + one carried `PROCESS_VM_READ`, which it acquired and never used. + +Measured here (541 processes, 405 openable), per detailed scan, before → after +dropping `Memory`: `OpenProcess` 1082 → 541. That halving is all the `Memory` +drop bought on its own — both handles carried `PROCESS_VM_READ` at the time, so +it moved the PEB traffic not at all. Replacing the PEB walk with the kernel +query is what took `PROCESS_VM_READ` and `ReadProcessMemory` out of the addon +altogether; the two changes compose, and neither substitutes for the other. + +So be precise about what these two flag sets buy now. A detailed scan is one +`OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION)` per process and no memory +access at all. What the split buys on top of that is the handle itself: an +identity scan opens nothing. + +So the module exposes two snapshots, and the row types differ so a cheap caller +cannot read what its flag set did not pay for: + +| reader | row type | flags | per-process handles | +| ------------------------------------------ | ---------------------------- | --------------------------- | ------------------- | +| `readWindowsProcessIdentityTable[Fresh]()` | `WindowsProcessIdentityRow` | `None \| CreationTime` | none | +| `readWindowsProcessTable[Fresh]()` | `WindowsProcessRow` | `+ CommandLine` | one `OpenProcess` | + +`Memory` is requested by neither. Nothing reads a working set off this table — +`windows-process-resource-collector.ts` runs its own sweep because it needs +commit and CPU counters in the same pass, and the addon stores `WorkingSetSize` +into a `DWORD` so anything above 4 GB wraps anyway. + +Measured on Windows 11 with 492 processes (p50 / p95): | | p50 | p95 | | -------------------------------- | ------- | ------- | -| pid + ppid + name | 15.9 ms | 17.5 ms | -| + memory + command line | 30.6 ms | 33.7 ms | +| identity (pid + ppid + name) | 6.3 ms | 7.0 ms | +| detailed (+ command line) | 12.3 ms | 13.4 ms | +| _retired_ (+ memory) | 13.1 ms | 14.1 ms | | `Get-CimInstance` via PowerShell | 706 ms | 723 ms | -Those are the module's published figures. The flag set this module actually -requests is `CommandLine | CreationTime` — **not** `Memory`, which cost a second -`OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ)` plus -`GetProcessMemoryInfo` per process (`src/process.cc:47-63`) for a value nothing -read. Dropping it halves the handles a snapshot opens. The remaining set sits -between the two rows above and has not been measured separately; on a real -Windows host, `Get-Counter '\Process(Orca)\Handle Count'` sampled across a -snapshot cadence is the check. +There are exactly **two** caches, never one per caller. The fan-out this module +exists to prevent is one scan per _caller_, and each reader still serves every +caller wanting its flag set, so a 32-wide teardown still collapses into one scan +of each. A third cache would need a third flag set, not a third caller. + +### Only one native read may be in flight, ever + +This is the price of having two flag sets, and it is not optional. + +The npm wrapper **coalesces rather than queues**. `getRawProcessList` pushes the +callback onto one list and calls the addon only when no request is in progress, +so a second concurrent caller's `flags` are **discarded** and it is handed the +first caller's rows. Measured against the real addon: issue identity first, both +callers get the same array, 0 of 541 rows carry a command line. A detailed read +that overlaps an identity read therefore returns a table with **every command +line empty**, and agent recognition reads that as "no agent" — silently, and +only under concurrency. + +Nothing else in this module prevents that. Each snapshot cache single-flights +only within itself (`inFlight` is a closure per reader), and the wedge set +latches only *after* a read misses its 3 s deadline, so through the healthy +~12 ms of a scan neither excludes the other. Overlap is the normal state rather +than an edge case: other panes keep polling detailed at 750 ms while a teardown +takes identity snapshots, and `codex-structured-turn-processes.ts` issues fresh +detailed scans on turn stop. + +`nativeReadGate` serializes every native read across both flag sets. It is also +what makes the relay's bare addon safe: `adaptAddon` has no queue at all, and +two simultaneous `CreateToolhelp32Snapshot` calls are the crash the vendor's +queue exists to prevent. Every link settles — a wedged read still rejects on its +deadline — so a waiter is never stranded; it re-checks the wedge and rejects. + +Because only one native call is ever outstanding, the wedge gate and the 3 s +deadline stay **shared** and retention stays bounded at exactly one callback, +not one per reader. Read ids are module-global and monotonic, so a late callback +can only clear its own wedge. + +`resetNativeReaderState` **chains** onto the gate rather than replacing it. A +replacement would let a waiter still holding the old chain run beside a read +queued on the new one; every link settles within the deadline, so chaining costs +a bounded wait and keeps the exclusion whole. That path is test-only, which is +exactly why it matters — it would otherwise hand a suite two concurrent calls +into its own mock, the condition these tests exist to detect. + +### Testing this module: assert a positive property, on the right mock + +Three defects have now shipped in this file's tests, all the same shape — a case +that passed for a reason other than the one it claimed to check: + +1. A loader that built a **fresh mock per call**, so the coalescing it was meant + to reproduce could never happen. +2. An identity-side assertion of only `!('command' in row)`, which a correctly + flagged read and a coalesced one satisfy equally, so the test would go green + on the very regression it guards. +3. A concurrency assertion placed on the **coalescing** mock, whose own + `requestInProgress` latch means it can never report more than one call in + flight — so it held whether or not this module excluded anything, and passed + against a read gate that had genuinely lost exclusion. + +The third arrived in the fix for the first two, which is the point: this is not a +mistake you make once. + +So: assert what each flag set **did** get, not only what it lacks, and put those +assertions in the helper both orderings run through, or the reverse order keeps +the blind spot. The identity set is checked on `creationTimeMs` because that is +the field it exists to carry. Keep both that check and the flags-array check — +they catch **different** failures and neither is redundant. The flags array +catches a read served another flag set's rows (the coalescing bug); the +positional `creationTimeMs` check catches field shaping — identity dropping +`CreationTime` from its flags, or `toIdentityRow` failing to forward it — which +no flags assertion would notice. + +And pick the mock to match the claim. The coalescing mock models the npm +wrapper's queue semantics and is the only place to assert those. Concurrency has +to be measured against the bare-addon mock, which has no queue and so makes +re-entry observable. + +With no native binding there is only one scan to run and it is the 1.4 s +PowerShell one, so the identity view rides the detailed snapshot — projected +through `toIdentityRow`, so an identity row carries no command line on any host. + +### Which callers need which + +| caller | reads | flag set | +| --------------------------------------------- | ------------------ | -------- | +| `windows-agent-foreground-process.ts` | `command` (agent recognition) | detailed | +| `local-workspace-platform-port-scanner.ts` | `command` (port attribution) | detailed | +| `codex-structured-turn-processes.ts` | `command` (turn-process identity) | detailed | +| `structured-tui-process-identity.ts` | `command` (child match) | detailed | +| `windows-pty-root-identity.ts` | `pid` / `ppid` only | identity | +| `agent-session-process-identity-probe.ts` | `creationTimeMs` only | identity | +| `relay/windows-port-scan.ts` | `name` (port owner label) | detailed | + +`windows-port-scan.ts` is the one mismatch in the table: it reads only `pid` and +`name`, which the identity set answers, but it calls the detailed reader. On a +host with a live pane that costs nothing extra — the detailed snapshot is +already cached — and on a headless relay it pays for a command line no caller +reads. Left as-is deliberately, because moving it to identity would trade that +for a second scan whenever a pane is polling; revisit if the relay ever scans +ports without one. + +The per-pane foreground tracker is the hot one (750 ms / 2 s cadence) and it +genuinely needs the command line, so the repeating per-process `OpenProcess` is +not something the split removes. What the split removes is that handle from +teardown identity and from the owner probe, which now open nothing. + +### `creationTimeMs` does not exist on any shipped build + +Nothing in the repo supplies a `CreationTime` flag. The package enum is +`None`/`Memory`/`CommandLine`, `process_worker.cc` emits no `creationTimeMs`, +the vendored patch adds none, and `adaptAddon`'s `PROCESS_DATA_FLAG` lacks the +bit. So `creationTimeMs` is always `undefined` in production and +`isWindowsProcessStartTimeAvailable()` is always `false` — a latent product gap +that predates the split and needs its own owner. + +Two consequences. `IDENTITY_PROJECTION.flags` evaluates to `0` today, so the +identity reader really does open zero handles. And +`agent-session-process-identity-probe.ts` early-returns on +`isWindowsProcessStartTimeAvailable()` rather than scanning the whole table to +produce `null`. Do not build anything on Windows start time working. Those CIM numbers are from a 1050-process host. The scan scales with process count: on a 1486-process Windows SSH host it measured **1.36 s** and produced @@ -345,9 +500,10 @@ ownership, and CPU accounting in the memory collector — still reads it through its own query. Those callers are not migrated. Committed private bytes have no equivalent either, and the one memory value the -snapshot _can_ carry is unusable for the sizes Orca now sees: `process.cc` stores -`pmc.WorkingSetSize` into a `DWORD`, so anything above 4 GB wraps. That is the -second reason `windows-process-resource-collector.ts` still runs its own +addon can produce is unusable for the sizes Orca now sees: `process.cc` stores +`pmc.WorkingSetSize` into a `DWORD`, so anything above 4 GB wraps — which is why +neither flag set asks for it. That is the second reason +`windows-process-resource-collector.ts` still runs its own `Get-CimInstance` sweep — it needs `PageFileUsage` (commit) and the CPU-time counters in the same pass. Migrating it to the native table would cost both, and it is why this module no longer sets the `Memory` flag at all: the field had no diff --git a/src/main/providers/windows-foreground-process-rows.test.ts b/src/main/providers/windows-foreground-process-rows.test.ts index 924c81789ce..42330dd6fe8 100644 --- a/src/main/providers/windows-foreground-process-rows.test.ts +++ b/src/main/providers/windows-foreground-process-rows.test.ts @@ -12,9 +12,13 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const getAllProcessesMock = vi.fn() -import { __setWindowsProcessTreeLoaderForTests } from '../windows/windows-process-table' +import { + __setWindowsProcessTreeLoaderForTests, + readWindowsProcessIdentityTable +} from '../windows/windows-process-table' import { queryWindowsProcessDescendants, + queryWindowsProcessLinksFresh, queryWindowsProcessRowsFresh, resetWindowsProcessRowsSnapshotForTests } from './windows-foreground-process-rows' @@ -117,4 +121,25 @@ describe('windows process rows', () => { expect(scanCount()).toBe(2) }) + + it('never answers the ancestry links from the identity TTL cache either', async () => { + // The identity table is a second reader with its own TTL, so the freshness + // the ancestry walk depends on has to be pinned on its own. + await readWindowsProcessIdentityTable() + getAllProcessesMock.mockImplementation((cb: (rows: unknown) => void) => { + cb(withSelf([{ pid: 300, ppid: 100, name: 'node.exe' }])) + }) + // Proves the cache the fresh read below ignores is live, not merely expired. + expect((await readWindowsProcessIdentityTable()).map((row) => row.pid)).toEqual([ + process.pid, + 100, + 200 + ]) + expect(scanCount()).toBe(1) + + const links = await queryWindowsProcessLinksFresh() + + expect(scanCount()).toBe(2) + expect(links.map((row) => row.pid)).toEqual([process.pid, 300]) + }) }) diff --git a/src/main/providers/windows-foreground-process-rows.ts b/src/main/providers/windows-foreground-process-rows.ts index e8320a6d00a..16f01d5fbe7 100644 --- a/src/main/providers/windows-foreground-process-rows.ts +++ b/src/main/providers/windows-foreground-process-rows.ts @@ -1,8 +1,10 @@ import { collectDescendantsFromIndex, getProcessTableIndex } from '../../shared/process-table-index' import { + readWindowsProcessIdentityTableFresh, readWindowsProcessTable, readWindowsProcessTableFresh, resetWindowsProcessTableForTests, + type WindowsProcessIdentityRow, type WindowsProcessRow as NativeWindowsProcessRow } from '../windows/windows-process-table' @@ -62,6 +64,16 @@ export async function queryWindowsProcessRowsFresh(): Promise { + return readWindowsProcessIdentityTableFresh() +} + export async function queryWindowsProcessDescendants( rootPid: number, options: { fresh?: boolean } = {} diff --git a/src/main/runtime/agent-session-process-identity-probe.ts b/src/main/runtime/agent-session-process-identity-probe.ts index 51577048d31..5d441782ca8 100644 --- a/src/main/runtime/agent-session-process-identity-probe.ts +++ b/src/main/runtime/agent-session-process-identity-probe.ts @@ -15,7 +15,10 @@ import type { } from '../../shared/agent-session-lease-adjudication' import type { AgentSessionProcessIdentity } from '../../shared/agent-session-record' import { runProcess } from '../../shared/child-process/run-process' -import { readWindowsProcessTableFresh } from '../windows/windows-process-table' +import { + isWindowsProcessStartTimeAvailable, + readWindowsProcessIdentityTableFresh +} from '../windows/windows-process-table' /** Start times drift by scheduler granularity and clock reads; compare with a tolerance. */ export const PROCESS_START_TIME_TOLERANCE_MS = 2_000 @@ -111,8 +114,17 @@ async function readDarwinProcessStartTimesMs( } async function readWindowsProcessStartTimeMs(pid: number): Promise { + // No shipped addon build exposes the creation-time flag, so without this the + // whole table gets scanned to produce `null` every time. + if (!isWindowsProcessStartTimeAvailable()) { + return null + } try { - const row = (await readWindowsProcessTableFresh()).find((candidate) => candidate.pid === pid) + // Identity flag set: only the creation time is read, so no command line is + // worth an `OpenProcess` per process here. + const row = (await readWindowsProcessIdentityTableFresh()).find( + (candidate) => candidate.pid === pid + ) return row?.creationTimeMs ?? null } catch { return null diff --git a/src/main/windows-pty-root-identity.ts b/src/main/windows-pty-root-identity.ts index c99224cb72a..28c632682d8 100644 --- a/src/main/windows-pty-root-identity.ts +++ b/src/main/windows-pty-root-identity.ts @@ -1,4 +1,4 @@ -import { queryWindowsProcessRowsFresh } from './providers/windows-foreground-process-rows' +import { queryWindowsProcessLinksFresh } from './providers/windows-foreground-process-rows' import { readOrcaChromiumProcessPids } from './orca-chromium-process-pids' /** @@ -138,7 +138,7 @@ export async function verifyWindowsTreeKillTarget( return 'unknown' } const rows = await readLinksBeforeDeadline( - deps.readRows ?? queryWindowsProcessRowsFresh, + deps.readRows ?? queryWindowsProcessLinksFresh, deps.timeoutMs ?? WINDOWS_ROOT_IDENTITY_TIMEOUT_MS ) if (!rows) { diff --git a/src/main/windows/windows-process-table-cim-scan.ts b/src/main/windows/windows-process-table-cim-scan.ts index 213f157f63b..b8d654ce238 100644 --- a/src/main/windows/windows-process-table-cim-scan.ts +++ b/src/main/windows/windows-process-table-cim-scan.ts @@ -75,6 +75,8 @@ export function parseWindowsCimProcessRows(stdout: string): WindowsProcessRow[] return [] } const name = fieldAsString(row.Name) + // No working set: Win32_Process reports one, but nothing reads memory off + // this table and asking widens an already costly scan. return [{ pid, ppid, name, command: fieldAsString(row.CommandLine) || name }] }) } diff --git a/src/main/windows/windows-process-table.test.ts b/src/main/windows/windows-process-table.test.ts index 96da6fcb4ef..bb5eda24385 100644 --- a/src/main/windows/windows-process-table.test.ts +++ b/src/main/windows/windows-process-table.test.ts @@ -8,12 +8,20 @@ import { __setWindowsProcessTreeRequireForTests, isWindowsProcessTableAvailable, isWindowsProcessStartTimeAvailable, + readWindowsProcessIdentityTable, + readWindowsProcessIdentityTableFresh, readWindowsProcessTable, readWindowsProcessTableFresh, - resetWindowsProcessTableForTests + resetWindowsProcessTableForTests, + type WindowsProcessIdentityRow, + type WindowsProcessRow } from './windows-process-table' import { resetWindowsCommandLineRecoveryHealthForTests } from './windows-command-line-recovery-health' +/** None | CreationTime, and CommandLine on top of it. Memory (1) is never asked for. */ +const IDENTITY_FLAGS = 4 +const DETAILED_FLAGS = 6 + const getAllProcesses = vi.fn() // A real snapshot always contains the querying process; the reader rejects a @@ -21,23 +29,115 @@ const getAllProcesses = vi.fn() // returns -- an empty list rather than an error. It also always carries our own // command line, since a process can always open itself -- an empty one there is // the host-wide-refusal signal, not a fixture detail. -const SELF = { pid: process.pid, ppid: 0, name: 'vitest.exe', commandLine: 'vitest.exe --run' } -const NATIVE = [ +type NativeRow = { + pid: number + ppid: number + name: string + commandLine?: string + creationTimeMs?: number +} + +const SELF: NativeRow = { pid: process.pid, ppid: 0, name: 'vitest.exe', commandLine: 'vitest.exe --run' } +const NATIVE: NativeRow[] = [ SELF, { pid: 100, ppid: 4, name: 'orca.exe', commandLine: '"C:/a b/orca.exe" --x', - memory: 4096, creationTimeMs: 1_700_000_000_000 } ] +/** + * The vendored wrapper, faithfully: one `requestInProgress` latch over a shared + * callback queue, resolved asynchronously. A second caller that arrives while a + * request is in flight has its `flags` DISCARDED and is served the first + * caller's rows -- the defect this module's read gate has to exclude. A + * synchronous mock cannot express it, because nothing ever overlaps. + */ +let coalescingCalls: { flags: number }[] = [] +let maxConcurrentNativeCalls = 0 + +function coalescingModule(): { + ProcessDataFlag: { None: number; Memory: number; CommandLine: number; CreationTime: number } + getAllProcesses: (cb: (rows: NativeRow[] | undefined) => void, flags?: number) => void +} { + let requestInProgress = false + const queue: ((rows: NativeRow[]) => void)[] = [] + return { + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2, CreationTime: 4 }, + getAllProcesses: (cb, flags) => { + queue.push(cb) + if (requestInProgress) { + return + } + requestInProgress = true + coalescingCalls.push({ flags: flags ?? 0 }) + // The rows the addon would produce for exactly these flags. Each field is + // gated on its OWN bit: reusing the CommandLine bit for both would strip + // creationTimeMs from an identity read that did request CreationTime, and + // no case could then tell a served-someone-else's-rows bug from a + // correctly-shaped cheap read. + const requested = flags ?? 0 + const rows: NativeRow[] = NATIVE.map((row) => ({ + pid: row.pid, + ppid: row.ppid, + name: row.name, + ...(requested & 2 && row.commandLine !== undefined ? { commandLine: row.commandLine } : {}), + ...(requested & 4 && row.creationTimeMs !== undefined + ? { creationTimeMs: row.creationTimeMs } + : {}) + })) + setTimeout(() => { + while (queue.length) { + queue.splice(0).forEach((callback) => callback(rows)) + } + requestInProgress = false + }, 0) + } + } +} + +/** One instance for the whole test: the latch it models is module-global. */ +function installCoalescingModule(): void { + const native = coalescingModule() + __setWindowsProcessTreeLoaderForTests(() => native) +} + +/** + * The relay's bare addon: `adaptAddon` over `getProcessList`, with no queue of + * any kind. Two simultaneous `CreateToolhelp32Snapshot` calls are the crash the + * vendor's queue exists to prevent, so here re-entry is observable rather than + * silently absorbed. + * + * Concurrency has to be measured against this and never against the coalescing + * mock, whose own latch means it can only ever report one call in flight -- an + * assertion that holds whether or not this module excludes anything. + */ +function installBareAddonModule(): void { + let inFlight = 0 + const native = { + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2, CreationTime: 4 }, + getAllProcesses: (cb: (rows: NativeRow[] | undefined) => void, flags?: number) => { + coalescingCalls.push({ flags: flags ?? 0 }) + inFlight += 1 + maxConcurrentNativeCalls = Math.max(maxConcurrentNativeCalls, inFlight) + setTimeout(() => { + inFlight -= 1 + cb(NATIVE) + }, 0) + } + } + __setWindowsProcessTreeLoaderForTests(() => native) +} + describe('windows process table', () => { let platform: PropertyDescriptor | undefined beforeEach(() => { + coalescingCalls = [] + maxConcurrentNativeCalls = 0 getAllProcesses.mockReset() getAllProcesses.mockImplementation((cb: (rows: unknown) => void) => cb(NATIVE)) platform = Object.getOwnPropertyDescriptor(process, 'platform') @@ -69,13 +169,157 @@ describe('windows process table', () => { ]) }) - it('requests the command line and creation time, never memory', async () => { + it('asks for the command line but never for memory', async () => { + // Memory costs a second OpenProcess(PROCESS_VM_READ) per process and no + // caller reads a working set off this table. await readWindowsProcessTableFresh() - // CommandLine (2) | CreationTime (4). The Memory bit (1) stays clear: the - // addon opens a second PROCESS_VM_READ handle per process to serve it and - // nothing reads a working set off this table. - expect(getAllProcesses.mock.calls[0]?.[1]).toBe(6) - expect((getAllProcesses.mock.calls[0]?.[1] as number) & 1).toBe(0) + expect(getAllProcesses.mock.calls[0]?.[1]).toBe(DETAILED_FLAGS) + }) + + it('reads the identity table with no per-process handle flag at all', async () => { + await readWindowsProcessIdentityTableFresh() + expect(getAllProcesses.mock.calls[0]?.[1]).toBe(IDENTITY_FLAGS) + }) + + it('drops the command line from identity rows rather than leaving it empty', async () => { + const rows = await readWindowsProcessIdentityTableFresh() + expect(rows).toEqual([ + { pid: process.pid, ppid: 0, name: 'vitest.exe' }, + { pid: 100, ppid: 4, name: 'orca.exe', creationTimeMs: 1_700_000_000_000 } + ]) + expect(rows.every((row) => !('command' in row))).toBe(true) + }) + + it('collapses a 32-wide burst into one scan per flag set', async () => { + installCoalescingModule() + const [identity, detailed] = await Promise.all([ + Promise.all(Array.from({ length: 16 }, () => readWindowsProcessIdentityTable())), + Promise.all(Array.from({ length: 16 }, () => readWindowsProcessTable())) + ]) + expect(coalescingCalls.map((call) => call.flags).sort()).toEqual([ + IDENTITY_FLAGS, + DETAILED_FLAGS + ]) + expect(identity).toHaveLength(16) + expect(detailed).toHaveLength(16) + }) + + // The npm wrapper coalesces rather than queues: a second concurrent caller's + // flags are discarded and it is served the first caller's rows. Overlapping an + // identity read with a detailed one therefore used to hand agent recognition a + // table with every command line empty. + async function expectEachViewGotItsOwnFlags( + identity: Promise, + detailed: Promise + ): Promise { + const [identityRows, detailedRows] = await Promise.all([identity, detailed]) + expect(detailedRows.some((row) => row.command === '"C:/a b/orca.exe" --x')).toBe(true) + expect(identityRows.every((row) => !('command' in row))).toBe(true) + // Both sets carry what their own flags asked for. Not redundant with the + // flags check below: that one catches a read served the OTHER set's rows, + // this one catches field shaping -- identity dropping CreationTime from its + // flags, or toIdentityRow failing to forward it. Neither sees the other's + // failure, so keep both. + expect(identityRows.map((row) => row.creationTimeMs)).toEqual([undefined, 1_700_000_000_000]) + expect(detailedRows.map((row) => row.creationTimeMs)).toEqual([undefined, 1_700_000_000_000]) + // Two calls, each with its own flags. Concurrency is asserted separately, + // against the bare addon: this mock's own latch means it could never report + // more than one call in flight, whatever this module did. + expect(coalescingCalls.map((call) => call.flags).sort()).toEqual([ + IDENTITY_FLAGS, + DETAILED_FLAGS + ]) + } + + it('gives each flag set its own data when the identity read is issued first', async () => { + installCoalescingModule() + const identity = readWindowsProcessIdentityTableFresh() + const detailed = readWindowsProcessTableFresh() + await expectEachViewGotItsOwnFlags(identity, detailed) + }) + + it('gives each flag set its own data when the detailed read is issued first', async () => { + installCoalescingModule() + const detailed = readWindowsProcessTableFresh() + const identity = readWindowsProcessIdentityTableFresh() + await expectEachViewGotItsOwnFlags(identity, detailed) + }) + + /** Microtasks only: the mocks call back on a timer, so nothing completes. */ + async function parkPendingReadsOnTheGate(): Promise { + for (let tick = 0; tick < 20; tick += 1) { + await Promise.resolve() + } + } + + it('never re-enters the bare relay addon when both flag sets overlap', async () => { + installBareAddonModule() + const detailed = readWindowsProcessTableFresh() + const identity = readWindowsProcessIdentityTableFresh() + await Promise.all([detailed, identity]) + expect(coalescingCalls.map((call) => call.flags).sort()).toEqual([ + IDENTITY_FLAGS, + DETAILED_FLAGS + ]) + expect(maxConcurrentNativeCalls).toBe(1) + }) + + it('keeps one read in flight across a test reset', async () => { + // Replacing the gate rather than chaining onto it lets a waiter still + // holding the old chain run beside a read queued on the new one. Reachable + // only from the test hooks -- which is the problem: it hands a suite two + // concurrent calls into its own mock, the exact condition the cases above + // exist to detect. + installBareAddonModule() + const inFlight = readWindowsProcessTableFresh() + const waiter = readWindowsProcessIdentityTableFresh() + await parkPendingReadsOnTheGate() + resetWindowsProcessTableForTests() + const afterReset = readWindowsProcessTableFresh() + + await Promise.allSettled([inFlight, waiter, afterReset]) + expect(maxConcurrentNativeCalls).toBe(1) + }) + + it('does not serve one flag set from the other cache', async () => { + await readWindowsProcessTable() + await readWindowsProcessIdentityTable() + expect(getAllProcesses).toHaveBeenCalledTimes(2) + }) + + it('rejects an empty identity snapshot rather than reporting an idle machine', async () => { + getAllProcesses.mockImplementation((cb: (rows: unknown) => void) => cb([])) + resetWindowsProcessTableForTests() + await expect(readWindowsProcessIdentityTableFresh()).rejects.toThrow(/unreadable/) + }) + + it('applies the deadline to the identity read too', async () => { + vi.useFakeTimers() + getAllProcesses.mockImplementation(() => {}) + resetWindowsProcessTableForTests() + const pending = readWindowsProcessIdentityTableFresh() + const assertion = expect(pending).rejects.toThrow(/timed out/) + await vi.advanceTimersByTimeAsync(3_000) + await assertion + vi.useRealTimers() + }) + + it('shares the wedge gate across flag sets, because they share one addon', async () => { + // One wedged read latches the vendored `requestInProgress` and pins the one + // libuv slot whichever flags asked for it, so a per-flag-set gate would let + // the other reader keep parking callbacks behind it. + vi.useFakeTimers() + getAllProcesses.mockImplementation(() => {}) + resetWindowsProcessTableForTests() + const wedge = readWindowsProcessIdentityTableFresh() + const wedgeAssertion = expect(wedge).rejects.toThrow(/timed out/) + await vi.advanceTimersByTimeAsync(3_000) + await wedgeAssertion + + await expect(readWindowsProcessTableFresh()).rejects.toThrow(/wedged/) + await expect(readWindowsProcessIdentityTableFresh()).rejects.toThrow(/wedged/) + expect(getAllProcesses).toHaveBeenCalledTimes(1) + vi.useRealTimers() }) it('only advertises PID-safe ownership when the native creation-time field exists', () => { @@ -173,6 +417,27 @@ describe('PowerShell fallback when the native binding is absent', () => { expect(cimScan).toHaveBeenCalledTimes(1) }) + it('serves the identity view from the one scan a relay can afford', async () => { + // With no binding there is only one scan to run and it costs ~1.4s and a + // powershell.exe, so the cheap view must ride it rather than fork a second. + __setWindowsProcessTreeLoaderForTests(() => null) + // Projected, not merely widened: an identity row carries no command line on + // any host, so nothing can come to depend on the fallback happening to have + // one. + await expect(readWindowsProcessIdentityTableFresh()).resolves.toEqual([ + { pid: process.pid, ppid: 0, name: 'node.exe' }, + { pid: 200, ppid: process.pid, name: 'claude.exe' } + ]) + await readWindowsProcessTable() + expect(cimScan).toHaveBeenCalledTimes(1) + }) + + it('rejects an identity read that omits our own pid', async () => { + __setWindowsProcessTreeLoaderForTests(() => null) + cimScan.mockResolvedValue([{ pid: 200, ppid: 4, name: 'claude.exe', command: 'claude' }]) + await expect(readWindowsProcessIdentityTableFresh()).rejects.toThrow(/unreadable/) + }) + it('does not engage when the native binding is present', async () => { const getAllProcesses = vi.fn() getAllProcesses.mockImplementation((cb: (rows: unknown) => void) => cb(NATIVE)) @@ -425,7 +690,7 @@ describe('resolving the native reader', () => { expect(isWindowsProcessTableAvailable()).toBe(true) }) - it('asks the addon for the command line but not memory, as the package path does', async () => { + it('asks the addon for the command line, as the package path does', async () => { const addon = addonReturning(NATIVE) __setWindowsProcessTreeRequireForTests((specifier: string) => { if (specifier === ADDON_SPECIFIER) { @@ -434,12 +699,26 @@ describe('resolving the native reader', () => { throw new Error('MODULE_NOT_FOUND') }) await readWindowsProcessTableFresh() - // CommandLine only: a bare snapshot would silently drop the command line - // every agent-recognition caller matches on first, and the relay addon - // exposes no CreationTime bit to add. + // CommandLine alone: a bare snapshot would silently drop the command line + // every agent-recognition caller matches on first, and Memory would add a + // second per-process handle nothing reads. expect(addon.getProcessList).toHaveBeenCalledWith(expect.any(Function), 2) }) + it('asks the addon for nothing per-process on the identity path', async () => { + const addon = addonReturning(NATIVE) + __setWindowsProcessTreeRequireForTests((specifier: string) => { + if (specifier === ADDON_SPECIFIER) { + return addon + } + throw new Error('MODULE_NOT_FOUND') + }) + await readWindowsProcessIdentityTableFresh() + // The relay addon exposes no CreationTime bit, so this is a bare Toolhelp32 + // walk: zero OpenProcess calls. + expect(addon.getProcessList).toHaveBeenCalledWith(expect.any(Function), 0) + }) + it('reaches the CIM scan when neither the package nor the addon is present', async () => { const cimScan = vi .fn() diff --git a/src/main/windows/windows-process-table.ts b/src/main/windows/windows-process-table.ts index 6683770435d..0a1acd7ae1c 100644 --- a/src/main/windows/windows-process-table.ts +++ b/src/main/windows/windows-process-table.ts @@ -21,30 +21,41 @@ import { readWindowsProcessRowsWithCim } from './windows-process-table-cim-scan' * A Toolhelp32 snapshot answers the same question in ~16 ms with no child * process at all, so none of those failure modes have anywhere to live. * - * Measured on Windows 11 (1050 processes), p50 / p95: - * pid+ppid+name 15.9 / 17.5 ms - * +memory +commandLine 30.6 / 33.7 ms - * PowerShell CIM 706 / 723 ms + * Two flag sets, because only some callers need a command line, and exactly one + * native read in flight at a time, because the vendored wrapper coalesces + * differing flags -- see docs/reference/windows-process-enumeration.md. * - * Those are the module's published figures for both extra fields together; the - * only flag set this module asks for is `CommandLine` (+ `CreationTime`, free), - * which sits between the two rows and has not been separately measured. + * Measured on Windows 11 (492 processes), p50 / p95: + * identity pid+ppid+name 6.3 / 7.0 ms 0 OpenProcess + * detailed +commandLine 12.3 / 13.4 ms 1 OpenProcess/process + * (retired) +memory +commandLine 13.1 / 14.1 ms 2 OpenProcess/process + * PowerShell CIM 706 / 723 ms * - * Both Toolhelp32 rows assume the optional `windows-process-tree.node` addon. + * Dropping Memory removed the second per-process handle: it took an + * OpenProcess(...|VM_READ) it never read through. CommandLine's own read is no + * longer a PEB walk either -- the patched addon asks the kernel, so identity is + * now the only flag set that opens nothing at all. + * + * All Toolhelp32 rows assume the optional `windows-process-tree.node` addon. * The desktop bundles it; no released relay carries it, so on an SSH host the * CIM row is the operative number and the child process is not avoided at all. */ -export type WindowsProcessRow = { +/** Everything a Toolhelp32 walk alone can answer. */ +export type WindowsProcessIdentityRow = { pid: number ppid: number name: string - /** Full command line. Empty when the process denied a query handle. */ - command: string /** Process creation time in Unix milliseconds, when the native snapshot provides it. */ creationTimeMs?: number } +/** Adds the kernel-supplied command line. Only ask for this if you read it. */ +export type WindowsProcessRow = WindowsProcessIdentityRow & { + /** Full command line. Empty when the process denied a query handle. */ + command: string +} + type NativeProcessInfo = { pid: number ppid: number @@ -82,9 +93,11 @@ let requireNative: NativeRequire = requireFromMain * * The published package's `lib/index.js` adds only a queue over this call, and * that queue is the wedge this module already defends against: it latches a - * module-global `requestInProgress` with no try/catch. We hold our own - * single-flight and deadline, so binding straight to the addon drops the - * duplicate queue rather than nesting inside it. + * module-global `requestInProgress` with no try/catch. `nativeReadGate` holds + * the mutual exclusion instead -- and must, because this addon has no queue of + * its own and two simultaneous `CreateToolhelp32Snapshot` calls are the crash + * the vendor's queue exists to prevent. With one native call ever outstanding, + * binding straight to the addon drops a duplicate rather than losing a guard. */ type WindowsProcessTreeAddon = { getProcessList: ( @@ -95,7 +108,8 @@ type WindowsProcessTreeAddon = { /** * Mirrors the package's enum; the addon takes the raw bit field. `Memory` (1) - * is listed for completeness and is deliberately never set — see `flags` below. + * is listed for completeness and is deliberately never set — see the projections + * below. */ const PROCESS_DATA_FLAG = { None: 0, Memory: 1, CommandLine: 2 } as const @@ -221,26 +235,122 @@ const WINDOWS_PROCESS_QUERY_TIMEOUT_MS = 3_000 * Reads that missed their deadline and have not called back yet. * Refusing re-entry bounds both vendored callbacks and relay addon workers to * one; read ids keep a late callback from clearing a newer wedge. + * + * One gate for both flag sets, not one each: they call the same addon, so a + * wedged read latches the one `requestInProgress` and pins the one libuv slot + * whichever flags asked for it. Retention stays at exactly one callback rather + * than one per reader because `nativeReadGate` below already admits only one + * native call at a time; read ids are module-global and monotonic, so a late + * callback can only clear its own wedge. */ const unreturnedReads = new Set() let readSequence = 0 let nativeReaderEpoch = 0 +/** + * Admits one native read at a time, across both flag sets. Nothing else does. + * + * The npm wrapper coalesces rather than queues: `getRawProcessList` pushes the + * callback onto one list and only calls the addon when no request is in + * progress, so a second concurrent caller's `flags` are DISCARDED and it is + * handed the first caller's rows. An identity read racing a detailed read + * therefore returns a table with every command line EMPTY, which agent + * recognition reads as "no agent" -- silently, and only under concurrency. + * Measured against the real addon: identity issued first, both callers got the + * same array, 0 of 541 rows with a command line. + * + * Nothing above stops that. Each cache single-flights only within itself + * (`inFlight` is a closure per reader) and the wedge set latches only after a + * read misses its 3s deadline, so through the healthy ~12ms of a scan neither + * excludes the other. Overlap is the normal state, not an edge case: panes poll + * detailed every 750ms while a teardown takes identity snapshots. + * + * It also has to be here for the relay's bare addon, which has no queue at all: + * two simultaneous `CreateToolhelp32Snapshot` calls are the crash the vendor's + * queue exists to prevent. + * + * Every link settles -- a wedged read still rejects on its deadline -- so a + * waiter is never stranded; it re-checks the wedge and rejects instead. + */ +let nativeReadGate: Promise = Promise.resolve() + function resetNativeReaderState(): void { nativeReaderEpoch += 1 unreturnedReads.clear() + // Chain, never replace. Dropping the old chain lets a waiter still holding it + // run against a read queued on the new one -- two concurrent calls into one + // mock addon, which is precisely the coalescing these suites exist to catch. + // Every link settles within the deadline, so the wait this costs is bounded. + nativeReadGate = nativeReadGate.then(ignoreSettlement, ignoreSettlement) } -function readNativeRows(): Promise { +/** A flag set and the row shape it can honestly produce. */ +type ProcessRowProjection = { + flags: (native: WindowsProcessTreeModule) => number + fromNative: (row: NativeProcessInfo) => Row + /** + * The no-binding scan, on the one flag set it can serve. Absent on the other, + * because a relay must never run two `Get-CimInstance` scans at ~1.4s each -- + * `readWindowsProcessIdentityTable` projects the detailed snapshot instead. + */ + cimFallback?: () => Promise +} + +function toIdentityRow(row: { + pid: number + ppid: number + name: string + creationTimeMs?: number +}): WindowsProcessIdentityRow { + return { + pid: row.pid, + ppid: row.ppid, + name: row.name, + ...(typeof row.creationTimeMs === 'number' ? { creationTimeMs: row.creationTimeMs } : {}) + } +} + +/** + * Toolhelp32 and nothing else: no `OpenProcess` per process, so this read has + * none of the shape an EDR scores as walking another process's memory. + */ +const IDENTITY_PROJECTION: ProcessRowProjection = { + flags: (native) => native.ProcessDataFlag.None | (native.ProcessDataFlag.CreationTime ?? 0), + fromNative: toIdentityRow +} + +/** + * Adds, per process, one `OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION)` and an + * `NtQueryInformationProcess(ProcessCommandLineInformation)` -- which is what + * agent recognition and port attribution match on. `Memory` is deliberately + * absent: it took a second handle carrying `PROCESS_VM_READ` and then never read + * through it, and no caller reads a working set off this table (the Resource + * Manager runs its own sweep, and the native field wraps above 4 GB anyway). + */ +const DETAILED_PROJECTION: ProcessRowProjection = { + flags: (native) => IDENTITY_PROJECTION.flags(native) | native.ProcessDataFlag.CommandLine, + fromNative: (row) => ({ ...toIdentityRow(row), command: row.commandLine ?? '' }), + cimFallback: readCimRows +} + +function ignoreSettlement(): void {} + +function readNativeRows(projection: ProcessRowProjection): Promise { + const attempt = nativeReadGate.then(() => readOneSnapshot(projection)) + nativeReadGate = attempt.then(ignoreSettlement, ignoreSettlement) + return attempt +} + +function readOneSnapshot(projection: ProcessRowProjection): Promise { const native = moduleLoader() if (!native) { - if (process.platform === 'win32') { + if (process.platform === 'win32' && projection.cimFallback) { // Why only when the module is absent: a binding that loads is the fast // path even when a read fails or wedges, so a failing native reader must // never silently start forking shells at the caller's poll rate. Absence // is the one condition that can never resolve itself — see // docs/reference/windows-process-enumeration.md. - return readCimRows() + return projection.cimFallback() } // Reject rather than resolve empty: an empty table is a claim that nothing // is running, and callers act on that by force-killing or by declaring a @@ -254,16 +364,7 @@ function readNativeRows(): Promise { } const readId = ++readSequence const readerEpoch = nativeReaderEpoch - // Why CommandLine but not Memory: each flag costs one OpenProcess per process - // inside the addon (CommandLine's is a kernel query, not a memory read), and - // every caller of this table matches on `command`, while nothing reads a - // working set off it -- the Resource Manager runs its own CIM sweep because it - // needs commit and CPU time in one pass, and `process.cc` truncates the working - // set into a DWORD anyway. Dropping Memory halves the per-snapshot handle - // count; the remaining flags stay in ONE flag set because every read shares one - // snapshot, so a 32-wide teardown collapses into a single scan. Splitting the - // cache per field set would restore exactly the fan-out it exists to prevent. - const flags = native.ProcessDataFlag.CommandLine | (native.ProcessDataFlag.CreationTime ?? 0) + const flags = projection.flags(native) return new Promise((resolve, reject) => { // Hoisted so a synchronous throw from getAllProcesses can clear it. An // orphaned timer would otherwise fire later and wedge a reader that had @@ -303,17 +404,7 @@ function readNativeRows(): Promise { if ((flags & native.ProcessDataFlag.CommandLine) !== 0) { reportWindowsCommandLineRecoveryHealth(processes) } - resolve( - processes.map((row) => ({ - pid: row.pid, - ppid: row.ppid, - name: row.name, - command: row.commandLine ?? '', - ...(typeof row.creationTimeMs === 'number' - ? { creationTimeMs: row.creationTimeMs } - : {}) - })) - ) + resolve(processes.map(projection.fromNative)) }, flags) } catch (error) { clearTimeout(deadline) @@ -340,14 +431,38 @@ async function readCimRows(): Promise { // Why still cache: the snapshot is cheap but not free, and a worktree delete // tears down PTYs 32-wide. The shared TTL + single-in-flight reader collapses // that burst into one scan, exactly as the PowerShell path had to. -const snapshotReader = createProcessTableSnapshotReader({ - runPs: readNativeRows, +// +// Why two caches are safe where N would not be: the fan-out this prevents is +// one scan per *caller*, and each reader below still serves every caller that +// wants its flag set, so a 32-wide teardown collapses into one scan per flag +// set. Two is the number of distinct native calls that exist -- a third cache +// would need a third flag set, never a third caller. +const identityReader = createProcessTableSnapshotReader({ + runPs: () => readNativeRows(IDENTITY_PROJECTION), + now: () => Date.now() +}) +const detailedReader = createProcessTableSnapshotReader({ + runPs: () => readNativeRows(DETAILED_PROJECTION), now: () => Date.now() }) -/** Cached snapshot, refreshed on the shared TTL. */ +/** + * With no binding there is only one scan to run and it is the expensive one, so + * the identity view rides the detailed snapshot rather than forking a second + * `powershell.exe` at ~1.4 s a scan. Projected, not merely widened: an identity + * row must not carry a command line on any host. + */ +async function readIdentityRows(fresh: boolean): Promise { + if (moduleLoader() === null) { + const rows = await (fresh ? detailedReader.getFreshSnapshot() : detailedReader.getSnapshot()) + return rows.map(toIdentityRow) + } + return fresh ? identityReader.getFreshSnapshot() : identityReader.getSnapshot() +} + +/** Cached command-line snapshot, refreshed on the shared TTL. */ export function readWindowsProcessTable(): Promise { - return snapshotReader.getSnapshot() + return detailedReader.getSnapshot() } /** @@ -357,7 +472,17 @@ export function readWindowsProcessTable(): Promise { * the very process exit it is being asked about. */ export function readWindowsProcessTableFresh(): Promise { - return snapshotReader.getFreshSnapshot() + return detailedReader.getFreshSnapshot() +} + +/** Cached pid/ppid/name snapshot. Prefer this whenever no command line is read. */ +export function readWindowsProcessIdentityTable(): Promise { + return readIdentityRows(false) +} + +/** The identity snapshot, from a scan that starts after this call. */ +export function readWindowsProcessIdentityTableFresh(): Promise { + return readIdentityRows(true) } /** Whether the native table can be read at all on this host. */ @@ -376,6 +501,11 @@ export function isWindowsProcessStartTimeAvailable(): boolean { return native !== null && typeof native.ProcessDataFlag.CreationTime === 'number' } +function resetSnapshotReaders(): void { + identityReader.reset() + detailedReader.reset() +} + /** * Test-only: substitute the native module. * @@ -389,7 +519,7 @@ export function __setWindowsProcessTreeLoaderForTests( moduleLoader = loader ?? loadWindowsProcessTree cachedModule = undefined resetNativeReaderState() - snapshotReader.reset() + resetSnapshotReaders() } /** @@ -403,7 +533,7 @@ export function __setWindowsProcessTreeRequireForTests(resolve?: NativeRequire): moduleLoader = loadWindowsProcessTree cachedModule = undefined resetNativeReaderState() - snapshotReader.reset() + resetSnapshotReaders() } /** Test-only: substitute the no-binding PowerShell scan, which spawns a child. */ @@ -411,12 +541,12 @@ export function __setWindowsProcessTableCimScanForTests( scan?: () => Promise ): void { cimScan = scan ?? readWindowsProcessRowsWithCim - snapshotReader.reset() + resetSnapshotReaders() } -/** Test-only: drop the shared snapshot so suites cannot serve each other's rows. */ +/** Test-only: drop the shared snapshots so suites cannot serve each other's rows. */ export function resetWindowsProcessTableForTests(): void { - snapshotReader.reset() + resetSnapshotReaders() cachedModule = undefined resetNativeReaderState() } From 8415d53a0549f53de629302f20e85993eb5b8f9d Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Sat, 5 Sep 2026 21:44:20 -0700 Subject: [PATCH 51/69] fix(release): stop shipping an unsigned elevate.exe on Windows (#18044) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(release): stop shipping an unsigned elevate.exe on Windows The release cut swaps the SignPath-signed elevate.exe into the electron-builder toolset cache so the NSIS rebuild's CopyElevateHelper re-copy becomes a no-op. It searched `\nsis`, a directory no app-builder-lib layout creates, and `-ErrorAction SilentlyContinue` plus `exit 0` turned that miss into a green step — v1.4.193 and v1.4.194 shipped an unsigned UAC elevation helper. Move the lookup into a script that covers the real layouts (`nsis-3.0.4.1/…`, `nsis@/…`, `ELECTRON_BUILDER_NSIS_DIR`), asks app-builder-lib for the authoritative path, and exits non-zero with an ::error:: annotation when it finds nothing. The step stays continue-on-error so the inner-signing chain remains fail-open. * fix(release): make the elevate.exe swap prove it replaced the packed copy Success was "some cached copy was replaced", which a stale release directory carried in by the `electron-builder-win-` prefix restore can satisfy on its own while the bundle the rebuild packs stays unsigned. The app-builder-lib probe returns the exact path CopyElevateHelper will pack, so make that the check and the directory scan the fallback: exit non-zero when the probed copy was not replaced, and annotate a warning when the probe could not run at all, so a green step never quietly means the authoritative check was skipped. Also pin both shebang scripts to LF: `core.autocrlf=true` gives a Windows checkout CRLF, and CRLF plus a shebang breaks vite's transform, so resolve-7za-path.test.mjs currently runs zero tests there. --------- Co-authored-by: Orca Worker --- .github/workflows/release-cut.yml | 31 +- .../scripts/replace-cached-nsis-elevate.mjs | 260 +++++++++++++ .../replace-cached-nsis-elevate.test.mjs | 364 ++++++++++++++++++ 3 files changed, 644 insertions(+), 11 deletions(-) create mode 100644 config/scripts/replace-cached-nsis-elevate.mjs create mode 100644 config/scripts/replace-cached-nsis-elevate.test.mjs diff --git a/.github/workflows/release-cut.yml b/.github/workflows/release-cut.yml index 001eee4e03c..a1b6784be18 100644 --- a/.github/workflows/release-cut.yml +++ b/.github/workflows/release-cut.yml @@ -1653,9 +1653,12 @@ jobs: # no-op. Known quirk: the cache persists across releases via actions/cache, # so later runs may see elevate.exe as already signed and skip staging it — # that is fine (the signature is timestamped) and the evidence gate checks - # elevate.exe in the shipped installer unconditionally. If this ever causes - # trouble, delete this step; the only effect is elevate.exe shipping - # unsigned again, which the evidence gate will flag. + # elevate.exe in the shipped installer unconditionally. + # + # The cache lookup lives in a script because the inline path this step used + # (`\nsis`) matches no app-builder-lib layout, and `SilentlyContinue` + # plus `exit 0` turned that miss into a green step — v1.4.193 and v1.4.194 + # shipped an unsigned elevate.exe that way. A miss now fails the step. - name: Replace cached elevate.exe with the signed copy id: sign-elevate-cache if: matrix.platform == 'win' && github.run_attempt == 1 && steps.restore-signed-inner.outcome == 'success' @@ -1667,20 +1670,26 @@ jobs: Write-Host '::warning::No elevate.exe in win-unpacked resources; nothing to protect from the rebuild clobber.' exit 0 } + # Why this guard stays: windows-signing-rehearsal.yml shares the + # electron-builder-win- cache key with this workflow, so a + # test-certificate elevate.exe must never be staged into a release cache. $signature = Get-AuthenticodeSignature -FilePath $signed $subject = if ($null -eq $signature.SignerCertificate) { '' } else { $signature.SignerCertificate.Subject } if ($signature.Status -ne 'Valid' -or $subject -notlike '*CN=SignPath Foundation*') { Write-Host "::warning::win-unpacked elevate.exe is not SignPath-signed ($($signature.Status), $subject); skipping cache swap." exit 0 } - $cached = @(Get-ChildItem "$env:LOCALAPPDATA\electron-builder\Cache\nsis" -Recurse -Filter elevate.exe -ErrorAction SilentlyContinue) - if ($cached.Count -eq 0) { - Write-Host '::warning::No cached elevate.exe found (electron-builder cache layout changed?); the rebuild will pack the unsigned copy and the evidence gate will flag it.' - exit 0 - } - foreach ($file in $cached) { - Copy-Item -Path $signed -Destination $file.FullName -Force - Write-Host "Replaced $($file.FullName) with the SignPath-signed copy." + node config/scripts/replace-cached-nsis-elevate.mjs $signed + if ($LASTEXITCODE -ne 0) { + $message = 'Cached elevate.exe swap found nothing to replace; the rebuilt installer ships an unsigned UAC elevation helper (issue #7785).' + if ($env:GITHUB_STEP_SUMMARY) { + try { + Add-Content -Path $env:GITHUB_STEP_SUMMARY -Value "**Windows elevate.exe cache swap:** FAILED — $message" -ErrorAction Stop + } catch { + Write-Host "::warning::Could not write the elevate.exe swap verdict to the job summary: $_" + } + } + throw $message } - name: Rebuild NSIS installer from signed unpacked app diff --git a/config/scripts/replace-cached-nsis-elevate.mjs b/config/scripts/replace-cached-nsis-elevate.mjs new file mode 100644 index 00000000000..fcd1a7323d4 --- /dev/null +++ b/config/scripts/replace-cached-nsis-elevate.mjs @@ -0,0 +1,260 @@ +#!/usr/bin/env node + +// Why: electron-builder re-runs `CopyElevateHelper.copy` on every NSIS pack, so the +// release rebuild overwrites the SignPath-signed `resources/elevate.exe` with the +// unsigned copy sitting in the electron-builder toolset cache. The release workflow +// swapped the cached copy first, but searched `/nsis` — a directory no current +// app-builder-lib layout creates (real ones are `/nsis-3.0.4.1/nsis-3.0.4.1-/` +// and `/nsis@/nsis-bundle--/`), so the swap silently found +// nothing and v1.4.193/v1.4.194 shipped an unsigned UAC elevation helper. + +import { copyFileSync, readdirSync, statSync } from 'node:fs' +import { createRequire } from 'node:module' +import { homedir, platform as osPlatform, tmpdir } from 'node:os' +import { join, parse, resolve } from 'node:path' + +const require = createRequire(import.meta.url) + +const ELEVATE_EXE = 'elevate.exe' + +// `nsis` (the layout the old hardcoded path assumed), `nsis-3.0.4.1` (legacy bundle via +// `getBinFromUrl`), `nsis@1.2.1` (unified bundle). Not `customNsisBinary`: the +// `nsis-` key `getBinFromCustomLoc` builds is only `getBin`'s in-process promise +// key, and the extract dir is named for the custom URL's parent segment, which need not +// start with `nsis` at all. Only the app-builder-lib probe covers that layout — which is +// why the probe, not this scan, is what decides whether the swap succeeded. +const NSIS_RELEASE_DIR = /^nsis(?:[-@].*)?$/i + +// elevate.exe lives at the bundle root, one level under the release dir. The legacy +// bundle carries thousands of files under Contrib/, so an unbounded walk is both slow +// and a way to match something that is not a toolset copy. +const MAX_DEPTH = 3 + +function isFile(path) { + try { + return statSync(path).isFile() + } catch { + return false + } +} + +/** + * Mirrors `getCacheDirectory` in app-builder-lib's `out/util/electronGet.js`, which is what + * decides where the NSIS bundle is unpacked. Kept as a local port rather than an import + * because the swap must still resolve a cache root when app-builder-lib cannot be loaded. + */ +export function resolveElectronBuilderCacheDir({ + env = process.env, + platform = osPlatform(), + home = homedir(), + temp = tmpdir() +} = {}) { + const override = env.ELECTRON_BUILDER_CACHE?.trim() + if (override && parse(override).root) { + return override + } + if (platform === 'darwin') { + return join(home, 'Library', 'Caches', 'electron-builder') + } + if (platform === 'win32') { + const localAppData = env.LOCALAPPDATA?.trim() + // https://github.com/electron-userland/electron-builder/issues/1164 + const isSystemUser = + localAppData?.toLowerCase().includes('\\windows\\system32\\') === true || + env.USERNAME?.trim().toLowerCase() === 'system' + if (!localAppData || isSystemUser) { + return join(temp, 'electron-builder-cache') + } + return join(localAppData, 'electron-builder', 'Cache') + } + const xdgCache = env.XDG_CACHE_HOME + return xdgCache && parse(xdgCache).root + ? join(xdgCache, 'electron-builder') + : join(home, '.cache', 'electron-builder') +} + +function collectElevateFiles(dir, depth, found) { + let entries + try { + entries = readdirSync(dir, { withFileTypes: true }) + } catch { + return found + } + for (const entry of entries) { + const path = join(dir, entry.name) + if (entry.isFile()) { + if (entry.name.toLowerCase() === ELEVATE_EXE) { + found.push(path) + } + } else if (entry.isDirectory() && depth > 1) { + collectElevateFiles(path, depth - 1, found) + } + } + return found +} + +/** + * Every cached `elevate.exe` under an NSIS release directory of `cacheDir`, plus the + * `ELECTRON_BUILDER_NSIS_DIR` override copy when that is set. + */ +export function findCachedElevatePaths(cacheDir, { env = process.env } = {}) { + const found = [] + const overrideDir = env.ELECTRON_BUILDER_NSIS_DIR?.trim() + if (overrideDir && isFile(join(overrideDir, ELEVATE_EXE))) { + found.push(join(overrideDir, ELEVATE_EXE)) + } + let entries + try { + entries = readdirSync(cacheDir, { withFileTypes: true }) + } catch { + return found + } + for (const entry of entries) { + if (entry.isDirectory() && NSIS_RELEASE_DIR.test(entry.name)) { + collectElevateFiles(join(cacheDir, entry.name), MAX_DEPTH, found) + } + } + return found +} + +/** + * The exact path `CopyElevateHelper` will pack, asked of app-builder-lib itself. Returns the + * failure instead of logging it: an unavailable probe leaves the directory scan as the only + * signal, and the caller has to say that out loud rather than quietly passing. + */ +export async function resolveToolsetElevatePath(projectDir = process.cwd()) { + try { + const configPath = require.resolve(resolve(projectDir, 'config/electron-builder.config.cjs')) + const config = require(configPath) + const { getNsisElevatePath } = require('app-builder-lib/out/toolsets/windows.js') + const path = await getNsisElevatePath(config.toolsets?.nsis, config.nsis?.customNsisBinary) + return { path, error: null } + } catch (error) { + return { path: null, error: error.message } + } +} + +/** + * Replaces every cached copy rather than picking one. Which bundle the rebuild packs + * depends on the toolset version resolved at pack time, and each cached copy is an + * unsigned `elevate.exe` that a later pack could reach for; the helper is a standalone + * UAC shim, not coupled to the NSIS version around it, so overwriting all of them is safe. + * + * `toolsetReplaced` is the signal that matters. A non-empty `replaced` only says that some + * cached copy was rewritten, which a stale release directory carried in by the + * `electron-builder-win-` prefix restore can satisfy on its own. + */ +export async function replaceCachedElevateHelpers({ + signedPath, + cacheDir = resolveElectronBuilderCacheDir(), + projectDir = process.cwd(), + env = process.env, + probe = resolveToolsetElevatePath +} = {}) { + if (!isFile(signedPath)) { + throw new Error(`Signed elevate.exe not found: ${signedPath}`) + } + const targets = new Set(findCachedElevatePaths(cacheDir, { env })) + const { path: toolsetPath, error: toolsetError } = await probe(projectDir) + if (toolsetPath != null && isFile(toolsetPath)) { + targets.add(toolsetPath) + } + + const replaced = [] + for (const target of targets) { + copyFileSync(signedPath, target) + replaced.push(target) + } + return { + replaced, + cacheDir, + toolsetPath, + toolsetError, + toolsetReplaced: toolsetPath != null && replaced.includes(toolsetPath) + } +} + +/** + * The annotations and exit code a swap result earns. Split out so every branch is testable + * without a subprocess — including the one that made this defect class possible, where the + * step passes because *a* cached copy was replaced while the copy the rebuild packs was not. + */ +export function summarizeSwap({ replaced, cacheDir, toolsetPath, toolsetError, toolsetReplaced }) { + if (toolsetPath != null && !toolsetReplaced) { + return { + annotations: [ + { + level: 'error', + message: + `app-builder-lib resolves the elevate.exe the NSIS rebuild will pack to ${toolsetPath}, ` + + 'but that path could not be replaced, so the installer will ship an unsigned UAC ' + + 'elevation helper.' + } + ], + exitCode: 1 + } + } + if (replaced.length === 0) { + return { + annotations: [ + { + level: 'error', + message: + `No cached elevate.exe found under ${cacheDir}; the NSIS rebuild will pack the unsigned ` + + 'helper and ship an unsigned UAC elevation binary. The electron-builder toolset cache ' + + 'layout has changed — update config/scripts/replace-cached-nsis-elevate.mjs.' + } + ], + exitCode: 1 + } + } + if (toolsetPath == null) { + // A green step must never quietly mean "the authoritative check did not run". The scan + // alone is satisfiable by a stale release directory that the `electron-builder-win-` + // prefix restore carried across a lockfile change, while the bundle the rebuild actually + // packs sits in a directory this scan does not match. + return { + annotations: [ + { + level: 'warning', + message: + 'Could not ask app-builder-lib which elevate.exe the NSIS rebuild will pack ' + + `(${toolsetError}); replaced ${replaced.length} copies found by scanning ${cacheDir} ` + + 'alone, which a stale release directory can satisfy while the packed copy stays unsigned.' + } + ], + exitCode: 0 + } + } + return { annotations: [], exitCode: 0 } +} + +// Why an exit code and not a warning: a swap that misses the copy the rebuild packs exits +// before that rebuild restores the unsigned helper, so a silent success here is +// indistinguishable from a release that shipped a signed one — which is how this went +// unnoticed for two releases. The workflow step is `continue-on-error`, so this annotates +// loudly without making a release unbuildable. +if (import.meta.filename === process.argv[1]) { + const signedPath = process.argv[2] + if (!signedPath) { + process.stderr.write('Usage: replace-cached-nsis-elevate.mjs \n') + process.exit(2) + } + try { + const result = await replaceCachedElevateHelpers({ signedPath }) + const { annotations, exitCode } = summarizeSwap(result) + for (const { level, message } of annotations) { + process.stdout.write(`::${level}::${message}\n`) + } + if (exitCode === 0) { + for (const path of result.replaced) { + const role = path === result.toolsetPath ? ' (the copy app-builder-lib will pack)' : '' + process.stdout.write(`Replaced ${path} with the SignPath-signed copy.${role}\n`) + } + } + process.exit(exitCode) + } catch (error) { + process.stdout.write(`::error::Could not replace the cached elevate.exe: ${error.message}\n`) + process.exit(1) + } +} diff --git a/config/scripts/replace-cached-nsis-elevate.test.mjs b/config/scripts/replace-cached-nsis-elevate.test.mjs new file mode 100644 index 00000000000..a88461703c3 --- /dev/null +++ b/config/scripts/replace-cached-nsis-elevate.test.mjs @@ -0,0 +1,364 @@ +import { spawnSync } from 'node:child_process' +import { + existsSync, + mkdirSync, + mkdtempSync, + readdirSync, + readFileSync, + rmSync, + writeFileSync +} from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { parse } from 'yaml' + +import { + findCachedElevatePaths, + replaceCachedElevateHelpers, + resolveElectronBuilderCacheDir, + summarizeSwap +} from './replace-cached-nsis-elevate.mjs' + +// The probe is app-builder-lib asking itself where the packed elevate.exe lives; injected +// here so no test needs the network or a warm toolset cache. +const probeFound = (path) => async () => ({ path, error: null }) +const probeUnavailable = async () => ({ path: null, error: 'app-builder-lib not loadable' }) + +const projectRoot = resolve(import.meta.dirname, '../..') +const scriptPath = join(projectRoot, 'config/scripts/replace-cached-nsis-elevate.mjs') + +let scratch + +beforeEach(() => { + scratch = mkdtempSync(join(tmpdir(), 'orca elevate swap ')) +}) + +afterEach(() => { + rmSync(scratch, { recursive: true, force: true }) +}) + +function makeCache(...relativeFiles) { + const cacheDir = join(scratch, 'Cache') + for (const relative of relativeFiles) { + const path = join(cacheDir, ...relative.split('/')) + mkdirSync(join(path, '..'), { recursive: true }) + writeFileSync(path, 'unsigned-elevate') + } + mkdirSync(cacheDir, { recursive: true }) + return cacheDir +} + +describe('cached elevate.exe swap covers the real electron-builder layouts', () => { + // Why these exact shapes: `downloadBuilderToolset` unpacks to + // `//-/`, and `releaseName` is + // `nsis-3.0.4.1` on the legacy bundle (`getBinFromUrl`) and `nsis@` on the + // unified bundle. The release workflow searched `/nsis`, which matches none of + // them. `customNsisBinary` is deliberately absent — see the probe suite below. + it.each([ + ['legacy bundle', 'nsis-3.0.4.1/nsis-3.0.4.1-1mx3n/elevate.exe'], + ['unified bundle', 'nsis@1.2.1/nsis-bundle-3.12-k4d9x/elevate.exe'], + ['bare nsis release dir', 'nsis/nsis-3.0.4.1/elevate.exe'] + ])('finds the cached helper in the %s layout', (_label, relative) => { + const cacheDir = makeCache(relative) + expect(findCachedElevatePaths(cacheDir, { env: {} })).toEqual([ + join(cacheDir, ...relative.split('/')) + ]) + }) + + it('leaves other toolsets and the raw download dir alone', () => { + const cacheDir = makeCache( + 'winCodeSign/winCodeSign-2.6.0-abc12/elevate.exe', + 'downloads/nsis/elevate.exe' + ) + expect(findCachedElevatePaths(cacheDir, { env: {} })).toEqual([]) + }) + + // `nsis-resources-3.4.1` matches the release-dir pattern and is scanned. Documented + // rather than excluded: `getLegacyNsisResourcesBin` ships plugins, never an elevate.exe, + // so the over-match costs one cheap directory read and nothing else. Narrowing the + // pattern to exclude it would be a guess about a name app-builder-lib owns. + it('scans the resources bundle too, which ships no helper to find', () => { + expect( + findCachedElevatePaths(makeCache('nsis-resources-3.4.1/plugins/x86-unicode/nsProcess.dll'), { + env: {} + }) + ).toEqual([]) + + const planted = 'nsis-resources-3.4.1/nsis-resources-3.4.1-p8w1z/elevate.exe' + const cacheDir = makeCache(planted) + expect(findCachedElevatePaths(cacheDir, { env: {} })).toEqual([ + join(cacheDir, ...planted.split('/')) + ]) + }) + + // The rebuild picks one bundle, and nothing outside app-builder-lib knows which. + // Replacing every cached copy is the deliberate answer to that ambiguity. + it('replaces every cached copy when several bundles are present', async () => { + const cacheDir = makeCache( + 'nsis-3.0.4.1/nsis-3.0.4.1-1mx3n/elevate.exe', + 'nsis@1.2.1/nsis-bundle-3.12-k4d9x/elevate.exe' + ) + const signed = join(scratch, 'signed-elevate.exe') + writeFileSync(signed, 'signpath-signed-elevate') + + const { replaced } = await replaceCachedElevateHelpers({ + signedPath: signed, + cacheDir, + env: {}, + probe: probeUnavailable + }) + + expect(replaced).toHaveLength(2) + for (const path of replaced) { + expect(readFileSync(path, 'utf8')).toBe('signpath-signed-elevate') + } + }) + + it('covers the ELECTRON_BUILDER_NSIS_DIR override copy', () => { + const overrideDir = join(scratch, 'nsis-override') + mkdirSync(overrideDir, { recursive: true }) + writeFileSync(join(overrideDir, 'elevate.exe'), 'unsigned-elevate') + const cacheDir = makeCache() + + expect( + findCachedElevatePaths(cacheDir, { env: { ELECTRON_BUILDER_NSIS_DIR: overrideDir } }) + ).toEqual([join(overrideDir, 'elevate.exe')]) + }) + + it('resolves the cache root the same way app-builder-lib does', () => { + expect( + resolveElectronBuilderCacheDir({ + env: { LOCALAPPDATA: 'C:\\Users\\runneradmin\\AppData\\Local' }, + platform: 'win32' + }) + ).toBe(join('C:\\Users\\runneradmin\\AppData\\Local', 'electron-builder', 'Cache')) + expect(resolveElectronBuilderCacheDir({ env: {}, platform: 'darwin', home: '/Users/a' })).toBe( + join('/Users/a', 'Library', 'Caches', 'electron-builder') + ) + expect(resolveElectronBuilderCacheDir({ env: { ELECTRON_BUILDER_CACHE: '/mnt/cache' } })).toBe( + '/mnt/cache' + ) + }) + + // Proof against the layout actually on disk, not just the fixtures. Cross-checked + // against an independent unbounded walk so a search that scopes itself wrongly + // cannot pass by finding nothing — which is exactly how the inline path passed. + // Skipped only where no NSIS bundle has been downloaded into the cache yet. + it('finds every elevate.exe the real electron-builder cache holds', (ctx) => { + const cacheDir = resolveElectronBuilderCacheDir() + if (!existsSync(cacheDir)) { + // Reported as skipped, never as passed: this is the one test that checks the scan + // against a layout nobody wrote down, and a silent no-op here is the suite + // confirming itself. The Linux unit-test job has no electron-builder cache. + ctx.skip() + return + } + const walk = (dir) => + readdirSync(dir, { withFileTypes: true }).flatMap((entry) => { + const path = join(dir, entry.name) + if (entry.isDirectory()) { + return walk(path) + } + return entry.name.toLowerCase() === 'elevate.exe' ? [path] : [] + }) + const onDisk = walk(cacheDir) + if (onDisk.length === 0) { + ctx.skip() + return + } + expect(findCachedElevatePaths(cacheDir, { env: {} }).sort()).toEqual(onDisk.sort()) + }) +}) + +describe('the probe, not the scan, decides whether the swap worked', () => { + // Why the probe is load-bearing: `getBinFromCustomLoc` passes `nsis-` to `getBin` + // as its in-process promise key only — the extract dir is named for the custom URL's parent + // segment, so a customNsisBinary bundle can sit outside `nsis*` entirely. + it('covers a custom bundle the directory scan cannot match', async () => { + const relative = 'orca-nsis-mirror/nsis-custom-3.11-0zqp2/elevate.exe' + const cacheDir = makeCache(relative) + const packed = join(cacheDir, ...relative.split('/')) + const signed = join(scratch, 'signed-elevate.exe') + writeFileSync(signed, 'signpath-signed-elevate') + + expect(findCachedElevatePaths(cacheDir, { env: {} })).toEqual([]) + + const result = await replaceCachedElevateHelpers({ + signedPath: signed, + cacheDir, + env: {}, + probe: probeFound(packed) + }) + + expect(result.toolsetReplaced).toBe(true) + expect(readFileSync(packed, 'utf8')).toBe('signpath-signed-elevate') + expect(summarizeSwap(result)).toEqual({ annotations: [], exitCode: 0 }) + }) + + // The shape that reproduced the hole: release-cut.yml restores the toolset cache with + // `restore-keys: electron-builder-win-`, so a stale release directory survives a lockfile + // change. Replacing that stale copy satisfies `replaced.length > 0` on its own while the + // bundle the rebuild packs sits in a directory the scan never matches. + it('does not call a stale directory a success when the packed bundle is unmatched', async () => { + const stale = 'nsis-3.0.4.1/nsis-3.0.4.1-1mx3n/elevate.exe' + const packed = 'builder-nsis@4.0.0/nsis-bundle-4.0-k4d9x/elevate.exe' + const cacheDir = makeCache(stale, packed) + const signed = join(scratch, 'signed-elevate.exe') + writeFileSync(signed, 'signpath-signed-elevate') + + const result = await replaceCachedElevateHelpers({ + signedPath: signed, + cacheDir, + env: {}, + probe: probeUnavailable + }) + + // The scan rewrote only the stale copy; the one that would be packed is untouched. + expect(result.replaced).toEqual([join(cacheDir, ...stale.split('/'))]) + expect(readFileSync(join(cacheDir, ...packed.split('/')), 'utf8')).toBe('unsigned-elevate') + + // So the run must not look clean. + const { annotations, exitCode } = summarizeSwap(result) + expect(exitCode).toBe(0) + expect(annotations).toHaveLength(1) + expect(annotations[0].level).toBe('warning') + expect(annotations[0].message).toContain('Could not ask app-builder-lib') + }) + + it('fails when the probe names a copy that could not be replaced', () => { + const summary = summarizeSwap({ + replaced: ['C:/cache/nsis-3.0.4.1/nsis-3.0.4.1-1mx3n/elevate.exe'], + cacheDir: 'C:/cache', + toolsetPath: 'C:/cache/nsis@2.0.0/nsis-bundle-4.0-k4d9x/elevate.exe', + toolsetError: null, + toolsetReplaced: false + }) + + expect(summary.exitCode).toBe(1) + expect(summary.annotations[0].level).toBe('error') + expect(summary.annotations[0].message).toContain('will pack') + }) + + it('fails when nothing at all was replaced', () => { + const summary = summarizeSwap({ + replaced: [], + cacheDir: 'C:/cache', + toolsetPath: null, + toolsetError: 'app-builder-lib not loadable', + toolsetReplaced: false + }) + + expect(summary.exitCode).toBe(1) + expect(summary.annotations[0].level).toBe('error') + expect(summary.annotations[0].message).toContain('No cached elevate.exe found') + }) +}) + +describe('a cached elevate.exe miss is not silent', () => { + // ELECTRON_BUILDER_NSIS_DIR short-circuits app-builder-lib's own resolution before + // any download, so the probe fails offline instead of fetching the NSIS bundle. + function runScript(cacheDir, nsisDir, signedPath) { + return spawnSync(process.execPath, [scriptPath, signedPath], { + cwd: projectRoot, + encoding: 'utf8', + env: { + ...process.env, + ELECTRON_BUILDER_CACHE: cacheDir, + ELECTRON_BUILDER_NSIS_DIR: nsisDir + } + }) + } + + it('exits non-zero with an ::error:: annotation when no cached copy is found', () => { + const cacheDir = makeCache() + const emptyNsisDir = join(scratch, 'empty-nsis') + mkdirSync(emptyNsisDir, { recursive: true }) + const signed = join(scratch, 'signed-elevate.exe') + writeFileSync(signed, 'signpath-signed-elevate') + + const result = runScript(cacheDir, emptyNsisDir, signed) + + expect(result.status).toBe(1) + expect(result.stdout).toContain('::error::No cached elevate.exe found') + }) + + it('warns on the scan-only path so green never means the probe was skipped', () => { + const cacheDir = makeCache('nsis-3.0.4.1/nsis-3.0.4.1-1mx3n/elevate.exe') + const emptyNsisDir = join(scratch, 'empty-nsis') + mkdirSync(emptyNsisDir, { recursive: true }) + const signed = join(scratch, 'signed-elevate.exe') + writeFileSync(signed, 'signpath-signed-elevate') + + const result = runScript(cacheDir, emptyNsisDir, signed) + + expect(result.status).toBe(0) + expect(result.stdout).not.toContain('::error::') + expect(result.stdout).toContain('::warning::Could not ask app-builder-lib') + expect( + readFileSync(join(cacheDir, 'nsis-3.0.4.1', 'nsis-3.0.4.1-1mx3n', 'elevate.exe'), 'utf8') + ).toBe('signpath-signed-elevate') + }) + + // The healthy release-job path: app-builder-lib answers, so the copy it will pack is the + // one that gets replaced and there is nothing to warn about. + it('exits clean when the probe resolves the copy the rebuild will pack', () => { + const cacheDir = makeCache() + const nsisDir = join(scratch, 'nsis-bundle') + mkdirSync(nsisDir, { recursive: true }) + writeFileSync(join(nsisDir, 'elevate.exe'), 'unsigned-elevate') + const signed = join(scratch, 'signed-elevate.exe') + writeFileSync(signed, 'signpath-signed-elevate') + + const result = runScript(cacheDir, nsisDir, signed) + + expect(result.status).toBe(0) + expect(result.stdout).not.toContain('::error::') + expect(result.stdout).not.toContain('::warning::') + expect(result.stdout).toContain('the copy app-builder-lib will pack') + expect(readFileSync(join(nsisDir, 'elevate.exe'), 'utf8')).toBe('signpath-signed-elevate') + }) +}) + +describe('release-cut.yml swaps the cached elevate.exe through the resolver', () => { + function swapStep() { + const workflow = parse( + readFileSync(join(projectRoot, '.github/workflows/release-cut.yml'), 'utf8') + ) + const step = workflow.jobs.build.steps.find( + (candidate) => candidate.name === 'Replace cached elevate.exe with the signed copy' + ) + expect(step).toBeDefined() + return step + } + + it('delegates the cache lookup to the script instead of an inline path', () => { + const step = swapStep() + expect(step.run).toContain('node config/scripts/replace-cached-nsis-elevate.mjs $signed') + // The hardcoded miss that shipped v1.4.193/v1.4.194 unsigned. + expect(step.run).not.toContain('electron-builder\\Cache\\nsis') + expect(step.run).not.toContain('-ErrorAction SilentlyContinue') + }) + + it('fails the step when the swap reports a miss', () => { + const step = swapStep() + // Matched as an executed statement: downgrading this to a Write-Host restores + // the silent fail-open that let the unsigned helper ship. + expect(step.run).toMatch(/if \(\$LASTEXITCODE -ne 0\) \{/) + expect(step.run).toMatch(/^\s*throw \$message\s*$/m) + expect(step.run).toContain('GITHUB_STEP_SUMMARY') + }) + + // Why kept: windows-signing-rehearsal.yml shares the electron-builder-win- + // cache key, so dropping this guard would let a test certificate reach a release cache. + it('still refuses to stage anything but a SignPath-signed helper', () => { + const step = swapStep() + expect(step.run).toContain("$signature.Status -ne 'Valid'") + expect(step.run).toContain("$subject -notlike '*CN=SignPath Foundation*'") + }) + + // The inner-signing chain stays fail-open: a loud red step, not an unbuildable release. + it('keeps the step unable to fail the release job', () => { + expect(swapStep()['continue-on-error']).toBe(true) + }) +}) From ec030f1d351e04231211e5c0ab6abd705e9af18b Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Sat, 5 Sep 2026 21:44:28 -0700 Subject: [PATCH 52/69] fix(windows): sign the NSIS uninstaller via SignPath (#17868) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(windows): sign the NSIS uninstaller via SignPath `Uninstall Orca.exe` ships NotSigned, and MDE's whole update cluster is that one file: electron-builder copies it to `old-uninstaller.exe` and runs it silently during every update. The cause is narrower than "NSIS generates the uninstaller at install time". app-builder-lib already builds the uninstaller in its own makensis pass and calls `packager.signIf(uninstallerPath)` on it before embedding it (NsisTarget.computeScriptAndSignUninstaller). Orca signs nothing during electron-builder — SignPath signs afterwards, behind a human approval — so that hook is a no-op and the file is deleted before CI can reach it. Use the hook as a relay instead of a signer: the first Windows build exports the uninstaller, it rides the existing inner-binaries SignPath request (no third approval wait), and the rebuild-from-signed-tree pass swaps the signed bytes back in before makensis embeds them. Every added step is fail-open. A missing export, a SignPath artifact configuration that does not cover `uninstaller/`, or a relay error costs only the uninstaller signature — the inner-binary chain and the shipped installer are unchanged. * fix(windows): keep the uninstaller relay out of the packed checkout Review fixes on the uninstaller signing chain. The export path lived at `${{ github.workspace }}\uninstaller-signing\`. `files` in the electron-builder config is all-negation, so app-builder prepends `**/*` and packs whatever is left in the checkout root, and the build step retries up to three times — attempt 1 wrote the file after packing, attempts 2 and 3 would have packed an unsigned `.exe` into app.asar. All seven relay sites move to `runner.temp`, and a contract test now fails if any of them points back into the checkout. The uninstaller staging block guarded with `Test-Path` but left `New-Item` and `Copy-Item` able to throw. That step's outcome gates the upload of every inner binary, so a locked file there would have cost all of them their signatures — worse than before the chain existed. It is wrapped in try/catch, asserted. Also: test `signWindowsUninstallerViaSignPath` itself (it runs in a step with no continue-on-error, so its no-throw property is load-bearing) and the sha1+sha256 double invocation; make the rehearsal verify the uninstaller the installer actually writes to disk rather than only the relay receipt, whose digest comparison is equal by construction; correct the staged-name comment, which asserted a collision that does not reproduce; count what was reported rather than what was extracted; and note two traps — a custom sign hook replaces signtool outright, and the single-env-var relay would race if a second NSIS target or arch is added. * fix(windows): stop the signing rehearsal failing on its own artefact The rehearsal is the merge gate for this chain, so it must not be able to fail on something that is not the thing under test. It trusted whatever 7-Zip's NSIS handler emitted. That handler produces partial or garbled output on some NSIS builds, and a truncated extract would score NotSigned and be reported as "the shipped uninstaller is unsigned" when nothing was wrong. It now has to reproduce the digest the sign hook recorded before its output is trusted; otherwise it falls through to the silent-install route, which is ground truth. A name miss falls through the same way. The install route only checked the signature. Comparing the on-disk file against the receipt is what actually proves the shipped installer embedded the SignPath-signed bytes — the release job's own comparison is equal by construction, so this is the only place the claim is really tested. Also: bound the silent install (a bare `-Wait` on an installer that ever prompts hangs to the 360-minute job cap) and poll before stopping Orca, since the oneClick installer launches the app as it finishes and the process can appear after the installer has already exited. Two smaller ones: `-ErrorAction Stop` on the staging New-Item/Copy-Item so the catch above them does not depend on GitHub's $ErrorActionPreference default; and the relay-path test now counts every occurrence rather than the first, so a step carrying two paths cannot root one in RUNNER_TEMP and leave the other bare-relative — the exact shape of the bug it guards. * test(windows): stop a pre-existing elevate.exe defect masking the gate The first real rehearsal (run 33484703381) proved the uninstaller relay works end to end — the 7-Zip route read the embedded uninstaller, the digest guard did not trip, SignPath accepted the new uninstaller/ zip entry, and the shipped `Uninstall Orca.exe` came back signed. It also failed, on `resources\elevate.exe`, for a reason that predates this PR. app-builder-lib re-copies the pristine cached elevate.exe over `resources\elevate.exe` on every nsis pack — `AppPackageHelper.packArch` calls `elevateHelper.copy()` before `buildAppPackage`, and `CopyElevateHelper.copy` does `copyFile(elevatePath, outFile, false)` then `signIf(outFile)`, which signs nothing because this build configures no certificate. The signed copy restored into win-unpacked is clobbered by the rebuild. That is not the sign hook displacing a signtool call: with no `sign` hook, `signFile` already returned false at "no signing info identified", so nothing was signing elevate.exe before either. release-cut.yml mitigates it separately by pre-seeding the electron-builder cache; this workflow has no such step, which is why the clobber is visible here and not there. Downgrade elevate.exe alone to advisory so it cannot mask the uninstaller result, and record it in the evidence artifact so downgrading stays distinguishable from deleting the check. Both uninstaller verdicts stay fatal, pinned by a contract test that also holds the escape hatch to exactly one file. The underlying defect gets its own PR — it is a UAC elevation helper and deserves more scrutiny than a footnote here. * docs(windows): warn against relaxing the elevate.exe cache guard The tempting edit, for anyone who finds the rehearsal red on resources\elevate.exe, is to relax release-cut's `Valid` + `CN=SignPath Foundation` guard so the cache swap runs under test-signing and the rehearsal goes green. That guard is the only thing stopping a test certificate from being seeded into a cache a real release restores from — both workflows share the key `electron-builder-win-`. Shipping users a binary signed by "Test certificate for 'Orca agent ide [OSS]'" is worse than shipping it unsigned, so say so at the place someone would make that edit. --------- Co-authored-by: Orca Worker --- .github/workflows/release-cut.yml | 114 ++++++++- .../workflows/windows-signing-rehearsal.yml | 215 +++++++++++++++- config/electron-builder.config.cjs | 15 +- .../verify-dev-channel-packaging.test.mjs | 13 + ...windows-signing-workflow-contract.test.mjs | 234 +++++++++++++++++ .../scripts/windows-uninstaller-signing.cjs | 111 +++++++++ .../windows-uninstaller-signing.test.mjs | 235 ++++++++++++++++++ 7 files changed, 920 insertions(+), 17 deletions(-) create mode 100644 config/scripts/windows-uninstaller-signing.cjs create mode 100644 config/scripts/windows-uninstaller-signing.test.mjs diff --git a/.github/workflows/release-cut.yml b/.github/workflows/release-cut.yml index a1b6784be18..c35999c7786 100644 --- a/.github/workflows/release-cut.yml +++ b/.github/workflows/release-cut.yml @@ -1427,6 +1427,17 @@ jobs: command: ${{ matrix.release_command }} env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + # Why: the NSIS uninstaller only exists inside electron-builder's + # uninstaller pass, which deletes it right after embedding it. The sign + # hook in config/scripts/windows-uninstaller-signing.cjs copies it out + # here so it can ride the inner-binaries SignPath request below. + # Why runner.temp and never the workspace: `files` in + # config/electron-builder.config.cjs is all-negation, so app-builder + # prepends `**/*` and packs whatever is left in the checkout root. This + # step retries up to 3 times; attempt 1 writes the file after packing, + # but attempts 2 and 3 would then pack the unsigned uninstaller into + # app.asar - the exact defect this chain exists to remove. + ORCA_WIN_UNINSTALLER_EXPORT_PATH: ${{ runner.temp }}\uninstaller-signing\unsigned\orca-uninstaller.exe - name: Verify Windows node-pty ConPTY runtime if: matrix.platform == 'win' && github.run_attempt == 1 @@ -1453,7 +1464,10 @@ jobs: # Why: SignPath cannot deep-sign inside NSIS installers, so inner PE # files (Orca.exe, node-pty *.node, DLLs) are signed via a separate zip # request, then the installer is rebuilt from the signed tree before the - # existing installer signing request below. Every step in this chain is + # existing installer signing request below. The NSIS uninstaller rides + # this same request (it is the MDE update cluster: old-uninstaller.exe / + # Uninstall Orca.exe), captured through electron-builder's sign hook and + # swapped back in during the rebuild — no third approval wait. Every step is # fail-open (continue-on-error + outcome gating): any failure ships the # original installer with unsigned inner binaries, exactly like releases # did before this chain existed. Rehearsed end to end in run 28988432001 @@ -1500,6 +1514,36 @@ jobs: Write-Host "Skipped $($skipped.Count) already-signed files:" $skipped | ForEach-Object { Write-Host " $_" } + # Why the uninstaller rides this request: it is the file MDE flagged in + # the whole update cluster (old-uninstaller.exe / Uninstall Orca.exe), + # and folding it in here costs no extra approval wait. Why it is kept + # out of inner-signing-list.txt: that list drives the copy-back into + # dist/win-unpacked, and the uninstaller does not live there — it is + # re-injected through the sign hook during the rebuild instead. + # Why this name and not "Uninstall Orca.exe": the restore loop below + # matches staged files by suffix (`-like "*$relative"`) and takes the + # first hit, so any staged path ending in "Orca.exe" is separated from + # the real Orca.exe only by Get-ChildItem's enumeration order. That + # order happens to favour the root file today, but it is not a + # documented guarantee; a name that cannot suffix-match is. + # Why the whole block is caught rather than just Test-Path'd: this + # step's outcome gates the upload of every inner binary, so a locked + # file or a full disk here would cost all of them their signatures - + # worse than shipping no uninstaller signature at all. + try { + $exportedUninstaller = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\unsigned\orca-uninstaller.exe' + if (Test-Path -LiteralPath $exportedUninstaller) { + $uninstallerStagePath = Join-Path $stage.FullName 'uninstaller\orca-uninstaller.exe' + New-Item -ItemType Directory -Force -Path (Split-Path $uninstallerStagePath) -ErrorAction Stop | Out-Null + Copy-Item -LiteralPath $exportedUninstaller -Destination $uninstallerStagePath -Force -ErrorAction Stop + Write-Host 'Staged the NSIS uninstaller for signing: uninstaller\orca-uninstaller.exe' + } else { + Write-Host "::warning::No exported NSIS uninstaller at $exportedUninstaller; this release ships an unsigned uninstaller (fail-open)." + } + } catch { + Write-Host "::warning::Could not stage the NSIS uninstaller ($_); this release ships an unsigned uninstaller (fail-open)." + } + - name: Upload unsigned inner binaries for SignPath id: upload-unsigned-inner if: matrix.platform == 'win' && github.run_attempt == 1 && steps.stage-inner.outcome == 'success' @@ -1644,6 +1688,31 @@ jobs: throw "Signed inner artifact did not round-trip cleanly ($($failures.Count) failures)." } + # Why gated separately from the inner restore above: if SignPath's + # windows-inner-binaries-zip artifact configuration does not (yet) cover the + # uninstaller/ directory, the uninstaller comes back missing. That must cost + # only the uninstaller signature — the rebuild below still runs and still + # ships the signed inner binaries, exactly as it does today. + - name: Restore signed uninstaller for the installer rebuild + id: restore-signed-uninstaller + if: matrix.platform == 'win' && github.run_attempt == 1 && steps.restore-signed-inner.outcome == 'success' + continue-on-error: true + shell: pwsh + run: | + $signed = Get-ChildItem -Path signed-inner -Recurse -File -Filter 'orca-uninstaller.exe' | + Select-Object -First 1 + if ($null -eq $signed) { + throw 'SignPath did not return uninstaller/orca-uninstaller.exe; check the windows-inner-binaries-zip artifact configuration covers it.' + } + $signature = Get-AuthenticodeSignature -FilePath $signed.FullName + if ($null -eq $signature.SignerCertificate) { + throw 'The returned NSIS uninstaller carries no signature.' + } + $signedDir = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\signed' + New-Item -ItemType Directory -Force -Path $signedDir | Out-Null + Copy-Item -LiteralPath $signed.FullName -Destination (Join-Path $signedDir 'orca-uninstaller.exe') -Force + Write-Host ("{0,-14} uninstaller <{1}>" -f $signature.Status, $signature.SignerCertificate.Subject) + # Why this step exists: electron-builder's CopyElevateHelper re-copies a # pristine elevate.exe from its download cache over resources\elevate.exe # on EVERY nsis pack — including the --prepackaged rebuild below — which @@ -1697,6 +1766,11 @@ jobs: if: matrix.platform == 'win' && github.run_attempt == 1 && steps.restore-signed-inner.outcome == 'success' continue-on-error: true shell: pwsh + env: + # Why unconditional: the sign hook keys off the file existing, which it + # only does when the restore step above succeeded. A missing file logs a + # warning and embeds the freshly built unsigned uninstaller instead. + ORCA_WIN_UNINSTALLER_SIGNED_PATH: ${{ runner.temp }}\uninstaller-signing\signed\orca-uninstaller.exe run: | # Why: keep the pre-rebuild artifacts so a failed rebuild can fall # back to shipping them unchanged (fail-open). @@ -1888,6 +1962,7 @@ jobs: env: ORCA_WINDOWS_INNER_SIGNATURE_REQUIRED: 'false' INNER_SIGNING_COMPLETED: ${{ steps.rebuild-nsis-signed.outcome == 'success' }} + UNINSTALLER_SIGNING_COMPLETED: ${{ steps.restore-signed-uninstaller.outcome == 'success' }} run: | $required = $env:ORCA_WINDOWS_INNER_SIGNATURE_REQUIRED -eq 'true' @@ -1968,6 +2043,39 @@ jobs: if ($targets -notcontains 'resources\elevate.exe') { $targets += 'resources\elevate.exe' } + # Why the uninstaller is not in $targets: NSIS embeds it in its own + # compressed data section (`File /oname=${UNINSTALL_FILENAME}` in + # app-builder-lib templates/nsis/include/installer.nsh), not in the + # app 7z payload extracted above - the bundled 7za cannot see it. + # What the receipt proves and does not: the digest comparison is + # equal by construction (the hook digests the bytes it copied from + # this same file), so the real signal is that the receipt exists at + # all - the import leg ran, and these are the bytes it embedded. The + # signature check below is the part with teeth. The shipped-artifact + # check lives in windows-signing-rehearsal.yml, which installs the + # installer and inspects the uninstaller it drops on disk. + if ($env:UNINSTALLER_SIGNING_COMPLETED -eq 'true') { + $signedUninstaller = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\signed\orca-uninstaller.exe' + $receipt = "$signedUninstaller.embedded-sha256" + if (-not (Test-Path -LiteralPath $receipt)) { + $failures.Add('the sign hook did not embed the signed uninstaller into the rebuilt installer') + } else { + $embedded = (Get-Content -LiteralPath $receipt -Raw).Trim() + $actual = (Get-FileHash -LiteralPath $signedUninstaller -Algorithm SHA256).Hash.ToLowerInvariant() + $signature = Get-AuthenticodeSignature -FilePath $signedUninstaller + $subject = if ($null -eq $signature.SignerCertificate) { '' } else { $signature.SignerCertificate.Subject } + $line = "{0,-14} {1} <{2}>" -f $signature.Status, 'Uninstall Orca.exe (embedded)', $subject + $report.Add($line) + Write-Host $line + if ($embedded -ne $actual) { + $failures.Add("the rebuilt installer embedded different uninstaller bytes than the signed one ($embedded vs $actual)") + } elseif ($signature.Status -ne 'Valid' -or $subject -notlike '*CN=SignPath Foundation*') { + $failures.Add("not signed by SignPath Foundation: Uninstall Orca.exe ($($signature.Status), $subject)") + } + } + } else { + Write-Host '::warning::The NSIS uninstaller was not signed on this run; it is excluded from the evidence gate (fail-open).' + } foreach ($relative in $targets) { $path = Join-Path $root $relative if (-not (Test-Path $path)) { @@ -2000,7 +2108,9 @@ jobs: Add-GateEvidence "VERDICT: FAILED — $message" Add-GateSummary "FAILED — $message" } else { - $ok = "All $($targets.Count) inner binaries in the shipped installer are signed by SignPath Foundation." + # $report, not $targets: the embedded uninstaller is reported but + # is not one of the extracted payload targets. + $ok = "All $($report.Count) checked binaries are signed by SignPath Foundation." Add-GateEvidence "VERDICT: PASSED — $ok" Add-GateSummary "PASSED — $ok" Write-Host $ok diff --git a/.github/workflows/windows-signing-rehearsal.yml b/.github/workflows/windows-signing-rehearsal.yml index 6fc6fab7193..90ab8db137c 100644 --- a/.github/workflows/windows-signing-rehearsal.yml +++ b/.github/workflows/windows-signing-rehearsal.yml @@ -3,9 +3,11 @@ # Why: SignPath cannot deep-sign inside NSIS installers, so shipping signed # inner binaries (Orca.exe, node-pty *.node, DLLs — see issue #7785) requires # a two-request flow: sign the unpacked PE files first, then build the NSIS -# installer from the signed tree, then sign the installer. This workflow -# rehearses that entire flow from a branch, end to end, without publishing -# anything — so the release pipeline on main is never at risk while we verify. +# installer from the signed tree, then sign the installer. The NSIS uninstaller +# rides that same first request — it is captured through electron-builder's sign +# hook and swapped back in during the rebuild — so it adds no third approval. +# This workflow rehearses that entire flow from a branch, end to end, without +# publishing anything — so the release pipeline on main is never at risk. # # Runs only via manual dispatch. Use the test-signing policy for iteration # (auto-approved test certificate) and release-signing to rehearse the @@ -81,15 +83,27 @@ jobs: env: NODE_OPTIONS: --max-old-space-size=4096 - - name: Package unpacked Windows app + # Why a full --win build and not --dir: the NSIS uninstaller only exists + # inside the installer build, and it is the file the MDE update cluster + # flags. --dir would never produce it, so the rehearsal would not rehearse + # the uninstaller leg at all. This mirrors release-cut's first Windows pass. + - name: Package Windows app and export the NSIS uninstaller shell: pwsh + env: + # runner.temp, never the workspace: the all-negation `files` list in + # config/electron-builder.config.cjs packs whatever is left in the + # checkout root into app.asar. + ORCA_WIN_UNINSTALLER_EXPORT_PATH: ${{ runner.temp }}\uninstaller-signing\unsigned\orca-uninstaller.exe run: | node config/scripts/ensure-native-runtime.mjs --runtime=electron if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } - pnpm exec electron-builder --config config/electron-builder.config.cjs --win --dir --publish never + pnpm exec electron-builder --config config/electron-builder.config.cjs --win --publish never if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } if (-not (Test-Path 'dist/win-unpacked/Orca.exe')) { - throw 'electron-builder --dir did not produce dist/win-unpacked/Orca.exe' + throw 'electron-builder --win did not produce dist/win-unpacked/Orca.exe' + } + if (-not (Test-Path -LiteralPath $env:ORCA_WIN_UNINSTALLER_EXPORT_PATH)) { + throw "The sign hook did not export the NSIS uninstaller to $env:ORCA_WIN_UNINSTALLER_EXPORT_PATH" } # Why: only unsigned PE files go to SignPath. Files that already carry a @@ -132,6 +146,17 @@ jobs: Write-Host "Skipped $($skipped.Count) already-signed files:" $skipped | ForEach-Object { Write-Host " $_" } + # Why kept out of inner-signing-list.txt: that list drives the copy-back + # into dist/win-unpacked, and the uninstaller does not live there — it is + # re-injected through the electron-builder sign hook during the rebuild. + # No catch here, unlike the release job: the rehearsal exists to prove + # the flow, so a staging failure must fail it loudly. + $exportedUninstaller = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\unsigned\orca-uninstaller.exe' + $uninstallerStagePath = Join-Path $stage.FullName 'uninstaller\orca-uninstaller.exe' + New-Item -ItemType Directory -Force -Path (Split-Path $uninstallerStagePath) | Out-Null + Copy-Item -LiteralPath $exportedUninstaller -Destination $uninstallerStagePath -Force + Write-Host 'Staged the NSIS uninstaller for signing: uninstaller\orca-uninstaller.exe' + - name: Upload unsigned inner binaries for SignPath id: upload-unsigned-inner uses: actions/upload-artifact@v7 @@ -200,8 +225,27 @@ jobs: throw "Signed inner artifact did not round-trip cleanly ($($failures.Count) failures)." } + - name: Restore signed uninstaller for the installer rebuild + shell: pwsh + run: | + $signed = Get-ChildItem -Path signed-inner -Recurse -File -Filter 'orca-uninstaller.exe' | + Select-Object -First 1 + if ($null -eq $signed) { + throw 'SignPath did not return uninstaller/orca-uninstaller.exe; check the inner-binaries artifact configuration covers it.' + } + $signature = Get-AuthenticodeSignature -FilePath $signed.FullName + if ($null -eq $signature.SignerCertificate) { + throw 'The returned NSIS uninstaller carries no signature.' + } + $signedDir = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\signed' + New-Item -ItemType Directory -Force -Path $signedDir | Out-Null + Copy-Item -LiteralPath $signed.FullName -Destination (Join-Path $signedDir 'orca-uninstaller.exe') -Force + Write-Host ("{0,-14} uninstaller <{1}>" -f $signature.Status, $signature.SignerCertificate.Subject) + - name: Build NSIS installer from signed unpacked app shell: pwsh + env: + ORCA_WIN_UNINSTALLER_SIGNED_PATH: ${{ runner.temp }}\uninstaller-signing\signed\orca-uninstaller.exe run: | pnpm exec electron-builder --config config/electron-builder.config.cjs --win --publish never --prepackaged "$env:GITHUB_WORKSPACE\dist\win-unpacked" if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } @@ -289,20 +333,33 @@ jobs: run: | $report = New-Object System.Collections.Generic.List[string] $failures = New-Object System.Collections.Generic.List[string] + $advisories = New-Object System.Collections.Generic.List[string] $requireValid = $env:SIGNING_POLICY -eq 'release-signing' - function Test-Signature([string]$label, [string]$path) { + # -Advisory records a problem without failing the run. It exists for + # exactly one file (resources\elevate.exe, below) and must not be + # widened casually: the point of this workflow is to fail when signing + # is broken. + function Test-Signature([string]$label, [string]$path, [switch]$Advisory) { $signature = Get-AuthenticodeSignature -FilePath $path $subject = if ($null -eq $signature.SignerCertificate) { '' } else { $signature.SignerCertificate.Subject } $line = "{0,-14} {1} <{2}>" -f $signature.Status, $label, $subject $script:report.Add($line) Write-Host $line + $problem = $null if ($null -eq $signature.SignerCertificate -or $signature.Status -eq 'NotSigned') { - $script:failures.Add("unsigned: $label") + $problem = "unsigned: $label" } elseif ($script:requireValid -and $signature.Status -ne 'Valid') { - $script:failures.Add("not Valid under release-signing: $label ($($signature.Status))") + $problem = "not Valid under release-signing: $label ($($signature.Status))" } elseif ($script:requireValid -and $subject -notlike '*CN=SignPath Foundation*') { - $script:failures.Add("unexpected signer: $label ($subject)") + $problem = "unexpected signer: $label ($subject)" + } + if ($null -eq $problem) { return } + if ($Advisory) { + $script:advisories.Add($problem) + Write-Host "::warning::$problem - known pre-existing issue, not failing the rehearsal" + } else { + $script:failures.Add($problem) } } @@ -324,21 +381,155 @@ jobs: & $7za x 'dist/orca-windows-setup.exe' '-oextracted-app' -y | Out-Null $root = Resolve-Path 'extracted-app' + # The receipt only proves the import leg ran; it cannot prove what NSIS + # embedded, because the uninstaller lives in a compressed NSIS data + # section rather than the app 7z payload above and the bundled 7za has + # no NSIS handler. So the rehearsal - unlike the release job, which + # must not mutate the runner it publishes from - goes all the way: it + # installs the installer silently and inspects the uninstaller the + # installer actually wrote to disk. That is the file MDE flags. + $signedUninstaller = Join-Path $env:RUNNER_TEMP 'uninstaller-signing\signed\orca-uninstaller.exe' + $receipt = "$signedUninstaller.embedded-sha256" + if (-not (Test-Path -LiteralPath $receipt)) { + $failures.Add('the sign hook did not embed the signed uninstaller into the rebuilt installer') + } else { + Test-Signature 'relayed: orca-uninstaller.exe' $signedUninstaller + } + + # Why a full 7-Zip attempt first: it is non-invasive. The runner image + # ships the complete 7z.exe, which - unlike the reduced 7za - has an + # NSIS handler. If it cannot read the section either, fall back to a + # real silent install. + $installedUninstaller = $null + $installedVia = $null + $expectedDigest = if (Test-Path -LiteralPath $receipt) { (Get-Content -LiteralPath $receipt -Raw).Trim() } else { $null } + $full7z = 'C:\Program Files\7-Zip\7z.exe' + if (Test-Path -LiteralPath $full7z) { + New-Item -ItemType Directory -Path nsis-extract -Force | Out-Null + & $full7z x -tnsis 'dist/orca-windows-setup.exe' '-onsis-extract' -y 2>&1 | Out-Null + $installedUninstaller = Get-ChildItem -Path nsis-extract -Recurse -File -Filter 'Uninstall*.exe' -ErrorAction SilentlyContinue | + Select-Object -First 1 + # Why the digest guard before trusting this route: 7-Zip's NSIS + # handler emits partial or garbled output on some NSIS builds, and a + # truncated extract would score NotSigned and fail the rehearsal as + # "the shipped uninstaller is unsigned" when nothing is wrong. Only + # trust it when it reproduces the bytes the relay embedded; otherwise + # fall through to the install route, which is ground truth. A name + # miss (the handler labelling the entry by its source name) falls + # through the same way. + if ($null -ne $installedUninstaller -and $null -ne $expectedDigest -and + (Get-FileHash -LiteralPath $installedUninstaller.FullName -Algorithm SHA256).Hash.ToLowerInvariant() -ne $expectedDigest) { + Write-Host "7-Zip's NSIS output did not match the relayed digest; falling back to a silent install." + $installedUninstaller = $null + } + if ($null -ne $installedUninstaller) { + $installedVia = "7-Zip's NSIS handler" + Write-Host "Read the embedded uninstaller with 7-Zip's NSIS handler: $($installedUninstaller.FullName)" + } else { + Write-Host "7-Zip's NSIS handler did not yield a usable uninstaller; falling back to a silent install." + } + } + + if ($null -eq $installedUninstaller) { + # Nothing here is published, so mutating this runner is free. + # Why -PassThru and a bounded wait rather than -Wait: a bare -Wait on + # an installer that ever prompts hangs to the job's 360-minute cap. + $installerProcess = Start-Process -FilePath (Resolve-Path 'dist/orca-windows-setup.exe') -ArgumentList '/S' -PassThru + if (-not $installerProcess.WaitForExit(300000)) { + $installerProcess | Stop-Process -Force -ErrorAction SilentlyContinue + $failures.Add('the silent install did not exit within 5 minutes; it is likely prompting') + } + # Why a poll rather than one Stop-Process: the oneClick installer + # launches the app as it finishes, so Orca.exe can appear *after* the + # installer process exits. A single silenced Stop-Process would miss + # it and leave Orca plus orca-terminal-daemon.exe holding handles + # under %LOCALAPPDATA%\Programs for the rest of the job. + for ($attempt = 0; $attempt -lt 20; $attempt++) { + $running = @(Get-Process -Name 'Orca' -ErrorAction SilentlyContinue) + if ($running.Count -gt 0) { + $running | Stop-Process -Force -ErrorAction SilentlyContinue + break + } + Start-Sleep -Milliseconds 500 + } + Get-Process -Name 'orca-terminal-daemon' -ErrorAction SilentlyContinue | + Stop-Process -Force -ErrorAction SilentlyContinue + $installedUninstaller = Get-ChildItem -Path "$env:LOCALAPPDATA\Programs" -Recurse -File -Filter 'Uninstall*.exe' -ErrorAction SilentlyContinue | + Where-Object { $_.FullName -like '*Orca*' } | + Select-Object -First 1 + if ($null -ne $installedUninstaller) { $installedVia = 'a silent install' } + } + + if ($null -eq $installedUninstaller) { + $failures.Add('could not obtain the uninstaller the installer ships; neither 7-Zip nor a silent install produced it') + } else { + # Why this digest comparison is the point of the whole rehearsal: + # unlike the release job's, it hashes a file NSIS itself wrote out + # rather than the file the hook copied, so it is the only check that + # proves the shipped installer embedded the SignPath-signed bytes. On + # the 7-Zip route the guard above already forced equality; on the + # install route this is the first time it is tested. + if ($null -ne $expectedDigest) { + $shippedDigest = (Get-FileHash -LiteralPath $installedUninstaller.FullName -Algorithm SHA256).Hash.ToLowerInvariant() + if ($shippedDigest -ne $expectedDigest) { + $failures.Add("the uninstaller the installer ships is not the relayed one (via $installedVia): $shippedDigest vs $expectedDigest") + } + } + Test-Signature "shipped: Uninstall Orca.exe (via $installedVia)" $installedUninstaller.FullName + } + foreach ($relative in Get-Content 'inner-signing-list.txt') { $path = Join-Path $root $relative if (-not (Test-Path $path)) { $failures.Add("missing from installer payload: $relative") continue } - Test-Signature "installed: $relative" $path + # Why elevate.exe alone is advisory: app-builder-lib re-copies the + # pristine cached elevate.exe over resources\elevate.exe on EVERY nsis + # pack - AppPackageHelper.packArch calls elevateHelper.copy() before + # buildAppPackage (nsisUtil.js), and CopyElevateHelper.copy does + # `copyFile(elevatePath, outFile, false)` then `signIf(outFile)`, which + # signs nothing because this build configures no certificate. So the + # signed copy restored into win-unpacked is clobbered by the rebuild. + # This predates the uninstaller relay and is not caused by it: with no + # `sign` hook, signIf already returned false at "no signing info + # identified" (windowsSignToolManager.js), so no signtool call was + # displaced. release-cut.yml mitigates it separately by pre-seeding the + # electron-builder cache ("Replace cached elevate.exe with the signed + # copy"); this workflow has no such step, which is why the clobber is + # visible here and not there. Mirroring that step here would not help: + # it only swaps when the copy is already Valid and SignPath-signed, so + # it no-ops under the test certificate. + # + # DO NOT relax that Valid + SignPath-signed guard to make this + # rehearsal go green. This workflow and release-cut.yml share the + # cache key `electron-builder-win-`, and that guard is + # the only thing stopping a test certificate from being seeded into + # the cache a real release restores from. Shipping users a binary + # signed by "Test certificate for 'Orca agent ide [OSS]'" is worse + # than shipping it unsigned. + # + # Fixing elevate.exe belongs in its own PR - it is a UAC elevation + # helper, and it deserves more scrutiny than a footnote in an + # uninstaller change. + if ($relative -eq 'resources\elevate.exe') { + Test-Signature "installed: $relative" $path -Advisory + } else { + Test-Signature "installed: $relative" $path + } } + if ($advisories.Count -gt 0) { + $report.Add('') + $report.Add('ADVISORY (known pre-existing, did not fail this run):') + $advisories | ForEach-Object { $report.Add(" $_") } + } Set-Content -Path 'signing-evidence.txt' -Value ($report -join "`n") if ($failures.Count -gt 0) { $failures | ForEach-Object { Write-Host "::error::$_" } throw "Signing rehearsal failed with $($failures.Count) problems." } - Write-Host "All $((Get-Content 'inner-signing-list.txt').Count) inner binaries plus the installer are signed." + Write-Host "All checked binaries are signed, including the uninstaller the installer writes to disk ($($advisories.Count) advisory)." - name: Upload rehearsal evidence and installer if: always() diff --git a/config/electron-builder.config.cjs b/config/electron-builder.config.cjs index ebf4d275678..7e0009b3a24 100644 --- a/config/electron-builder.config.cjs +++ b/config/electron-builder.config.cjs @@ -19,6 +19,7 @@ const { } = require('./scripts/verify-packaged-node-pty-job-ownership.cjs') const { verifySkillsCliRuntime } = require('./scripts/verify-skills-cli-runtime.cjs') const { verifyStaticAppImagePackage } = require('./scripts/static-appimage-package-contract.cjs') +const { signWindowsUninstallerViaSignPath } = require('./scripts/windows-uninstaller-signing.cjs') // Why: dev-channel builds must carry the *release* identity — same bundle id, // Developer ID signature, and notarization ticket — or Squirrel.Mac refuses to @@ -401,9 +402,17 @@ module.exports = { // name is absent. An unsigned build that still claimed 'SignPath Foundation' // would therefore reject its own channel's next build — and its way back to // stable with it. Dropping it is what makes dev→dev and dev→stable work. - ...(isWinDevChannel - ? { verifyUpdateCodeSignature: false } - : { signtoolOptions: { publisherName: 'SignPath Foundation' } }), + // Why a sign hook on a build that does not sign: it is the only moment + // electron-builder exposes the NSIS uninstaller (built in its own makensis + // pass, embedded, then deleted). The hook signs nothing — it relays the file + // to and from the CI SignPath request, and is inert when the relay env vars + // are unset, so local and dev builds are unaffected. publisherName stays on + // its existing channel split above. + signtoolOptions: { + sign: signWindowsUninstallerViaSignPath, + ...(isWinDevChannel ? {} : { publisherName: 'SignPath Foundation' }) + }, + ...(isWinDevChannel ? { verifyUpdateCodeSignature: false } : {}), extraResources: [ ...commonExtraResources, ...createPackagedRuntimeNodeModuleResources('win32'), diff --git a/config/scripts/verify-dev-channel-packaging.test.mjs b/config/scripts/verify-dev-channel-packaging.test.mjs index 63e1c7d5b0c..8e5a00f48e1 100644 --- a/config/scripts/verify-dev-channel-packaging.test.mjs +++ b/config/scripts/verify-dev-channel-packaging.test.mjs @@ -53,6 +53,19 @@ describe('electron-builder dev-channel identity', () => { expect(config.win.verifyUpdateCodeSignature).toBe(false) }) + // Why on every channel: the hook is the only handle electron-builder gives on + // the NSIS uninstaller, and it signs nothing — it relays the file to and from + // the CI SignPath request. Carrying it must not drag a publisherName onto a + // dev build, which is the failure the split above exists to prevent. + it('carries the uninstaller sign hook without changing publisherName semantics', () => { + for (const env of [{}, WIN_ADHOC_ENV]) { + const config = loadConfigWithEnv(env) + expect(typeof config.win.signtoolOptions.sign).toBe('function') + } + expect(loadConfigWithEnv({}).win.signtoolOptions.publisherName).toBe('SignPath Foundation') + expect(loadConfigWithEnv(WIN_ADHOC_ENV).win.signtoolOptions.publisherName).toBeUndefined() + }) + it.each([ ['hourly', { ORCA_WIN_HOURLY: '1' }, 'orca-hourly'], ['daily', { ORCA_WIN_DAILY: '1' }, 'orca-daily'], diff --git a/config/scripts/windows-signing-workflow-contract.test.mjs b/config/scripts/windows-signing-workflow-contract.test.mjs index 37edc2196d4..c321db8cfd2 100644 --- a/config/scripts/windows-signing-workflow-contract.test.mjs +++ b/config/scripts/windows-signing-workflow-contract.test.mjs @@ -1,4 +1,5 @@ import { readFileSync } from 'node:fs' +import { createRequire } from 'node:module' import { join, resolve } from 'node:path' import { describe, expect, it } from 'vitest' import { parse } from 'yaml' @@ -212,6 +213,7 @@ describe('Windows signing workflow contract', () => { 'Notify Slack that inner-binary signing is waiting for approval', 'Download signed inner binaries from SignPath', 'Restore signed inner binaries into unpacked app', + 'Restore signed uninstaller for the installer rebuild', 'Replace cached elevate.exe with the signed copy', 'Rebuild NSIS installer from signed unpacked app' ] @@ -222,3 +224,235 @@ describe('Windows signing workflow contract', () => { } }) }) + +// Why these exist: the NSIS uninstaller is generated inside electron-builder's +// uninstaller pass and deleted immediately after being embedded, so the only way +// CI can sign it is the export/import relay through win.signtoolOptions.sign. +// Every link is asserted here the way Orca.exe and conpty_console_list.node are. +describe('Windows NSIS uninstaller signing', () => { + const releaseSteps = () => readWorkflow('.github/workflows/release-cut.yml').jobs.build.steps + const stepNamed = (steps, name) => steps.find((step) => step.name === name) + + const EXPORT_ENV = 'ORCA_WIN_UNINSTALLER_EXPORT_PATH' + const SIGNED_ENV = 'ORCA_WIN_UNINSTALLER_SIGNED_PATH' + + it('exports the uninstaller from the first Windows build', () => { + const build = stepNamed(releaseSteps(), 'Build Windows release artifacts') + + expect(build.env[EXPORT_ENV]).toContain('uninstaller-signing') + expect(build.env[EXPORT_ENV]).toContain('orca-uninstaller.exe') + }) + + // Why this is a test and not a comment: `files` in the electron-builder config + // is all-negation, so app-builder packs whatever is left in the checkout root. + // These steps retry, and a retried attempt would pack an unsigned .exe into + // app.asar — the very defect this chain removes. Every relay path must live + // outside the checkout. + it('keeps every relay path out of the packed checkout', () => { + const relayEnvValues = [ + ...releaseSteps(), + ...readWorkflow('.github/workflows/windows-signing-rehearsal.yml').jobs.rehearse.steps + ].flatMap((step) => [step.env?.[EXPORT_ENV], step.env?.[SIGNED_ENV]].filter(Boolean)) + + expect(relayEnvValues.length).toBe(4) + for (const value of relayEnvValues) { + expect(value).toContain('runner.temp') + expect(value).not.toContain('github.workspace') + } + + const relayScripts = [ + ...releaseSteps(), + ...readWorkflow('.github/workflows/windows-signing-rehearsal.yml').jobs.rehearse.steps + ] + .map((step) => step.run ?? '') + .filter((run) => run.includes('uninstaller-signing')) + + expect(relayScripts.length).toBeGreaterThan(0) + for (const run of relayScripts) { + // Why count occurrences rather than assert `toContain` once: a step + // carrying two relay paths could root the first in RUNNER_TEMP and leave + // the second bare-relative — which resolves against the checkout, and is + // exactly the shape of the defect this test exists to catch. + const mentions = run.match(/uninstaller-signing/g) ?? [] + const rooted = run.match(/Join-Path \$env:RUNNER_TEMP 'uninstaller-signing/g) ?? [] + + expect(rooted.length, run).toBe(mentions.length) + expect(run).not.toContain('$env:GITHUB_WORKSPACE') + } + }) + + it('stages the uninstaller into the same request as the inner binaries', () => { + const stage = stepNamed(releaseSteps(), 'Stage unsigned inner PE files for signing') + + expect(stage.run).toContain('uninstaller-signing\\unsigned\\orca-uninstaller.exe') + expect(stage.run).toContain('uninstaller\\orca-uninstaller.exe') + // No third SignPath request: exactly two submissions, as budgeted for the + // 1h + 4h approval waits inside the 360-minute job cap. + const submissions = releaseSteps().filter( + (step) => step.uses === 'signpath/github-action-submit-signing-request@v2' + ) + expect(submissions).toHaveLength(2) + }) + + // A staged-but-unreturned uninstaller must not fail the inner chain, or a + // SignPath artifact-configuration gap would cost the inner-binary signatures. + it('keeps the uninstaller out of the inner-binary copy-back list', () => { + const stage = stepNamed(releaseSteps(), 'Stage unsigned inner PE files for signing') + const restoreInner = stepNamed( + releaseSteps(), + 'Restore signed inner binaries into unpacked app' + ) + + expect(stage.run).not.toMatch(/\$list\.Add\(['"]uninstaller/) + expect(restoreInner.run).not.toContain('orca-uninstaller.exe') + }) + + // This step's outcome gates the upload of every inner binary, so a filesystem + // error while staging the uninstaller must not escape — otherwise one + // uninstaller-specific failure costs every inner-binary signature, which is + // strictly worse than the behaviour before this chain existed. + it('cannot let an uninstaller staging failure cost the inner-binary signatures', () => { + const stage = stepNamed(releaseSteps(), 'Stage unsigned inner PE files for signing') + const uninstallerBlock = stage.run.slice(stage.run.indexOf('$exportedUninstaller')) + + expect(stage.run).toMatch(/try \{[\s\S]*\$exportedUninstaller[\s\S]*\} catch \{/) + expect(uninstallerBlock).toContain('::warning::Could not stage the NSIS uninstaller') + expect(uninstallerBlock).not.toContain('throw') + // Explicit, so the catch does not silently depend on GitHub's + // $ErrorActionPreference='Stop' default for `shell: pwsh`. + expect(uninstallerBlock).toContain('New-Item -ItemType Directory -Force -Path (Split-Path') + expect(uninstallerBlock).toMatch(/New-Item[^\r\n]*-ErrorAction Stop/) + expect(uninstallerBlock).toMatch(/Copy-Item[^\r\n]*-ErrorAction Stop/) + // The upload it gates still keys off this step, so the catch is load-bearing. + expect(stepNamed(releaseSteps(), 'Upload unsigned inner binaries for SignPath').if).toContain( + "steps.stage-inner.outcome == 'success'" + ) + }) + + it('re-injects the signed uninstaller into the rebuilt installer', () => { + const steps = releaseSteps() + const restore = stepNamed(steps, 'Restore signed uninstaller for the installer rebuild') + const rebuild = stepNamed(steps, 'Rebuild NSIS installer from signed unpacked app') + const names = steps.map((step) => step.name) + + expect(restore.if).toContain('github.run_attempt == 1') + expect(restore.if).toContain("steps.restore-signed-inner.outcome == 'success'") + expect(restore.run).toContain('orca-uninstaller.exe') + expect(names.indexOf(restore.name)).toBeLessThan(names.indexOf(rebuild.name)) + expect(rebuild.env[SIGNED_ENV]).toContain('uninstaller-signing') + // The rebuild must not depend on the uninstaller leg: a missing signed + // uninstaller ships today's installer, it does not skip the rebuild. + expect(rebuild.if).not.toContain('restore-signed-uninstaller') + }) + + // NSIS hides the uninstaller in a compressed data section the bundled 7za + // cannot read, so the gate proves it from the sign hook's digest receipt + // instead of extracting it — and only when the relay actually ran. + it('reports the embedded uninstaller in the inner-binary evidence gate', () => { + const gate = stepNamed(releaseSteps(), 'Verify Windows inner binary signatures') + + expect(gate.env.UNINSTALLER_SIGNING_COMPLETED).toBe( + "${{ steps.restore-signed-uninstaller.outcome == 'success' }}" + ) + expect(gate.run).toContain('.embedded-sha256') + expect(gate.run).toContain("$env:UNINSTALLER_SIGNING_COMPLETED -eq 'true'") + expect(gate.run).toContain('not signed by SignPath Foundation: Uninstall Orca.exe') + // The uninstaller must not join the 7z payload loop, which cannot see it. + expect(gate.run).not.toContain("$targets += 'Uninstall Orca.exe'") + }) + + it('rehearses the uninstaller leg end to end', () => { + const steps = readWorkflow('.github/workflows/windows-signing-rehearsal.yml').jobs.rehearse + .steps + const names = steps.map((step) => step.name) + const pack = stepNamed(steps, 'Package Windows app and export the NSIS uninstaller') + const rebuild = stepNamed(steps, 'Build NSIS installer from signed unpacked app') + const verify = stepNamed(steps, 'Verify signatures end to end') + + // --dir never produces an uninstaller, so the rehearsal has to build the + // installer the way release-cut's first Windows pass does. + expect(pack.run).toContain('--win --publish never') + expect(pack.run).not.toContain('--dir') + expect(pack.env[EXPORT_ENV]).toContain('orca-uninstaller.exe') + expect(names).toContain('Restore signed uninstaller for the installer rebuild') + expect(rebuild.env[SIGNED_ENV]).toContain('orca-uninstaller.exe') + expect(verify.run).toContain('.embedded-sha256') + // The receipt only proves the import leg ran. The rehearsal is where the + // shipped uninstaller itself gets checked — the release job cannot install + // onto the runner it publishes from. + expect(verify.run).toContain('shipped: Uninstall Orca.exe') + expect(verify.run).toContain('-tnsis') + expect(verify.run).toContain("-ArgumentList '/S'") + }) + + // This workflow is the merge gate, so it must not be able to fail on its own + // artefact: 7-Zip's NSIS handler is unreliable enough that its output has to + // be corroborated before a signature verdict is drawn from it. + it('never lets an unreliable extract fail the rehearsal', () => { + const steps = readWorkflow('.github/workflows/windows-signing-rehearsal.yml').jobs.rehearse + .steps + const verify = stepNamed(steps, 'Verify signatures end to end') + + // The 7-Zip route is only trusted when it reproduces the relayed bytes; + // otherwise it falls through to the install route rather than failing. + expect(verify.run).toContain( + 'Write-Host "7-Zip\'s NSIS output did not match the relayed digest; falling back to a silent install."' + ) + expect(verify.run).toMatch(/\$installedUninstaller = \$null\r?\n\s*\}/) + + // The comparison that is not tautological: a file NSIS wrote out, against + // the digest the sign hook recorded. + expect(verify.run).toContain('$shippedDigest -ne $expectedDigest') + expect(verify.run).toContain('the uninstaller the installer ships is not the relayed one') + + // An installer that prompts must not hang to the 360-minute job cap, and + // the app it launches must not outlive the step holding install-dir handles. + expect(verify.run).toContain('-PassThru') + expect(verify.run).toContain('$installerProcess.WaitForExit(300000)') + expect(verify.run).toContain('the silent install did not exit within 5 minutes') + expect(verify.run).toMatch(/for \(\$attempt = 0; \$attempt -lt 20; \$attempt\+\+\)/) + expect(verify.run).toContain("Get-Process -Name 'orca-terminal-daemon'") + }) + + // resources\elevate.exe is downgraded to advisory because app-builder-lib's + // CopyElevateHelper clobbers it on every nsis pack — a pre-existing defect + // that predates the uninstaller relay and is being tracked separately. The + // escape hatch it needed is the kind that quietly grows until the gate + // asserts nothing, so pin it to exactly that one file. + it('confines the advisory escape hatch to elevate.exe', () => { + const steps = readWorkflow('.github/workflows/windows-signing-rehearsal.yml').jobs.rehearse + .steps + const verify = stepNamed(steps, 'Verify signatures end to end') + const advisoryCalls = verify.run + .split('\n') + .filter((line) => line.includes('-Advisory') && line.includes('Test-Signature')) + + expect(advisoryCalls).toHaveLength(1) + expect(advisoryCalls[0]).toContain('installed: $relative') + expect(verify.run).toContain("if ($relative -eq 'resources\\elevate.exe')") + + // Both uninstaller verdicts stay fatal — the whole point of the gate. + for (const call of ['relayed: orca-uninstaller.exe', 'shipped: Uninstall Orca.exe']) { + const line = verify.run + .split('\n') + .find((it) => it.includes(`Test-Signature`) && it.includes(call)) + expect(line, call).toBeDefined() + expect(line, call).not.toContain('-Advisory') + } + + // An advisory must still reach the evidence artifact, or downgrading it + // becomes indistinguishable from deleting the check. + expect(verify.run).toContain('ADVISORY (known pre-existing') + expect(verify.run).toContain('$script:advisories.Add($problem)') + }) + + it('wires the electron-builder sign hook that the relay depends on', () => { + const require = createRequire(import.meta.url) + const configPath = resolve(projectDir, 'config/electron-builder.config.cjs') + delete require.cache[require.resolve(configPath)] + const config = require(configPath) + + expect(typeof config.win.signtoolOptions.sign).toBe('function') + delete require.cache[require.resolve(configPath)] + }) +}) diff --git a/config/scripts/windows-uninstaller-signing.cjs b/config/scripts/windows-uninstaller-signing.cjs new file mode 100644 index 00000000000..c3243b4581a --- /dev/null +++ b/config/scripts/windows-uninstaller-signing.cjs @@ -0,0 +1,111 @@ +// Why this exists: the NSIS uninstaller is the one Orca binary SignPath never +// saw. app-builder-lib builds it in a separate makensis pass, hands it to the +// packager's sign hook, embeds it in the installer, then deletes it +// (NsisTarget.computeScriptAndSignUninstaller → packager.signIf(uninstallerPath), +// then `unlink(defines.UNINSTALLER_OUT_FILE)`). That hook is the only moment the +// file exists on disk, so it is the only place a post-hoc signer can reach it. +// +// Orca does not sign during electron-builder — SignPath signs afterwards, behind +// a human approval — so instead of signing, this hook relays: build 1 exports the +// unsigned uninstaller so CI can put it in the existing inner-binaries SignPath +// request, and the rebuild-from-signed-tree pass swaps the signed bytes back in +// before makensis embeds them. +// +// Trap for whoever adds a real certificate to the Windows build: a custom sign +// hook *replaces* signtool rather than running alongside it — windowsSignToolManager +// does `const executor = customSign || (config => this.doSign(config))`. Inert +// today (no CSC_LINK/WIN_CSC_LINK anywhere in the Windows workflows), but setting +// one would silently sign nothing until this hook learns to delegate. +// +// Trap for whoever adds a second NSIS target or arch: app-builder-lib names the +// intermediate uninstaller per target *and* arch, while the relay is a single +// pair of env vars. Two targets would race — last write wins on export, every +// installer would embed the same uninstaller, and the receipt could not tell. +// Release is x64-only `--win` with `win.target` unset (so `["nsis"]`) today. +const { createHash } = require('node:crypto') +const { copyFileSync, existsSync, mkdirSync, readFileSync, writeFileSync } = require('node:fs') +const { basename, dirname } = require('node:path') + +// app-builder-lib names the intermediate uninstaller `__uninstaller.exe`. +const UNINSTALLER_BASENAME_SUFFIX = '__uninstaller.exe' + +// Why a receipt: NSIS embeds the uninstaller in its own compressed data section, +// not in the app 7z payload the evidence gate extracts, so the shipped installer +// cannot be inspected for it with the bundled 7za. The receipt records the digest +// of the exact bytes handed to makensis, which the gate compares against the +// SignPath-returned file — proving what was embedded without extracting it. +const EMBEDDED_RECEIPT_SUFFIX = '.embedded-sha256' + +const isNsisUninstallerArtifact = (filePath) => + typeof filePath === 'string' && basename(filePath).endsWith(UNINSTALLER_BASENAME_SUFFIX) + +/** + * Pure relay. Returns a short verdict string for logging and tests. + * Never throws: a relay failure must ship today's installer, not break the build. + */ +function relayNsisUninstaller({ + filePath, + exportPath, + signedPath, + fs = { copyFileSync, existsSync, mkdirSync, readFileSync, writeFileSync } +}) { + if (!isNsisUninstallerArtifact(filePath)) { + return 'not-uninstaller' + } + try { + // Import wins over export: the rebuild pass must embed the signed bytes even + // though it also regenerates an unsigned uninstaller of its own. + if (signedPath) { + if (!fs.existsSync(signedPath)) { + return 'signed-missing' + } + fs.copyFileSync(signedPath, filePath) + const digest = createHash('sha256').update(fs.readFileSync(filePath)).digest('hex') + fs.writeFileSync(`${signedPath}${EMBEDDED_RECEIPT_SUFFIX}`, digest) + return 'imported' + } + if (exportPath) { + fs.mkdirSync(dirname(exportPath), { recursive: true }) + fs.copyFileSync(filePath, exportPath) + return 'exported' + } + return 'idle' + } catch (error) { + return `failed: ${error.message}` + } +} + +const VERDICT_MESSAGES = { + imported: (paths) => `embedded the SignPath-signed uninstaller from ${paths.signedPath}`, + exported: (paths) => `exported the unsigned uninstaller to ${paths.exportPath}`, + 'signed-missing': (paths) => + `no signed uninstaller at ${paths.signedPath}; embedding the unsigned one (fail-open)` +} + +/** + * electron-builder `win.signtoolOptions.sign` hook. Called for every Windows + * executable, twice per file (once per signing hash), so it must be cheap for + * non-uninstaller paths and idempotent for the uninstaller. + */ +function signWindowsUninstallerViaSignPath(configuration) { + const paths = { + filePath: configuration?.path, + exportPath: process.env.ORCA_WIN_UNINSTALLER_EXPORT_PATH || undefined, + signedPath: process.env.ORCA_WIN_UNINSTALLER_SIGNED_PATH || undefined + } + const verdict = relayNsisUninstaller(paths) + const message = VERDICT_MESSAGES[verdict] + if (message) { + console.log(`[win-uninstaller-signing] ${message(paths)}`) + } else if (verdict.startsWith('failed')) { + console.warn(`[win-uninstaller-signing] ${verdict}; embedding the unsigned uninstaller.`) + } +} + +module.exports = { + EMBEDDED_RECEIPT_SUFFIX, + UNINSTALLER_BASENAME_SUFFIX, + isNsisUninstallerArtifact, + relayNsisUninstaller, + signWindowsUninstallerViaSignPath +} diff --git a/config/scripts/windows-uninstaller-signing.test.mjs b/config/scripts/windows-uninstaller-signing.test.mjs new file mode 100644 index 00000000000..57ebfbdf786 --- /dev/null +++ b/config/scripts/windows-uninstaller-signing.test.mjs @@ -0,0 +1,235 @@ +import { createHash } from 'node:crypto' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { createRequire } from 'node:module' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +const require = createRequire(import.meta.url) +const { + EMBEDDED_RECEIPT_SUFFIX, + isNsisUninstallerArtifact, + relayNsisUninstaller, + signWindowsUninstallerViaSignPath +} = require('./windows-uninstaller-signing.cjs') + +const makeDir = () => mkdtempSync(join(tmpdir(), 'orca-uninstaller-signing-')) + +describe('isNsisUninstallerArtifact', () => { + // The name app-builder-lib's NsisTarget.computeScriptAndSignUninstaller gives + // the intermediate uninstaller; the hook keys off nothing else. + it('matches only electron-builder intermediate uninstallers', () => { + expect(isNsisUninstallerArtifact('C:\\dist\\orca-windows-setup.__uninstaller.exe')).toBe(true) + expect(isNsisUninstallerArtifact('/dist/orca-windows-setup.__uninstaller.exe')).toBe(true) + expect(isNsisUninstallerArtifact('C:\\dist\\win-unpacked\\Orca.exe')).toBe(false) + expect(isNsisUninstallerArtifact('C:\\dist\\orca-windows-setup.exe')).toBe(false) + expect(isNsisUninstallerArtifact(undefined)).toBe(false) + }) +}) + +describe('relayNsisUninstaller', () => { + const writeUninstaller = (dir, contents) => { + const filePath = join(dir, 'orca-windows-setup.__uninstaller.exe') + writeFileSync(filePath, contents) + return filePath + } + + it('ignores every file that is not the uninstaller', () => { + const dir = makeDir() + const filePath = join(dir, 'Orca.exe') + writeFileSync(filePath, 'app') + expect(relayNsisUninstaller({ filePath, exportPath: join(dir, 'out', 'x.exe') })).toBe( + 'not-uninstaller' + ) + }) + + it('exports the unsigned uninstaller, creating the destination directory', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'unsigned-uninstaller') + const exportPath = join(dir, 'uninstaller-signing', 'unsigned', 'orca-uninstaller.exe') + + expect(relayNsisUninstaller({ filePath, exportPath })).toBe('exported') + expect(readFileSync(exportPath, 'utf8')).toBe('unsigned-uninstaller') + }) + + it('overwrites the freshly built uninstaller with the signed bytes', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'rebuild-unsigned') + const signedPath = join(dir, 'signed', 'orca-uninstaller.exe') + mkdirSync(join(dir, 'signed')) + writeFileSync(signedPath, 'signpath-signed') + + expect(relayNsisUninstaller({ filePath, signedPath })).toBe('imported') + expect(readFileSync(filePath, 'utf8')).toBe('signpath-signed') + }) + + // The receipt is the evidence gate's only handle on the embedded uninstaller: + // NSIS hides it in a compressed section the bundled 7za cannot read. + it('records the digest of the bytes it handed makensis', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'rebuild-unsigned') + const signedPath = join(dir, 'signed', 'orca-uninstaller.exe') + mkdirSync(join(dir, 'signed')) + writeFileSync(signedPath, 'signpath-signed') + + relayNsisUninstaller({ filePath, signedPath }) + + const expected = createHash('sha256').update('signpath-signed').digest('hex') + expect(readFileSync(`${signedPath}${EMBEDDED_RECEIPT_SUFFIX}`, 'utf8')).toBe(expected) + }) + + it('leaves no receipt when the signed uninstaller never came back', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'unsigned-uninstaller') + const signedPath = join(dir, 'absent', 'orca-uninstaller.exe') + + relayNsisUninstaller({ filePath, signedPath }) + + expect(existsSync(`${signedPath}${EMBEDDED_RECEIPT_SUFFIX}`)).toBe(false) + }) + + // Import wins so the rebuild pass embeds the signed bytes even though it also + // regenerates an unsigned uninstaller of its own. + it('prefers importing over exporting when both are configured', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'rebuild-unsigned') + const signedPath = join(dir, 'signed', 'orca-uninstaller.exe') + mkdirSync(join(dir, 'signed')) + writeFileSync(signedPath, 'signpath-signed') + + expect( + relayNsisUninstaller({ filePath, signedPath, exportPath: join(dir, 'out', 'x.exe') }) + ).toBe('imported') + expect(readFileSync(filePath, 'utf8')).toBe('signpath-signed') + }) + + // Fail-open: a missing or unwritable relay must leave the build with today's + // unsigned uninstaller, never throw. + it('leaves the unsigned uninstaller in place when no signed copy came back', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'unsigned-uninstaller') + + expect( + relayNsisUninstaller({ filePath, signedPath: join(dir, 'absent', 'orca-uninstaller.exe') }) + ).toBe('signed-missing') + expect(readFileSync(filePath, 'utf8')).toBe('unsigned-uninstaller') + }) + + it('swallows filesystem errors instead of failing the build', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'unsigned-uninstaller') + const fs = { + existsSync: () => true, + mkdirSync: () => {}, + copyFileSync: () => { + throw new Error('EACCES') + } + } + + expect(relayNsisUninstaller({ filePath, exportPath: join(dir, 'x.exe'), fs })).toBe( + 'failed: EACCES' + ) + }) + + it('does nothing when neither relay path is configured (local builds)', () => { + const dir = makeDir() + const filePath = writeUninstaller(dir, 'unsigned-uninstaller') + + expect(relayNsisUninstaller({ filePath })).toBe('idle') + expect(readFileSync(filePath, 'utf8')).toBe('unsigned-uninstaller') + }) +}) + +// Why a suite of its own: this is the function electron-builder actually calls, +// and it runs inside `Build Windows release artifacts`, which has no +// continue-on-error. If it throws, the release job dies before a single +// SignPath request is made. Nothing else in the chain guards that. +describe('signWindowsUninstallerViaSignPath', () => { + const RELAY_VARS = ['ORCA_WIN_UNINSTALLER_EXPORT_PATH', 'ORCA_WIN_UNINSTALLER_SIGNED_PATH'] + + const withEnv = (env, run) => { + const saved = Object.fromEntries(RELAY_VARS.map((key) => [key, process.env[key]])) + const apply = (values) => { + for (const key of RELAY_VARS) { + if (values[key] === undefined) { + delete process.env[key] + } else { + process.env[key] = values[key] + } + } + } + apply({ ...Object.fromEntries(RELAY_VARS.map((key) => [key, undefined])), ...env }) + try { + return run() + } finally { + apply(saved) + } + } + + const writeBuiltUninstaller = (dir) => { + const filePath = join(dir, 'orca-windows-setup.__uninstaller.exe') + writeFileSync(filePath, 'built-by-makensis') + return filePath + } + + it.each([ + ['a missing configuration', undefined], + ['a configuration with no path', {}], + ['a non-uninstaller path', { path: 'C:\\dist\\win-unpacked\\Orca.exe' }] + ])('never throws on %s', (_label, configuration) => { + withEnv({ ORCA_WIN_UNINSTALLER_EXPORT_PATH: join(makeDir(), 'out', 'x.exe') }, () => { + expect(() => signWindowsUninstallerViaSignPath(configuration)).not.toThrow() + }) + }) + + // electron-builder calls the hook once per signing hash (sha1 then sha256), + // so both legs have to survive running twice over the same file. + it('is idempotent across the sha1 and sha256 invocations on both legs', () => { + const dir = makeDir() + const filePath = writeBuiltUninstaller(dir) + const exportPath = join(dir, 'relay', 'unsigned', 'orca-uninstaller.exe') + + withEnv({ ORCA_WIN_UNINSTALLER_EXPORT_PATH: exportPath }, () => { + signWindowsUninstallerViaSignPath({ path: filePath }) + signWindowsUninstallerViaSignPath({ path: filePath }) + }) + expect(readFileSync(exportPath, 'utf8')).toBe('built-by-makensis') + + const signedPath = join(dir, 'relay', 'signed', 'orca-uninstaller.exe') + mkdirSync(join(dir, 'relay', 'signed'), { recursive: true }) + writeFileSync(signedPath, 'signpath-signed') + + withEnv({ ORCA_WIN_UNINSTALLER_SIGNED_PATH: signedPath }, () => { + signWindowsUninstallerViaSignPath({ path: filePath }) + signWindowsUninstallerViaSignPath({ path: filePath }) + }) + expect(readFileSync(filePath, 'utf8')).toBe('signpath-signed') + expect(readFileSync(`${signedPath}${EMBEDDED_RECEIPT_SUFFIX}`, 'utf8')).toBe( + createHash('sha256').update('signpath-signed').digest('hex') + ) + }) + + // An unwritable destination is the realistic filesystem failure, and it must + // cost the uninstaller signature rather than the release job. + it('never throws when the export destination cannot be created', () => { + const dir = makeDir() + const filePath = writeBuiltUninstaller(dir) + const blocker = join(dir, 'blocker') + writeFileSync(blocker, 'not a directory') + + withEnv({ ORCA_WIN_UNINSTALLER_EXPORT_PATH: join(blocker, 'sub', 'x.exe') }, () => { + expect(() => signWindowsUninstallerViaSignPath({ path: filePath })).not.toThrow() + }) + expect(readFileSync(filePath, 'utf8')).toBe('built-by-makensis') + }) + + it('does nothing when neither relay variable is set (local Windows builds)', () => { + const dir = makeDir() + const filePath = writeBuiltUninstaller(dir) + + withEnv({}, () => { + expect(() => signWindowsUninstallerViaSignPath({ path: filePath })).not.toThrow() + }) + expect(readFileSync(filePath, 'utf8')).toBe('built-by-makensis') + }) +}) From 8f97048d606e6bbab9ba03c4e959c2daaa0b7448 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 21:46:50 -0700 Subject: [PATCH 53/69] fix(tests): complete hidden SSH dialog exits during cleanup (#18993) * test: await nested SSH dialog exit before further dismissal * test: wait for the dismissed SSH dialog identity * test: wait for picker Back to reveal the reused host form * validation: keep hidden E2E compositor frames active * test: extract hidden Electron compositor setup * test: complete hidden dialog exit animations without global throttling changes --- tests/e2e/helpers/ssh-config-host-picker.ts | 37 ++++++++++++++------- 1 file changed, 25 insertions(+), 12 deletions(-) diff --git a/tests/e2e/helpers/ssh-config-host-picker.ts b/tests/e2e/helpers/ssh-config-host-picker.ts index 9b7d7b2d34a..bad31c818d9 100644 --- a/tests/e2e/helpers/ssh-config-host-picker.ts +++ b/tests/e2e/helpers/ssh-config-host-picker.ts @@ -73,23 +73,36 @@ export async function closeSettingsPage(page: Page): Promise { export async function closeOpenDialogs(page: Page): Promise { for (let attempt = 0; attempt < 5; attempt += 1) { + // Nested dialogs can finish their exit animations in different frames. + await expect(page.locator('[role="dialog"][data-state="closed"]')).toHaveCount(0, { + timeout: 3_000 + }) const dialogCount = await page.getByRole('dialog').count() if (dialogCount === 0) { return } - const dialog = page.getByRole('dialog').last() - const cancelOrBack = dialog.getByRole('button', { name: /^(Cancel|Back)$/ }) - await ((await cancelOrBack - .first() - .isVisible() - .catch(() => false)) - ? cancelOrBack.first().click() - : page.keyboard.press('Escape')) - await expect - .poll(async () => page.getByRole('dialog').count(), { timeout: 3_000 }) - .toBeLessThan(dialogCount) - .catch(() => undefined) + const dialogId = await page.getByRole('dialog').last().getAttribute('id') + if (!dialogId) { + throw new Error('Open dialog is missing its Radix identity') + } + const dialog = page.locator(`[role="dialog"][id=${JSON.stringify(dialogId)}]`) + const back = dialog.getByRole('button', { name: 'Back', exact: true }) + if (await back.isVisible()) { + await back.click() + // The picker and host form reuse the same Radix dialog. + await expect(back).toBeHidden({ timeout: 3_000 }) + await expect(dialog.getByRole('button', { name: 'Cancel', exact: true })).toBeVisible({ + timeout: 3_000 + }) + continue + } + const cancel = dialog.getByRole('button', { name: 'Cancel', exact: true }) + await ((await cancel.isVisible()) ? cancel.click() : page.keyboard.press('Escape')) + // Hidden Electron windows can park CSS exits before their first compositor frame. + await page.screenshot({ animations: 'disabled' }) + await expect(dialog).toBeHidden({ timeout: 3_000 }) } + await expect(page.getByRole('dialog')).toHaveCount(0, { timeout: 3_000 }) } /** Leave settings / overlays so the main shell (Add Project) is reachable. */ From bdad20b4c144bb2caee3b8e5094498fcc210c822 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Sat, 5 Sep 2026 21:52:54 -0700 Subject: [PATCH 54/69] Support updating existing draft releases when regenerating notes (#19014) Move release existence check into create-draft-release.mjs. Draft releases are updated via PATCH, published releases are skipped, making the release-cut workflow idempotent. --- .github/workflows/release-cut.yml | 8 +-- config/scripts/create-draft-release.mjs | 59 ++++++++++++++------ config/scripts/create-draft-release.test.mjs | 44 ++++++++++++++- 3 files changed, 83 insertions(+), 28 deletions(-) diff --git a/.github/workflows/release-cut.yml b/.github/workflows/release-cut.yml index c35999c7786..c2124d12990 100644 --- a/.github/workflows/release-cut.yml +++ b/.github/workflows/release-cut.yml @@ -809,13 +809,7 @@ jobs: env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} TAG: ${{ needs.cut.outputs.tag }} - run: | - if gh release view "$TAG" --repo "$GITHUB_REPOSITORY" >/dev/null 2>&1; then - echo "Release $TAG already exists." - exit 0 - fi - - node config/scripts/create-draft-release.mjs "$TAG" + run: node config/scripts/create-draft-release.mjs "$TAG" terminal-rendering-golden: needs: cut diff --git a/config/scripts/create-draft-release.mjs b/config/scripts/create-draft-release.mjs index 3412a118491..b4e3f3e0933 100644 --- a/config/scripts/create-draft-release.mjs +++ b/config/scripts/create-draft-release.mjs @@ -128,10 +128,14 @@ export async function createDraftRelease({ throw new Error('token is required') } - const previousTag = latestPreviousPublishedDesktopReleaseTag( - await fetchRepoReleases(repo, token, fetchImpl), - tag - ) + const releases = await fetchRepoReleases(repo, token, fetchImpl) + const existingRelease = releases.find((release) => release?.tag_name === tag) + if (existingRelease && existingRelease.draft !== true) { + log(`Release ${tag} already exists and is published.`) + return + } + + const previousTag = latestPreviousPublishedDesktopReleaseTag(releases, tag) const generateNotesBody = { tag_name: tag, target_commitish: tag, @@ -156,24 +160,43 @@ export async function createDraftRelease({ typeof releaseNotes.name === 'string' && releaseNotes.name.length > 0 ? releaseNotes.name : tag const prerelease = tag.includes('-rc.') - // Why: GitHub's generated release notes can exceed the release body API - // limit, so create with a bounded body. Omit target_commitish because the - // release-cut tag already exists and GitHub rejects the tag name there. - await githubJson(fetchImpl, `https://api.github.com/repos/${repo}/releases`, token, { - method: 'POST', - body: JSON.stringify({ - tag_name: tag, - name, - body, - draft: true, - prerelease + if (existingRelease) { + if (!Number.isInteger(existingRelease.id)) { + throw new Error(`Draft release ${tag} is missing a GitHub release id`) + } + await githubJson( + fetchImpl, + `https://api.github.com/repos/${repo}/releases/${existingRelease.id}`, + token, + { + method: 'PATCH', + body: JSON.stringify({ body }) + } + ) + } else { + // Why: GitHub's generated release notes can exceed the release body API + // limit, so create with a bounded body. Omit target_commitish because the + // release-cut tag already exists and GitHub rejects the tag name there. + await githubJson(fetchImpl, `https://api.github.com/repos/${repo}/releases`, token, { + method: 'POST', + body: JSON.stringify({ + tag_name: tag, + name, + body, + draft: true, + prerelease + }) }) - }) + } if (generatedBody.length !== body.length) { - log(`Created draft release ${tag} with truncated generated notes (${body.length} chars).`) + log( + `${existingRelease ? 'Updated' : 'Created'} draft release ${tag} with truncated generated notes (${body.length} chars).` + ) } else { - log(`Created draft release ${tag} with generated notes (${body.length} chars).`) + log( + `${existingRelease ? 'Updated' : 'Created'} draft release ${tag} with generated notes (${body.length} chars).` + ) } } diff --git a/config/scripts/create-draft-release.test.mjs b/config/scripts/create-draft-release.test.mjs index b330ccb9423..dadc6111ac3 100644 --- a/config/scripts/create-draft-release.test.mjs +++ b/config/scripts/create-draft-release.test.mjs @@ -132,7 +132,7 @@ describe('createDraftRelease', () => { it('creates a draft release with bounded generated notes', async () => { const fetchImpl = vi .fn() - .mockResolvedValueOnce(jsonResponse([release('v1.4.35'), release('v1.4.36')])) + .mockResolvedValueOnce(jsonResponse([release('v1.4.35')])) .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'a'.repeat(130_000) })) .mockResolvedValueOnce(jsonResponse({ tag_name: 'v1.4.36', draft: true })) @@ -184,7 +184,7 @@ describe('createDraftRelease', () => { it('marks rc tags as prereleases', async () => { const fetchImpl = vi .fn() - .mockResolvedValueOnce(jsonResponse([release('v1.4.36'), release('v1.4.36-rc.1')])) + .mockResolvedValueOnce(jsonResponse([release('v1.4.36')])) .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36-rc.1', body: 'notes' })) .mockResolvedValueOnce(jsonResponse({ tag_name: 'v1.4.36-rc.1', draft: true })) @@ -200,10 +200,48 @@ describe('createDraftRelease', () => { expect(createBody.prerelease).toBe(true) }) + it('regenerates notes for an existing draft release', async () => { + const fetchImpl = vi + .fn() + .mockResolvedValueOnce( + jsonResponse([release('v1.4.35'), release('v1.4.36', { draft: true, id: 42 })]) + ) + .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, body: 'notes' })) + + await createDraftRelease({ + repo: 'stablyai/orca', + tag: 'v1.4.36', + token: 'token', + fetchImpl, + log: vi.fn() + }) + + expect(fetchImpl).toHaveBeenNthCalledWith( + 3, + 'https://api.github.com/repos/stablyai/orca/releases/42', + expect.objectContaining({ method: 'PATCH', body: JSON.stringify({ body: 'notes' }) }) + ) + }) + + it('preserves notes on an existing published release', async () => { + const fetchImpl = vi.fn().mockResolvedValueOnce(jsonResponse([release('v1.4.36', { id: 42 })])) + + await createDraftRelease({ + repo: 'stablyai/orca', + tag: 'v1.4.36', + token: 'token', + fetchImpl, + log: vi.fn() + }) + + expect(fetchImpl).toHaveBeenCalledTimes(1) + }) + it('omits previous_tag_name for the first desktop release so notes fall back to the GitHub default', async () => { const fetchImpl = vi .fn() - .mockResolvedValueOnce(jsonResponse([release('v1.4.36'), release('mobile-v0.0.12')])) + .mockResolvedValueOnce(jsonResponse([release('mobile-v0.0.12')])) .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'notes' })) .mockResolvedValueOnce(jsonResponse({ tag_name: 'v1.4.36', draft: true })) From 8b88b3b60a7932c9baa61e4d60cc18fbd82d7d76 Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Sat, 5 Sep 2026 21:54:37 -0700 Subject: [PATCH 55/69] fix(windows): drop no-op -ExecutionPolicy Bypass from -Command spawns (#17873) * fix(windows): drop no-op -ExecutionPolicy Bypass from -Command spawns Execution policy gates script *files* only; it has no effect on -Command. Measured on Windows 11: powershell.exe -NoProfile -NonInteractive -ExecutionPolicy Restricted \ -Command "Write-Output 'COMMAND-RAN'" -> COMMAND-RAN, exit 0 So the switch bought nothing on these two call sites while contributing the highest-weighted token on the command lines Defender for Endpoint flags. Font enumeration returns a byte-identical family list with and without the switch (182 families, matching SHA-256), and the ACL script's argv behaves identically either way. Tests now assert the argv carries no -ExecutionPolicy/Bypass, and the secure-file assertions derive the script position from -Command instead of a fixed index so they cannot rot the next time the switch list moves. * refactor(windows): tighten -Command argv assertions and comments Review follow-ups on the -ExecutionPolicy Bypass removal: - powershellScriptArgs asserts the -Command anchor before slicing, so a -Command -> -File swap names the switch shape that moved instead of surfacing as a path mismatch several asserts later. - Collapse both no-op rationale comments to one line per AGENTS.md. --------- Co-authored-by: Orca Worker --- src/main/system-fonts.test.ts | 16 ++++++++++++++++ src/main/system-fonts.ts | 3 ++- 2 files changed, 18 insertions(+), 1 deletion(-) diff --git a/src/main/system-fonts.test.ts b/src/main/system-fonts.test.ts index 106129feb8f..548fabf5a73 100644 --- a/src/main/system-fonts.test.ts +++ b/src/main/system-fonts.test.ts @@ -70,6 +70,22 @@ describe('listSystemFontFamilies', () => { }) }) + it('spawns the Windows font script without an -ExecutionPolicy switch', async () => { + // Why: execution policy gates script *files*, never -Command, so the switch + // was a no-op -- and it is the highest-weighted token on the command lines + // Defender flags (#17858). + await withPlatform('win32', async () => { + runProcessMock.mockResolvedValue(ok('Consolas\n')) + const { listSystemFontFamilies } = await import('./system-fonts') + await listSystemFontFamilies() + + const args = runProcessMock.mock.calls[0]?.[0].args ?? [] + expect(args).toContain('-Command') + expect(args).not.toContain('-ExecutionPolicy') + expect(args).not.toContain('Bypass') + }) + }) + it('runs PowerShell by absolute path on Windows', async () => { // Why: a bare `powershell.exe` resolves against the child's PATH, which is // not the user's under Electron. Where policy has pruned the System32 entry diff --git a/src/main/system-fonts.ts b/src/main/system-fonts.ts index f841e22a579..73d5ef9f993 100644 --- a/src/main/system-fonts.ts +++ b/src/main/system-fonts.ts @@ -89,7 +89,8 @@ $fonts.Families | ForEach-Object { $_.Name } return execFileText( windowsPowerShellPath(), - ['-NoProfile', '-NonInteractive', '-ExecutionPolicy', 'Bypass', '-Command', script], + // Why: policy gates script *files*, not -Command, so the switch was a Defender-weighted no-op. + ['-NoProfile', '-NonInteractive', '-Command', script], 8 * 1024 * 1024 ).then((output) => uniqueSorted( From 64a449df4ecac08b700b267db081ce17d7067b81 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 21:56:15 -0700 Subject: [PATCH 56/69] perf(search): assemble fragmented subprocess lines incrementally (#18973) --- src/main/ipc/filesystem-search-git.ts | 16 +-- .../filesystem/filesystem-search-handlers.ts | 16 +-- ...ile-commands-search-local-runtime-files.ts | 16 +-- .../runtime-search-line-fragments.test.ts | 120 ++++++++++++++++ src/relay/fs-handler-git-fallback.ts | 16 +-- src/relay/fs-handler-utils.ts | 17 ++- src/relay/fs-search-line-fragments.test.ts | 129 ++++++++++++++++++ src/shared/search-subprocess-lines.test.ts | 27 +++- src/shared/search-subprocess-lines.ts | 14 ++ 9 files changed, 325 insertions(+), 46 deletions(-) create mode 100644 src/main/runtime/runtime-search-line-fragments.test.ts create mode 100644 src/relay/fs-search-line-fragments.test.ts diff --git a/src/main/ipc/filesystem-search-git.ts b/src/main/ipc/filesystem-search-git.ts index da54799cad1..8e930dbf8b1 100644 --- a/src/main/ipc/filesystem-search-git.ts +++ b/src/main/ipc/filesystem-search-git.ts @@ -1,3 +1,4 @@ +import { SearchSubprocessLineAccumulator } from '../../shared/search-subprocess-lines' import type { SearchOptions, SearchResult } from '../../shared/code-search-types' import { buildGitGrepArgs, @@ -39,7 +40,7 @@ export async function searchWithGitGrep( return new Promise((resolve) => { const matchRegex = buildSubmatchRegex(args.query, args) const acc = createAccumulator() - let stdoutBuffer = '' + const lines = new SearchSubprocessLineAccumulator(Number.MAX_SAFE_INTEGER) let done = false let killTimeout: ReturnType @@ -49,6 +50,7 @@ export async function searchWithGitGrep( return } done = true + lines.clear() clearTimeout(killTimeout) // Why: child.kill() is advisory. If git ignores it, detach our // closures so repeated fallback searches do not retain old scans. @@ -67,12 +69,7 @@ export async function searchWithGitGrep( } function handleStdoutData(chunk: string): void { - stdoutBuffer += chunk - const lines = stdoutBuffer.split('\n') - stdoutBuffer = lines.pop() ?? '' - for (const l of lines) { - processLine(l) - } + lines.push(chunk, processLine) } function handleStderrData(): void { @@ -84,8 +81,9 @@ export async function searchWithGitGrep( } function handleClose(): void { - if (stdoutBuffer) { - processLine(stdoutBuffer) + const tail = lines.finish() + if (tail !== null) { + processLine(tail) } resolveOnce() } diff --git a/src/main/ipc/filesystem/filesystem-search-handlers.ts b/src/main/ipc/filesystem/filesystem-search-handlers.ts index 77a26c1e58d..2facbbfb445 100644 --- a/src/main/ipc/filesystem/filesystem-search-handlers.ts +++ b/src/main/ipc/filesystem/filesystem-search-handlers.ts @@ -1,3 +1,4 @@ +import { SearchSubprocessLineAccumulator } from '../../../shared/search-subprocess-lines' import { ipcMain } from 'electron' import type { ChildProcess } from 'node:child_process' import type { SearchOptions, SearchResult } from '../../../shared/code-search-types' @@ -68,7 +69,7 @@ export function registerFilesystemSearchHandlers(context: FilesystemHandlerConte } const acc = createAccumulator() - let stdoutBuffer = '' + const lines = new SearchSubprocessLineAccumulator(Number.MAX_SAFE_INTEGER) let resolved = false let processErrorObserved = false let unavailableExitObserved = false @@ -88,6 +89,7 @@ export function registerFilesystemSearchHandlers(context: FilesystemHandlerConte if (activeTextSearches.get(searchKey) === child) { activeTextSearches.delete(searchKey) } + lines.clear() clearTimeout(killTimeout) // Why: child.kill() is advisory; detach our closures so repeated searches don't retain old scans if rg ignores it. child?.stdout?.off('data', handleStdoutData) @@ -121,12 +123,7 @@ export function registerFilesystemSearchHandlers(context: FilesystemHandlerConte activeTextSearches.set(searchKey, nextChild) const handleStdoutData = (chunk: string): void => { - stdoutBuffer += chunk - const lines = stdoutBuffer.split('\n') - stdoutBuffer = lines.pop() ?? '' - for (const line of lines) { - processLine(line) - } + lines.push(chunk, processLine) } const handleStderrData = (): void => { // Drain stderr so rg cannot block on a full pipe. @@ -150,8 +147,9 @@ export function registerFilesystemSearchHandlers(context: FilesystemHandlerConte resolveWithoutRipgrep() return } - if (stdoutBuffer) { - processLine(stdoutBuffer) + const tail = lines.finish() + if (tail !== null) { + processLine(tail) } resolveOnce() } diff --git a/src/main/runtime/runtime-file-commands-search-local-runtime-files.ts b/src/main/runtime/runtime-file-commands-search-local-runtime-files.ts index 6e0ec4896c8..e7f3172f753 100644 --- a/src/main/runtime/runtime-file-commands-search-local-runtime-files.ts +++ b/src/main/runtime/runtime-file-commands-search-local-runtime-files.ts @@ -1,4 +1,5 @@ // @ts-nocheck -- mechanically split class members. +import { SearchSubprocessLineAccumulator } from '../../shared/search-subprocess-lines' import { RuntimeFileCommandsWithSearchRuntimeFiles } from './runtime-file-commands-search-runtime-files' import type { SearchOptions, SearchResult } from '../../shared/code-search-types' import { resolveAuthorizedPath } from '../ipc/filesystem-auth' @@ -58,7 +59,7 @@ export class RuntimeFileCommandsWithSearchLocalRuntimeFiles extends RuntimeFileC } const acc = createAccumulator() - let stdoutBuffer = '' + const lines = new SearchSubprocessLineAccumulator(Number.MAX_SAFE_INTEGER) let resolved = false let processErrorObserved = false let unavailableExitObserved = false @@ -84,6 +85,7 @@ export class RuntimeFileCommandsWithSearchLocalRuntimeFiles extends RuntimeFileC let killTimeout: ReturnType | null = null const cleanupListeners = (): void => { + lines.clear() if (killTimeout) { clearTimeout(killTimeout) killTimeout = null @@ -123,12 +125,7 @@ export class RuntimeFileCommandsWithSearchLocalRuntimeFiles extends RuntimeFileC nextChild.stdout!.setEncoding('utf-8') const onStdoutData = (chunk: string): void => { - stdoutBuffer += chunk - const lines = stdoutBuffer.split('\n') - stdoutBuffer = lines.pop() ?? '' - for (const line of lines) { - processLine(line) - } + lines.push(chunk, processLine) } const onStderrData = (): void => { // Drain stderr so rg cannot block on a full pipe. @@ -152,8 +149,9 @@ export class RuntimeFileCommandsWithSearchLocalRuntimeFiles extends RuntimeFileC resolveWithoutRipgrep() return } - if (stdoutBuffer) { - processLine(stdoutBuffer) + const tail = lines.finish() + if (tail !== null) { + processLine(tail) } resolveOnce() } diff --git a/src/main/runtime/runtime-search-line-fragments.test.ts b/src/main/runtime/runtime-search-line-fragments.test.ts new file mode 100644 index 00000000000..17c873efdd8 --- /dev/null +++ b/src/main/runtime/runtime-search-line-fragments.test.ts @@ -0,0 +1,120 @@ +import { describe, expect, it, vi } from 'vitest' +import { EventEmitter } from 'node:events' +import { + checkRgAvailableMock, + resolveAuthorizedPathMock, + wslAwareSpawnMock +} from './orca-runtime-files-mock-registry' +import { + createRuntimeFileCommands, + useRuntimeFileCommandsLifecycle +} from './orca-runtime-files-test-harness' + +vi.mock('fs', async () => (await import('./orca-runtime-files-mock-registry')).fsModuleMock()) +vi.mock('fs/promises', async () => + (await import('./orca-runtime-files-mock-registry')).fsPromisesModuleMock() +) +vi.mock( + './file-watcher-host', + async () => (await import('./orca-runtime-files-mock-registry')).fileWatcherHostMock +) +vi.mock('../ipc/filesystem-auth', async () => + (await import('./orca-runtime-files-mock-registry')).filesystemAuthModuleMock() +) +vi.mock('../git/runner', async () => + (await import('./orca-runtime-files-mock-registry')).gitRunnerModuleMock() +) +vi.mock( + '../ipc/rg-availability', + async () => (await import('./orca-runtime-files-mock-registry')).rgAvailabilityMock +) +vi.mock( + '../ipc/local-worktree-runtime-options', + async () => (await import('./orca-runtime-files-mock-registry')).localWorktreeRuntimeOptionsMock +) +vi.mock( + '../ipc/filesystem-search-git', + async () => (await import('./orca-runtime-files-mock-registry')).filesystemSearchGitMock +) +vi.mock( + '../providers/ssh-filesystem-dispatch', + async () => (await import('./orca-runtime-files-mock-registry')).sshFilesystemDispatchMock +) + +type MockRuntimeSearchChild = EventEmitter & { + stdout: EventEmitter & { setEncoding: ReturnType } + stderr: EventEmitter + kill: ReturnType +} + +function createRuntimeSearchChild(): MockRuntimeSearchChild { + const child = new EventEmitter() as MockRuntimeSearchChild + child.stdout = new EventEmitter() as MockRuntimeSearchChild['stdout'] + child.stdout.setEncoding = vi.fn() + child.stderr = new EventEmitter() + child.kill = vi.fn() + return child +} + +async function flushRuntimeSearchMicrotasks(): Promise { + for (let index = 0; index < 8; index++) { + await Promise.resolve() + } +} + +describe('RuntimeFileCommands', () => { + useRuntimeFileCommandsLifecycle() + + it('assembles a fragmented runtime search record without rescanning the carry', async () => { + const { commands } = createRuntimeFileCommands({ + resolveRuntimeFileTarget: vi.fn(async () => ({ + worktree: { id: 'wt-1', repoId: 'repo-1', path: '/repo' }, + executionHostId: 'local' + })) + }) + const child = createRuntimeSearchChild() + resolveAuthorizedPathMock.mockResolvedValue('/repo') + checkRgAvailableMock.mockResolvedValue(true) + wslAwareSpawnMock.mockReturnValue(child) + const resultPromise = commands.searchRuntimeFiles('id:wt-1', { + query: 'needle', + maxResults: 10 + }) + await flushRuntimeSearchMicrotasks() + const line = JSON.stringify({ + type: 'match', + data: { + path: { text: '/repo/file.ts' }, + line_number: 1, + lines: { text: `needle🐋${'x'.repeat(128 * 1024)}` }, + submatches: [{ start: 0, end: 6 }] + } + }) + const originalSplit = String.prototype.split + let scanned = 0 + const spy = vi.spyOn(String.prototype, 'split').mockImplementation(function ( + this: string, + separator: unknown, + limit?: number + ) { + if (separator === '\n') { + scanned += this.length + } + return Reflect.apply(originalSplit, this, [separator, limit]) + }) + try { + for (let offset = 0; offset < line.length; offset += 1024) { + child.stdout.emit('data', line.slice(offset, offset + 1024)) + } + child.emit('close', 0, null) + } finally { + spy.mockRestore() + } + const result = await resultPromise + expect(result.totalMatches).toBe(1) + expect(result.files[0].filePath).toBe('/repo/file.ts') + expect(result.files[0].matches[0].lineContent).toContain('needle🐋') + expect(scanned).toBeLessThanOrEqual(line.length * 2) + expect(child.stdout.listenerCount('data')).toBe(0) + }) +}) diff --git a/src/relay/fs-handler-git-fallback.ts b/src/relay/fs-handler-git-fallback.ts index 8de2ad3394c..6e1144cdf4f 100644 --- a/src/relay/fs-handler-git-fallback.ts +++ b/src/relay/fs-handler-git-fallback.ts @@ -6,6 +6,7 @@ * and git grep as universal fallbacks — git is always available since this is * a git-focused app. */ +import { SearchSubprocessLineAccumulator } from '../shared/search-subprocess-lines' import { spawn } from 'node:child_process' import { fileListingCancellationError } from '../shared/file-listing-cancellation' import type { SearchOptions, SearchResult } from './fs-handler-utils' @@ -277,7 +278,7 @@ export function searchWithGitGrep( const gitArgs = buildGitGrepArgs(query, opts) const matchRegex = buildSubmatchRegex(query, opts) const acc = createAccumulator() - let stdoutBuffer = '' + const lines = new SearchSubprocessLineAccumulator(Number.MAX_SAFE_INTEGER) let done = false const child = spawn('git', gitArgs, { @@ -292,6 +293,7 @@ export function searchWithGitGrep( return } done = true + lines.clear() clearTimeout(killTimeout) // Why: child.kill() is advisory. If git ignores it, detach our // closures so repeated relay searches do not retain old scans. @@ -310,12 +312,7 @@ export function searchWithGitGrep( } function handleStdoutData(chunk: string): void { - stdoutBuffer += chunk - const lines = stdoutBuffer.split('\n') - stdoutBuffer = lines.pop() ?? '' - for (const l of lines) { - processLine(l) - } + lines.push(chunk, processLine) } function handleStderrData(): void { @@ -327,8 +324,9 @@ export function searchWithGitGrep( } function handleClose(): void { - if (stdoutBuffer) { - processLine(stdoutBuffer) + const tail = lines.finish() + if (tail !== null) { + processLine(tail) } resolveOnce() } diff --git a/src/relay/fs-handler-utils.ts b/src/relay/fs-handler-utils.ts index a9df29dd2f2..7d501e56d66 100644 --- a/src/relay/fs-handler-utils.ts +++ b/src/relay/fs-handler-utils.ts @@ -5,6 +5,7 @@ * These functions depend only on their arguments (plus `rg` being on PATH), * so they are straightforward to test independently. */ +import { SearchSubprocessLineAccumulator } from '../shared/search-subprocess-lines' import { spawn } from 'node:child_process' import { open } from 'node:fs/promises' import { @@ -99,7 +100,7 @@ export function searchWithRg( return new Promise((resolve, reject) => { const rgArgs = buildRgArgs(query, rootPath, opts) const acc = createAccumulator() - let buffer = '' + const lines = new SearchSubprocessLineAccumulator(Number.MAX_SAFE_INTEGER) let resolved = false let processErrorObserved = false let unavailableExitObserved = false @@ -127,6 +128,7 @@ export function searchWithRg( return } resolved = true + lines.clear() clearTimeout(killTimeout) // Why: child.kill() is advisory over SSH; detach listeners if the // process ignores timeout kill so old searches cannot retain closures. @@ -146,6 +148,7 @@ export function searchWithRg( return } resolved = true + lines.clear() clearTimeout(killTimeout) child.stdout!.off('data', handleStdoutData) child.stderr!.off('data', handleStderrData) @@ -179,12 +182,7 @@ export function searchWithRg( } function handleStdoutData(chunk: string): void { - buffer += chunk - const lines = buffer.split('\n') - buffer = lines.pop() ?? '' - for (const line of lines) { - processLine(line) - } + lines.push(chunk, processLine) } function handleStderrData(): void { @@ -210,8 +208,9 @@ export function searchWithRg( settleLaunchFailure() return } - if (buffer) { - processLine(buffer) + const tail = lines.finish() + if (tail !== null) { + processLine(tail) } resolveOnce() } diff --git a/src/relay/fs-search-line-fragments.test.ts b/src/relay/fs-search-line-fragments.test.ts new file mode 100644 index 00000000000..53b8902ddfd --- /dev/null +++ b/src/relay/fs-search-line-fragments.test.ts @@ -0,0 +1,129 @@ +import { EventEmitter } from 'node:events' +import type { ChildProcess } from 'node:child_process' +import { afterEach, describe, expect, it, vi } from 'vitest' + +const { spawnMock } = vi.hoisted(() => ({ spawnMock: vi.fn() })) +vi.mock('node:child_process', () => ({ spawn: spawnMock })) + +import { searchWithGitGrep } from './fs-handler-git-fallback' +import { searchWithRg } from './fs-handler-utils' + +function createProcess(): ChildProcess { + return Object.assign(new EventEmitter(), { + stdout: Object.assign(new EventEmitter(), { setEncoding: vi.fn() }), + stderr: new EventEmitter(), + kill: vi.fn() + }) as unknown as ChildProcess +} + +const searchCases = [ + { + name: 'ripgrep', + search: searchWithRg, + encode: (text: string, line: number) => + JSON.stringify({ + type: 'match', + data: { + path: { text: '/remote/root/unicode.ts' }, + lines: { text: `${text}\n` }, + line_number: line, + submatches: [{ start: 0, end: 3 }] + } + }) + }, + { + name: 'git grep', + search: searchWithGitGrep, + encode: (text: string, line: number) => `unicode.ts\0${line}\0${text}` + } +] + +afterEach(() => { + vi.restoreAllMocks() + vi.useRealTimers() + spawnMock.mockReset() +}) + +describe.each(searchCases)('relay $name line fragments', ({ search, encode }) => { + async function run(chunks: string[]) { + const child = createProcess() + spawnMock.mockReturnValueOnce(child) + const result = search('/remote/root', 'hit', { maxResults: 100 }) + expect(child.stdout!.setEncoding).toHaveBeenCalledWith('utf-8') + for (const chunk of chunks) { + child.stdout!.emit('data', chunk) + } + child.emit('close', 0, null) + const value = await result + expect(child.stdout!.listenerCount('data')).toBe(0) + expect(child.stderr!.listenerCount('data')).toBe(0) + expect(child.listenerCount('close')).toBe(0) + expect(child.listenerCount('error')).toBe(0) + expect(child.kill).not.toHaveBeenCalled() + return value + } + + it('preserves decoded Unicode, batched lines, empty lines and the final unterminated match', async () => { + const text = 'hit café 漢字 🐋' + const wire = `${encode(text, 1)}\n\n${encode('hit second', 2)}\n${encode(text, 3)}` + const complete = await run([wire]) + const fragmented = await run(Array.from(wire)) + expect(fragmented).toEqual(complete) + expect(fragmented.totalMatches).toBe(3) + expect(fragmented.truncated).toBe(false) + expect(fragmented.files[0].matches.map((match) => match.line)).toEqual([1, 2, 3]) + expect(fragmented.files[0].matches[0].lineContent).toBe(text) + expect(fragmented.files[0].matches[2].lineContent).toBe(text) + }) + + it('does not repeatedly split the growing partial output of a large matching line', async () => { + const wire = `${encode(`hit ${'x'.repeat(1024 * 1024)}`, 7)}\n` + const complete = await run([wire]) + const chunks: string[] = [] + for (let offset = 0; offset < wire.length; offset += 4096) { + chunks.push(wire.slice(offset, offset + 4096)) + } + const originalSplit = String.prototype.split + let scannedCharacters = 0 + const spy = vi.spyOn(String.prototype, 'split').mockImplementation(function ( + this: string, + separator: unknown, + limit?: number + ) { + if (separator === '\n') { + scannedCharacters += this.length + } + return Reflect.apply(originalSplit, this, [separator, limit]) + }) + let fragmented + try { + fragmented = await run(chunks) + } finally { + spy.mockRestore() + } + expect(fragmented).toEqual(complete) + expect(fragmented.totalMatches).toBe(1) + expect(fragmented.files[0].matches[0].line).toBe(7) + expect(scannedCharacters).toBe(0) + }) + + it('discards an unfinished line on timeout and detaches the output listeners', async () => { + vi.useFakeTimers() + const child = createProcess() + spawnMock.mockReturnValueOnce(child) + const result = search('/remote/root', 'hit', { maxResults: 100 }) + child.stdout!.emit('data', `${encode('hit complete', 1)}\n${encode('hit partial', 2)}`) + await vi.runOnlyPendingTimersAsync() + const value = await result + expect(value.totalMatches).toBe(1) + expect(value.truncated).toBe(true) + expect(child.kill).toHaveBeenCalled() + expect(child.stdout!.listenerCount('data')).toBe(0) + expect(child.stderr!.listenerCount('data')).toBe(0) + expect(child.listenerCount('close')).toBe(0) + expect(vi.getTimerCount()).toBe(0) + child.stdout!.emit('data', '\n') + child.emit('close', 0, null) + expect(value.totalMatches).toBe(1) + }) +}) diff --git a/src/shared/search-subprocess-lines.test.ts b/src/shared/search-subprocess-lines.test.ts index 3344776072e..34425801105 100644 --- a/src/shared/search-subprocess-lines.test.ts +++ b/src/shared/search-subprocess-lines.test.ts @@ -1,7 +1,32 @@ -import { describe, expect, it } from 'vitest' +import { describe, expect, it, vi } from 'vitest' import { SearchSubprocessLineAccumulator } from './search-subprocess-lines' describe('SearchSubprocessLineAccumulator', () => { + it('keeps complete decoded batches as strings without allocating byte copies', () => { + const parser = new SearchSubprocessLineAccumulator() + const lines: string[] = [] + const from = vi.spyOn(Buffer, 'from') + let copies: number + try { + parser.push('first🐋\n\nlast\n', (line) => lines.push(line)) + copies = from.mock.calls.length + } finally { + from.mockRestore() + } + expect(copies).toBe(0) + expect(lines).toEqual(['first🐋', '', 'last']) + expect(parser.finish()).toBeNull() + }) + + it('still enforces per-line UTF-8 byte limits for decoded batches', () => { + const parser = new SearchSubprocessLineAccumulator(4) + const lines: string[] = [] + expect(parser.push('éé\n漢\n', (line) => lines.push(line))).toBe(true) + expect(parser.push('漢é\n', (line) => lines.push(line))).toBe(false) + expect(lines).toEqual(['éé', '漢']) + expect(parser.finish()).toBeNull() + }) + it('preserves UTF-8 records split across raw byte chunks', () => { const parser = new SearchSubprocessLineAccumulator(32) const bytes = Buffer.from('first🐋\nsecond') diff --git a/src/shared/search-subprocess-lines.ts b/src/shared/search-subprocess-lines.ts index 5c04d9e8292..26f98b4d348 100644 --- a/src/shared/search-subprocess-lines.ts +++ b/src/shared/search-subprocess-lines.ts @@ -12,6 +12,20 @@ export class SearchSubprocessLineAccumulator { } push(rawChunk: Buffer | string, onLine: (line: string) => void): boolean { + // Three bytes per UTF-16 code unit bounds UTF-8 size without re-encoding complete batches. + if ( + typeof rawChunk === 'string' && + this.bytes === 0 && + rawChunk.endsWith('\n') && + rawChunk.length * 3 <= this.maxLineBytes + ) { + const lines = rawChunk.split('\n') + lines.pop() + for (const line of lines) { + onLine(line) + } + return true + } const chunk = Buffer.isBuffer(rawChunk) ? rawChunk : Buffer.from(rawChunk, 'utf8') let cursor = 0 while (cursor < chunk.length) { From b107f42c4c39bd025157f32ecc9b32c2edeb25dc Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 22:03:19 -0700 Subject: [PATCH 57/69] test: keep reveal filter valid through catalog refresh (#19017) --- tests/e2e/worktree-scroll-to-current.spec.ts | 56 +++++++++++++++----- 1 file changed, 43 insertions(+), 13 deletions(-) diff --git a/tests/e2e/worktree-scroll-to-current.spec.ts b/tests/e2e/worktree-scroll-to-current.spec.ts index d61d96847a0..07f9f87d05c 100644 --- a/tests/e2e/worktree-scroll-to-current.spec.ts +++ b/tests/e2e/worktree-scroll-to-current.spec.ts @@ -1,3 +1,5 @@ +import { mkdirSync } from 'node:fs' +import { runProcess } from '../../src/shared/child-process/run-process' import type { Page } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' @@ -41,7 +43,42 @@ test.describe('Reveal active workspace button', () => { test('clears sidebar filters before revealing a hidden current workspace', async ({ orcaPage, testRepoPath - }) => { + }, testInfo) => { + const filterRepoPath = testInfo.outputPath('filter-repo') + mkdirSync(filterRepoPath, { recursive: true }) + for (const args of [ + ['init', filterRepoPath], + [ + '-C', + filterRepoPath, + '-c', + 'user.name=E2E', + '-c', + 'user.email=e2e@test.local', + 'commit', + '--allow-empty', + '-m', + 'Filter fixture' + ] + ]) { + const result = await runProcess({ program: 'git', args }) + expect(result.code, result.stderr).toBe(0) + } + const filterRepoId = await orcaPage.evaluate(async (repoPath) => { + const result = await window.api.repos.add({ path: repoPath }) + if ('error' in result) { + throw new Error(result.error) + } + return result.repo.id + }, filterRepoPath) + await expect + .poll(() => + orcaPage.evaluate(async (id) => { + await window.__store!.getState().fetchRepos() + return window.__store!.getState().repos.some((repo) => repo.id === id) + }, filterRepoId) + ) + .toBe(true) await prepareSidebarForScrollTest(orcaPage) // Other specs can add worktrees to the shared repository before this test runs. @@ -86,18 +123,11 @@ test.describe('Reveal active workspace button', () => { }, targetId) await expect(targetRow).toHaveAttribute('aria-current', 'page') - await orcaPage.evaluate(() => { - const store = window.__store - if (!store) { - throw new Error('window.__store is not available') - } - store.getState().setFilterRepoIds(['__filtered_repo__']) - }) - - // Why: the filter's row-hiding side effect is covered deterministically by - // visible-worktrees.test.ts. Asserting an empty DOM here over-specifies an - // incidental render-settle state that flakes under the shared page; the - // contract under test is that reveal clears the filter (asserted below). + // Catalog refreshes prune nonexistent IDs, so use a real repo to keep the filter applied. + await orcaPage.evaluate((repoId) => { + window.__store!.getState().setFilterRepoIds([repoId]) + }, filterRepoId) + await expect(targetRows).toHaveCount(0) await revealButton.click() await orcaPage From bf5f3c2ec428e01d227038d28c443d845f6761fa Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 22:12:21 -0700 Subject: [PATCH 58/69] test: cover native X11 Hangul-plus-digit PTY bytes in CI (#19013) * test: run the native Hangul terminating-digit regression in CI * test: distinguish X11 byte coverage from the manual Wayland repro * test: require native IME engagement proof for the digit case --- config/scripts/pr-e2e-gate-contract.test.mjs | 16 +++++------ .../scripts/run-terminal-ibus-hangul-e2e.mjs | 3 ++- .../terminal-ime-engagement-receipt.mjs | 3 ++- .../terminal-ime-engagement-receipt.test.mjs | 27 ++++++++++++------- ...al-hangul-terminating-digit-native.spec.ts | 21 ++++++++------- 5 files changed, 41 insertions(+), 29 deletions(-) diff --git a/config/scripts/pr-e2e-gate-contract.test.mjs b/config/scripts/pr-e2e-gate-contract.test.mjs index 67e271868df..41f9338ab75 100644 --- a/config/scripts/pr-e2e-gate-contract.test.mjs +++ b/config/scripts/pr-e2e-gate-contract.test.mjs @@ -636,13 +636,8 @@ describe('PR E2E gate contract', () => { .filter((spec) => nativeGateExpression.test(readFileSync(join(projectDir, spec), 'utf8'))) expect(nativeGatedSpecs.length).toBeGreaterThan(0) - // Why exempt: the digit repro needs a nested gnome-shell, which no hosted runner provides - // (headless mutter never answers RemoteDesktop.CreateSession); the macOS spec needs a real - // macOS input source, and no macOS runner exists on any PR or scheduled lane. - const unreachableSpecs = new Set([ - 'tests/e2e/terminal-hangul-terminating-digit-native.spec.ts', - 'tests/e2e/terminal-macos-2set-korean-native.spec.ts' - ]) + // The macOS spec needs a native input source; PR and scheduled IME lanes use Linux. + const unreachableSpecs = new Set(['tests/e2e/terminal-macos-2set-korean-native.spec.ts']) const unclaimed = nativeGatedSpecs.filter( (spec) => !unreachableSpecs.has(spec) && !nativeImeRunner.includes(spec) ) @@ -682,8 +677,13 @@ describe('PR E2E gate contract', () => { // Why pin the titles: the runner requires one receipt per name, so a rename that nobody // mirrored here would fail the lane loudly instead of quietly halving it. + const nativeDigitSpec = readFileSync( + join(projectDir, 'tests/e2e/terminal-hangul-terminating-digit-native.spec.ts'), + 'utf8' + ) + expect(nativeDigitSpec).toContain('appendImeEngagementReceipt(testInfo.title, trace)') for (const title of EXPECTED_NATIVE_IME_TESTS) { - expect(nativeImeSpec, title).toContain(title) + expect(nativeImeSpec + nativeDigitSpec, title).toContain(title) } }) diff --git a/config/scripts/run-terminal-ibus-hangul-e2e.mjs b/config/scripts/run-terminal-ibus-hangul-e2e.mjs index 8bfdb0e2ae6..669f7744b39 100644 --- a/config/scripts/run-terminal-ibus-hangul-e2e.mjs +++ b/config/scripts/run-terminal-ibus-hangul-e2e.mjs @@ -199,7 +199,8 @@ async function runInsideSession(evidenceDir) { 'test:e2e:headful', '--workers=1', '--', - 'tests/e2e/terminal-ibus-hangul-native.spec.ts' + 'tests/e2e/terminal-ibus-hangul-native.spec.ts', + 'tests/e2e/terminal-hangul-terminating-digit-native.spec.ts' ], { cwd: projectDir, diff --git a/config/scripts/terminal-ime-engagement-receipt.mjs b/config/scripts/terminal-ime-engagement-receipt.mjs index 9ad5255d235..8f0732908c1 100644 --- a/config/scripts/terminal-ime-engagement-receipt.mjs +++ b/config/scripts/terminal-ime-engagement-receipt.mjs @@ -13,7 +13,8 @@ export const IME_ENGAGEMENT_RECEIPT_ENV = 'ORCA_E2E_IME_ENGAGEMENT_RECEIPT' /** The tests that must each leave a receipt. Pinned so deleting one cannot quietly shrink the lane. */ export const EXPECTED_NATIVE_IME_TESTS = [ 'forwards the issue exact-byte sequence without loss or duplication', - 'forwards the issue sentence stress sequence without leaked ASCII' + 'forwards the issue sentence stress sequence without leaked ASCII', + 'a digit typed right after a Hangul syllable reaches the pty' ] function parseReceipts(text) { diff --git a/config/scripts/terminal-ime-engagement-receipt.test.mjs b/config/scripts/terminal-ime-engagement-receipt.test.mjs index 04161339a0f..613abc2ffae 100644 --- a/config/scripts/terminal-ime-engagement-receipt.test.mjs +++ b/config/scripts/terminal-ime-engagement-receipt.test.mjs @@ -4,7 +4,7 @@ import { verifyImeEngagementReceipts } from './terminal-ime-engagement-receipt.mjs' -const [firstTest, secondTest] = EXPECTED_NATIVE_IME_TESTS +const [firstTest, secondTest, thirdTest] = EXPECTED_NATIVE_IME_TESTS function receipt(test, overrides = {}) { return JSON.stringify({ @@ -18,9 +18,11 @@ function receipt(test, overrides = {}) { describe('verifyImeEngagementReceipts', () => { it('accepts a run where every expected test observed real composition', () => { - expect(verifyImeEngagementReceipts(`${receipt(firstTest)}\n${receipt(secondTest)}\n`)).toEqual( - [] - ) + expect( + verifyImeEngagementReceipts( + `${receipt(firstTest)}\n${receipt(secondTest)}\n${receipt(thirdTest)}\n` + ) + ).toEqual([]) }) // The failure this whole mechanism exists for: Playwright reports a skipped test as a pass, so @@ -35,13 +37,20 @@ describe('verifyImeEngagementReceipts', () => { it('rejects a partial run where only one test reached the engine', () => { expect(verifyImeEngagementReceipts(`${receipt(firstTest)}\n`)).toEqual([ - `no engagement receipt for "${secondTest}" — it was skipped, filtered out, or renamed` + `no engagement receipt for "${secondTest}" — it was skipped, filtered out, or renamed`, + `no engagement receipt for "${thirdTest}" — it was skipped, filtered out, or renamed` + ]) + }) + + it('requires the digit receipt even when both original native tests passed', () => { + expect(verifyImeEngagementReceipts(`${receipt(firstTest)}\n${receipt(secondTest)}\n`)).toEqual([ + `no engagement receipt for "${thirdTest}" — it was skipped, filtered out, or renamed` ]) }) it('rejects a run that typed keys but never opened a composition', () => { const problems = verifyImeEngagementReceipts( - `${receipt(firstTest, { compositionStart: 0 })}\n${receipt(secondTest)}\n` + `${receipt(firstTest, { compositionStart: 0 })}\n${receipt(secondTest)}\n${receipt(thirdTest)}\n` ) expect(problems).toEqual([ `"${firstTest}" recorded no compositionstart — the IME never engaged` @@ -50,7 +59,7 @@ describe('verifyImeEngagementReceipts', () => { it('rejects a composition that produced no Hangul, which a latin passthrough would satisfy', () => { const problems = verifyImeEngagementReceipts( - `${receipt(firstTest, { hangulComposition: 0 })}\n${receipt(secondTest)}\n` + `${receipt(firstTest, { hangulComposition: 0 })}\n${receipt(secondTest)}\n${receipt(thirdTest)}\n` ) expect(problems).toEqual([ `"${firstTest}" recorded no Hangul composition data — the engine produced no syllables` @@ -59,7 +68,7 @@ describe('verifyImeEngagementReceipts', () => { it('rejects a renamed test rather than counting it toward coverage', () => { const problems = verifyImeEngagementReceipts( - `${receipt(firstTest)}\n${receipt(secondTest)}\n${receipt('some new scenario')}\n` + `${receipt(firstTest)}\n${receipt(secondTest)}\n${receipt(thirdTest)}\n${receipt('some new scenario')}\n` ) expect(problems).toEqual([ 'unexpected engagement receipt for "some new scenario" — update EXPECTED_NATIVE_IME_TESTS' @@ -68,7 +77,7 @@ describe('verifyImeEngagementReceipts', () => { it('reports a truncated receipt rather than parsing around it', () => { const problems = verifyImeEngagementReceipts( - `${receipt(firstTest)}\n{"test":"trunc\n${receipt(secondTest)}\n` + `${receipt(firstTest)}\n{"test":"trunc\n${receipt(secondTest)}\n${receipt(thirdTest)}\n` ) expect(problems).toEqual(['malformed receipt line: {"test":"trunc']) }) diff --git a/tests/e2e/terminal-hangul-terminating-digit-native.spec.ts b/tests/e2e/terminal-hangul-terminating-digit-native.spec.ts index 2fb90a9c992..4344f94adaf 100644 --- a/tests/e2e/terminal-hangul-terminating-digit-native.spec.ts +++ b/tests/e2e/terminal-hangul-terminating-digit-native.spec.ts @@ -3,20 +3,18 @@ * the pty. Written to reproduce #15299, where a digit typed straight after a Hangul syllable was * dropped under Wayland but not under X11. * - * THIS DOES NOT RUN IN CI. It is gated on ORCA_E2E_NATIVE_IBUS_HANGUL=1 and needs a compositor - * session that CI does not have, so it is a manual reproduction harness rather than coverage. - * That is stated plainly because this repo already carries native IME specs that are skipped - * everywhere and were mistaken for coverage they never provided. + * CI runs the default xdotool injector under X11, checking exact Hangul-plus-digit PTY bytes. + * That path passed even before the Wayland fix; it does not prove #15299 is fixed. + * Reproducing #15299 still requires the nested Wayland session below. * - * To run it, on a machine with gnome-shell and ibus-hangul: + * To run the Wayland reproduction on a machine with gnome-shell and ibus-hangul: * * Xvfb :65 -extension GLX & * DISPLAY=:65 gnome-shell --nested --wayland # nested, NOT --headless * ORCA_E2E_NATIVE_IBUS_HANGUL=1 ORCA_E2E_IME_INJECTOR=nested npx playwright test \ * tests/e2e/terminal-hangul-terminating-digit-native.spec.ts * - * Eight things that decide whether a run is real or a silent false negative, each of which cost a - * failed attempt: + * Nested Wayland prerequisites: * * - Nested, not headless. A headless mutter never answers RemoteDesktop.CreateSession, so there * is no way to inject input; nested makes the whole compositor an X window that xdotool can @@ -45,6 +43,7 @@ import { mkdirSync, writeFileSync } from 'node:fs' import path from 'node:path' import type { Page, TestInfo } from '@stablyai/playwright-test' import { test, expect } from './helpers/orca-app' +import { appendImeEngagementReceipt } from './terminal-ime-engagement-receipt' import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { focusActiveTerminalInput, @@ -232,6 +231,11 @@ test.describe('Hangul terminating digit @headful', () => { } receivedBytes = await waitForTerminalImeBytes(page, reader, 20_000) + expect(receivedBytes.map((hex) => Buffer.from(hex, 'hex').toString('utf8'))).toEqual( + Array.from({ length: REPETITIONS }, () => `${EXPECTED_LINE}\n`) + ) + const trace = await readTerminalImeBoundaryTrace(page) + appendImeEngagementReceipt(testInfo.title, trace) } finally { await writeEvidence(page, testInfo, 'hangul-terminating-digit', { expectedHex, @@ -243,8 +247,5 @@ test.describe('Hangul terminating digit @headful', () => { await sendToTerminal(page, ptyId, '\x03').catch(() => undefined) removeTerminalImeByteReader(reader) } - expect(receivedBytes.map((hex) => Buffer.from(hex, 'hex').toString('utf8'))).toEqual( - Array.from({ length: REPETITIONS }, () => `${EXPECTED_LINE}\n`) - ) }) }) From e2270fe94d1699207a2cf2a3f80bd5acf6c49e52 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 22:16:57 -0700 Subject: [PATCH 59/69] Stop evicted and expired worktree preparations (#18951) * Stop obsolete worktree preparations when evicted or expired * fix(worktree): skip discard retries for registrations an aborted checkout already removed An evicted or expired preparation now aborts its checkout, which self-discards the registration before the pool's own discard runs. That second discard failed with "is not a working tree" and was enrolled for up to three retries on later preparations for the same host, spawning Git only to fail again and warning that the path stays registered when it was already gone. Also accept fs.watch events without a filename in the abort real-Git test, and add an opt-in bench (ORCA_WORKTREE_PREPARATION_CANCEL_BENCH=1) that measures a fresh checkout's wall time with obsolete checkouts left running versus aborted. --- ...rktree-create-preparation-real-git.test.ts | 58 ++++++- ...e-preparation-cancel-latency.bench.test.ts | 158 +++++++++++++++++ src/main/worktree-create-preparation-pool.ts | 27 ++- src/main/worktree-create-preparation.test.ts | 164 +++++++++++++++++- .../worktree-preparation-discard-retry.ts | 6 + 5 files changed, 403 insertions(+), 10 deletions(-) create mode 100644 src/main/git/worktree-preparation-cancel-latency.bench.test.ts diff --git a/src/main/git/worktree-create-preparation-real-git.test.ts b/src/main/git/worktree-create-preparation-real-git.test.ts index 63f8021a7ae..f28349f77ca 100644 --- a/src/main/git/worktree-create-preparation-real-git.test.ts +++ b/src/main/git/worktree-create-preparation-real-git.test.ts @@ -1,8 +1,10 @@ import { execFileSync } from 'node:child_process' +import { existsSync, watch, type FSWatcher } from 'node:fs' import { mkdir, mkdtemp, readFile, realpath, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' -import { afterEach, describe, expect, it } from 'vitest' +import { afterEach, describe, expect, it, vi } from 'vitest' +import * as gitRunner from './runner' import { createWorktreePreparationLockReason, isWorktreeCreatePreparation, @@ -46,6 +48,60 @@ afterEach(async () => { }) describe('prepared worktree creation with real Git', () => { + it('removes partial checkout files and registration after materialization is aborted', async () => { + const { repoPath, root } = await createRepo() + await Promise.all( + Array.from({ length: 1000 }, (_, index) => + writeFile( + join(repoPath, `payload-${index.toString().padStart(4, '0')}.txt`), + 'payload'.repeat(128) + ) + ) + ) + git(repoPath, ['add', '.']) + git(repoPath, ['commit', '--quiet', '-m', 'materialization fixture']) + const preparationRoot = join(root, WORKTREE_CREATE_PREPARATION_DIRECTORY) + const preparedPath = join(preparationRoot, `${process.pid}-partial`) + await mkdir(preparationRoot, { recursive: true }) + const controller = new AbortController() + const original = gitRunner.gitExecFileAsync + let watcher: FSWatcher | undefined + let observedMaterialization = false + const calls: string[][] = [] + const spy = vi.spyOn(gitRunner, 'gitExecFileAsync').mockImplementation((args, options) => { + calls.push([...args]) + if (args.includes('reset')) { + watcher = watch(preparedPath, (_event, filename) => { + // Only the reset writes here, so an event without a filename is still materialization. + if (filename === null || filename.toString().startsWith('payload-')) { + observedMaterialization = true + watcher?.close() + controller.abort() + } + }) + } + return original(args, options) + }) + try { + await expect( + prepareWorktreeCreateCheckout( + repoPath, + preparedPath, + 'main', + createWorktreePreparationLockReason('partial-test'), + { signal: controller.signal } + ) + ).rejects.toThrow() + expect(observedMaterialization).toBe(true) + expect(calls.some((args) => args[args.indexOf('worktree') + 1] === 'lock')).toBe(false) + expect(existsSync(preparedPath)).toBe(false) + expect(await listWorktrees(repoPath, { includeCreatePreparations: true })).toHaveLength(1) + } finally { + watcher?.close() + spy.mockRestore() + } + }) + it('cleans up when the create signal is canceled', async () => { const { repoPath, root } = await createRepo() const preparationRoot = join(root, WORKTREE_CREATE_PREPARATION_DIRECTORY) diff --git a/src/main/git/worktree-preparation-cancel-latency.bench.test.ts b/src/main/git/worktree-preparation-cancel-latency.bench.test.ts new file mode 100644 index 00000000000..e2750df839d --- /dev/null +++ b/src/main/git/worktree-preparation-cancel-latency.bench.test.ts @@ -0,0 +1,158 @@ +// Opt in: ORCA_WORKTREE_PREPARATION_CANCEL_BENCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/git/worktree-preparation-cancel-latency.bench.test.ts +// +// Measures the create-side cost of an obsolete preparation: the wall time of a fresh checkout +// (the next Create's critical path) while an evicted preparation's checkout is either left running +// (main before #18951) or aborted (after). Same code, same fixture; only the abort differs. +import { execFileSync } from 'node:child_process' +import { existsSync } from 'node:fs' +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { performance } from 'node:perf_hooks' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { + createWorktreePreparationLockReason, + WORKTREE_CREATE_PREPARATION_DIRECTORY +} from '../../shared/worktree/create-preparation' +import { + discardPreparedWorktree, + prepareWorktreeCreateCheckout +} from './worktree-create-preparation' + +const describeBench = process.env.ORCA_WORKTREE_PREPARATION_CANCEL_BENCH ? describe : describe.skip +const FILE_COUNT = Number(process.env.ORCA_WORKTREE_PREPARATION_CANCEL_BENCH_FILES ?? 6000) +const FILE_BYTES = 48 * 1024 +const TRIALS = Number(process.env.ORCA_WORKTREE_PREPARATION_CANCEL_BENCH_TRIALS ?? 5) +const OBSOLETE_COUNTS = [1, 3] +const RESULT_PATH = process.env.ORCA_WORKTREE_PREPARATION_CANCEL_BENCH_RESULT + +type Variant = 'running' | 'aborted' +type Sample = { variant: Variant; obsolete: number; freshCheckoutMs: number } + +let root = '' +let repoPath = '' +let preparationRoot = '' +let sequence = 0 + +function git(cwd: string, args: string[]): void { + execFileSync('git', args, { cwd, stdio: ['ignore', 'ignore', 'pipe'] }) +} + +function nextPreparedPath(label: string): string { + sequence += 1 + return join(preparationRoot, `${process.pid}-${label}-${sequence}`) +} + +function checkout(preparedPath: string, signal?: AbortSignal): Promise { + return prepareWorktreeCreateCheckout( + repoPath, + preparedPath, + 'main', + createWorktreePreparationLockReason(`bench-${sequence}`), + signal ? { signal } : {} + ) +} + +async function runTrial(variant: Variant, obsolete: number): Promise { + const controllers = Array.from({ length: obsolete }, () => new AbortController()) + const obsoletePaths = controllers.map(() => nextPreparedPath('obsolete')) + const obsoleteWork = obsoletePaths.map((path, index) => + checkout(path, controllers[index].signal).catch(() => {}) + ) + if (variant === 'aborted') { + // Eviction aborts in the same turn the incoming preparation is armed, so abort before the + // fresh checkout starts. + controllers.forEach((controller) => controller.abort()) + } + const freshPath = nextPreparedPath('fresh') + const started = performance.now() + await checkout(freshPath) + const freshCheckoutMs = performance.now() - started + await Promise.all(obsoleteWork) + await Promise.all( + [...obsoletePaths, freshPath].map((path) => + discardPreparedWorktree(repoPath, path).catch(() => {}) + ) + ) + return { variant, obsolete, freshCheckoutMs } +} + +function median(values: number[]): number { + const sorted = [...values].sort((left, right) => left - right) + const middle = Math.floor(sorted.length / 2) + return sorted.length % 2 ? sorted[middle] : (sorted[middle - 1] + sorted[middle]) / 2 +} + +describeBench('obsolete preparation cancellation latency', () => { + beforeAll(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-preparation-cancel-bench-')) + repoPath = join(root, 'repo') + preparationRoot = join(root, WORKTREE_CREATE_PREPARATION_DIRECTORY) + await mkdir(preparationRoot, { recursive: true }) + execFileSync('git', ['init', '--quiet', repoPath]) + git(repoPath, ['symbolic-ref', 'HEAD', 'refs/heads/main']) + git(repoPath, ['config', 'user.email', 'bench@example.com']) + git(repoPath, ['config', 'user.name', 'Bench']) + git(repoPath, ['config', 'core.autocrlf', 'false']) + // Unique content per file so the object store cannot dedupe the materialization work. + for (let batch = 0; batch < FILE_COUNT; batch += 500) { + await Promise.all( + Array.from({ length: Math.min(500, FILE_COUNT - batch) }, (_, offset) => { + const index = batch + offset + return writeFile( + join(repoPath, `payload-${index.toString().padStart(5, '0')}.txt`), + `${index}\n`.repeat(Math.ceil(FILE_BYTES / `${index}\n`.length)) + ) + }) + ) + } + git(repoPath, ['add', '.']) + git(repoPath, ['commit', '--quiet', '-m', 'bench fixture']) + }, 600_000) + + afterAll(async () => { + await rm(root, { recursive: true, force: true }) + }) + + it('reports fresh checkout wall time with obsolete checkouts running vs aborted', async () => { + // Warm the object store and page cache once so the first variant is not penalised. + const warm = nextPreparedPath('warm') + await checkout(warm) + await discardPreparedWorktree(repoPath, warm) + + const samples: Sample[] = [] + for (const obsolete of OBSOLETE_COUNTS) { + for (let trial = 0; trial < TRIALS; trial += 1) { + // Alternate order so drift in cache or thermal state does not favour one variant. + const order: Variant[] = trial % 2 ? ['aborted', 'running'] : ['running', 'aborted'] + for (const variant of order) { + samples.push(await runTrial(variant, obsolete)) + } + } + } + const summary = OBSOLETE_COUNTS.map((obsolete) => { + const pick = (variant: Variant): number[] => + samples + .filter((sample) => sample.variant === variant && sample.obsolete === obsolete) + .map((sample) => sample.freshCheckoutMs) + const running = median(pick('running')) + const aborted = median(pick('aborted')) + return { + obsolete, + trials: TRIALS, + freshCheckoutMedianMs: { obsoleteRunning: running, obsoleteAborted: aborted }, + speedup: running / aborted + } + }) + const report = JSON.stringify( + { fixture: { files: FILE_COUNT, bytesPerFile: FILE_BYTES }, samples, summary }, + null, + 2 + ) + console.log(report) + if (RESULT_PATH) { + await writeFile(RESULT_PATH, `${report}\n`) + } + expect(existsSync(preparationRoot)).toBe(true) + }, 900_000) +}) diff --git a/src/main/worktree-create-preparation-pool.ts b/src/main/worktree-create-preparation-pool.ts index 7539c6076e4..1581f404b20 100644 --- a/src/main/worktree-create-preparation-pool.ts +++ b/src/main/worktree-create-preparation-pool.ts @@ -38,6 +38,8 @@ export type PreparationEntry = { createdAt: number ready: Promise expiration: NodeJS.Timeout + controller: AbortController + checkoutStarted: boolean } export type StartPreparationArgs = { @@ -68,6 +70,9 @@ async function discardEntry(entry: PreparationEntry): Promise { // A failed checkout self-discards, but that self-discard is best-effort too, so it can strand the // registration for the same reason the discard here can. Enrol either way. await entry.ready.catch(() => {}) + if (!entry.checkoutStarted) { + return + } await discardPreparationWithRetry({ hostKey: preparationHostKey(entry.repoPathKey, entry.wslDistro), repoPath: entry.repoPath, @@ -86,6 +91,7 @@ function expireEntry(entry: PreparationEntry): void { return } preparations.delete(entry.key) + entry.controller.abort() discardEntryInBackground(entry) } @@ -118,6 +124,7 @@ function enforcePreparationLimit( } preparations.delete(victim.key) clearTimeout(victim.expiration) + victim.controller.abort() discardEntryInBackground(victim) } } @@ -163,6 +170,10 @@ export function startPreparation({ WORKTREE_CREATE_PREPARATION_DIRECTORY ) const preparedPath = pathOps(workspaceRoot).join(preparationRoot, preparationId) + const controller = new AbortController() + const signal = options.signal + ? AbortSignal.any([options.signal, controller.signal]) + : controller.signal const entry = {} as PreparationEntry const expiration = setTimeout(() => expireEntry(entry), WORKTREE_CREATE_PREPARATION_TTL_MS) expiration.unref() @@ -179,17 +190,19 @@ export function startPreparation({ options, createdAt: Date.now(), expiration, + controller, + checkoutStarted: false, ready: (async () => { await cleanupStalePreparations(preparationHostKey(repoPathKey, wslDistro), repoPath, options) + signal.throwIfAborted() await mkdir(toHostFilesystemPath(preparationRoot), { recursive: true }) + signal.throwIfAborted() // Already canonical, so the add re-resolves nothing. - await prepareWorktreeCreateCheckout( - repoPath, - preparedPath, - canonicalBase, - lockReason, - options - ) + entry.checkoutStarted = true + await prepareWorktreeCreateCheckout(repoPath, preparedPath, canonicalBase, lockReason, { + ...options, + signal + }) })() } satisfies PreparationEntry) preparations.set(key, entry) diff --git a/src/main/worktree-create-preparation.test.ts b/src/main/worktree-create-preparation.test.ts index 06818fec422..f6f7e295497 100644 --- a/src/main/worktree-create-preparation.test.ts +++ b/src/main/worktree-create-preparation.test.ts @@ -1,5 +1,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type * as WorktreeLogic from './ipc/worktree-logic' import type { Store } from './persistence' +import { WORKTREE_CREATE_PREPARATION_TTL_MS } from './worktree-create-preparation-pool' import type { Repo } from '../shared/repo-types' import { WORKTREE_CREATE_PREPARATION_DIRECTORY } from '../shared/worktree/create-preparation' import { resolveWorktreeAddBaseRef } from '../shared/worktree/base-ref' @@ -36,7 +38,8 @@ vi.mock('./project-runtime-git-options', () => ({ getLocalProjectWorktreeGitOptions: mocks.getWorktreeOptions, getWorktreeMirrorDistro: () => undefined })) -vi.mock('./ipc/worktree-logic', () => ({ +vi.mock('./ipc/worktree-logic', async (importOriginal) => ({ + isOrphanedWorktreeError: (await importOriginal()).isOrphanedWorktreeError, computeWorkspaceRoot: mocks.computeWorkspaceRoot, computeWorkspaceRootAsync: mocks.computeWorkspaceRootAsync, getWorktreePathSettings: () => ({ @@ -96,6 +99,163 @@ afterEach(async () => { }) describe('worktree create preparation registry', () => { + it('cancels an evicted checkout and cleans up with the original options', async () => { + let signal: AbortSignal | undefined + mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, options) => { + signal = options.signal + return new Promise((_resolve, reject) => { + signal!.addEventListener('abort', () => reject(signal!.reason), { once: true }) + }) + }) + const obsolete = prepareWorktreeCreateForRepo(store, repo, 'origin/main') + const settled = Promise.allSettled([obsolete]) + await flushBackgroundWork() + const obsoletePath = mocks.prepareCheckout.mock.calls[0][1] + for (const base of ['origin/one', 'origin/two', 'origin/three']) { + await prepareWorktreeCreateForRepo(store, repo, base) + } + expect(signal?.aborted).toBe(true) + expect((await settled)[0].status).toBe('rejected') + await flushBackgroundWork() + expect(mocks.discard).toHaveBeenCalledWith(repo.path, obsoletePath, {}) + }) + + it('does not retry a discard whose registration the aborted checkout already removed', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, options) => { + const signal = options.signal! + return new Promise((_resolve, reject) => { + signal.addEventListener('abort', () => reject(signal.reason), { once: true }) + }) + }) + try { + const obsolete = prepareWorktreeCreateForRepo(store, repo, 'origin/main').catch(() => {}) + await flushBackgroundWork() + const obsoletePath = mocks.prepareCheckout.mock.calls[0][1] as string + mocks.discard.mockImplementation(async (_repoPath: string, path: string) => { + if (path === obsoletePath) { + throw Object.assign(new Error(`fatal: '${path}' is not a working tree`), { + stderr: `fatal: '${path}' is not a working tree` + }) + } + }) + for (const base of ['origin/one', 'origin/two', 'origin/three']) { + await prepareWorktreeCreateForRepo(store, repo, base) + } + await obsolete + await flushBackgroundWork() + const obsoleteDiscards = (): number => + mocks.discard.mock.calls.filter((call) => call[1] === obsoletePath).length + expect(obsoleteDiscards()).toBe(1) + + for (const base of ['origin/four', 'origin/five']) { + await prepareWorktreeCreateForRepo(store, repo, base) + await flushBackgroundWork() + } + expect(obsoleteDiscards()).toBe(1) + expect(warn).not.toHaveBeenCalled() + } finally { + warn.mockRestore() + } + }) + + it('does not start obsolete checkout work after shared cleanup finishes', async () => { + let releaseCleanup!: () => void + mocks.listWorktreeGraph.mockImplementationOnce( + () => + new Promise<[]>((resolve) => { + releaseCleanup = () => resolve([]) + }) + ) + const requests = ['main', 'one', 'two', 'three'].map((base) => + prepareWorktreeCreateForRepo(store, repo, `origin/${base}`) + ) + const settled = Promise.allSettled(requests) + await flushBackgroundWork() + expect(mocks.prepareCheckout).not.toHaveBeenCalled() + releaseCleanup() + const results = await settled + expect(results.map((result) => result.status)).toEqual([ + 'rejected', + 'fulfilled', + 'fulfilled', + 'fulfilled' + ]) + expect(mocks.prepareCheckout).toHaveBeenCalledTimes(3) + await flushBackgroundWork() + expect(mocks.discard).not.toHaveBeenCalled() + }) + + it('keeps a claimed in-flight checkout alive when new preparations fill the pool', async () => { + let signal: AbortSignal | undefined + let finishCheckout!: () => void + mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, options) => { + signal = options.signal + return new Promise((resolve) => { + finishCheckout = resolve + }) + }) + const preparation = prepareWorktreeCreateForRepo(store, repo, 'origin/main') + await flushBackgroundWork() + const create = consumePreparedWorktreeCreate({ + repoPath: repo.path, + workspaceRoot: '/workspace', + worktreePath: '/workspace/claimed', + branch: 'claimed', + baseBranch: 'origin/main' + }) + await flushBackgroundWork() + for (const base of ['origin/one', 'origin/two', 'origin/three', 'origin/four']) { + await prepareWorktreeCreateForRepo(store, repo, base) + } + expect(signal?.aborted).toBe(false) + finishCheckout() + await preparation + expect(await create).toMatchObject({ status: 'hit' }) + }) + + it('cancels an expired in-flight checkout', async () => { + vi.useFakeTimers() + let signal: AbortSignal | undefined + mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, options) => { + signal = options.signal + return new Promise((_resolve, reject) => { + signal!.addEventListener('abort', () => reject(signal!.reason), { once: true }) + }) + }) + try { + const settled = Promise.allSettled([prepareWorktreeCreateForRepo(store, repo, 'origin/main')]) + await vi.advanceTimersByTimeAsync(0) + expect(signal?.aborted).toBe(false) + await vi.advanceTimersByTimeAsync(WORKTREE_CREATE_PREPARATION_TTL_MS) + expect(signal?.aborted).toBe(true) + expect((await settled)[0].status).toBe('rejected') + } finally { + vi.useRealTimers() + } + }) + + it('preserves caller cancellation without mutating its options', async () => { + const controller = new AbortController() + const options = { signal: controller.signal } + mocks.getWorktreeOptions.mockReturnValue(options) + let signal: AbortSignal | undefined + mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, executionOptions) => { + signal = executionOptions.signal + return new Promise((_resolve, reject) => { + signal!.addEventListener('abort', () => reject(signal!.reason), { once: true }) + }) + }) + const preparation = prepareWorktreeCreateForRepo(store, repo, 'origin/main') + const settled = Promise.allSettled([preparation]) + await flushBackgroundWork() + controller.abort() + expect(signal?.aborted).toBe(true) + expect((await settled)[0].status).toBe('rejected') + expect(options.signal).toBe(controller.signal) + expect(signal).not.toBe(controller.signal) + }) + it('starts the checkout only once the async workspace root resolves', async () => { let resolveRoot!: (root: string) => void mocks.computeWorkspaceRootAsync.mockReturnValue( @@ -364,7 +524,7 @@ describe('worktree create preparation registry', () => { expect.any(String), 'refs/remotes/origin/main', expect.any(String), - options + { ...options, signal: expect.any(AbortSignal) } ) expect(mocks.finalize).toHaveBeenCalledWith( repo.path, diff --git a/src/main/worktree-preparation-discard-retry.ts b/src/main/worktree-preparation-discard-retry.ts index e18f890084c..8602bebb39a 100644 --- a/src/main/worktree-preparation-discard-retry.ts +++ b/src/main/worktree-preparation-discard-retry.ts @@ -1,5 +1,6 @@ import type { AddWorktreeOptions } from './git/worktree' import { discardPreparedWorktree } from './git/worktree-create-preparation' +import { isOrphanedWorktreeError } from './ipc/worktree-logic' // Stale cleanup only reclaims preparations whose owner pid is dead, so a discard that fails inside // the live process would strand its scratch checkout until the app restarts. Remember the failure @@ -31,6 +32,11 @@ async function runDiscard(target: PreparationDiscardTarget, attempts: number): P try { await discardPreparedWorktree(target.repoPath, target.preparedPath, target.options) } catch (error) { + // An aborted or failed checkout self-discards first, so the registration is usually already + // gone by the time the pool discards; retrying that would only spawn Git to fail again. + if (isOrphanedWorktreeError(error)) { + return + } // Bounded: a path that never becomes removable must not tax every later preparation. if (attempts >= PREPARATION_DISCARD_ATTEMPT_LIMIT) { console.warn( From a63a4579cf5c7f3d964333dd1b550eb1d54e4ce1 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 22:30:01 -0700 Subject: [PATCH 60/69] Let worktree creation proceed during stale preparation reclamation (#18967) * Stop obsolete worktree preparations when evicted or expired * Let worktree preparation proceed during stale reclamation * Verify creation during stalled stale worktree reclamation * Preserve preparation ownership until Git removal starts * test: keep artifact share fixtures unexpired across calendar dates (#18955) --- ...rktree-create-preparation-real-git.test.ts | 106 +++++++ src/main/git/worktree-create-preparation.ts | 18 +- ...ee-create-preparation-cancellation.test.ts | 259 ++++++++++++++++++ src/main/worktree-create-preparation-pool.ts | 10 +- ...rktree-create-preparation-stale-cleanup.ts | 51 ++-- src/main/worktree-create-preparation.test.ts | 213 ++++---------- src/shared/git-binary-compatibility.test.ts | 10 + 7 files changed, 473 insertions(+), 194 deletions(-) create mode 100644 src/main/worktree-create-preparation-cancellation.test.ts diff --git a/src/main/git/worktree-create-preparation-real-git.test.ts b/src/main/git/worktree-create-preparation-real-git.test.ts index f28349f77ca..c309e35f14f 100644 --- a/src/main/git/worktree-create-preparation-real-git.test.ts +++ b/src/main/git/worktree-create-preparation-real-git.test.ts @@ -17,6 +17,13 @@ import { prepareWorktreeCreateCheckout } from './worktree-create-preparation' import { areWorktreePathsEqual } from './worktree-path-comparison' +import { + _resetPreparationPoolForTests, + listPreparations, + startPreparation, + takePreparation +} from '../worktree-create-preparation-pool' +import { hasPendingStalePreparationCleanup } from '../worktree-create-preparation-stale-cleanup' const tempRoots: string[] = [] @@ -48,6 +55,105 @@ afterEach(async () => { }) describe('prepared worktree creation with real Git', () => { + it('retains preparation ownership when the removal command cannot start', async () => { + const fixture = await createRepo() + const repoPath = await realpath(fixture.repoPath) + const root = await realpath(fixture.root) + const preparedPath = join(root, WORKTREE_CREATE_PREPARATION_DIRECTORY, 'owned-removal') + await mkdir(join(root, WORKTREE_CREATE_PREPARATION_DIRECTORY), { recursive: true }) + const lockReason = createWorktreePreparationLockReason('removal-failure') + await prepareWorktreeCreateCheckout(repoPath, preparedPath, 'main', lockReason) + const original = gitRunner.gitExecFileAsync + const spy = vi.spyOn(gitRunner, 'gitExecFileAsync').mockImplementation((args, options) => { + if (args.includes('remove') && args.includes(preparedPath)) { + return Promise.reject(new Error('injected removal launch failure')) + } + return original(args, options) + }) + try { + await expect(discardPreparedWorktree(repoPath, preparedPath)).rejects.toThrow( + 'injected removal launch failure' + ) + const remaining = await listWorktrees(repoPath, { includeCreatePreparations: true }) + const prepared = remaining.find((worktree) => + areWorktreePathsEqual(worktree.path, preparedPath) + ) + expect(prepared).toBeDefined() + expect(prepared?.lockReason).toBe(lockReason) + expect(await readFile(join(preparedPath, 'version.txt'), 'utf8')).toBe('one\n') + } finally { + spy.mockRestore() + await discardPreparedWorktree(repoPath, preparedPath) + } + expect(existsSync(preparedPath)).toBe(false) + }) + + it('creates and finalizes while dead-owner reclamation is stalled', async () => { + const fixture = await createRepo() + const repoPath = await realpath(fixture.repoPath) + const root = await realpath(fixture.root) + const preparationRoot = join(root, WORKTREE_CREATE_PREPARATION_DIRECTORY) + const stalePath = join(preparationRoot, '999999999-11111111-1111-4111-8111-111111111111') + await mkdir(preparationRoot, { recursive: true }) + await prepareWorktreeCreateCheckout( + repoPath, + stalePath, + 'main', + 'orca-create-preparation:v1:999999999:stale' + ) + let releaseRemoval!: () => void + const removalGate = new Promise((resolve) => { + releaseRemoval = resolve + }) + let markRemovalStarted!: () => void + const removalStarted = new Promise((resolve) => { + markRemovalStarted = resolve + }) + const original = gitRunner.gitExecFileAsync + const spy = vi + .spyOn(gitRunner, 'gitExecFileAsync') + .mockImplementation(async (args, options) => { + if (args.includes('remove') && args.includes(stalePath)) { + markRemovalStarted() + await removalGate + } + return original(args, options) + }) + try { + const preparing = startPreparation({ + repoPath, + workspaceRoot: root, + baseBranch: 'main', + canonicalBase: 'refs/heads/main', + options: {} + }) + await removalStarted + expect(hasPendingStalePreparationCleanup()).toBe(true) + await preparing + const [entry] = listPreparations() + expect(entry).toBeDefined() + takePreparation(entry) + const finalPath = join(root, 'fresh-worktree') + await finalizePreparedWorktree(repoPath, entry.preparedPath, finalPath, 'fresh', 'main') + expect(git(finalPath, ['status', '--porcelain'])).toBe('') + expect(git(finalPath, ['symbolic-ref', '--short', 'HEAD'])).toBe('fresh') + expect(await readFile(join(finalPath, 'version.txt'), 'utf8')).toBe('one\n') + expect(existsSync(stalePath)).toBe(true) + expect(hasPendingStalePreparationCleanup()).toBe(true) + releaseRemoval() + await _resetPreparationPoolForTests() + expect(existsSync(stalePath)).toBe(false) + const remaining = await listWorktrees(repoPath, { includeCreatePreparations: true }) + expect(remaining).toHaveLength(2) + expect(remaining.map((w) => w.path)).toEqual(expect.arrayContaining([repoPath, finalPath])) + expect(hasPendingStalePreparationCleanup()).toBe(false) + } finally { + releaseRemoval() + await _resetPreparationPoolForTests() + spy.mockRestore() + } + }) + it('removes partial checkout files and registration after materialization is aborted', async () => { const { repoPath, root } = await createRepo() await Promise.all( diff --git a/src/main/git/worktree-create-preparation.ts b/src/main/git/worktree-create-preparation.ts index 78713957690..0f27f045287 100644 --- a/src/main/git/worktree-create-preparation.ts +++ b/src/main/git/worktree-create-preparation.ts @@ -45,16 +45,16 @@ async function performDiscardPreparedWorktree( timeout: options.timeout ?? WORKTREE_REMOVAL_REGISTRATION_TIMEOUT_MS } try { + // Preserve the ownership lock if removal cannot start; Git 2.25 supports locked removal. await gitExecFileAsync( - [...windowsLongPathGitArgs(repoPath), 'worktree', 'unlock', worktreePath], - cleanupGitOptions - ) - } catch { - // It may be unlocked already or only partially registered. - } - try { - await gitExecFileAsync( - [...windowsLongPathGitArgs(repoPath), 'worktree', 'remove', '--force', worktreePath], + [ + ...windowsLongPathGitArgs(repoPath), + 'worktree', + 'remove', + '--force', + '--force', + worktreePath + ], cleanupGitOptions ) } finally { diff --git a/src/main/worktree-create-preparation-cancellation.test.ts b/src/main/worktree-create-preparation-cancellation.test.ts new file mode 100644 index 00000000000..7974dcb7e3d --- /dev/null +++ b/src/main/worktree-create-preparation-cancellation.test.ts @@ -0,0 +1,259 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type * as WorktreeLogic from './ipc/worktree-logic' +import type { Store } from './persistence' +import { WORKTREE_CREATE_PREPARATION_TTL_MS } from './worktree-create-preparation-pool' +import type { Repo } from '../shared/repo-types' +import { resolveWorktreeAddBaseRef } from '../shared/worktree/base-ref' + +const mocks = vi.hoisted(() => ({ + mkdir: vi.fn(), + listWorktreeGraph: vi.fn(), + prepareCheckout: vi.fn(), + finalize: vi.fn(), + discard: vi.fn(), + unlock: vi.fn(), + getWorktreeOptions: vi.fn(), + computeWorkspaceRoot: vi.fn(), + computeWorkspaceRootAsync: vi.fn(), + resolveBaseRef: vi.fn(), + measureDivergence: vi.fn() +})) + +vi.mock('node:fs/promises', () => ({ mkdir: mocks.mkdir })) +vi.mock('./git/worktree', () => ({ listWorktreeGraph: mocks.listWorktreeGraph })) +vi.mock('./git/worktree-create-preparation', () => ({ + prepareWorktreeCreateCheckout: mocks.prepareCheckout, + finalizePreparedWorktree: mocks.finalize, + discardPreparedWorktree: mocks.discard, + unlockPreparedWorktree: mocks.unlock +})) +vi.mock('./git/worktree-base-ref-probe', () => ({ + resolveLocalWorktreeBaseRef: mocks.resolveBaseRef +})) +vi.mock('./git/worktree-base-divergence', () => ({ + measureRetargetDivergence: mocks.measureDivergence +})) +vi.mock('./project-runtime-git-options', () => ({ + getLocalProjectWorktreeGitOptions: mocks.getWorktreeOptions, + getWorktreeMirrorDistro: () => undefined +})) +vi.mock('./ipc/worktree-logic', async (importOriginal) => ({ + isOrphanedWorktreeError: (await importOriginal()).isOrphanedWorktreeError, + computeWorkspaceRoot: mocks.computeWorkspaceRoot, + computeWorkspaceRootAsync: mocks.computeWorkspaceRootAsync, + getWorktreePathSettings: () => ({ + workspaceDir: process.platform === 'win32' ? 'C:\\workspace' : '/workspace', + nestWorkspaces: false + }) +})) + +import { + _resetWorktreeCreatePreparationsForTests, + consumePreparedWorktreeCreate, + prepareWorktreeCreateForRepo +} from './worktree-create-preparation' + +// Evictions and retries are fire-and-forget, so let them settle before asserting. +function flushBackgroundWork(ms = 0): Promise { + return new Promise((resolve) => setTimeout(resolve, ms)) +} + +const EXISTING_REFS = new Set([ + 'refs/heads/main', + 'refs/remotes/origin/main', + 'refs/remotes/origin/release' +]) +const repo = { id: 'repo-1', path: '/repo' } as Repo +const store = { getSettings: () => ({}) } as unknown as Store + +beforeEach(() => { + mocks.mkdir.mockReset().mockResolvedValue(undefined) + mocks.listWorktreeGraph.mockReset().mockResolvedValue([]) + mocks.prepareCheckout.mockReset().mockResolvedValue(undefined) + mocks.finalize.mockReset().mockResolvedValue({}) + mocks.discard.mockReset().mockResolvedValue(undefined) + mocks.unlock.mockReset().mockResolvedValue(undefined) + mocks.getWorktreeOptions.mockReset().mockReturnValue({}) + mocks.measureDivergence.mockReset().mockResolvedValue('within') + mocks.resolveBaseRef + .mockReset() + .mockImplementation((_repoPath: string, baseRef: string) => + resolveWorktreeAddBaseRef(baseRef, async (candidate) => EXISTING_REFS.has(candidate)) + ) + mocks.computeWorkspaceRoot.mockReset().mockImplementation(() => { + throw new Error('synchronous workspace-root lookup must not run on the main thread') + }) + mocks.computeWorkspaceRootAsync + .mockReset() + .mockImplementation(async (repoPath: string) => + process.platform === 'win32' && /^[A-Za-z]:[\\/]/.test(repoPath) + ? 'C:\\workspace' + : '/workspace' + ) +}) + +afterEach(async () => { + await _resetWorktreeCreatePreparationsForTests() +}) + +// Why this file exists separately from worktree-create-preparation.test.ts: it holds the +// in-flight checkout cancellation paths (eviction, expiry, caller abort) and the discard that +// follows, keeping both suites under the test-file line limit. +describe('worktree create preparation cancellation', () => { + it('cancels an evicted checkout and cleans up with the original options', async () => { + let signal: AbortSignal | undefined + mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, options) => { + signal = options.signal + return new Promise((_resolve, reject) => { + signal!.addEventListener('abort', () => reject(signal!.reason), { once: true }) + }) + }) + const obsolete = prepareWorktreeCreateForRepo(store, repo, 'origin/main') + const settled = Promise.allSettled([obsolete]) + await flushBackgroundWork() + const obsoletePath = mocks.prepareCheckout.mock.calls[0][1] + for (const base of ['origin/one', 'origin/two', 'origin/three']) { + await prepareWorktreeCreateForRepo(store, repo, base) + } + expect(signal?.aborted).toBe(true) + expect((await settled)[0].status).toBe('rejected') + await flushBackgroundWork() + expect(mocks.discard).toHaveBeenCalledWith(repo.path, obsoletePath, {}) + }) + + it('does not retry a discard whose registration the aborted checkout already removed', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, options) => { + const signal = options.signal! + return new Promise((_resolve, reject) => { + signal.addEventListener('abort', () => reject(signal.reason), { once: true }) + }) + }) + try { + const obsolete = prepareWorktreeCreateForRepo(store, repo, 'origin/main').catch(() => {}) + await flushBackgroundWork() + const obsoletePath = mocks.prepareCheckout.mock.calls[0][1] as string + mocks.discard.mockImplementation(async (_repoPath: string, path: string) => { + if (path === obsoletePath) { + throw Object.assign(new Error(`fatal: '${path}' is not a working tree`), { + stderr: `fatal: '${path}' is not a working tree` + }) + } + }) + for (const base of ['origin/one', 'origin/two', 'origin/three']) { + await prepareWorktreeCreateForRepo(store, repo, base) + } + await obsolete + await flushBackgroundWork() + const obsoleteDiscards = (): number => + mocks.discard.mock.calls.filter((call) => call[1] === obsoletePath).length + expect(obsoleteDiscards()).toBe(1) + + for (const base of ['origin/four', 'origin/five']) { + await prepareWorktreeCreateForRepo(store, repo, base) + await flushBackgroundWork() + } + expect(obsoleteDiscards()).toBe(1) + expect(warn).not.toHaveBeenCalled() + } finally { + warn.mockRestore() + } + }) + + it('does not start obsolete checkout work after shared cleanup finishes', async () => { + let releaseCleanup!: () => void + mocks.listWorktreeGraph.mockImplementationOnce( + () => + new Promise<[]>((resolve) => { + releaseCleanup = () => resolve([]) + }) + ) + const requests = ['main', 'one', 'two', 'three'].map((base) => + prepareWorktreeCreateForRepo(store, repo, `origin/${base}`) + ) + const settled = Promise.allSettled(requests) + await flushBackgroundWork() + expect(mocks.prepareCheckout).not.toHaveBeenCalled() + releaseCleanup() + const results = await settled + expect(results.map((result) => result.status)).toEqual([ + 'rejected', + 'fulfilled', + 'fulfilled', + 'fulfilled' + ]) + expect(mocks.prepareCheckout).toHaveBeenCalledTimes(3) + await flushBackgroundWork() + expect(mocks.discard).not.toHaveBeenCalled() + }) + + it('keeps a claimed in-flight checkout alive when new preparations fill the pool', async () => { + let signal: AbortSignal | undefined + let finishCheckout!: () => void + mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, options) => { + signal = options.signal + return new Promise((resolve) => { + finishCheckout = resolve + }) + }) + const preparation = prepareWorktreeCreateForRepo(store, repo, 'origin/main') + await flushBackgroundWork() + const create = consumePreparedWorktreeCreate({ + repoPath: repo.path, + workspaceRoot: '/workspace', + worktreePath: '/workspace/claimed', + branch: 'claimed', + baseBranch: 'origin/main' + }) + await flushBackgroundWork() + for (const base of ['origin/one', 'origin/two', 'origin/three', 'origin/four']) { + await prepareWorktreeCreateForRepo(store, repo, base) + } + expect(signal?.aborted).toBe(false) + finishCheckout() + await preparation + expect(await create).toMatchObject({ status: 'hit' }) + }) + + it('cancels an expired in-flight checkout', async () => { + vi.useFakeTimers() + let signal: AbortSignal | undefined + mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, options) => { + signal = options.signal + return new Promise((_resolve, reject) => { + signal!.addEventListener('abort', () => reject(signal!.reason), { once: true }) + }) + }) + try { + const settled = Promise.allSettled([prepareWorktreeCreateForRepo(store, repo, 'origin/main')]) + await vi.advanceTimersByTimeAsync(0) + expect(signal?.aborted).toBe(false) + await vi.advanceTimersByTimeAsync(WORKTREE_CREATE_PREPARATION_TTL_MS) + expect(signal?.aborted).toBe(true) + expect((await settled)[0].status).toBe('rejected') + } finally { + vi.useRealTimers() + } + }) + + it('preserves caller cancellation without mutating its options', async () => { + const controller = new AbortController() + const options = { signal: controller.signal } + mocks.getWorktreeOptions.mockReturnValue(options) + let signal: AbortSignal | undefined + mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, executionOptions) => { + signal = executionOptions.signal + return new Promise((_resolve, reject) => { + signal!.addEventListener('abort', () => reject(signal!.reason), { once: true }) + }) + }) + const preparation = prepareWorktreeCreateForRepo(store, repo, 'origin/main') + const settled = Promise.allSettled([preparation]) + await flushBackgroundWork() + controller.abort() + expect(signal?.aborted).toBe(true) + expect((await settled)[0].status).toBe('rejected') + expect(options.signal).toBe(controller.signal) + expect(signal).not.toBe(controller.signal) + }) +}) diff --git a/src/main/worktree-create-preparation-pool.ts b/src/main/worktree-create-preparation-pool.ts index 1581f404b20..8251e59a94b 100644 --- a/src/main/worktree-create-preparation-pool.ts +++ b/src/main/worktree-create-preparation-pool.ts @@ -11,7 +11,7 @@ import { prepareWorktreeCreateCheckout } from './git/worktree-create-preparation import { toHostFilesystemPath } from './host-tree-removal' import { preparationEntryKey, preparationPathKey } from './worktree-create-preparation-claim' import { - cleanupStalePreparations, + startStalePreparationCleanup, hasPendingStalePreparationCleanup, resetStalePreparationCleanupForTests } from './worktree-create-preparation-stale-cleanup' @@ -193,7 +193,11 @@ export function startPreparation({ controller, checkoutStarted: false, ready: (async () => { - await cleanupStalePreparations(preparationHostKey(repoPathKey, wslDistro), repoPath, options) + await startStalePreparationCleanup( + preparationHostKey(repoPathKey, wslDistro), + repoPath, + options + ) signal.throwIfAborted() await mkdir(toHostFilesystemPath(preparationRoot), { recursive: true }) signal.throwIfAborted() @@ -218,7 +222,7 @@ export function startPreparation({ export async function _resetPreparationPoolForTests(): Promise { const entries = [...preparations.values()] preparations.clear() - resetStalePreparationCleanupForTests() + await resetStalePreparationCleanupForTests() await Promise.all( entries.map(async (entry) => { clearTimeout(entry.expiration) diff --git a/src/main/worktree-create-preparation-stale-cleanup.ts b/src/main/worktree-create-preparation-stale-cleanup.ts index fd02b7cc0d1..1606e62ed8f 100644 --- a/src/main/worktree-create-preparation-stale-cleanup.ts +++ b/src/main/worktree-create-preparation-stale-cleanup.ts @@ -10,7 +10,7 @@ import { retryPendingPreparationDiscards } from './worktree-preparation-discard- const STALE_PREPARATION_CLEANUP_CONCURRENCY = 4 -const staleCleanupInFlight = new Map>() +const staleCleanupInFlight = new Map; settled: Promise }>() function isProcessAlive(pid: number): boolean { try { @@ -21,26 +21,24 @@ function isProcessAlive(pid: number): boolean { } } -/** Reclaims preparations a crashed process left registered. Single-flighted per host key so a burst - * of arming calls shares one worktree listing. */ -export async function cleanupStalePreparations( +/** Returns after the shared host scan; file reclamation stays tracked in the background. */ +export async function startStalePreparationCleanup( cleanupKey: string, repoPath: string, options: AddWorktreeOptions ): Promise { const existing = staleCleanupInFlight.get(cleanupKey) if (existing) { - await existing.catch(() => {}) + await existing.scanned.catch(() => {}) return } - const cleanup = (async () => { - // Not awaited: the create path awaits this cleanup, and one stranded discard costs an unlock plus - // a `worktree remove --force` bounded at 30s each. Reclaiming leaked scratch must not delay create. - void retryPendingPreparationDiscards(cleanupKey) - const worktrees = await listWorktreeGraph(repoPath, { - ...options, - includeCreatePreparations: true - }) + void retryPendingPreparationDiscards(cleanupKey) + const scan = listWorktreeGraph(repoPath, { + ...options, + includeCreatePreparations: true + }) + const scanned = scan.then(() => {}) + const cleanup = scan.then(async (worktrees) => { const staleWorktrees = worktrees.filter(isWorktreeCreatePreparation) let nextIndex = 0 async function discardNextStalePreparation(): Promise { @@ -63,22 +61,27 @@ export async function cleanupStalePreparations( } const workerCount = Math.min(STALE_PREPARATION_CLEANUP_CONCURRENCY, staleWorktrees.length) await Promise.all(Array.from({ length: workerCount }, () => discardNextStalePreparation())) - })() - staleCleanupInFlight.set(cleanupKey, cleanup) - try { - await cleanup.catch(() => {}) - } finally { - if (staleCleanupInFlight.get(cleanupKey) === cleanup) { - staleCleanupInFlight.delete(cleanupKey) - } - } + }) + // Keep reclamation single-flighted, but do not make a new checkout wait for old file removal. + const entry = { scanned, settled: cleanup } + staleCleanupInFlight.set(cleanupKey, entry) + void cleanup + .catch(() => {}) + .finally(() => { + if (staleCleanupInFlight.get(cleanupKey) === entry) { + staleCleanupInFlight.delete(cleanupKey) + } + }) + await scanned.catch(() => {}) } -/** True while a crash-recovery scan is running, which means a create is in flight or imminent. */ +/** Keeps repo maintenance paused through crash-recovery scanning and reclamation. */ export function hasPendingStalePreparationCleanup(): boolean { return staleCleanupInFlight.size > 0 } -export function resetStalePreparationCleanupForTests(): void { +export async function resetStalePreparationCleanupForTests(): Promise { + const cleanups = [...staleCleanupInFlight.values()].map((entry) => entry.settled) staleCleanupInFlight.clear() + await Promise.allSettled(cleanups) } diff --git a/src/main/worktree-create-preparation.test.ts b/src/main/worktree-create-preparation.test.ts index f6f7e295497..785ed094b33 100644 --- a/src/main/worktree-create-preparation.test.ts +++ b/src/main/worktree-create-preparation.test.ts @@ -1,7 +1,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type * as WorktreeLogic from './ipc/worktree-logic' import type { Store } from './persistence' -import { WORKTREE_CREATE_PREPARATION_TTL_MS } from './worktree-create-preparation-pool' +import { hasPendingStalePreparationCleanup } from './worktree-create-preparation-stale-cleanup' import type { Repo } from '../shared/repo-types' import { WORKTREE_CREATE_PREPARATION_DIRECTORY } from '../shared/worktree/create-preparation' import { resolveWorktreeAddBaseRef } from '../shared/worktree/base-ref' @@ -99,163 +99,6 @@ afterEach(async () => { }) describe('worktree create preparation registry', () => { - it('cancels an evicted checkout and cleans up with the original options', async () => { - let signal: AbortSignal | undefined - mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, options) => { - signal = options.signal - return new Promise((_resolve, reject) => { - signal!.addEventListener('abort', () => reject(signal!.reason), { once: true }) - }) - }) - const obsolete = prepareWorktreeCreateForRepo(store, repo, 'origin/main') - const settled = Promise.allSettled([obsolete]) - await flushBackgroundWork() - const obsoletePath = mocks.prepareCheckout.mock.calls[0][1] - for (const base of ['origin/one', 'origin/two', 'origin/three']) { - await prepareWorktreeCreateForRepo(store, repo, base) - } - expect(signal?.aborted).toBe(true) - expect((await settled)[0].status).toBe('rejected') - await flushBackgroundWork() - expect(mocks.discard).toHaveBeenCalledWith(repo.path, obsoletePath, {}) - }) - - it('does not retry a discard whose registration the aborted checkout already removed', async () => { - const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) - mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, options) => { - const signal = options.signal! - return new Promise((_resolve, reject) => { - signal.addEventListener('abort', () => reject(signal.reason), { once: true }) - }) - }) - try { - const obsolete = prepareWorktreeCreateForRepo(store, repo, 'origin/main').catch(() => {}) - await flushBackgroundWork() - const obsoletePath = mocks.prepareCheckout.mock.calls[0][1] as string - mocks.discard.mockImplementation(async (_repoPath: string, path: string) => { - if (path === obsoletePath) { - throw Object.assign(new Error(`fatal: '${path}' is not a working tree`), { - stderr: `fatal: '${path}' is not a working tree` - }) - } - }) - for (const base of ['origin/one', 'origin/two', 'origin/three']) { - await prepareWorktreeCreateForRepo(store, repo, base) - } - await obsolete - await flushBackgroundWork() - const obsoleteDiscards = (): number => - mocks.discard.mock.calls.filter((call) => call[1] === obsoletePath).length - expect(obsoleteDiscards()).toBe(1) - - for (const base of ['origin/four', 'origin/five']) { - await prepareWorktreeCreateForRepo(store, repo, base) - await flushBackgroundWork() - } - expect(obsoleteDiscards()).toBe(1) - expect(warn).not.toHaveBeenCalled() - } finally { - warn.mockRestore() - } - }) - - it('does not start obsolete checkout work after shared cleanup finishes', async () => { - let releaseCleanup!: () => void - mocks.listWorktreeGraph.mockImplementationOnce( - () => - new Promise<[]>((resolve) => { - releaseCleanup = () => resolve([]) - }) - ) - const requests = ['main', 'one', 'two', 'three'].map((base) => - prepareWorktreeCreateForRepo(store, repo, `origin/${base}`) - ) - const settled = Promise.allSettled(requests) - await flushBackgroundWork() - expect(mocks.prepareCheckout).not.toHaveBeenCalled() - releaseCleanup() - const results = await settled - expect(results.map((result) => result.status)).toEqual([ - 'rejected', - 'fulfilled', - 'fulfilled', - 'fulfilled' - ]) - expect(mocks.prepareCheckout).toHaveBeenCalledTimes(3) - await flushBackgroundWork() - expect(mocks.discard).not.toHaveBeenCalled() - }) - - it('keeps a claimed in-flight checkout alive when new preparations fill the pool', async () => { - let signal: AbortSignal | undefined - let finishCheckout!: () => void - mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, options) => { - signal = options.signal - return new Promise((resolve) => { - finishCheckout = resolve - }) - }) - const preparation = prepareWorktreeCreateForRepo(store, repo, 'origin/main') - await flushBackgroundWork() - const create = consumePreparedWorktreeCreate({ - repoPath: repo.path, - workspaceRoot: '/workspace', - worktreePath: '/workspace/claimed', - branch: 'claimed', - baseBranch: 'origin/main' - }) - await flushBackgroundWork() - for (const base of ['origin/one', 'origin/two', 'origin/three', 'origin/four']) { - await prepareWorktreeCreateForRepo(store, repo, base) - } - expect(signal?.aborted).toBe(false) - finishCheckout() - await preparation - expect(await create).toMatchObject({ status: 'hit' }) - }) - - it('cancels an expired in-flight checkout', async () => { - vi.useFakeTimers() - let signal: AbortSignal | undefined - mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, options) => { - signal = options.signal - return new Promise((_resolve, reject) => { - signal!.addEventListener('abort', () => reject(signal!.reason), { once: true }) - }) - }) - try { - const settled = Promise.allSettled([prepareWorktreeCreateForRepo(store, repo, 'origin/main')]) - await vi.advanceTimersByTimeAsync(0) - expect(signal?.aborted).toBe(false) - await vi.advanceTimersByTimeAsync(WORKTREE_CREATE_PREPARATION_TTL_MS) - expect(signal?.aborted).toBe(true) - expect((await settled)[0].status).toBe('rejected') - } finally { - vi.useRealTimers() - } - }) - - it('preserves caller cancellation without mutating its options', async () => { - const controller = new AbortController() - const options = { signal: controller.signal } - mocks.getWorktreeOptions.mockReturnValue(options) - let signal: AbortSignal | undefined - mocks.prepareCheckout.mockImplementationOnce((_repo, _path, _base, _lock, executionOptions) => { - signal = executionOptions.signal - return new Promise((_resolve, reject) => { - signal!.addEventListener('abort', () => reject(signal!.reason), { once: true }) - }) - }) - const preparation = prepareWorktreeCreateForRepo(store, repo, 'origin/main') - const settled = Promise.allSettled([preparation]) - await flushBackgroundWork() - controller.abort() - expect(signal?.aborted).toBe(true) - expect((await settled)[0].status).toBe('rejected') - expect(options.signal).toBe(controller.signal) - expect(signal).not.toBe(controller.signal) - }) - it('starts the checkout only once the async workspace root resolves', async () => { let resolveRoot!: (root: string) => void mocks.computeWorkspaceRootAsync.mockReturnValue( @@ -545,6 +388,60 @@ describe('worktree create preparation registry', () => { expect(mocks.listWorktreeGraph).toHaveBeenCalledTimes(2) }) + it('prepares while stale removal is stalled, shares its scan, and settles removal on reset', async () => { + const stalePath = '/workspace/.orca-preparing/999999999-11111111-1111-4111-8111-111111111111' + let releaseRemoval!: () => void + const removal = new Promise((resolve) => { + releaseRemoval = resolve + }) + mocks.listWorktreeGraph.mockResolvedValueOnce([ + { + path: stalePath, + branch: undefined, + lockReason: 'orca-create-preparation:v1:999999999:stale', + head: 'deadbeef', + isBare: false, + isMainWorktree: false + } + ]) + mocks.discard.mockImplementation((_repo, path) => + path === stalePath ? removal : Promise.resolve() + ) + let ready = false + let reset: Promise | undefined + const preparation = prepareWorktreeCreateForRepo(store, repo, 'origin/main').then(() => { + ready = true + }) + try { + await flushBackgroundWork() + expect(mocks.discard).toHaveBeenCalledWith(repo.path, stalePath, {}) + expect(ready).toBe(true) + await prepareWorktreeCreateForRepo(store, repo, 'origin/release') + expect(mocks.prepareCheckout).toHaveBeenCalledTimes(2) + expect(mocks.listWorktreeGraph).toHaveBeenCalledTimes(1) + expect(hasPendingStalePreparationCleanup()).toBe(true) + mocks.getWorktreeOptions.mockReturnValue({ wslDistro: 'Ubuntu' }) + await prepareWorktreeCreateForRepo(store, repo, 'origin/main') + expect(mocks.prepareCheckout).toHaveBeenCalledTimes(3) + expect(mocks.listWorktreeGraph).toHaveBeenCalledTimes(2) + expect(mocks.listWorktreeGraph).toHaveBeenLastCalledWith(repo.path, { + wslDistro: 'Ubuntu', + includeCreatePreparations: true + }) + let resetFinished = false + reset = _resetWorktreeCreatePreparationsForTests().then(() => { + resetFinished = true + }) + await flushBackgroundWork() + expect(resetFinished).toBe(false) + } finally { + releaseRemoval() + await preparation + await reset + } + expect(hasPendingStalePreparationCleanup()).toBe(false) + }) + it('unlocks a stale branch-attached final path instead of deleting user work', async () => { mocks.listWorktreeGraph.mockResolvedValueOnce([ { diff --git a/src/shared/git-binary-compatibility.test.ts b/src/shared/git-binary-compatibility.test.ts index fb8161b9f90..e11fa6ed2c4 100644 --- a/src/shared/git-binary-compatibility.test.ts +++ b/src/shared/git-binary-compatibility.test.ts @@ -204,6 +204,16 @@ describeBinaryCompatibility('real Git binary compatibility', () => { await rm(join(repoPath, 'deferred-trash'), { recursive: true, force: true }) }) + it('removes locked prepared worktrees without a separate unlock', async () => { + await runGit(['worktree', 'add', '--detach', '--no-checkout', 'compat-discard', 'HEAD']) + await runGit(['-C', 'compat-discard', 'reset', '--hard', 'HEAD']) + await runGit(['worktree', 'lock', '--reason', 'owned preparation', 'compat-discard']) + await runGit(['worktree', 'remove', '--force', '--force', 'compat-discard']) + expect((await runGit(['worktree', 'list', '--porcelain'])).stdout).not.toContain( + 'compat-discard' + ) + }) + it('supports prepared worktree creation and finalization', async () => { await runGit(['worktree', 'add', '--detach', '--no-checkout', 'compat-prepared', 'HEAD']) await runGit(['-C', 'compat-prepared', 'reset', '--hard', 'HEAD']) From 891ae62df5e7ece868c62be86bd82911c4dbc3b7 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Sat, 5 Sep 2026 22:34:56 -0700 Subject: [PATCH 61/69] fix(release): revalidate draft state before patching generated notes (#19019) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(release): revalidate draft state before patching generated notes The release listing is a snapshot taken before generate-notes runs. If the draft is published in that window, the PATCH overwrote a live release body. Re-read the release by id immediately before the update and skip it when the release is no longer a draft. * Handle publication race during draft release notes patch Between the draft status check and the PATCH request, a release can be published. The PATCH succeeds but now modifies published content. Check the PATCH response—if draft=false, publication won; restore the published body and leave generated notes unapplied. * fix(release): only roll back the draft body we actually wrote Re-read the release before the compensating PATCH and skip the rollback when the body no longer matches the notes we patched in, so a body written after our PATCH is not clobbered. --- config/scripts/create-draft-release.mjs | 49 ++++++++++- config/scripts/create-draft-release.test.mjs | 90 +++++++++++++++++++- 2 files changed, 137 insertions(+), 2 deletions(-) diff --git a/config/scripts/create-draft-release.mjs b/config/scripts/create-draft-release.mjs index b4e3f3e0933..1732e9a1e8a 100644 --- a/config/scripts/create-draft-release.mjs +++ b/config/scripts/create-draft-release.mjs @@ -164,7 +164,21 @@ export async function createDraftRelease({ if (!Number.isInteger(existingRelease.id)) { throw new Error(`Draft release ${tag} is missing a GitHub release id`) } - await githubJson( + // Why: the listing is a snapshot; the draft can be published while notes + // generate, and patching then overwrites a live release body. + const currentRelease = await githubJson( + fetchImpl, + `https://api.github.com/repos/${repo}/releases/${existingRelease.id}`, + token + ) + if (currentRelease?.draft !== true) { + log(`Release ${tag} was published while notes were generated; leaving it unchanged.`) + return + } + // Why: the PATCH endpoint supports no conditional/versioned update, so the + // GET above cannot close the window. The PATCH response reports the state we + // actually wrote to; if publication won, put the published body back. + const patchedRelease = await githubJson( fetchImpl, `https://api.github.com/repos/${repo}/releases/${existingRelease.id}`, token, @@ -173,6 +187,39 @@ export async function createDraftRelease({ body: JSON.stringify({ body }) } ) + if (patchedRelease?.draft !== true) { + const publishedBody = typeof currentRelease.body === 'string' ? currentRelease.body : '' + if (publishedBody === body) { + log(`Release ${tag} was published while notes were patched; its body is unchanged.`) + return + } + // Why: the rollback must not clobber a body written after our PATCH, so + // restore only while the release still carries exactly what we wrote. + const releaseBeforeRollback = await githubJson( + fetchImpl, + `https://api.github.com/repos/${repo}/releases/${existingRelease.id}`, + token + ) + if (releaseBeforeRollback?.body !== body) { + log( + `Release ${tag} was published and its body changed again while notes were patched; leaving the newer body in place.` + ) + return + } + await githubJson( + fetchImpl, + `https://api.github.com/repos/${repo}/releases/${existingRelease.id}`, + token, + { + method: 'PATCH', + body: JSON.stringify({ body: publishedBody }) + } + ) + log( + `Release ${tag} was published while notes were patched; restored its published body and left the generated notes unapplied.` + ) + return + } } else { // Why: GitHub's generated release notes can exceed the release body API // limit, so create with a bounded body. Omit target_commitish because the diff --git a/config/scripts/create-draft-release.test.mjs b/config/scripts/create-draft-release.test.mjs index dadc6111ac3..911ac00be63 100644 --- a/config/scripts/create-draft-release.test.mjs +++ b/config/scripts/create-draft-release.test.mjs @@ -207,7 +207,8 @@ describe('createDraftRelease', () => { jsonResponse([release('v1.4.35'), release('v1.4.36', { draft: true, id: 42 })]) ) .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'notes' })) - .mockResolvedValueOnce(jsonResponse({ id: 42, body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: true, body: 'stale' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: true, body: 'notes' })) await createDraftRelease({ repo: 'stablyai/orca', @@ -220,10 +221,97 @@ describe('createDraftRelease', () => { expect(fetchImpl).toHaveBeenNthCalledWith( 3, 'https://api.github.com/repos/stablyai/orca/releases/42', + expect.not.objectContaining({ method: expect.anything() }) + ) + expect(fetchImpl).toHaveBeenNthCalledWith( + 4, + 'https://api.github.com/repos/stablyai/orca/releases/42', expect.objectContaining({ method: 'PATCH', body: JSON.stringify({ body: 'notes' }) }) ) }) + it('skips the update when the draft was published while notes were generated', async () => { + const fetchImpl = vi + .fn() + .mockResolvedValueOnce( + jsonResponse([release('v1.4.35'), release('v1.4.36', { draft: true, id: 42 })]) + ) + .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: false })) + + await createDraftRelease({ + repo: 'stablyai/orca', + tag: 'v1.4.36', + token: 'token', + fetchImpl, + log: vi.fn() + }) + + expect(fetchImpl).toHaveBeenCalledTimes(3) + expect(fetchImpl).toHaveBeenNthCalledWith( + 3, + 'https://api.github.com/repos/stablyai/orca/releases/42', + expect.not.objectContaining({ method: expect.anything() }) + ) + }) + + it('restores the published body when publication lands between the check and the patch', async () => { + const log = vi.fn() + const fetchImpl = vi + .fn() + .mockResolvedValueOnce( + jsonResponse([release('v1.4.35'), release('v1.4.36', { draft: true, id: 42 })]) + ) + .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: true, body: 'hand-written notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: false, body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: false, body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: false, body: 'hand-written notes' })) + + await createDraftRelease({ + repo: 'stablyai/orca', + tag: 'v1.4.36', + token: 'token', + fetchImpl, + log + }) + + expect(fetchImpl).toHaveBeenCalledTimes(6) + expect(fetchImpl).toHaveBeenNthCalledWith( + 6, + 'https://api.github.com/repos/stablyai/orca/releases/42', + expect.objectContaining({ + method: 'PATCH', + body: JSON.stringify({ body: 'hand-written notes' }) + }) + ) + expect(log).toHaveBeenCalledWith(expect.stringContaining('restored its published body')) + }) + + it('leaves a body written after the patch in place instead of rolling it back', async () => { + const log = vi.fn() + const fetchImpl = vi + .fn() + .mockResolvedValueOnce( + jsonResponse([release('v1.4.35'), release('v1.4.36', { draft: true, id: 42 })]) + ) + .mockResolvedValueOnce(jsonResponse({ name: 'v1.4.36', body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: true, body: 'hand-written notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: false, body: 'notes' })) + .mockResolvedValueOnce(jsonResponse({ id: 42, draft: false, body: 'newer published body' })) + + await createDraftRelease({ + repo: 'stablyai/orca', + tag: 'v1.4.36', + token: 'token', + fetchImpl, + log + }) + + expect(fetchImpl).toHaveBeenCalledTimes(5) + expect(log).toHaveBeenCalledWith(expect.stringContaining('leaving the newer body in place')) + }) + it('preserves notes on an existing published release', async () => { const fetchImpl = vi.fn().mockResolvedValueOnce(jsonResponse([release('v1.4.36', { id: 42 })])) From 15dabf8d0ba3ab80fc80b19f7e49f3d418ac7b35 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 22:42:06 -0700 Subject: [PATCH 62/69] perf(worktree): overlap base refresh with prepared checkout (#18998) * Stop obsolete worktree preparations when evicted or expired * Let worktree preparation proceed during stale reclamation * Verify creation during stalled stale worktree reclamation * Preserve preparation ownership until Git removal starts * test: keep artifact share fixtures unexpired across calendar dates (#18955) * perf(worktree): overlap base refresh with prepared checkout --- ...rktree-create-preparation-real-git.test.ts | 52 +++++++ .../register-worktree-prefetch-handler.ts | 8 +- .../orca-runtime-create-base-prefetch.test.ts | 5 +- ...get-worktree-terminal-provisioning-host.ts | 12 +- .../worktree-create-base-prefetch.test.ts | 142 ++++++++++++++++++ src/main/worktree-create-base-prefetch.ts | 41 ++++- 6 files changed, 242 insertions(+), 18 deletions(-) diff --git a/src/main/git/worktree-create-preparation-real-git.test.ts b/src/main/git/worktree-create-preparation-real-git.test.ts index c309e35f14f..19c023a2403 100644 --- a/src/main/git/worktree-create-preparation-real-git.test.ts +++ b/src/main/git/worktree-create-preparation-real-git.test.ts @@ -279,6 +279,58 @@ describe('prepared worktree creation with real Git', () => { ) }) + it.each(['before reset', 'after reset'])( + 'finalizes refreshed content when the base moves %s', + async (when) => { + const { repoPath, root } = await createRepo() + const preparedPath = join(root, WORKTREE_CREATE_PREPARATION_DIRECTORY, 'fetch-overlap') + const finalPath = join(root, 'final-overlap') + await mkdir(join(root, WORKTREE_CREATE_PREPARATION_DIRECTORY), { recursive: true }) + const original = git(repoPath, ['rev-parse', 'HEAD']) + await writeFile(join(repoPath, 'version.txt'), 'refreshed\n') + git(repoPath, ['commit', '-am', 'remote update']) + const refreshed = git(repoPath, ['rev-parse', 'HEAD']) + git(repoPath, ['update-ref', 'refs/remotes/origin/main', original]) + const exec = gitRunner.gitExecFileAsync + let moved = false + const spy = vi + .spyOn(gitRunner, 'gitExecFileAsync') + .mockImplementation(async (args, options) => { + if (!moved && args.includes('reset') && when === 'before reset') { + git(repoPath, ['update-ref', 'refs/remotes/origin/main', refreshed]) + moved = true + } + const result = await exec(args, options) + if (!moved && args.includes('reset') && when === 'after reset') { + git(repoPath, ['update-ref', 'refs/remotes/origin/main', refreshed]) + moved = true + } + return result + }) + try { + await prepareWorktreeCreateCheckout( + repoPath, + preparedPath, + 'refs/remotes/origin/main', + createWorktreePreparationLockReason('fetch-overlap') + ) + expect(moved).toBe(true) + await finalizePreparedWorktree( + repoPath, + preparedPath, + finalPath, + 'feature/overlap', + 'refs/remotes/origin/main' + ) + expect(git(finalPath, ['rev-parse', 'HEAD'])).toBe(refreshed) + expect(await readFile(join(finalPath, 'version.txt'), 'utf8')).toBe('refreshed\n') + expect(git(finalPath, ['status', '--porcelain'])).toBe('') + } finally { + spy.mockRestore() + } + } + ) + it('hides the preparation, retargets an advanced base, and attaches the final branch', async () => { const { repoPath, root } = await createRepo() const preparationRoot = join(root, WORKTREE_CREATE_PREPARATION_DIRECTORY) diff --git a/src/main/ipc/worktrees/create/register-worktree-prefetch-handler.ts b/src/main/ipc/worktrees/create/register-worktree-prefetch-handler.ts index 2f1ab7abad9..c1bcbbf3d59 100644 --- a/src/main/ipc/worktrees/create/register-worktree-prefetch-handler.ts +++ b/src/main/ipc/worktrees/create/register-worktree-prefetch-handler.ts @@ -15,15 +15,13 @@ export function registerWorktreePrefetchHandler(context: WorktreeIpcContext): vo return } try { - const baseBranch = await prefetchWorktreeCreateBase({ + await prefetchWorktreeCreateBase({ repo, baseBranch: args.baseBranch, runtime, - gitOptions: getWorktreeCreatePrefetchGitOptions(store, repo) + gitOptions: getWorktreeCreatePrefetchGitOptions(store, repo), + prepareCheckout: (base) => prepareWorktreeCreateForRepo(store, repo, base) }) - if (baseBranch) { - await prepareWorktreeCreateForRepo(store, repo, baseBranch) - } } catch { // Why: optimistic warm-up; the real create path awaits the same refresh and reports failures there. } diff --git a/src/main/runtime/orca-runtime-create-base-prefetch.test.ts b/src/main/runtime/orca-runtime-create-base-prefetch.test.ts index 7dcd1092625..d65209b9cc4 100644 --- a/src/main/runtime/orca-runtime-create-base-prefetch.test.ts +++ b/src/main/runtime/orca-runtime-create-base-prefetch.test.ts @@ -107,7 +107,10 @@ describe('prefetchManagedWorktreeCreateBase (orca-runtime-get-worktree-terminal- it('prepares the checkout the prefetch resolved', async () => { _setWslCachesForTests({ available: true, distros: ['Ubuntu'] }) setPlatform('win32') - mocks.prefetchWorktreeCreateBase.mockResolvedValue('origin/main') + mocks.prefetchWorktreeCreateBase.mockImplementation(async ({ prepareCheckout }) => { + await prepareCheckout('origin/main') + return 'origin/main' + }) const runtime = new OrcaRuntimeService(makeStore() as never) await runtime.prefetchManagedWorktreeCreateBase({ repoSelector: 'repo-1' }) diff --git a/src/main/runtime/orca-runtime-get-worktree-terminal-provisioning-host.ts b/src/main/runtime/orca-runtime-get-worktree-terminal-provisioning-host.ts index 1591f344bda..d9d6b2acfb9 100644 --- a/src/main/runtime/orca-runtime-get-worktree-terminal-provisioning-host.ts +++ b/src/main/runtime/orca-runtime-get-worktree-terminal-provisioning-host.ts @@ -50,18 +50,12 @@ export class OrcaRuntimeWithGetWorktreeTerminalProvisioningHost extends OrcaRunt const repo = await this.resolveRepoSelector(args.repoSelector) const store = this.requireStore() - const baseBranch = await prefetchWorktreeCreateBase({ + await prefetchWorktreeCreateBase({ repo, baseBranch: args.baseBranch, runtime: this, - gitOptions: getWorktreeCreatePrefetchGitOptions(store, repo) + gitOptions: getWorktreeCreatePrefetchGitOptions(store, repo), + prepareCheckout: (base) => prepareWorktreeCreateForRepo(store, repo, base) }) - if (baseBranch) { - try { - await prepareWorktreeCreateForRepo(store, repo, baseBranch) - } catch { - // Why: speculative preparation is an optimistic warm-up; the real create path reports failures. - } - } } } diff --git a/src/main/worktree-create-base-prefetch.test.ts b/src/main/worktree-create-base-prefetch.test.ts index cfe8a76799a..af80afe4b41 100644 --- a/src/main/worktree-create-base-prefetch.test.ts +++ b/src/main/worktree-create-base-prefetch.test.ts @@ -160,12 +160,14 @@ describe('prefetchWorktreeCreateBase local git routing', () => { }) it('does not resolve a local base for SSH repos', async () => { + const prepareCheckout = vi.fn() const provider = { exec: vi.fn() } mocks.getSshGitProvider.mockReturnValue(provider) await expect( prefetchWorktreeCreateBase({ repo: { ...repo, connectionId: 'conn-1' }, + prepareCheckout, baseBranch: 'origin/main', runtime: runtime(), gitOptions: WSL @@ -178,5 +180,145 @@ describe('prefetchWorktreeCreateBase local git routing', () => { { baseBranch: 'origin/main' } ) expect(mocks.gitExecFileAsync).not.toHaveBeenCalled() + expect(prepareCheckout).not.toHaveBeenCalled() }) }) + +describe('checkout and refresh overlap', () => { + it.each([{}, WSL])( + 'starts one checkout before a blocked refresh finishes on %j', + async (gitOptions) => { + const base = { + remote: 'origin', + branch: 'main', + ref: 'refs/remotes/origin/main', + base: 'origin/main' + } + mocks.resolveRemoteTrackingBase.mockResolvedValue(base) + mocks.hasRemoteTrackingRef.mockResolvedValue(true) + let release!: () => void + mocks.getOrStartRemoteTrackingBaseRefresh.mockImplementation( + () => + new Promise((resolve) => { + release = resolve + }) + ) + const prepareCheckout = vi.fn().mockResolvedValue(undefined) + let settled = false + const result = prefetchWorktreeCreateBase({ + repo, + baseBranch: 'origin/main', + runtime: runtime(), + gitOptions, + prepareCheckout + }).finally(() => { + settled = true + }) + await vi.waitFor(() => expect(prepareCheckout).toHaveBeenCalledWith('origin/main')) + expect(settled).toBe(false) + release() + await expect(result).resolves.toBe('origin/main') + expect(prepareCheckout).toHaveBeenCalledTimes(1) + } + ) + + it('waits for refresh when the selected base is not local', async () => { + mocks.resolveRemoteTrackingBase.mockResolvedValue({ + remote: 'origin', + branch: 'main', + ref: 'refs/remotes/origin/main', + base: 'origin/main' + }) + let release!: () => void + mocks.getOrStartRemoteTrackingBaseRefresh.mockImplementation( + () => + new Promise((resolve) => { + release = resolve + }) + ) + const prepareCheckout = vi.fn().mockResolvedValue(undefined) + const result = prefetchWorktreeCreateBase({ + repo, + baseBranch: 'origin/main', + runtime: runtime(), + gitOptions: {}, + prepareCheckout + }) + await vi.waitFor(() => + expect(mocks.getOrStartRemoteTrackingBaseRefresh).toHaveBeenCalledTimes(1) + ) + expect(prepareCheckout).not.toHaveBeenCalled() + release() + await result + expect(prepareCheckout).toHaveBeenCalledTimes(1) + }) + + it('does not fail a successful refresh because preparation fails', async () => { + mocks.resolveRemoteTrackingBase.mockResolvedValue({ + remote: 'origin', + branch: 'main', + ref: 'refs/remotes/origin/main', + base: 'origin/main' + }) + mocks.hasRemoteTrackingRef.mockResolvedValue(true) + const prepareCheckout = vi.fn().mockRejectedValue(new Error('checkout failed')) + await expect( + prefetchWorktreeCreateBase({ + repo, + baseBranch: 'origin/main', + runtime: runtime(), + gitOptions: {}, + prepareCheckout + }) + ).resolves.toBe('origin/main') + expect(prepareCheckout).toHaveBeenCalledTimes(1) + }) + + it('settles preparation before propagating refresh failure', async () => { + mocks.resolveRemoteTrackingBase.mockResolvedValue({ + remote: 'origin', + branch: 'main', + ref: 'refs/remotes/origin/main', + base: 'origin/main' + }) + mocks.hasRemoteTrackingRef.mockResolvedValue(true) + const error = new Error('refresh failed') + mocks.getOrStartRemoteTrackingBaseRefresh.mockRejectedValue(error) + let release!: () => void + const prepareCheckout = vi.fn( + () => + new Promise((resolve) => { + release = resolve + }) + ) + let settled = false + const result = prefetchWorktreeCreateBase({ + repo, + baseBranch: 'origin/main', + runtime: runtime(), + gitOptions: {}, + prepareCheckout + }).finally(() => { + settled = true + }) + const assertion = expect(result).rejects.toBe(error) + await vi.waitFor(() => expect(prepareCheckout).toHaveBeenCalledTimes(1)) + expect(settled).toBe(false) + release() + await assertion + }) +}) + +it('does not prepare folder repositories', async () => { + const prepareCheckout = vi.fn() + await expect( + prefetchWorktreeCreateBase({ + repo: { ...repo, kind: 'folder' }, + runtime: runtime(), + gitOptions: {}, + prepareCheckout + }) + ).resolves.toBeUndefined() + expect(prepareCheckout).not.toHaveBeenCalled() + expect(mocks.gitExecFileAsync).not.toHaveBeenCalled() +}) diff --git a/src/main/worktree-create-base-prefetch.ts b/src/main/worktree-create-base-prefetch.ts index a0ce1822d9c..8db177a3e9c 100644 --- a/src/main/worktree-create-base-prefetch.ts +++ b/src/main/worktree-create-base-prefetch.ts @@ -45,7 +45,8 @@ async function prefetchLocalWorktreeCreateBase( repo: Repo, baseBranch: string | undefined, runtime: WorktreeCreateBasePrefetchRuntime, - options: WorktreeCreateBaseGitOptions + options: WorktreeCreateBaseGitOptions, + prepareLocalCheckout: (base: string) => void ): Promise { // Keep host-routed calls at their original arity so they stay on the runtime's default options. const optionArgs: [] | [WorktreeCreateBaseGitOptions] = options.wslDistro ? [options] : [] @@ -83,8 +84,17 @@ async function prefetchLocalWorktreeCreateBase( ...optionArgs ) if (remoteTrackingBase) { + const hasTrackingRef = await runtime.hasRemoteTrackingRef( + repo.path, + remoteTrackingBase, + ...optionArgs + ) + if (hasTrackingRef) { + // Finalization revalidates the refreshed commit before exposing the checkout. + prepareLocalCheckout(resolvedBaseBranch) + } if ( - (await runtime.hasRemoteTrackingRef(repo.path, remoteTrackingBase, ...optionArgs)) || + hasTrackingRef || !(await hasLocalWorktreeBaseRef(repo.path, resolvedBaseBranch, options)) ) { await runtime.getOrStartRemoteTrackingBaseRefresh( @@ -114,6 +124,7 @@ export async function prefetchWorktreeCreateBase(args: { /** Routing for the project's Git host; required so a caller cannot silently * warm up the wrong ref store — pass `{}` for host Git. */ gitOptions: WorktreeCreateBaseGitOptions + prepareCheckout?: (base: string) => Promise }): Promise { if (isFolderRepo(args.repo)) { return undefined @@ -126,5 +137,29 @@ export async function prefetchWorktreeCreateBase(args: { await prefetchRemoteWorktreeCreateBase(provider, args.repo, { baseBranch: args.baseBranch }) return undefined } - return prefetchLocalWorktreeCreateBase(args.repo, args.baseBranch, args.runtime, args.gitOptions) + const prepareCheckout = args.prepareCheckout + let preparation: Promise | undefined + const prepare = (base: string): void => { + if (!preparation && prepareCheckout) { + preparation = Promise.resolve() + .then(() => prepareCheckout(base)) + .catch(() => {}) + } + } + try { + const base = await prefetchLocalWorktreeCreateBase( + args.repo, + args.baseBranch, + args.runtime, + args.gitOptions, + prepare + ) + if (base) { + prepare(base) + } + return base + } finally { + // Settle speculative work even if refresh fails; Create owns error reporting. + await preparation + } } From a37a0b50d105b95abdb99e025b2906d6adaea8a1 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 23:44:45 -0700 Subject: [PATCH 63/69] test: await fresh inventory after headless terminal materialization (#19028) * test: restore headless folder terminal materialization coverage * test: await a fresh terminal census after materialization --- .../helpers/startup-exec-readiness-oracle.ts | 32 ++++++----------- .../helpers/terminal-inventory-observation.ts | 14 ++++++++ ...terminal-materialization-reconnect.spec.ts | 34 +++++++------------ 3 files changed, 37 insertions(+), 43 deletions(-) create mode 100644 tests/e2e/helpers/terminal-inventory-observation.ts diff --git a/tests/e2e/helpers/startup-exec-readiness-oracle.ts b/tests/e2e/helpers/startup-exec-readiness-oracle.ts index 577702c8ea0..7aa56e38863 100644 --- a/tests/e2e/helpers/startup-exec-readiness-oracle.ts +++ b/tests/e2e/helpers/startup-exec-readiness-oracle.ts @@ -9,6 +9,7 @@ import type { import { toWebTerminalSurfaceTabId } from '../../../src/shared/terminal-surface-id' import { expect } from './orca-app' import { getTerminalContent, waitForActivePanePtyId } from './terminal' +import { readFreshTerminalInventory } from './terminal-inventory-observation' const RECOVERY_DEADLINE_MS = 8_000 @@ -76,10 +77,6 @@ function count(text: string, marker: string): number { return text.split(marker).length - 1 } -function isTransientPtyLivenessError(error: unknown): boolean { - return error instanceof Error && error.message.includes('terminal_liveness_unavailable') -} - async function expectSingleOwningPty( page: Page, worktreeId: string, @@ -90,24 +87,15 @@ async function expectSingleOwningPty( await expect .poll( async () => { - try { - const listed = await callStartupExecRuntime( - page, - 'terminal.list', - { - worktree: `id:${worktreeId}`, - requireFreshPtyLiveness: true - } - ) - return listed.terminals - .filter((candidate) => candidate.tabId === tabId) - .map((candidate) => ({ handle: candidate.handle, ptyId: candidate.ptyId })) - } catch (error) { - if (isTransientPtyLivenessError(error)) { - return [] - } - throw error - } + const listed = await readFreshTerminalInventory(() => + callStartupExecRuntime(page, 'terminal.list', { + worktree: `id:${worktreeId}`, + requireFreshPtyLiveness: true + }) + ) + return (listed?.terminals ?? []) + .filter((candidate) => candidate.tabId === tabId) + .map((candidate) => ({ handle: candidate.handle, ptyId: candidate.ptyId })) }, { timeout: 30_000 } ) diff --git a/tests/e2e/helpers/terminal-inventory-observation.ts b/tests/e2e/helpers/terminal-inventory-observation.ts new file mode 100644 index 00000000000..0fcefdb23c8 --- /dev/null +++ b/tests/e2e/helpers/terminal-inventory-observation.ts @@ -0,0 +1,14 @@ +import type { RuntimeTerminalListResult } from '../../../src/shared/runtime-types' + +export async function readFreshTerminalInventory( + read: () => Promise +): Promise { + try { + return await read() + } catch (error) { + if (error instanceof Error && error.message.includes('terminal_liveness_unavailable')) { + return null + } + throw error + } +} diff --git a/tests/e2e/paired-remote-terminal-materialization-reconnect.spec.ts b/tests/e2e/paired-remote-terminal-materialization-reconnect.spec.ts index ef331ee6277..fda18fc74c6 100644 --- a/tests/e2e/paired-remote-terminal-materialization-reconnect.spec.ts +++ b/tests/e2e/paired-remote-terminal-materialization-reconnect.spec.ts @@ -16,6 +16,7 @@ import { launchPairedElectronClient } from './helpers/paired-electron-client' import { getTerminalContent, waitForActivePanePtyId } from './helpers/terminal' +import { readFreshTerminalInventory } from './helpers/terminal-inventory-observation' const scratch = mkdtempSync(path.join(os.tmpdir(), 'orca-paired-materialize-')) const fixturePath = path.join(scratch, 'materialize-terminal.mjs') @@ -316,18 +317,17 @@ async function runMaterializationJourney( await tab.click() await expect.poll(() => getTerminalContent(page), { timeout: 10_000 }).toContain(marker) - const listed = await callRuntime( - page, - environmentId, - 'terminal.list', - { - worktree: `id:${worktreeId}`, - requireFreshPtyLiveness: true - } - ) - expect( - listed.terminals.filter((terminal) => terminal.tabId === created.tab.parentTabId) - ).toHaveLength(1) + await expect + .poll(async () => { + const listed = await readFreshTerminalInventory(() => + callRuntime(page, environmentId, 'terminal.list', { + worktree: `id:${worktreeId}`, + requireFreshPtyLiveness: true + }) + ) + return listed?.terminals.filter((terminal) => terminal.tabId === created.tab.parentTabId) + }) + .toHaveLength(1) await callRuntime(page, environmentId, 'terminal.closeTab', { terminal: replacementHandle }) } @@ -354,15 +354,7 @@ test('materializes a stopped terminal on reconnect from a headed paired host', a } }) -// Why fixme: this journey's fault injection cannot be set up on a headless `orca serve` host. -// `terminal.stopExact` keeps returning terminal_exact_stop_failed because stopAndWait's -// keep-history verification window expires before the parked PTY is observed gone, so the pane -// never reaches pending-handle and the reconnect behavior is never exercised. That precondition -// fails identically on this PR's base, so it is a pre-existing exact-stop defect rather than a -// reconnect-activation one. The recovery behavior itself was confirmed by hand in this topology -// (the host materializes the pending surface and the client rebinds to the replacement PTY); -// re-enable once exact stop settles deterministically against a serve host. -test.fixme('materializes a stopped terminal on reconnect from a headless folder host', async ({ +test('materializes a stopped terminal on reconnect from a headless folder host', async ({ testRepoPath }, testInfo) => { test.setTimeout(150_000) From ced8a93bfdb7ddb85405d312cc305ac31fe61766 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 23:47:47 -0700 Subject: [PATCH 64/69] fix(sidebar): stop a missed pointerup from hiding a remote host section (#19032) Clicking a host header arms a drag session on pointerdown, but the window pointermove/pointerup listeners attach from an effect gated on that state -- a render and a paint later. On a heavy sidebar a quick click's pointerup can land inside that window and never be seen, so the session survives the click and the next bare mouse move clears the 4px threshold and promotes a drag the user is not doing. The host tier is the only header that hides itself while dragging (opacity-0, plus forceCollapseHosts on every section), so the host the user just collapsed vanishes outright. It only returns on a stray later pointerup or when the viewport remounts -- which is why toggling the host filter fixes it: the viewport's React key includes visibleWorkspaceHostIds. Treat a pointermove with no button held as a released pointer and end the session instead of promoting. The repo and project-group header drags share the race, where it commits an unintended reorder on the next click, so they get the same guard. Extracting their duplicated click-swallow block keeps project-header-drag.ts under the max-lines ceiling. --- .../sidebar/header-drag-click-swallow.ts | 20 ++++ .../sidebar/header-drag-pointer-release.ts | 14 +++ .../sidebar/host-header-drag.test.tsx | 92 +++++++++++++++++++ .../components/sidebar/host-header-drag.ts | 18 ++-- .../sidebar/project-group-header-drag.ts | 21 ++--- .../components/sidebar/project-header-drag.ts | 21 ++--- 6 files changed, 147 insertions(+), 39 deletions(-) create mode 100644 src/renderer/src/components/sidebar/header-drag-click-swallow.ts create mode 100644 src/renderer/src/components/sidebar/header-drag-pointer-release.ts create mode 100644 src/renderer/src/components/sidebar/host-header-drag.test.tsx diff --git a/src/renderer/src/components/sidebar/header-drag-click-swallow.ts b/src/renderer/src/components/sidebar/header-drag-click-swallow.ts new file mode 100644 index 00000000000..1e875558b70 --- /dev/null +++ b/src/renderer/src/components/sidebar/header-drag-click-swallow.ts @@ -0,0 +1,20 @@ +/** + * Swallow the click that follows a completed header drag. + * + * Why: the pointerup that ends a promoted drag is followed by a click on the + * drag handle, which would also toggle the section the user just reordered. + * The listener removes itself on the first click; the returned timeout handle + * is the fallback for a drop that produces no click. + */ +export function swallowNextClickOnDragHandle(handleEl: HTMLElement): ReturnType { + const swallow = (event: MouseEvent): void => { + const target = event.target as Node | null + if (target && handleEl.contains(target)) { + event.stopPropagation() + event.preventDefault() + } + window.removeEventListener('click', swallow, true) + } + window.addEventListener('click', swallow, true) + return setTimeout(() => window.removeEventListener('click', swallow, true), 0) +} diff --git a/src/renderer/src/components/sidebar/header-drag-pointer-release.ts b/src/renderer/src/components/sidebar/header-drag-pointer-release.ts new file mode 100644 index 00000000000..4e8466ed3e7 --- /dev/null +++ b/src/renderer/src/components/sidebar/header-drag-pointer-release.ts @@ -0,0 +1,14 @@ +/** + * True when a pointermove arrives with no button held, meaning the pointerup + * that should have ended the armed drag never reached us. + * + * Why: the header drag hooks subscribe to window pointer events from an effect + * armed by pointerdown state, so a fast click's release can land before that + * effect runs (a heavy sidebar render sits between them). The session then + * survives the click and the next hover promotes a drag the user is not doing — + * for host sections that hides the header outright and force-collapses every + * host. A capture-phase listener swallowing pointerup has the same effect. + */ +export function hasPointerBeenReleased(event: PointerEvent): boolean { + return event.buttons === 0 +} diff --git a/src/renderer/src/components/sidebar/host-header-drag.test.tsx b/src/renderer/src/components/sidebar/host-header-drag.test.tsx new file mode 100644 index 00000000000..6c9e0f66e33 --- /dev/null +++ b/src/renderer/src/components/sidebar/host-header-drag.test.tsx @@ -0,0 +1,92 @@ +// @vitest-environment happy-dom +import React from 'react' +import { act, render } from '@testing-library/react' +import { describe, expect, it, vi } from 'vitest' + +import { useHostHeaderDrag } from './host-header-drag' +import type { ExecutionHostId } from '../../../../shared/execution-host' + +function setup() { + const scrollContainer = document.createElement('div') + document.body.append(scrollContainer) + const controller: { current: ReturnType | null } = { current: null } + + function Harness(): React.JSX.Element { + const drag = useHostHeaderDrag({ + orderedHostIds: ['ssh:host-a', 'ssh:host-b'] as ExecutionHostId[], + onCommit: vi.fn(), + getScrollContainer: () => scrollContainer + }) + controller.current = drag + return ( +
drag.onHandlePointerDown(event, 'ssh:host-a')} + /> + ) + } + + const view = render() + const header = view.container.querySelector('[data-host-header-drag-id]')! + header.setPointerCapture = vi.fn() + header.releasePointerCapture = vi.fn() + return { controller, header } +} + +function pointer(type: string, init: PointerEventInit): PointerEvent { + return new PointerEvent(type, { bubbles: true, pointerId: 1, ...init }) +} + +describe('useHostHeaderDrag', () => { + it('does not start a drag when the pointer is released before the window listeners attach', () => { + const { controller, header } = setup() + + // A click: pointerdown arms the session, pointerup lands before React has + // flushed the passive effect that subscribes to window pointer events. + act(() => { + header.dispatchEvent(pointer('pointerdown', { button: 0, clientX: 10, clientY: 10 })) + window.dispatchEvent(pointer('pointerup', { clientX: 10, clientY: 10 })) + }) + + // Moving the mouse afterwards, with no button held, must not promote a drag. + act(() => { + window.dispatchEvent(pointer('pointermove', { clientX: 200, clientY: 400, buttons: 0 })) + }) + + expect(controller.current?.state.draggingHostId).toBeNull() + }) + + it('clears a session whose pointerup was missed so a later drag still works', () => { + const { controller, header } = setup() + + act(() => { + header.dispatchEvent(pointer('pointerdown', { button: 0, clientX: 10, clientY: 10 })) + window.dispatchEvent(pointer('pointerup', { clientX: 10, clientY: 10 })) + }) + act(() => { + window.dispatchEvent(pointer('pointermove', { clientX: 200, clientY: 400, buttons: 0 })) + }) + + act(() => { + header.dispatchEvent(pointer('pointerdown', { button: 0, clientX: 10, clientY: 10 })) + }) + act(() => { + window.dispatchEvent(pointer('pointermove', { clientX: 40, clientY: 60, buttons: 1 })) + }) + + expect(controller.current?.state.draggingHostId).toBe('ssh:host-a') + }) + + it('still promotes a drag while the pointer stays down', () => { + const { controller, header } = setup() + + act(() => { + header.dispatchEvent(pointer('pointerdown', { button: 0, clientX: 10, clientY: 10 })) + }) + act(() => { + window.dispatchEvent(pointer('pointermove', { clientX: 40, clientY: 60, buttons: 1 })) + }) + + expect(controller.current?.state.draggingHostId).toBe('ssh:host-a') + }) +}) diff --git a/src/renderer/src/components/sidebar/host-header-drag.ts b/src/renderer/src/components/sidebar/host-header-drag.ts index 8b6ea676a46..cb1955dda02 100644 --- a/src/renderer/src/components/sidebar/host-header-drag.ts +++ b/src/renderer/src/components/sidebar/host-header-drag.ts @@ -16,6 +16,8 @@ import { readHostHeaderRects, type HostHeaderRect } from './host-header-drag-dom' +import { hasPointerBeenReleased } from './header-drag-pointer-release' +import { swallowNextClickOnDragHandle } from './header-drag-click-swallow' export type HostDragState = { draggingHostId: ExecutionHostId | null @@ -157,17 +159,7 @@ export function useHostHeaderDrag({ session.preview?.remove() setSidebarPointerDragDocumentStyles(false) if (session.promoted) { - const handleEl = session.handleEl - const swallow = (e: MouseEvent): void => { - const target = e.target as Node | null - if (target && handleEl.contains(target)) { - e.stopPropagation() - e.preventDefault() - } - window.removeEventListener('click', swallow, true) - } - window.addEventListener('click', swallow, true) - setTimeout(() => window.removeEventListener('click', swallow, true), 0) + swallowNextClickOnDragHandle(session.handleEl) } const finalIndex = commit && session.promoted @@ -206,6 +198,10 @@ export function useHostHeaderDrag({ if (!session || e.pointerId !== session.pointerId) { return } + if (hasPointerBeenReleased(e)) { + endDrag(false) + return + } if (!session.promoted) { const dx = e.clientX - session.startX const dy = e.clientY - session.startY diff --git a/src/renderer/src/components/sidebar/project-group-header-drag.ts b/src/renderer/src/components/sidebar/project-group-header-drag.ts index e2a879aebdc..853adfc751f 100644 --- a/src/renderer/src/components/sidebar/project-group-header-drag.ts +++ b/src/renderer/src/components/sidebar/project-group-header-drag.ts @@ -15,6 +15,8 @@ import { } from './project-group-header-drag-contract' import { createProjectGroupHeaderDragSession } from './project-group-header-drag-start' import { getWorktreeSidebarDragAutoscroll } from './worktree-sidebar-drag-autoscroll' +import { hasPointerBeenReleased } from './header-drag-pointer-release' +import { swallowNextClickOnDragHandle } from './header-drag-click-swallow' // Why pointer events instead of HTML5 DnD: Project Group rows are virtualized // and may unmount while scrolling; cached row-model indices keep drops stable. @@ -114,20 +116,7 @@ export function useProjectGroupHeaderDrag({ // capture may already be released (pointercancel, element unmounted) } if (session.promoted) { - const handleEl = session.handleEl - const swallow = (event: MouseEvent): void => { - const target = event.target as Node | null - if (target && handleEl.contains(target)) { - event.stopPropagation() - event.preventDefault() - } - window.removeEventListener('click', swallow, true) - } - window.addEventListener('click', swallow, true) - clickSwallowTimeoutRef.current = setTimeout(() => { - window.removeEventListener('click', swallow, true) - clickSwallowTimeoutRef.current = null - }, 0) + clickSwallowTimeoutRef.current = swallowNextClickOnDragHandle(session.handleEl) } const sidebarDropIndex = commit && session.promoted && latestDropIndexRef.current !== null @@ -199,6 +188,10 @@ export function useProjectGroupHeaderDrag({ if (!session || event.pointerId !== session.pointerId) { return } + if (hasPointerBeenReleased(event)) { + endDrag(false) + return + } session.latestPointerY = event.clientY if (!session.promoted) { const dx = event.clientX - session.startX diff --git a/src/renderer/src/components/sidebar/project-header-drag.ts b/src/renderer/src/components/sidebar/project-header-drag.ts index 4d13ddfa0af..68ab299d616 100644 --- a/src/renderer/src/components/sidebar/project-header-drag.ts +++ b/src/renderer/src/components/sidebar/project-header-drag.ts @@ -15,6 +15,8 @@ import { } from './project-header-drag-contract' import { createProjectHeaderDragSession } from './project-header-drag-start' import { getWorktreeSidebarDragAutoscroll } from './worktree-sidebar-drag-autoscroll' +import { hasPointerBeenReleased } from './header-drag-pointer-release' +import { swallowNextClickOnDragHandle } from './header-drag-click-swallow' // Why pointer events instead of HTML5 DnD: rows are absolutely-positioned by // react-virtual and unmount/remount as scroll changes, so DnD enter/leave fire @@ -124,20 +126,7 @@ export function useRepoHeaderDrag({ // capture may already be released (pointercancel, element unmounted) } if (session.promoted) { - const handleEl = session.handleEl - const swallow = (e: MouseEvent): void => { - const target = e.target as Node | null - if (target && handleEl.contains(target)) { - e.stopPropagation() - e.preventDefault() - } - window.removeEventListener('click', swallow, true) - } - window.addEventListener('click', swallow, true) - clickSwallowTimeoutRef.current = setTimeout(() => { - window.removeEventListener('click', swallow, true) - clickSwallowTimeoutRef.current = null - }, 0) + clickSwallowTimeoutRef.current = swallowNextClickOnDragHandle(session.handleEl) } const sidebarDropIndex = commit && session.promoted && latestDropIndexRef.current !== null @@ -212,6 +201,10 @@ export function useRepoHeaderDrag({ if (!session || e.pointerId !== session.pointerId) { return } + if (hasPointerBeenReleased(e)) { + endDrag(false) + return + } session.latestPointerY = e.clientY if (!session.promoted) { const dx = e.clientX - session.startX From e73f8dfa0f49246aa651fc0f2c5a85615d088aab Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 5 Sep 2026 23:50:30 -0700 Subject: [PATCH 65/69] test: await board pointer readiness before marquee selection (#19029) --- ...orkspace-board-lane-virtualization.spec.ts | 53 ++++++++++--------- 1 file changed, 28 insertions(+), 25 deletions(-) diff --git a/tests/e2e/workspace-board-lane-virtualization.spec.ts b/tests/e2e/workspace-board-lane-virtualization.spec.ts index 69a18e06937..36da44dd357 100644 --- a/tests/e2e/workspace-board-lane-virtualization.spec.ts +++ b/tests/e2e/workspace-board-lane-virtualization.spec.ts @@ -307,7 +307,6 @@ test.describe('Workspace board lane virtualization', () => { }) test('selects the full lane across a single large marquee scroll jump', async ({ orcaPage }) => { - test.skip(true, 'Quarantined by https://github.com/stablyai/orca/issues/12415') const statusId = 'virtual-marquee' const emptyStatusId = 'virtual-marquee-start' await orcaPage.evaluate( @@ -380,32 +379,36 @@ test.describe('Workspace board lane virtualization', () => { } // Why: CI can overlay individual lane pixels, so choose a live board-owned point. - const startPoint = await emptyLaneScroll.evaluate((element) => { - const ignored = [ - '[data-workspace-board-card-id]', - 'a', - 'button', - 'input', - 'select', - 'textarea', - '[role="button"]', - '[role="menu"]', - '[role="menuitem"]' - ].join(',') - const rect = element.getBoundingClientRect() - for (let y = Math.ceil(rect.top) + 6; y <= Math.floor(rect.top) + 40; y += 6) { - for (let x = Math.ceil(rect.left) + 8; x <= Math.floor(rect.right) - 8; x += 8) { - const target = document.elementFromPoint(x, y) - if ( - target?.closest('[data-workspace-board-selection-surface]') && - !target.closest(ignored) - ) { - return { x, y } + const findStartPoint = () => + emptyLaneScroll.evaluate((element) => { + const ignored = [ + '[data-workspace-board-card-id]', + 'a', + 'button', + 'input', + 'select', + 'textarea', + '[role="button"]', + '[role="menu"]', + '[role="menuitem"]' + ].join(',') + const rect = element.getBoundingClientRect() + for (let y = Math.ceil(rect.top) + 6; y <= Math.floor(rect.top) + 40; y += 6) { + for (let x = Math.ceil(rect.left) + 8; x <= Math.floor(rect.right) - 8; x += 8) { + const target = document.elementFromPoint(x, y) + if ( + target?.closest('[data-workspace-board-selection-surface]') && + !target.closest(ignored) + ) { + return { x, y } + } } } - } - return null - }) + return null + }) + // The board's clip animation can expose cards before the empty lane accepts pointer hits. + await expect.poll(findStartPoint).not.toBeNull() + const startPoint = await findStartPoint() expect(startPoint, 'the empty start lane must expose board-owned space').not.toBeNull() if (!startPoint) { throw new Error('Expected empty board space for the marquee start') From 0f27445789ddd3c0d2a3656bb176fb0cd3200431 Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Sun, 6 Sep 2026 00:02:05 -0700 Subject: [PATCH 66/69] fix(build): pin config/relay-assets LF so one release is one relay hash (#19024) * fix(build): pin config/relay-assets LF so one release is one relay hash * test(build): correct why the negative fixtures exist Review measured it: the first assertion checks the eol attribute via check-attr, not file content, so it fails first without the pin. The fixtures add over-broadness coverage, they do not carry the test. * test(build): key the relay line-ending pin off the manifest, not a directory A path glob proves the directory is non-empty, not that it is still the directory build-relay reads from. Relocating an asset into config/scripts (where only **/*.mjs is pinned) reintroduced the CRLF bug with the suite fully green. RELAY_ARTIFACTS is the right anchor: build-relay refuses to emit an artifact absent from it, so a relocated or new asset cannot slip past. Bundles have no tracked source and drop out with zero hits. --------- Co-authored-by: Orca Worker --- .gitattributes | 5 ++ .../relay-asset-line-ending-pin.test.mjs | 80 +++++++++++++++++++ 2 files changed, 85 insertions(+) create mode 100644 config/scripts/relay-asset-line-ending-pin.test.mjs diff --git a/.gitattributes b/.gitattributes index 1ce5b29ee45..8f4f884295d 100644 --- a/.gitattributes +++ b/.gitattributes @@ -8,6 +8,11 @@ /src/cli/bundled-skill-guides.ts text eol=lf # Bundled plugin trees are byte-hashed; CRLF checkout would break the pinned hash. /resources/plugins/** text eol=lf +# Relay assets are copied verbatim into the bundle and hashed byte-for-byte into +# .version, which names the immutable remote install dir. A CRLF checkout makes a +# Windows-built client disagree with a mac/Linux-built one on the same release, +# so one host ends up with two relay trees (#17886 review). +/config/relay-assets/** text eol=lf # Pin the bytes so a patch reads and diffs identically on every host. It is NOT # what makes the hash right: pnpm hashes a patch LF-normalized, so a CRLF checkout # cannot change it. Believing otherwise put a hand-computed raw digest in the diff --git a/config/scripts/relay-asset-line-ending-pin.test.mjs b/config/scripts/relay-asset-line-ending-pin.test.mjs new file mode 100644 index 00000000000..6e0f358f333 --- /dev/null +++ b/config/scripts/relay-asset-line-ending-pin.test.mjs @@ -0,0 +1,80 @@ +import { execFileSync } from 'node:child_process' +import { resolve } from 'node:path' +import { RELAY_ARTIFACTS } from '../../src/shared/relay-artifacts.ts' +import { describe, expect, it } from 'vitest' + +/** + * Guard the `.gitattributes` pin that keeps `config/relay-assets` on LF. + * + * `core.autocrlf=true` ships in the Git-for-Windows system config, so without a + * pin a Windows runner checks these out as CRLF. build-relay.mjs copies them + * verbatim into the bundle and hashes them byte-for-byte into `.version`, which + * names the immutable remote relay directory -- so a Windows-built client and a + * mac/Linux-built one disagree on the same release, and one SSH host ends up with + * two relay trees, each paying its own remote native-dep compile. + * + * Measured on v1.4.197: master-cloexec-patch.cjs shipped at 11229 bytes from the + * mac runner and 11547 (= 11229 + 318 lines) from the Windows one. + */ +const projectDir = resolve(import.meta.dirname, '../..') + +function git(args) { + return execFileSync('git', args, { cwd: projectDir, encoding: 'utf8' }) +} + +/** `git check-attr -z` emits NUL-separated path/attr/value triples. */ +function eolAttributes(paths) { + const fields = git(['check-attr', '-z', 'eol', '--', ...paths]).split('\0') + const found = new Map() + for (let index = 0; index + 2 < fields.length; index += 3) { + found.set(fields[index], fields[index + 2]) + } + return found +} + +/** + * Keyed off the manifest, not a directory: build-relay refuses to emit an + * artifact absent from RELAY_ARTIFACTS, so relocating an asset cannot slip + * past this the way a path glob would. esbuild bundles have no tracked + * source and contribute no hits, so they need no classifying. + */ +function trackedManifestSources() { + const paths = new Set() + for (const { filename } of RELAY_ARTIFACTS) { + const hits = git(['ls-files', '-z', '--', `*/${filename}`]).split('\0').filter(Boolean) + for (const path of hits) { + paths.add(path) + } + } + return [...paths] +} + +describe('config/relay-assets line-ending pin', () => { + it('pins every tracked relay artifact source to LF', () => { + const assets = trackedManifestSources() + expect(assets.length).toBeGreaterThan(0) + + const attributes = eolAttributes(assets) + const unpinned = assets.filter((path) => attributes.get(path) !== 'lf') + + expect( + unpinned, + 'A relay asset left on the platform default gets CRLF on a Windows runner, ' + + 'which changes the .version hash and splits one release across two remote ' + + 'relay directories. Pin it in .gitattributes.' + ).toEqual([]) + }) + + // Why: the assertion above only sees files that exist today. These fix the + // pattern itself -- broad enough to cover a file added tomorrow, narrow enough + // not to claim neighbours. + it.each([ + ['config/relay-assets/example.cjs', 'lf'], + ['config/relay-assets/nested/deeper/example.cjs', 'lf'], + ['config/relay-assets/example.txt', 'lf'], + ['config/relay-assets-extra/example.cjs', 'unspecified'], + ['vendor/config/relay-assets/example.cjs', 'unspecified'] + ])('resolves %s to eol=%s', (path, expected) => { + expect(eolAttributes([path]).get(path)).toBe(expected) + }) +}) From 1326d6b40ccca0fbd1743e023bd2adec803ba09c Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:02:13 -0400 Subject: [PATCH 67/69] docs(relay): record Roll 2 phase 0/1 (code merge, image, director deploy) (#18979) --- cloud/docs/relay-reconnect-2026-09-findings.md | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/cloud/docs/relay-reconnect-2026-09-findings.md b/cloud/docs/relay-reconnect-2026-09-findings.md index 426a120c251..580a4da84d8 100644 --- a/cloud/docs/relay-reconnect-2026-09-findings.md +++ b/cloud/docs/relay-reconnect-2026-09-findings.md @@ -989,3 +989,14 @@ Owner: "sure, feel free to drive these." Sequence chosen: Roll 1 first (highest | Monitor dry-run #50 | Dispatched 21:54Z at gen 146 on main `51eed5a1bc`, run 33994385666. **Green** 22:10Z, 16/16 samples. Main had moved to `d7767fb196`; trusted paths identical to `a3c1d32995`. Chain dispatched the c29 `canary-apply` (run 33995164002, protocol 0) 12 s after green. | | c29 canary (run 33995164002, `canary-apply`) | **Success** 22:27Z. Isolate → migration-only at **gen 147**, verifier passed on the old image (1 199 assignments), Terraform applied same-cap template `…20260905221622`, new incarnation on `519f4914` at protocol 0, verifier passed at migration-only, activate → **gen 148**, c29 general, verifier passed (1 199 assignments carried). No `container die` fleet-wide 22:11Z–22:30Z. | | **Roll 1 complete** | Image census 22:30Z from MIG templates: c8–c10, c13–c16, c19–c29 on `519f4914` (18 cells); c7 on `85bf6799` (the earlier rehearsal image, carries the same fix); existing-only c1–c6, c11, c12 and migration-only c17, c18 untouched by design. No serving cell remains on `5aedbca5`. Selector gen 148, membership unchanged from the start of the roll. Zero relay container exits fleet-wide across the roll (01:14Z–22:30Z). Gates used: #19–#50; freezes were all monitor-side (provenance, freshness, flat Asia latency bar, one Cloud Monitoring collector failure), none a fleet health finding. Roll 2 (fresh image with #18722 + #18720) is the next data-plane step and waits on the owner's private-IP window decision. | + +## Roll 2 (image `4916ed67`, 2026-09-06) + +| Step | Result | Evidence | +|---|---|---| +| Docs split | #18958 merged `3bb038a185` (findings, checklist, roadmap, Roll 2 plan). | | +| Code PR | #18959 merged `61b09b7a02` (rebase of #18565 onto main; desktop rotation change dropped since #18719 shipped a proportional version). Two Opus review rounds: round 1 caught the mobile fail-fast rejecting on any socket close (one AP flap would book the 60 s cooldown) → 2 s grace, re-armed once on `handshaking`; round 2 caught a removed jitter assertion that let a one-sided jitter pass → exact pin on the top of the band. Control lease 55 min → 6 h ± 30 min. | | +| Image publish | run 34002233801 → `sha256:4916ed676d8389f694a648e750f1112d9002d68c84a1e0c7af828d5af129de62`; mirrored to staging (run 34002326150). | | +| Staging cell smoke | **Dropped.** Staging C4 is pinned to the Asia launch digest by `relay-staging-c4-refresh-workflow.test.mjs` (with production c27–c29 tfvars and the C4 recovery workflow) and the only C4 image-refresh path pins its accepted predecessor to an older digest. Re-pinning all of it for a smoke widens into the Asia launch machinery; #18969 closed. Roll 2 follows the Roll 1 path: director first, c7 as the rehearsal cell. | | +| Director deploy | run 34002673626 **success** 01:02Z: serving `orca-cloud-relay-00575-leq` on `4916ed67`, `00574-wag` (same image) tagged `selector-rollback`, `00569-ret` (`519f4914`) still deployable. Baseline before: 1 director Postgres retry in the prior hour, 0 `container die`. | | +| c7 `verify` (read-only) | run 34002885408 dispatched 01:03Z, target `4916ed67`, rollback `85bf6799`, protocol 1, gen 148. | | From a567e33bf7d8b1bd5eee0306fd4f81d75cffa8e2 Mon Sep 17 00:00:00 2001 From: NaoyaTatetsu Date: Sun, 6 Sep 2026 16:03:41 +0900 Subject: [PATCH 68/69] feat(github-projects): render Roadmap project views as a timeline (#17795) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Add roadmap timeline view for GitHub Projects - Renders roadmap-layout project views as a scrollable timeline with date/iteration-based placement, zoom levels, and grouped lanes, instead of surfacing them as unsupported - Derives placement fields from view config or row-carried field values since GitHub's API never exposes a roadmap's date source directly - Falls back to the existing table list when no field can place items * Fix roadmap timeline edge cases: reject invalid calendar dates and refre - parseRoadmapDate previously let Date.UTC silently normalize overflowing dates (e.g. 2026-02-30 → Mar 2); now round-trips components to reject them - ProjectRoadmap's "today" marker was frozen at mount, so panes left open across midnight showed the wrong day; now re-derives and re-arms a timer * fix(github-projects): center roadmaps when dated rows arrive * fix(github-projects): keep pinned roadmap header opaque * fix: remove stale pnpm executable lockfile entries * fix(i18n): retain replaced project labels in runtime catalog --------- Co-authored-by: Neil <4138956+nwparker@users.noreply.github.com> --- .../project-view-field-normalization.ts | 10 +- .../project-view/project-view-table.test.ts | 107 +++++ .../github/project-view/project-view-table.ts | 16 +- .../github-project/ProjectGroupHeader.tsx | 42 +- .../github-project/ProjectPickerPanels.tsx | 24 +- .../github-project/ProjectRoadmap.test.tsx | 273 ++++++++++++ .../github-project/ProjectRoadmap.tsx | 419 ++++++++++++++++++ .../github-project/ProjectRoadmapBar.tsx | 92 ++++ .../github-project/ProjectViewStates.tsx | 27 +- .../github-project/ProjectViewWrapper.tsx | 9 +- .../github-project/roadmap-tick-format.ts | 68 +++ .../github-project/roadmap-zoom-preference.ts | 27 ++ .../src/i18n/en-runtime-required.json | 4 + src/renderer/src/i18n/locales/en.json | 24 + .../github/project-roadmap-timeline.test.ts | 310 +++++++++++++ src/shared/github/project-roadmap-timeline.ts | 301 +++++++++++++ src/shared/github/project-types.ts | 5 +- 17 files changed, 1720 insertions(+), 38 deletions(-) create mode 100644 src/main/github/project-view/project-view-table.test.ts create mode 100644 src/renderer/src/components/github-project/ProjectRoadmap.test.tsx create mode 100644 src/renderer/src/components/github-project/ProjectRoadmap.tsx create mode 100644 src/renderer/src/components/github-project/ProjectRoadmapBar.tsx create mode 100644 src/renderer/src/components/github-project/roadmap-tick-format.ts create mode 100644 src/renderer/src/components/github-project/roadmap-zoom-preference.ts create mode 100644 src/shared/github/project-roadmap-timeline.test.ts create mode 100644 src/shared/github/project-roadmap-timeline.ts diff --git a/src/main/github/project-view/project-view-field-normalization.ts b/src/main/github/project-view/project-view-field-normalization.ts index 1441a999809..ab41b812e08 100644 --- a/src/main/github/project-view/project-view-field-normalization.ts +++ b/src/main/github/project-view/project-view-field-normalization.ts @@ -145,7 +145,8 @@ export function normalizeFieldValue( iterationId: raw.iterationId, title: raw.title ?? '', startDate: raw.startDate ?? '', - duration: typeof raw.duration === 'number' ? raw.duration : 0 + duration: typeof raw.duration === 'number' ? raw.duration : 0, + ...(typeof raw.field.name === 'string' ? { fieldName: raw.field.name } : {}) } case 'ProjectV2ItemFieldTextValue': return { kind: 'text', fieldId, text: raw.text ?? '' } @@ -155,7 +156,12 @@ export function normalizeFieldValue( } return { kind: 'number', fieldId, number: raw.number } case 'ProjectV2ItemFieldDateValue': - return { kind: 'date', fieldId, date: raw.date ?? '' } + return { + kind: 'date', + fieldId, + date: raw.date ?? '', + ...(typeof raw.field.name === 'string' ? { fieldName: raw.field.name } : {}) + } case 'ProjectV2ItemFieldLabelValue': { const labels = (raw.labels?.nodes ?? []) .map(normalizeLabel) diff --git a/src/main/github/project-view/project-view-table.test.ts b/src/main/github/project-view/project-view-table.test.ts new file mode 100644 index 00000000000..b51a08d4bf0 --- /dev/null +++ b/src/main/github/project-view/project-view-table.test.ts @@ -0,0 +1,107 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { fetchProjectViewsPage, type RawProjectView } from './project-view-config' +import type * as ProjectViewConfig from './project-view-config' +import { fetchAllItems, fetchItemsCountOnly } from './project-view-items' +import { getProjectViewTable } from './project-view-table' + +vi.mock('./project-view-config', async (importOriginal) => ({ + ...(await importOriginal()), + fetchProjectViewsPage: vi.fn() +})) +vi.mock('./project-view-items', () => ({ + fetchAllItems: vi.fn(), + fetchItemsCountOnly: vi.fn() +})) + +const args = { + owner: 'acme', + ownerType: 'organization', + projectNumber: 1, + host: 'github.acme.test' +} as const +const view = (id: string, layout: string): RawProjectView => ({ + id, + number: 1, + name: id, + layout, + filter: 'status:open', + fields: { nodes: [] }, + groupByFields: { nodes: [] }, + sortByFields: { nodes: [] } +}) +function page(views: RawProjectView[], hasNextPage = false) { + return { + ok: true as const, + project: { id: 'project', title: 'Plan', url: 'https://github.acme.test/orgs/acme/projects/1' }, + views, + hasNextPage, + endCursor: hasNextPage ? 'next' : null + } +} + +beforeEach(() => { + vi.resetAllMocks() + vi.mocked(fetchAllItems).mockResolvedValue({ + ok: true, + rows: [], + totalCount: 0, + parentFieldDropped: false + }) + vi.mocked(fetchItemsCountOnly).mockResolvedValue(12) +}) + +describe('project view layout selection', () => { + it('fetches roadmap items with the selected host and filter', async () => { + vi.mocked(fetchProjectViewsPage).mockResolvedValue(page([view('roadmap', 'ROADMAP_LAYOUT')])) + const result = await getProjectViewTable({ ...args, viewId: 'roadmap' }) + expect(result).toMatchObject({ ok: true, data: { selectedView: { layout: 'ROADMAP_LAYOUT' } } }) + expect(fetchAllItems).toHaveBeenCalledWith({ ...args, query: 'status:open' }) + expect(fetchItemsCountOnly).not.toHaveBeenCalled() + }) + + it('defaults to a roadmap when no table exists across all view pages', async () => { + vi.mocked(fetchProjectViewsPage) + .mockResolvedValueOnce(page([view('roadmap', 'ROADMAP_LAYOUT')], true)) + .mockResolvedValueOnce(page([view('board', 'BOARD_LAYOUT')])) + expect(await getProjectViewTable(args)).toMatchObject({ + ok: true, + data: { selectedView: { id: 'roadmap' } } + }) + expect(fetchProjectViewsPage).toHaveBeenLastCalledWith({ ...args, after: 'next' }) + }) + + it('prefers a table on a later page over an earlier roadmap', async () => { + vi.mocked(fetchProjectViewsPage) + .mockResolvedValueOnce(page([view('roadmap', 'ROADMAP_LAYOUT')], true)) + .mockResolvedValueOnce(page([view('table', 'TABLE_LAYOUT')])) + expect(await getProjectViewTable(args)).toMatchObject({ + ok: true, + data: { selectedView: { id: 'table' } } + }) + }) + + it('does not substitute a roadmap for a missing explicit selection', async () => { + vi.mocked(fetchProjectViewsPage).mockResolvedValue(page([view('roadmap', 'ROADMAP_LAYOUT')])) + expect(await getProjectViewTable({ ...args, viewId: 'missing' })).toMatchObject({ + ok: false, + error: { type: 'not_found' } + }) + expect(fetchAllItems).not.toHaveBeenCalled() + }) + + it.each(['BOARD_LAYOUT', 'FUTURE_LAYOUT'])( + 'rejects %s without fetching items', + async (layout) => { + vi.mocked(fetchProjectViewsPage).mockResolvedValue(page([view('unsupported', layout)])) + expect( + await getProjectViewTable({ ...args, viewId: 'unsupported', queryOverride: '' }) + ).toMatchObject({ + ok: false, + error: { type: 'unsupported_layout' }, + totalCount: 12 + }) + expect(fetchAllItems).not.toHaveBeenCalled() + expect(fetchItemsCountOnly).toHaveBeenCalledWith({ ...args, query: '' }) + } + ) +}) diff --git a/src/main/github/project-view/project-view-table.ts b/src/main/github/project-view/project-view-table.ts index 042aeb333cf..bb580cb78e0 100644 --- a/src/main/github/project-view/project-view-table.ts +++ b/src/main/github/project-view/project-view-table.ts @@ -84,6 +84,14 @@ export async function getProjectViewTable( if (!project) { return { ok: false, error: { type: 'not_found', message: 'Project not found.' } } } + const noSelector = + args.viewId === undefined && args.viewNumber === undefined && args.viewName === undefined + if (!selectedRaw && noSelector) { + // Why: `matchesSelector` only defaults to a table view, so a project whose + // views are all roadmaps resolved to nothing even though we can now render + // one. Table stays the preferred default; this is the empty-handed case. + selectedRaw = viewsSeen.find((v) => v.layout === 'ROADMAP_LAYOUT') ?? null + } if (!selectedRaw) { return { ok: false, error: { type: 'not_found', message: 'Could not find the selected view.' } } } @@ -109,8 +117,10 @@ export async function getProjectViewTable( const effectiveQuery = typeof args.queryOverride === 'string' ? args.queryOverride : selectedView.filter - // Unsupported layout: skip item pagination; best-effort count-only query. - if (selectedView.layout !== 'TABLE_LAYOUT') { + // Why: roadmaps read the same item stream as a table — only the renderer + // differs. Allowlist, not `=== 'BOARD_LAYOUT'`: raw.layout is cast unchecked, + // so a future GitHub layout must reject cleanly, not render as a table. + if (selectedView.layout !== 'TABLE_LAYOUT' && selectedView.layout !== 'ROADMAP_LAYOUT') { const count = await fetchItemsCountOnly({ owner: args.owner, ownerType: args.ownerType, @@ -122,7 +132,7 @@ export async function getProjectViewTable( ok: false, error: { type: 'unsupported_layout', - message: `Orca only renders table views. This is a ${selectedView.layout.replace('_LAYOUT', '').toLowerCase()} view.` + message: `Orca renders table and roadmap views. This is a ${selectedView.layout.replace('_LAYOUT', '').toLowerCase()} view.` }, ...(typeof count === 'number' ? { totalCount: count } : {}) } diff --git a/src/renderer/src/components/github-project/ProjectGroupHeader.tsx b/src/renderer/src/components/github-project/ProjectGroupHeader.tsx index add12fb4e99..7eb26d51fde 100644 --- a/src/renderer/src/components/github-project/ProjectGroupHeader.tsx +++ b/src/renderer/src/components/github-project/ProjectGroupHeader.tsx @@ -8,12 +8,16 @@ type Props = { group: ProjectGroup expanded: boolean onToggle: () => void + /** Total band width for horizontally scrolling surfaces (the roadmap). The + * label pins to the viewport so it stays readable when scrolled off. */ + bandWidth?: number } export default function ProjectGroupHeader({ group, expanded, - onToggle + onToggle, + bandWidth }: Props): React.JSX.Element { const isCurrent = group.iteration ? isIterationCurrent(group.iteration) : false const dateRange = group.iteration @@ -24,24 +28,30 @@ export default function ProjectGroupHeader({ type="button" onClick={onToggle} className={cn( - 'flex w-full items-center gap-2 border-b border-border/50 bg-muted/40 px-3 py-1.5 text-left text-xs', - 'hover:bg-muted/60' + 'flex items-center border-b border-border/50 bg-muted/40 px-3 py-1.5 text-left text-xs', + 'hover:bg-muted/60', + // Why: min-w-full lets the band keep painting to the pane's right + // edge when the pane is wider than the timeline grid. + bandWidth == null ? 'w-full' : 'min-w-full' )} + style={bandWidth == null ? undefined : { width: bandWidth }} > - {expanded ? : } - - {group.label || - translate('auto.components.github.project.ProjectGroupHeader.244c9e7d06', 'All')} - - - {group.rows.length} - - {dateRange ? {dateRange} : null} - {isCurrent ? ( - - {translate('auto.components.github.project.ProjectGroupHeader.82a22d2079', 'Current')} + + {expanded ? : } + + {group.label || + translate('auto.components.github.project.ProjectGroupHeader.244c9e7d06', 'All')} - ) : null} + + {group.rows.length} + + {dateRange ? {dateRange} : null} + {isCurrent ? ( + + {translate('auto.components.github.project.ProjectGroupHeader.82a22d2079', 'Current')} + + ) : null} + ) } diff --git a/src/renderer/src/components/github-project/ProjectPickerPanels.tsx b/src/renderer/src/components/github-project/ProjectPickerPanels.tsx index f82ff38d7cb..a0183ee6968 100644 --- a/src/renderer/src/components/github-project/ProjectPickerPanels.tsx +++ b/src/renderer/src/components/github-project/ProjectPickerPanels.tsx @@ -126,19 +126,23 @@ function ProjectViewPickerRow({ view: GitHubProjectViewSummary onPick: (view: GitHubProjectViewSummary) => void | Promise }): React.JSX.Element { - const supported = view.layout === 'TABLE_LAYOUT' + const supported = view.layout === 'TABLE_LAYOUT' || view.layout === 'ROADMAP_LAYOUT' const layoutLabel = view.layout === 'TABLE_LAYOUT' ? translate('auto.components.github.project.ProjectPicker.1a2b8e512e', 'Table') - : view.layout === 'BOARD_LAYOUT' - ? translate( - 'auto.components.github.project.ProjectPicker.d34ef9b554', - 'Board (unsupported)' - ) - : translate( - 'auto.components.github.project.ProjectPicker.ab1a2c357d', - 'Roadmap (unsupported)' - ) + : view.layout === 'ROADMAP_LAYOUT' + ? translate('auto.components.github.project.ProjectPickerPanels.04ec212ccb', 'Roadmap') + : view.layout === 'BOARD_LAYOUT' + ? translate( + 'auto.components.github.project.ProjectPicker.d34ef9b554', + 'Board (unsupported)' + ) + : // Why: raw.layout is cast unchecked, so a future GitHub layout value + // lands here — keep it disabled instead of mislabeling it. + translate( + 'auto.components.github.project.ProjectPickerPanels.9fe1ac868c', + 'Unsupported' + ) return (
} + /> + ) + const scroller = screen.getByTestId('project-roadmap-scroller') + const marker = scroller.querySelector('.sticky.top-0 .absolute')! + const before = Number.parseFloat(marker.style.left) + scroller.scrollLeft = 123 + act(() => vi.advanceTimersByTime(2100)) + expect(Number.parseFloat(marker.style.left) - before).toBeCloseTo(148 / 30) + expect(scroller.scrollLeft).toBe(123) + expect(vi.getTimerCount()).toBe(1) + unmount() + expect(vi.getTimerCount()).toBe(0) + }) + + it.each([false, true])( + 'centers when an initially empty view gains dated rows (fields hidden: %s)', + (hidden) => { + vi.useFakeTimers() + vi.setSystemTime(new Date(2026, 8, 5, 12)) + const fields = hidden ? [TITLE_FIELD] : [TITLE_FIELD, START_FIELD, TARGET_FIELD] + const { rerender } = render( + list
} /> + ) + const populated = table(fields, [ + row('one', 'Arrived', [ + { kind: 'date', fieldId: 'f_start', date: '2026-01-01' }, + { kind: 'date', fieldId: 'f_end', date: '2026-09-10' } + ]) + ]) + rerender(list
} />) + const scroller = screen.getByTestId('project-roadmap-scroller') + expect(scroller.scrollLeft).toBeGreaterThan(1000) + scroller.scrollLeft = 123 + rerender( + list} + /> + ) + expect(scroller.scrollLeft).toBe(123) + fireEvent.click(screen.getByRole('button', { name: 'Year' })) + expect(scroller.scrollLeft).not.toBe(123) + expect(window.localStorage.getItem('orca.githubProject.roadmapZoom')).toBe('year') + } + ) + + it('places a dated row on the timeline and names the fields driving it', () => { + render( + list} + /> + ) + expect(screen.getByText('Placed by Start date → Target date')).toBeTruthy() + expect(screen.getByLabelText(/^Ship the thing — /)).toBeTruthy() + expect(screen.queryByText('list')).toBeNull() + }) + + it('keeps an undated row in place and flags it rather than hiding it', () => { + render( + list} + /> + ) + expect(screen.getByText('No dates')).toBeTruthy() + expect(screen.getByText('1 without dates')).toBeTruthy() + expect(screen.queryByLabelText(/^Undated — /)).toBeNull() + }) + + it('opens the row dialog when a bar is clicked', () => { + const onOpenDialog = vi.fn() + render( + list} + /> + ) + fireEvent.click(screen.getByLabelText(/^Ship the thing — /)) + expect(onOpenDialog).toHaveBeenCalledTimes(1) + expect(onOpenDialog.mock.calls[0]?.[0]).toMatchObject({ id: 'PVTI_1' }) + }) + + it('places items from row-carried dates when the view hides its date fields', () => { + render( + list} + /> + ) + expect(screen.getByText('Placed by Start date → Target date')).toBeTruthy() + expect(screen.getByLabelText(/^Hidden-field item — /)).toBeTruthy() + expect(screen.queryByText('list')).toBeNull() + }) + + it('announces restricted items by name in the bar label', () => { + const redacted: GitHubProjectRow = { + ...row('PVTI_9', '', [ + { kind: 'date', fieldId: 'f_start', date: '2026-03-02' }, + { kind: 'date', fieldId: 'f_end', date: '2026-03-05' } + ]), + itemType: 'REDACTED' + } + render( + list} + /> + ) + expect(screen.getByLabelText(/^Restricted item — /)).toBeTruthy() + }) + + it('falls back to the caller-supplied list when no field can place items', () => { + render(list} />) + expect(screen.getByText('list')).toBeTruthy() + expect( + screen.getByText( + 'This roadmap view has no date or iteration field to place items on, so Orca is listing them instead.' + ) + ).toBeTruthy() + }) + + it('reports an empty filter result instead of drawing an empty grid', () => { + render( + list} + /> + ) + expect(screen.getByText("No items match this view's filter.")).toBeTruthy() + expect(screen.queryByText('list')).toBeNull() + }) +}) diff --git a/src/renderer/src/components/github-project/ProjectRoadmap.tsx b/src/renderer/src/components/github-project/ProjectRoadmap.tsx new file mode 100644 index 00000000000..d06ef409dff --- /dev/null +++ b/src/renderer/src/components/github-project/ProjectRoadmap.tsx @@ -0,0 +1,419 @@ +import React, { useCallback, useEffect, useLayoutEffect, useMemo, useRef, useState } from 'react' +import { CalendarClock } from 'lucide-react' +import { Button } from '@/components/ui/button' +import { usePrefersReducedMotion } from '@/hooks/usePrefersReducedMotion' +import { cn } from '@/lib/utils' +import { i18n, translate } from '@/i18n/i18n' +import ProjectGroupHeader from './ProjectGroupHeader' +import ProjectRoadmapBar from './ProjectRoadmapBar' +import { ProjectTitleCell } from './ProjectCellIdentity' +import { formatRoadmapTick } from './roadmap-tick-format' +import { loadRoadmapZoom, saveRoadmapZoom } from './roadmap-zoom-preference' +import { groupRows, sortRows } from '../../../../shared/github/project-group-sort' +import { + buildRoadmapTicks, + getRoadmapSpan, + resolveRoadmapDateSource, + roadmapOffsetPx, + roadmapSourceFieldNames, + type RoadmapSpan, + type RoadmapTick, + type RoadmapZoom +} from '../../../../shared/github/project-roadmap-timeline' +import type { GitHubProjectRow, GitHubProjectTable } from '../../../../shared/github/project-types' + +const LABEL_WIDTH_PX = 280 +const LANE_HEIGHT_PX = 36 +const TICK_WIDTH_PX: Record = { month: 148, quarter: 128, year: 160 } +const ZOOMS: RoadmapZoom[] = ['month', 'quarter', 'year'] + +function localTodayAsUtcMidnightMs(): number { + const now = new Date() + return Date.UTC(now.getFullYear(), now.getMonth(), now.getDate()) +} + +type Props = { + table: GitHubProjectTable + onOpenDialog?: (row: GitHubProjectRow) => void + /** Rendered instead of the timeline when the view has no field to place + * items on — the caller supplies the table list so the items stay usable. */ + fallback: React.ReactNode +} + +export default function ProjectRoadmap({ + table, + onOpenDialog, + fallback +}: Props): React.JSX.Element { + const view = table.selectedView + const prefersReducedMotion = usePrefersReducedMotion() + const locale = i18n.resolvedLanguage ?? i18n.language + // Why: the grid lives on UTC calendar days (parseRoadmapDate), so "today" + // must be the viewer's LOCAL calendar date mapped to UTC midnight — the raw + // instant would shift the marker into the wrong day off UTC. + const [todayMs, setTodayMs] = useState(localTodayAsUtcMidnightMs) + // Why: a pane left open across midnight would otherwise keep yesterday's + // marker; re-arm after each fire so multi-day sessions stay honest. + useEffect(() => { + const now = new Date() + const nextLocalMidnight = new Date( + now.getFullYear(), + now.getMonth(), + now.getDate() + 1 + ).getTime() + // Why: the +1s pad absorbs timer drift so the callback lands after the + // date change, not just before it. + const timer = setTimeout( + () => setTodayMs(localTodayAsUtcMidnightMs()), + nextLocalMidnight - now.getTime() + 1000 + ) + return () => clearTimeout(timer) + }, [todayMs]) + const [zoom, setZoom] = useState(loadRoadmapZoom) + const [collapsed, setCollapsed] = useState>(() => new Set()) + const scrollRef = useRef(null) + + const source = useMemo(() => resolveRoadmapDateSource(view, table.rows), [view, table.rows]) + const groups = useMemo(() => groupRows(table, sortRows(table, table.rows)), [table]) + const spans = useMemo(() => { + const bySpan = new Map() + if (!source) { + return bySpan + } + for (const row of table.rows) { + const span = getRoadmapSpan(row, source) + if (span) { + bySpan.set(row.id, span) + } + } + return bySpan + }, [source, table.rows]) + + const tickWidth = TICK_WIDTH_PX[zoom] + const ticks = useMemo( + () => buildRoadmapTicks(Array.from(spans.values()), zoom, todayMs), + [spans, todayMs, zoom] + ) + const timelineWidth = ticks.length * tickWidth + const todayPx = roadmapOffsetPx(todayMs, ticks, tickWidth) + const hasTimeline = source !== null && table.rows.length > 0 + + // Why: the interesting part of a roadmap is around now — open there instead + // of at the padded left edge, and re-centre when the zoom changes scale. + const scrollToToday = useCallback(() => { + const scroller = scrollRef.current + if (!scroller) { + return + } + const lead = (scroller.clientWidth - LABEL_WIDTH_PX) / 3 + scroller.scrollTo({ + left: Math.max(0, todayPx - lead), + behavior: prefersReducedMotion ? 'instant' : 'smooth' + }) + }, [todayPx, prefersReducedMotion]) + const todayPxRef = useRef(todayPx) + useLayoutEffect(() => { + todayPxRef.current = todayPx + }) + // Center when the timeline appears or zoom changes; refetches must preserve user scroll. + useEffect(() => { + const scroller = scrollRef.current + if (!scroller) { + return + } + const lead = (scroller.clientWidth - LABEL_WIDTH_PX) / 3 + scroller.scrollLeft = Math.max(0, todayPxRef.current - lead) + }, [zoom, hasTimeline]) + + const colorFieldId = useMemo(() => { + const grouped = view.groupByFields.find((field) => field.kind === 'single-select') + return (grouped ?? view.fields.find((field) => field.kind === 'single-select'))?.id ?? null + }, [view]) + + if (!source) { + return ( +
+
+ {translate( + 'auto.components.github.project.ProjectRoadmap.be52f7b6db', + 'This roadmap view has no date or iteration field to place items on, so Orca is listing them instead.' + )} +
+ {fallback} +
+ ) + } + + if (table.rows.length === 0) { + return ( +
+ {translate( + 'auto.components.github.project.ProjectViewList.4f57d2e0b1', + "No items match this view's filter." + )} +
+ ) + } + + const undatedCount = table.rows.length - spans.size + const bandWidth = LABEL_WIDTH_PX + timelineWidth + return ( +
+ { + setZoom(next) + saveRoadmapZoom(next) + }} + onToday={scrollToToday} + /> +
+
+ +
+
+
+ {groups.map((group) => { + const expanded = !collapsed.has(group.key) + return ( +
+ {view.groupByFields[0] ? ( + + setCollapsed((previous) => { + const next = new Set(previous) + if (!next.delete(group.key)) { + next.add(group.key) + } + return next + }) + } + /> + ) : null} + {expanded + ? group.rows.map((row) => ( + + )) + : null} +
+ ) + })} +
+
+
+
+ ) +} + +function RoadmapControls({ + placedBy, + undatedCount, + zoom, + onZoom, + onToday +}: { + placedBy: string + undatedCount: number + zoom: RoadmapZoom + onZoom: (zoom: RoadmapZoom) => void + onToday: () => void +}): React.JSX.Element { + const zoomLabels: Record = { + month: translate('auto.components.github.project.ProjectRoadmap.6405e036e0', 'Month'), + quarter: translate('auto.components.github.project.ProjectRoadmap.f2b1cabef7', 'Quarter'), + year: translate('auto.components.github.project.ProjectRoadmap.b6afc6fe45', 'Year') + } + return ( +
+ + {translate( + 'auto.components.github.project.ProjectRoadmap.343888b143', + 'Placed by {{value0}}', + { + value0: placedBy + } + )} + + {undatedCount > 0 ? ( + + {translate( + 'auto.components.github.project.ProjectRoadmap.6a088a5da1', + '{{value0}} without dates', + { value0: undatedCount } + )} + + ) : null} +
+ +
+ {ZOOMS.map((option) => ( + + ))} +
+
+
+ ) +} + +function RoadmapHeaderRow({ + ticks, + tickWidth, + zoom, + locale, + todayPx +}: { + ticks: RoadmapTick[] + tickWidth: number + zoom: RoadmapZoom + locale: string + todayPx: number +}): React.JSX.Element { + return ( +
+
+ {translate('auto.components.github.project.ProjectRoadmap.e304235879', 'Item')} +
+
+ {ticks.map((tick, index) => { + const { label, sublabel } = formatRoadmapTick(tick, zoom, index, locale) + return ( +
+ {label} + {sublabel ? {sublabel} : null} +
+ ) + })} +
+
+
+ ) +} + +function RoadmapLane({ + row, + span, + ticks, + tickWidth, + timelineWidth, + colorFieldId, + locale, + onOpenDialog +}: { + row: GitHubProjectRow + span: RoadmapSpan | null + ticks: RoadmapTick[] + tickWidth: number + timelineWidth: number + colorFieldId: string | null + locale: string + onOpenDialog?: (row: GitHubProjectRow) => void +}): React.JSX.Element { + const statusValue = colorFieldId ? row.fieldValuesByFieldId[colorFieldId] : undefined + const chipColor = statusValue?.kind === 'single-select' ? statusValue.color : null + const left = span ? roadmapOffsetPx(span.startMs, ticks, tickWidth) : 0 + const width = span ? roadmapOffsetPx(span.endMs, ticks, tickWidth) - left : 0 + return ( +
+
+ onOpenDialog?.(row)} /> + {span ? null : ( + + {translate('auto.components.github.project.ProjectRoadmap.e077c79083', 'No dates')} + + )} +
+
+ {span ? ( + onOpenDialog?.(row)} + /> + ) : null} +
+
+ ) +} diff --git a/src/renderer/src/components/github-project/ProjectRoadmapBar.tsx b/src/renderer/src/components/github-project/ProjectRoadmapBar.tsx new file mode 100644 index 00000000000..361378c0710 --- /dev/null +++ b/src/renderer/src/components/github-project/ProjectRoadmapBar.tsx @@ -0,0 +1,92 @@ +import React from 'react' +import { GitPullRequest, Lock } from 'lucide-react' +import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' +import { cn } from '@/lib/utils' +import { translate } from '@/i18n/i18n' +import { chipStyle, labelChipColors, singleSelectChipColors } from './project-cell-chip-colors' +import { formatRoadmapSpan } from './roadmap-tick-format' +import type { RoadmapSpan } from '../../../../shared/github/project-roadmap-timeline' +import type { GitHubProjectRow } from '../../../../shared/github/project-types' + +const MIN_BAR_WIDTH_PX = 24 + +type Props = { + row: GitHubProjectRow + span: RoadmapSpan + leftPx: number + widthPx: number + /** GitHub single-select color token for the row's status, when it has one. */ + chipColor: string | null + locale: string + onOpen?: () => void +} + +export default function ProjectRoadmapBar({ + row, + span, + leftPx, + widthPx, + chipColor, + locale, + onOpen +}: Props): React.JSX.Element { + const colors = chipColor ? singleSelectChipColors(chipColor) : labelChipColors('') + const interactive = row.itemType !== 'REDACTED' && row.itemType !== 'DRAFT_ISSUE' + const dates = formatRoadmapSpan(span, locale) + // Why: shared by the visible text, aria-label, and tooltip — a redacted row + // must never announce or render an empty name. + const title = + row.itemType === 'REDACTED' + ? translate('auto.components.github.project.ProjectRoadmapBar.7d1220d979', 'Restricted item') + : row.content.title + const bar = ( + + ) + return ( + + {bar} + +
+
{title}
+
{dates}
+
+
+
+ ) +} diff --git a/src/renderer/src/components/github-project/ProjectViewStates.tsx b/src/renderer/src/components/github-project/ProjectViewStates.tsx index 83e088b2759..e3184878851 100644 --- a/src/renderer/src/components/github-project/ProjectViewStates.tsx +++ b/src/renderer/src/components/github-project/ProjectViewStates.tsx @@ -42,13 +42,17 @@ function ProjectViewTab({ active: boolean onPick: (viewId: string) => void }): React.JSX.Element { - const supported = view.layout === 'TABLE_LAYOUT' + // Why: allowlist, not denylist — raw.layout is cast unchecked, so a future + // GitHub layout value must stay disabled instead of masquerading as a table. + const supported = view.layout === 'TABLE_LAYOUT' || view.layout === 'ROADMAP_LAYOUT' const layoutLabel = view.layout === 'BOARD_LAYOUT' ? 'Board' : view.layout === 'ROADMAP_LAYOUT' ? 'Roadmap' - : 'Table' + : view.layout === 'TABLE_LAYOUT' + ? 'Table' + : formatUnknownLayout(view.layout) const Icon = view.layout === 'BOARD_LAYOUT' ? KanbanSquare @@ -106,8 +110,8 @@ function ProjectViewTab({

{message}{' '} {translate( - 'auto.components.github.project.ProjectViewWrapper.1bf8c01c8b', - 'Switch to a Table view to work with this project in Orca.' + 'auto.components.github.project.ProjectViewStates.ac83c45672', + 'Switch to a Table or Roadmap view to work with this project in Orca.' )}