From a624e7cd5df69f54910141c81afa88cbde89cd62 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 26 Aug 2026 14:33:02 -0700 Subject: [PATCH 01/19] test(agent-status): inventory legacy pane identity surfaces (#16575) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(agent-status): measure identity evidence before migrating any consumer PR 1 of the identity migration. It changes no displayed or routed identity — it only measures. Why measure first: the hierarchy shipped in #16148/#16157 has zero consumers, while ~31 sites still derive identity independently. Every migration decision after this is currently a guess, including the one that matters most — how often a real pane has no evidence at all. A live P0 reports "No Claude status shown", and this design trades toward showing nothing when uncertain, so the blank rate has to be a number before any surface moves. - `pane-agent-identity-evidence.ts` — one assembler that gathers a pane's evidence, so consumers stop each inventing their own ladder. - `pane-agent-identity-census.ts` — shadow-only counters keyed by host kind (native / wsl-host / wsl-distro / ssh / relay) and launch mode (typed / orca-launch / resume). Records a bitmask of which sources were present and whether the resolver returned null or ambiguous. No titles, prompts, paths, handles, or agent text. - `pane-agent-identity-inventory.test.ts` — a ratchet that fails when a legacy identity helper gains a new production caller, so the surface cannot grow while the migration runs. Three review findings are encoded rather than deferred: launch stays above run-key-less completed hooks (promoting the hook lets a stale record hijack a pane); OMP/Pi evidence is owner-normalized before assembly, since OMP emits Pi-compatible frames and a wrapper's hook would otherwise be read as the agent it wraps; and Windows-side `wsl.exe` is rejected as process evidence, because the host observes the distro wrapper rather than the agent inside it. The census cannot be completed from a worktree. It needs representative native, SSH, WSL and relay cohorts collected from real use, and that review is the gate on PR 3 — not this PR. * test(agent-status): keep identity migration inventory-only * test(agent-status): reuse reliable source scanner * test(agent-status): bound inventory scan work * test(agent-status): avoid inventory path false negatives * test(agent-status): refresh identity inventory after base repair * test(agent-status): correct inventory classifications * test(agent-status): correct action boundary inventory * test(agent-status): pin inventory occurrence counts * test(agent-status): fail closed on scanner desync --- src/renderer/src/lib/pane-agent-evidence.ts | 5 - .../pane-agent-identity-inventory.test.ts | 363 ++++++++++++++++++ .../pane-agent-identity-resolver.test.ts | 8 + 3 files changed, 371 insertions(+), 5 deletions(-) create mode 100644 src/shared/pane-agent-identity-inventory.test.ts diff --git a/src/renderer/src/lib/pane-agent-evidence.ts b/src/renderer/src/lib/pane-agent-evidence.ts index 8230bb51522..cfed24bfa33 100644 --- a/src/renderer/src/lib/pane-agent-evidence.ts +++ b/src/renderer/src/lib/pane-agent-evidence.ts @@ -115,8 +115,3 @@ export function resolvePaneAgentActivity( livePtyRequired: false } } - -// Deliberately absent: a resolvePaneAgentOwner precedence resolver. The only -// Phase 2 identity consumer (native-chat toggle) reads hook identity without a -// freshness gate, so a gated owner resolver would change its behavior; the -// owner resolver lands with its first real consumer in a later slice. diff --git a/src/shared/pane-agent-identity-inventory.test.ts b/src/shared/pane-agent-identity-inventory.test.ts new file mode 100644 index 00000000000..730c6b0ce05 --- /dev/null +++ b/src/shared/pane-agent-identity-inventory.test.ts @@ -0,0 +1,363 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { glob } from 'tinyglobby' +import { + blankStringContents, + blankStringContentsDesynced, + isTestFile, + stripComments +} from './source-scan/source-tree-scan' + +const HELPERS = [ + 'getAgentLabel', + 'isClaudeAgent', + 'titleHasAgentName', + 'buildAgentNameRe', + 'resolveTerminalTitleAgentType', + 'resolveExplicitTerminalTitleAgentType', + 'resolveCommittedTitleAgentType', + 'resolvePaneAgentOwner', + 'resolveCompatibleAgentTypeForOwner' +] as const + +const TEST_SUPPORT_PATHS = new Set([ + 'src/renderer/src/components/terminal-pane/pty-connection-test-environment.ts' +]) + +type Helper = (typeof HELPERS)[number] +type InventoryPath = string | readonly [path: string, occurrences: number] +type Classification = + | 'parser-implementation' + | 'activity-only' + | 'enum-formatter' + | 'evidence-producer' + | 'identity-consumer' + | 'action-consumer' + +type InventoryGroup = { + helper: Helper + classification: Classification + paths: readonly InventoryPath[] +} + +const INVENTORY: readonly InventoryGroup[] = [ + { + helper: 'getAgentLabel', + classification: 'enum-formatter', + paths: [ + [ + 'src/renderer/src/components/agent-session-continuation/AgentSessionContinuationDialog.tsx', + 2 + ], + ['src/renderer/src/components/automations/AutomationListLocalRows.tsx', 2], + 'src/renderer/src/components/automations/automation-draft-model.ts', + ['src/renderer/src/components/automations/automation-list-search-rows.ts', 2], + ['src/renderer/src/components/dashboard-popout/AgentMapSnapshotWorkspaceMenu.tsx', 2], + ['src/renderer/src/components/dashboard-popout/AgentMapWorktreeRingNode.tsx', 2], + ['src/renderer/src/components/settings/QuickCommandsList.tsx', 2], + ['src/renderer/src/components/tab-bar/TabBarQuickCommandItem.tsx', 2], + ['src/renderer/src/components/tab-bar/TabBarQuickCommandsMenu.tsx', 2], + 'src/renderer/src/lib/agent-catalog.tsx', + ['src/renderer/src/lib/launch-agent-session-continuation.ts', 3], + ['src/renderer/src/lib/orchestration-skill-coverage.ts', 2] + ] + }, + { + helper: 'getAgentLabel', + classification: 'activity-only', + paths: [['src/renderer/src/lib/pane-agent-evidence.ts', 2]] + }, + { + helper: 'getAgentLabel', + classification: 'parser-implementation', + paths: [ + 'src/renderer/src/lib/agent-status.ts', + 'src/shared/agent-detection.ts', + 'src/shared/agent-title-identity.ts', + ['src/shared/agent-title-owner.ts', 2], + ['src/shared/terminal-title-agent-type.ts', 2] + ] + }, + { + helper: 'isClaudeAgent', + classification: 'action-consumer', + paths: [ + ['src/renderer/src/components/terminal-pane/cache-timer-seeding.ts', 2], + ['src/renderer/src/components/terminal-pane/parked-terminal-byte-watcher.ts', 2], + ['src/renderer/src/components/terminal-pane/pty-connection/agent-task-complete-notify.ts', 2], + ['src/renderer/src/store/terminals/terminal-ephemeral-state.ts', 2] + ] + }, + { + helper: 'isClaudeAgent', + classification: 'action-consumer', + paths: [ + ['src/renderer/src/components/terminal-pane/pty-connection/command-inferred-pane-agent.ts', 2] + ] + }, + { + helper: 'isClaudeAgent', + classification: 'parser-implementation', + paths: [ + 'src/renderer/src/lib/agent-status.ts', + 'src/shared/agent-detection.ts', + ['src/shared/agent-title-identity.ts', 2], + ['src/shared/terminal-title-agent-type.ts', 2] + ] + }, + { + helper: 'titleHasAgentName', + classification: 'parser-implementation', + paths: [ + 'src/shared/agent-detection.ts', + 'src/shared/agent-name-token-match.ts', + ['src/shared/agent-title-core.ts', 4], + ['src/shared/agent-title-evidence.ts', 2], + ['src/shared/agent-title-identity.ts', 11], + ['src/shared/terminal-title-agent-type.ts', 14] + ] + }, + { + helper: 'titleHasAgentName', + classification: 'evidence-producer', + paths: [['src/renderer/src/hooks/ipc-events/agent-status-routing.ts', 2]] + }, + { + helper: 'buildAgentNameRe', + classification: 'action-consumer', + paths: [['src/main/runtime/orchestration/groups.ts', 2]] + }, + { + helper: 'buildAgentNameRe', + classification: 'parser-implementation', + paths: [['src/shared/agent-name-token-match.ts', 2]] + }, + { + helper: 'resolveTerminalTitleAgentType', + classification: 'identity-consumer', + paths: [['src/renderer/src/lib/notes-send-agent-targets.ts', 2]] + }, + { + helper: 'resolveTerminalTitleAgentType', + classification: 'parser-implementation', + paths: [['src/shared/terminal-title-agent-type.ts', 2]] + }, + { + helper: 'resolveExplicitTerminalTitleAgentType', + classification: 'identity-consumer', + paths: [ + ['mobile/src/session/mobile-terminal-tab-agent.ts', 2], + ['src/renderer/src/lib/open-tab-occupant-agent.ts', 2], + ['src/renderer/src/lib/use-tab-agent.ts', 3] + ] + }, + { + helper: 'resolveExplicitTerminalTitleAgentType', + classification: 'parser-implementation', + paths: [ + ['src/renderer/src/lib/pane-agent-evidence.ts', 2], + 'src/shared/terminal-title-agent-type.ts' + ] + }, + { + helper: 'resolveCommittedTitleAgentType', + classification: 'action-consumer', + paths: [ + ['src/renderer/src/components/native-chat/use-native-chat-toggle-shortcut.ts', 3], + ['src/renderer/src/components/terminal-pane/pty-connection/connect-pane-pty.ts', 2], + ['src/renderer/src/components/terminal-pane/terminal-ctrl-enter.ts', 2], + ['src/renderer/src/components/terminal-pane/terminal-windows-shift-enter.ts', 2], + ['src/renderer/src/components/terminal-pane/use-notification-dispatch.ts', 2] + ] + }, + { + helper: 'resolveCommittedTitleAgentType', + classification: 'identity-consumer', + paths: [ + ['src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx', 3], + ['src/renderer/src/components/terminal-pane/native-chat-leaf-title-agent.ts', 4], + ['src/renderer/src/components/terminal-pane/pty-connection/pane-agent-identity.ts', 2] + ] + }, + { + helper: 'resolveCommittedTitleAgentType', + classification: 'parser-implementation', + paths: ['src/renderer/src/lib/pane-agent-evidence.ts'] + }, + { + helper: 'resolveCommittedTitleAgentType', + classification: 'action-consumer', + paths: [ + ['src/renderer/src/components/terminal-pane/pty-connection/command-inferred-pane-agent.ts', 2] + ] + }, + { + helper: 'resolvePaneAgentOwner', + classification: 'parser-implementation', + paths: ['src/shared/pane-agent-owner.ts'] + }, + { + helper: 'resolvePaneAgentOwner', + classification: 'evidence-producer', + paths: [['src/renderer/src/components/terminal-pane/parked-terminal-command-status.ts', 2]] + }, + { + helper: 'resolvePaneAgentOwner', + classification: 'identity-consumer', + paths: [ + ['src/main/runtime/orca-runtime.ts', 3], + ['src/renderer/src/components/sidebar/worktree-title-derived-agent-rows.ts', 2], + ['src/renderer/src/components/terminal-pane/pty-connection/shell-command-inference.ts', 2], + ['src/renderer/src/lib/use-tab-agent.ts', 2], + ['src/renderer/src/runtime/web-session-tabs-sync.ts', 3] + ] + }, + { + helper: 'resolveCompatibleAgentTypeForOwner', + classification: 'parser-implementation', + paths: [['src/shared/agent-title-owner.ts', 2]] + }, + { + helper: 'resolveCompatibleAgentTypeForOwner', + classification: 'identity-consumer', + paths: [ + ['src/main/runtime/orca-runtime.ts', 3], + ['src/renderer/src/components/sidebar/worktree-agent-rows.ts', 2], + ['src/renderer/src/components/sidebar/worktree-title-derived-agent-rows.ts', 2], + ['src/renderer/src/lib/use-tab-agent.ts', 2] + ] + }, + { + helper: 'resolveCompatibleAgentTypeForOwner', + classification: 'action-consumer', + paths: [ + ['src/renderer/src/components/terminal-pane/pty-connection/agent-task-complete-notify.ts', 2], + [ + 'src/renderer/src/components/terminal-pane/pty-connection/command-inferred-pane-agent.ts', + 3 + ], + ['src/renderer/src/components/terminal-pane/pty-connection/terminal-keydown-fit.ts', 3], + ['src/renderer/src/components/terminal-pane/use-notification-dispatch.ts', 2] + ] + }, + { + helper: 'resolveCompatibleAgentTypeForOwner', + classification: 'evidence-producer', + paths: [ + ['src/renderer/src/components/terminal-pane/pty-connection/direct-ssh-retry-status.ts', 2], + ['src/renderer/src/components/terminal-pane/pty-connection/title-spawn-bell.ts', 2] + ] + } +] + +const DIRECT_SINGLE_SOURCE_SURFACES: readonly { + path: string + classification: Classification + marker: string +}[] = [ + { + path: 'src/renderer/src/components/terminal-pane/terminal-renderer-policy.ts', + classification: 'identity-consumer', + marker: 'resolveGeminiCompatFallback' + }, + { + path: 'src/renderer/src/components/terminal-pane/terminal-title-evidence.ts', + classification: 'identity-consumer', + marker: 'resolvePaneTitleDecision' + }, + { + path: 'src/renderer/src/components/terminal/terminal-close-copy-kind.ts', + classification: 'identity-consumer', + marker: 'resolveLeafCloseCopyKind' + }, + { + path: 'src/main/runtime/orchestration/mailbox-pointer-delivery.ts', + classification: 'action-consumer', + marker: 'isCursorAgentTitle' + }, + { + path: 'src/main/providers/local-pty-provider.ts', + classification: 'action-consumer', + marker: 'launchAgent' + }, + { + path: 'src/renderer/src/components/terminal-pane/pty-connection/pane-serializer-settle.ts', + classification: 'action-consumer', + marker: 'sendStartupDraftPaste' + }, + { + path: 'src/renderer/src/lib/active-agent-note-send.ts', + classification: 'action-consumer', + marker: 'sendNotesToActiveAgentSession' + }, + { + path: 'src/renderer/src/components/native-chat/native-chat-runtime-send.ts', + classification: 'action-consumer', + marker: 'sendNativeChatMessage' + }, + { + path: 'mobile/src/session/mobile-native-chat-send.ts', + classification: 'action-consumer', + marker: 'sendMobileNativeChatMessageWithOutcome' + }, + { + path: 'mobile/src/session/mobile-native-chat-image-send.ts', + classification: 'action-consumer', + marker: 'pasteMobileNativeChatImagePaths' + }, + { + path: 'mobile/src/session/pr-ai-triage-launch.ts', + classification: 'action-consumer', + marker: 'createTerminalAndSendPrompt' + } +] + +describe('pane agent identity inventory ratchet', () => { + it('classifies every legacy helper definition, import, and callsite in src and mobile/src', async () => { + const files = await glob(['src/**/*.{ts,tsx}', 'mobile/src/**/*.{ts,tsx}'], { + ignore: ['**/*.test.*', '**/*.spec.*'] + }) + const actual: { helper: Helper; path: string; occurrences: number }[] = [] + for (const path of files) { + if (isTestFile(path) || TEST_SUPPORT_PATHS.has(path)) { + continue + } + const rawSource = readFileSync(join(process.cwd(), path), 'utf8') + if (!HELPERS.some((helper) => rawSource.includes(helper))) { + continue + } + const decommentedSource = stripComments(rawSource) + if (blankStringContentsDesynced(decommentedSource)) { + throw new Error(`String scanner desynchronized while inventorying ${path}`) + } + const source = blankStringContents(decommentedSource) + for (const helper of HELPERS) { + const occurrences = source.match(new RegExp(`\\b${helper}\\b`, 'g'))?.length ?? 0 + if (occurrences > 0) { + actual.push({ helper, path, occurrences }) + } + } + } + const expected = INVENTORY.flatMap(({ helper, paths }) => + paths.map((site) => { + const [path, occurrences] = typeof site === 'string' ? [site, 1] : site + return { helper, path, occurrences } + }) + ) + const byHelperAndPath = (left: (typeof actual)[number], right: (typeof actual)[number]) => + left.helper.localeCompare(right.helper) || left.path.localeCompare(right.path) + expect(actual.sort(byHelperAndPath)).toEqual(expected.sort(byHelperAndPath)) + }) + + it('pins direct single-source identity and action branches outside named helpers', () => { + for (const site of DIRECT_SINGLE_SOURCE_SURFACES) { + const source = stripComments(readFileSync(join(process.cwd(), site.path), 'utf8')) + expect({ + path: site.path, + classification: site.classification, + hasMarker: source.includes(site.marker) + }).toEqual({ path: site.path, classification: site.classification, hasMarker: true }) + } + }) +}) diff --git a/src/shared/pane-agent-identity-resolver.test.ts b/src/shared/pane-agent-identity-resolver.test.ts index e11ef249c65..e28fb0cbd22 100644 --- a/src/shared/pane-agent-identity-resolver.test.ts +++ b/src/shared/pane-agent-identity-resolver.test.ts @@ -85,6 +85,14 @@ describe('resolvePaneAgentIdentity', () => { }) describe('mixed-version peers', () => { + it('keeps a run-key-less completed row from hijacking launch evidence', () => { + const result = resolve([ + { source: 'completed-hook', agent: 'claude' }, + { source: 'launch', agent: 'codex' } + ]) + expect(result).toMatchObject({ agent: 'codex', source: 'launch' }) + }) + it('treats evidence with no run id as eligible', () => { // An old host publishes no run ids. Treating unknown as stale would blank every row. const result = resolvePaneAgentIdentity({ From e8005c332574835ed18d88359d050dbcaad441d9 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 26 Aug 2026 14:33:34 -0700 Subject: [PATCH 02/19] fix(codex): preserve WSL account home trust (#16496) * fix(codex): preserve WSL account home trust * fix(codex): preserve WSL drive path semantics * fix(codex): preserve mounted-drive WSL config paths * test(codex): preserve WSL path helpers in mock --- .../codex-accounts/runtime-home-service.ts | 8 +- .../runtime-home-wsl-session-bridge.test.ts | 23 ++- .../runtime-home-wsl-system-default.test.ts | 44 +++++ .../service-wsl-accounts.test.ts | 158 +++++++++++++++++- src/main/codex-accounts/service.ts | 69 ++------ src/main/codex/codex-config-mirror.test.ts | 9 + src/main/codex/codex-config-mirror.ts | 15 +- src/main/codex/config-settings-promotion.ts | 2 + 8 files changed, 253 insertions(+), 75 deletions(-) diff --git a/src/main/codex-accounts/runtime-home-service.ts b/src/main/codex-accounts/runtime-home-service.ts index c9be97fc641..3049a2a70b6 100644 --- a/src/main/codex-accounts/runtime-home-service.ts +++ b/src/main/codex-accounts/runtime-home-service.ts @@ -59,7 +59,7 @@ import { prepareSystemConfigForFreshRuntimeMirror, syncSystemConfigIntoManagedCodexHome } from '../codex/codex-config-mirror' -import { parseWslUncPath } from '../../shared/wsl-paths' +import { parseWslUncPath, toLinuxPath } from '../../shared/wsl-paths' import { getWslSelectionKey, getSelectedCodexAccountIdForTarget, @@ -757,7 +757,11 @@ export class CodexRuntimeHomeService { systemHomePath, managedHomePath: runtimeHomePath }) - syncSystemConfigIntoManagedCodexHome({ runtimeHomePath, systemHomePath }) + syncSystemConfigIntoManagedCodexHome({ + runtimeHomePath, + systemHomePath, + systemConfigDir: toLinuxPath(systemHomePath) + }) } // Why: `null` is a real value here — it means "use the system-default lane". diff --git a/src/main/codex-accounts/runtime-home-wsl-session-bridge.test.ts b/src/main/codex-accounts/runtime-home-wsl-session-bridge.test.ts index 0c3b418b06a..05cfa12e240 100644 --- a/src/main/codex-accounts/runtime-home-wsl-session-bridge.test.ts +++ b/src/main/codex-accounts/runtime-home-wsl-session-bridge.test.ts @@ -1,6 +1,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { existsSync, lstatSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs' import { join } from 'node:path' +import type * as WslPaths from '../../shared/wsl-paths' import { createSettings } from './runtime-home-settings-test-fixtures' import { createManagedAuth, @@ -281,15 +282,19 @@ describe('CodexRuntimeHomeService', () => { getDefaultWslDistro: () => null, getWslHome: (distro: string) => (distro === 'Debian' ? wslHome : null) })) - vi.doMock('../../shared/wsl-paths', () => ({ - parseWslUncPath: (candidate: string) => - candidate === wslRuntimeHomePath - ? { - distro: 'Debian', - linuxPath: '/home/alice/.local/share/orca/codex-runtime-home/home' - } - : null - })) + vi.doMock('../../shared/wsl-paths', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + parseWslUncPath: (candidate: string) => + candidate === wslRuntimeHomePath + ? { + distro: 'Debian', + linuxPath: '/home/alice/.local/share/orca/codex-runtime-home/home' + } + : null + } + }) const managedHomePath = createManagedAuth( testState.userDataDir, 'debian-account', diff --git a/src/main/codex-accounts/runtime-home-wsl-system-default.test.ts b/src/main/codex-accounts/runtime-home-wsl-system-default.test.ts index bfac7707f5a..a1ad86a0a5f 100644 --- a/src/main/codex-accounts/runtime-home-wsl-system-default.test.ts +++ b/src/main/codex-accounts/runtime-home-wsl-system-default.test.ts @@ -1,6 +1,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { mkdirSync, readFileSync, writeFileSync } from 'node:fs' import { join } from 'node:path' +import type * as CodexConfigMirror from '../codex/codex-config-mirror' import { createSettings } from './runtime-home-settings-test-fixtures' import { createCodexAuthJson, @@ -327,4 +328,47 @@ describe('CodexRuntimeHomeService', () => { } } }) + + it('passes the Linux source config directory for mounted-drive WSL homes', async () => { + const originalPlatform = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + vi.doMock('../wsl', () => ({ + getDefaultWslDistro: () => 'Ubuntu', + getWslHome: () => 'C:\\Users\\alice' + })) + const syncConfig = vi.fn() + vi.doMock('../codex/codex-config-mirror', async () => ({ + ...(await vi.importActual('../codex/codex-config-mirror')), + syncSystemConfigIntoManagedCodexHome: syncConfig + })) + + try { + const { CodexRuntimeHomeService } = await import('./runtime-home-service') + const service = new CodexRuntimeHomeService(createStore(createSettings()) as never) + const syncWslConfig = ( + service as unknown as { + syncWslConfigAndGlobalInstructionsForLaunch: ( + target: { runtime: 'wsl'; wslDistro?: string | null }, + runtimeHomePath: string | null + ) => void + } + ).syncWslConfigAndGlobalInstructionsForLaunch + + syncWslConfig.call( + service, + { runtime: 'wsl', wslDistro: 'Ubuntu' }, + join(testState.userDataDir, 'runtime-home') + ) + + expect(syncConfig).toHaveBeenCalledWith({ + runtimeHomePath: join(testState.userDataDir, 'runtime-home'), + systemHomePath: 'C:\\Users\\alice/.codex', + systemConfigDir: '/mnt/c/Users/alice/.codex' + }) + } finally { + if (originalPlatform) { + Object.defineProperty(process, 'platform', originalPlatform) + } + } + }) }) diff --git a/src/main/codex-accounts/service-wsl-accounts.test.ts b/src/main/codex-accounts/service-wsl-accounts.test.ts index 2dbf44026a2..0a1e69a81cd 100644 --- a/src/main/codex-accounts/service-wsl-accounts.test.ts +++ b/src/main/codex-accounts/service-wsl-accounts.test.ts @@ -45,6 +45,148 @@ function wslFailed(code: number, stderr = ''): WslResult { describe('CodexAccountService config sync', () => { registerCodexAccountsTestHomes() + it('preserves WSL account-home project trust while refreshing canonical settings', async () => { + const wslManagedHomePath = join(testState.userDataDir, 'wsl-account', 'home') + const wslCanonicalHomePath = join(testState.userDataDir, 'wsl-home', '.codex') + const wslLinuxHomePath = '/home/alice/.local/share/orca/codex-accounts/account-1/home' + const wslLinuxCanonicalHomePath = '/home/alice/.codex' + mkdirSync(wslManagedHomePath, { recursive: true }) + mkdirSync(wslCanonicalHomePath, { recursive: true }) + writeFileSync(join(wslManagedHomePath, '.orca-managed-home'), 'account-1\n', 'utf-8') + writeFileSync( + join(wslManagedHomePath, 'config.toml'), + 'approval_policy = "untrusted"\n[projects."/workspace"]\ntrust_level = "trusted"\n', + 'utf-8' + ) + writeFileSync( + join(wslCanonicalHomePath, 'config.toml'), + 'sandbox_mode = "danger-full-access"\n', + 'utf-8' + ) + + vi.doMock('../../shared/wsl-paths', () => ({ + parseWslUncPath: (path: string) => { + if (path === wslManagedHomePath) { + return { distro: 'Ubuntu', linuxPath: wslLinuxHomePath } + } + if (path === wslCanonicalHomePath) { + return { distro: 'Ubuntu', linuxPath: wslLinuxCanonicalHomePath } + } + return null + } + })) + vi.doMock('../wsl', () => ({ + toWindowsWslPath: (linuxPath: string) => + linuxPath === wslLinuxCanonicalHomePath || + linuxPath === `${wslLinuxCanonicalHomePath}/config.toml` + ? linuxPath.endsWith('/config.toml') + ? join(wslCanonicalHomePath, 'config.toml') + : wslCanonicalHomePath + : wslManagedHomePath + })) + + const settings = createSettings({ + codexManagedAccounts: [ + { + id: 'account-1', + email: 'wsl@example.com', + managedHomePath: wslManagedHomePath, + managedHomeRuntime: 'wsl', + wslDistro: 'Ubuntu', + wslLinuxHomePath, + providerAccountId: null, + workspaceLabel: null, + workspaceAccountId: null, + createdAt: 1, + updatedAt: 1, + lastAuthenticatedAt: 1 + } + ] + }) + + const { CodexAccountService } = await import('./service') + new CodexAccountService( + createStore(settings) as never, + createRateLimits() as never, + createRuntimeHome() as never + ) + + expect(readFileSync(join(wslManagedHomePath, 'config.toml'), 'utf-8')).toBe( + 'sandbox_mode = "danger-full-access"\n\n' + + '[projects."/workspace"]\ntrust_level = "trusted"\n' + ) + }) + + it('keeps Linux-relative config paths when a WSL home is under a mounted drive', async () => { + vi.resetModules() + const originalPlatform = process.platform + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + + const wslManagedHomePath = join(testState.userDataDir, 'wsl-account', 'home') + const wslCanonicalHomePath = join(testState.userDataDir, 'wsl-home', '.codex') + const wslCanonicalConfigPath = join(wslCanonicalHomePath, 'config.toml') + const wslLinuxHomePath = '/mnt/c/Users/alice/.local/share/orca/codex-accounts/account-1/home' + const wslLinuxCanonicalHomePath = '/mnt/c/Users/alice/.codex' + mkdirSync(wslManagedHomePath, { recursive: true }) + mkdirSync(wslCanonicalHomePath, { recursive: true }) + writeFileSync(join(wslManagedHomePath, '.orca-managed-home'), 'account-1\n', 'utf-8') + writeFileSync(wslCanonicalConfigPath, 'model_instructions_file = "instructions.md"\n', 'utf-8') + + vi.doMock('node:child_process', () => ({ + execFileSync: vi.fn(() => `${wslLinuxHomePath}\n`), + spawn: vi.fn() + })) + vi.doMock('../../shared/wsl-paths', () => ({ + parseWslUncPath: (path: string) => + path === wslManagedHomePath ? { distro: 'Ubuntu', linuxPath: wslLinuxHomePath } : null + })) + vi.doMock('../wsl', () => ({ + toWindowsWslPath: (linuxPath: string) => + linuxPath.endsWith('/config.toml') + ? wslCanonicalConfigPath + : linuxPath === wslLinuxCanonicalHomePath + ? wslCanonicalHomePath + : wslManagedHomePath + })) + + const settings = createSettings({ + codexManagedAccounts: [ + { + id: 'account-1', + email: 'wsl@example.com', + managedHomePath: wslManagedHomePath, + managedHomeRuntime: 'wsl', + wslDistro: 'Ubuntu', + wslLinuxHomePath, + providerAccountId: null, + workspaceLabel: null, + workspaceAccountId: null, + createdAt: 1, + updatedAt: 1, + lastAuthenticatedAt: 1 + } + ] + }) + + try { + const { CodexAccountService } = await import('./service') + new CodexAccountService( + createStore(settings) as never, + createRateLimits() as never, + createRuntimeHome() as never + ) + + expect(readFileSync(join(wslManagedHomePath, 'config.toml'), 'utf-8')).toContain( + "model_instructions_file = '/mnt/c/Users/alice/.codex/instructions.md'" + ) + } finally { + Object.defineProperty(process, 'platform', { + configurable: true, + value: originalPlatform + }) + } + }) + it('adds a managed Codex account inside WSL when the account context is WSL', async () => { vi.resetModules() const originalPlatform = process.platform @@ -54,8 +196,10 @@ describe('CodexAccountService config sync', () => { }) const wslManagedHomePath = join(testState.userDataDir, 'wsl-managed-home') - const wslConfigPath = join(testState.userDataDir, 'wsl-config.toml') + const wslConfigHomePath = join(testState.userDataDir, 'wsl-config-home') + const wslConfigPath = join(wslConfigHomePath, 'config.toml') const wslLinuxHomePath = '/home/alice/.local/share/orca/codex-accounts/account-id-for-test/home' + mkdirSync(wslConfigHomePath, { recursive: true }) writeFileSync( wslConfigPath, 'sandbox_mode = "danger-full-access"\nmodel_instructions_file = "instructions.md"\n', @@ -130,11 +274,19 @@ describe('CodexAccountService config sync', () => { vi.doMock('../wsl/wsl-runner', () => ({ runWslProcess: runWslProcessMock })) vi.doMock('../../shared/wsl-paths', () => ({ parseWslUncPath: (path: string) => - path === wslManagedHomePath ? { distro: 'Debian', linuxPath: wslLinuxHomePath } : null + path === wslManagedHomePath + ? { distro: 'Debian', linuxPath: wslLinuxHomePath } + : path === wslConfigHomePath + ? { distro: 'Debian', linuxPath: '/home/alice/.codex' } + : null })) vi.doMock('../wsl', () => ({ toWindowsWslPath: (linuxPath: string) => - linuxPath.endsWith('/.codex/config.toml') ? wslConfigPath : wslManagedHomePath + linuxPath.endsWith('/.codex/config.toml') + ? wslConfigPath + : linuxPath.endsWith('/.codex') + ? wslConfigHomePath + : wslManagedHomePath })) const settings = createSettings() diff --git a/src/main/codex-accounts/service.ts b/src/main/codex-accounts/service.ts index 56cfd630f2d..97a3985c701 100644 --- a/src/main/codex-accounts/service.ts +++ b/src/main/codex-accounts/service.ts @@ -32,12 +32,8 @@ import type { } from '../../shared/codex-reset-credit-attempt-ledger' import type { CodexRuntimeHomeService } from './runtime-home-service' import { writeFileAtomically } from './fs-utils' -import { rewriteRelativePathConfigValues } from '../codex/codex-config-path-reference-rewrite' -import { stripCodexManagedHookTrustEntriesFromConfig } from '../codex/codex-managed-trust-reconciliation' -import { getCodexManagedHookInstallMaterial } from '../codex/hook-service' import { syncSystemConfigIntoManagedCodexHome } from '../codex/codex-config-mirror' import { getSystemCodexHomePath } from '../codex/codex-home-paths' -import { MANAGED_HOOK_TIMEOUT_SECONDS } from '../agent-hooks/installer-utils' import { readCodexTopLevelModelProvider } from '../codex/codex-model-provider-config' import { resolveCodexCommand } from '../codex-cli/command' import { withCliRuntimeOnPath } from '../../shared/node-cli-command-resolution' @@ -93,9 +89,10 @@ type ResolvedCodexIdentity = { type CanonicalCodexConfig = { contents: string - /** Home the config was read from, in the path style Codex sees (Linux-side for WSL); relative settings resolve against it. */ + /** Host-readable source home; the mirror resolves WSL UNC paths to their Linux spelling. */ sourceHomePath: string - sourceHooksPath: string + /** Preserve Linux path semantics when WSL $HOME is under /mnt/. */ + sourceConfigDir?: string } export type CodexAccountAddTarget = { @@ -1303,12 +1300,6 @@ export class CodexAccountService { } } - private isSelfContainedHostManagedHome(managedHomePath: string): boolean { - // Why: each host account home is its own launch CODEX_HOME. WSL homes keep - // their distro-local seed lane. - return !parseWslUncPath(managedHomePath) - } - private syncCanonicalConfigIntoManagedHome( managedHomePath: string, canonicalConfig = this.readCanonicalConfigForManagedHome(managedHomePath), @@ -1319,36 +1310,13 @@ export class CodexAccountService { } const trustedManagedHomePath = this.assertManagedHomePath(managedHomePath, expectedAccountId) - if (this.isSelfContainedHostManagedHome(trustedManagedHomePath)) { - // Why: this home is codex's live CODEX_HOME, so mirror config with the - // trust-preserving merge — the plain overwrite below would wipe the - // hook/project trust codex granted in this home, forcing a re-approval and - // an app-server re-grant on every account switch. - syncSystemConfigIntoManagedCodexHome({ - runtimeHomePath: trustedManagedHomePath, - systemHomePath: getSystemCodexHomePath() - }) - return - } - // Why: Orca account switching is meant to swap Codex credentials and quota - // identity, not silently fork the user's sandbox/config defaults. Syncing - // one canonical config into every managed home keeps auth isolated per - // account while preserving consistent Codex behavior. Managed homes are - // real CODEX_HOMEs for `codex login`, so relative path-valued settings - // must keep resolving against the home the config was read from. - const material = getCodexManagedHookInstallMaterial() - // Why: source-home Orca trust is foreign to each managed home's hooks.json. - const sanitizedConfig = stripCodexManagedHookTrustEntriesFromConfig(canonicalConfig.contents, { - runtimeHomePath: canonicalConfig.sourceHomePath, - sourcePath: canonicalConfig.sourceHooksPath, - command: material.command, - managedEventLabels: new Set(Object.values(material.eventLabel)), - timeoutSec: MANAGED_HOOK_TIMEOUT_SECONDS + // Why: every account home is Codex's own CODEX_HOME. Preserve trust Codex + // granted there while refreshing ordinary settings from the lane's source. + syncSystemConfigIntoManagedCodexHome({ + runtimeHomePath: trustedManagedHomePath, + systemHomePath: canonicalConfig.sourceHomePath, + systemConfigDir: canonicalConfig.sourceConfigDir }) - this.writeManagedConfig( - trustedManagedHomePath, - rewriteRelativePathConfigValues(sanitizedConfig, canonicalConfig.sourceHomePath) - ) } private readCanonicalConfig(): CanonicalCodexConfig | null { @@ -1361,8 +1329,7 @@ export class CodexAccountService { try { return { contents: readFileSync(primaryConfigPath, 'utf-8'), - sourceHomePath, - sourceHooksPath: join(sourceHomePath, 'hooks.json') + sourceHomePath } } catch (error) { console.warn('[codex-accounts] Failed to read canonical config:', error) @@ -1392,8 +1359,8 @@ export class CodexAccountService { // path rewrites must anchor to the Linux-side ~/.codex, not the UNC path. return { contents: readFileSync(configPath, 'utf-8'), - sourceHomePath: `${wslHome}/.codex`, - sourceHooksPath: `${wslHome}/.codex/hooks.json` + sourceHomePath: toWindowsWslPath(`${wslHome}/.codex`, wslInfo.distro), + sourceConfigDir: `${wslHome}/.codex` } } catch (error) { console.warn('[codex-accounts] Failed to read WSL canonical config:', error) @@ -1416,18 +1383,6 @@ export class CodexAccountService { ) } - private writeManagedConfig(managedHomePath: string, contents: string): void { - const configPath = join(managedHomePath, 'config.toml') - try { - if (existsSync(configPath) && readFileSync(configPath, 'utf-8') === contents) { - return - } - } catch { - // Why: a read error must not make a stale config look current; atomic write owns ACL repair and error surfacing. - } - writeFileAtomically(configPath, contents) - } - private getManagedAccountsRoot(): string { const root = join(app.getPath('userData'), 'codex-accounts') mkdirSync(root, { recursive: true }) diff --git a/src/main/codex/codex-config-mirror.test.ts b/src/main/codex/codex-config-mirror.test.ts index 7fcd1754992..ee1d375582f 100644 --- a/src/main/codex/codex-config-mirror.test.ts +++ b/src/main/codex/codex-config-mirror.test.ts @@ -768,6 +768,15 @@ describe('syncSystemConfigIntoLegacySharedCodexHome', () => { }) describe('prepareSystemConfigForFreshRuntimeMirror', () => { + it('allows WSL callers to retain Linux semantics for mounted-drive homes', () => { + expect( + resolveCodexConfigMirrorSourceDirectory( + 'C:\\Users\\alice\\.codex', + '/mnt/c/Users/alice/.codex' + ) + ).toBe('/mnt/c/Users/alice/.codex') + }) + it('uses the Linux-side directory for WSL UNC source homes', () => { const sourceDir = resolveCodexConfigMirrorSourceDirectory( '\\\\wsl.localhost\\Ubuntu\\home\\alice\\.codex' diff --git a/src/main/codex/codex-config-mirror.ts b/src/main/codex/codex-config-mirror.ts index f3d96a1064c..0b0e66c35be 100644 --- a/src/main/codex/codex-config-mirror.ts +++ b/src/main/codex/codex-config-mirror.ts @@ -155,7 +155,7 @@ type CodexConfigMirrorResult = | { status: 'mirrored'; preservedConflictKeys: ReadonlySet } function syncSystemConfigIntoManagedCodexHomeUnsafe( - { runtimeHomePath, systemHomePath }: CodexSettingsPromotionHomes, + { runtimeHomePath, systemHomePath, systemConfigDir }: CodexSettingsPromotionHomes, promotionPlan: CodexSettingsPromotionPlan ): CodexConfigMirrorResult { const systemConfigPath = join(systemHomePath, 'config.toml') @@ -184,7 +184,7 @@ function syncSystemConfigIntoManagedCodexHomeUnsafe( : { status: 'mirrored', preservedConflictKeys: new Set() } } - const sourceConfigDir = resolveCodexConfigMirrorSourceDirectory(systemHomePath) + const sourceConfigDir = resolveCodexConfigMirrorSourceDirectory(systemHomePath, systemConfigDir) if (!runtimeConfigExists) { writeFileAtomically( runtimeConfigPath, @@ -207,8 +207,15 @@ function syncSystemConfigIntoManagedCodexHomeUnsafe( return { status: 'mirrored', preservedConflictKeys: preserved.keys } } -export function resolveCodexConfigMirrorSourceDirectory(systemHomePath: string): string { - return parseWslUncPath(systemHomePath)?.linuxPath ?? dirname(join(systemHomePath, 'config.toml')) +export function resolveCodexConfigMirrorSourceDirectory( + systemHomePath: string, + systemConfigDir?: string +): string { + return ( + systemConfigDir ?? + parseWslUncPath(systemHomePath)?.linuxPath ?? + dirname(join(systemHomePath, 'config.toml')) + ) } function prepareSystemConfigForRuntimeMirror(config: string, systemConfigDir: string): string { diff --git a/src/main/codex/config-settings-promotion.ts b/src/main/codex/config-settings-promotion.ts index 9ff7383291d..45d95a266b1 100644 --- a/src/main/codex/config-settings-promotion.ts +++ b/src/main/codex/config-settings-promotion.ts @@ -184,6 +184,8 @@ export function snapshotCodexRuntimeSettingsBaseline( export type CodexSettingsPromotionHomes = { runtimeHomePath: string systemHomePath: string + /** Linux spelling of the source config directory when its host path is a drvfs drive. */ + systemConfigDir?: string } export type CodexSettingsPromotionPlan = { From d1a11b32992bc10fba7d8fb91c53f84b2380c5ad Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 26 Aug 2026 14:36:45 -0700 Subject: [PATCH 03/19] fix(pty): match the echo shapes a real tty actually produces (#16542) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Reply echo suppression modelled two echo shapes from the spec rather than from a tty. Captured under node-pty against real bash, at a readline prompt and under `read`: - Readline mangles CSI replies, not just OSC: `ESC [ ?` becomes BEL and the residue echoes. The projection was gated on an OSC introducer, so a private DSR echo was never matched at a readline prompt. This is the reachable one: a mode-2031 theme push (`CSI ?997;1n`) left latched by an exited TUI paints `997;1n` on a bash prompt (#9993's scenario). - ECHOCTL carets EVERY control, not just ESC. A BEL-terminated OSC reply echoes as `^G`, but the needle kept a literal BEL — a string no tty produces. Hardening only: every in-tree OSC reply is ST-terminated (terminal-osc-color-reply.ts:112, xterm's own reply), so the changed byte is unreachable except from a foreign or older emulator. Why this is not the CSI projection #13160 review dropped: that one was the identity (`replaceAll('\x1b]', …)` is a no-op on a CSI reply), so it was ESC-led and 500ms-held bare-ESC tails away from the query parser. This one is BEL-led. The rule is now asserted for every shape rather than implied by the gate: holdPartial iff the needle does not start with ESC. The readline branch is keyed on the private-DSR grammar with a non-empty parameter list, plus a floor on needle length. The containment grammar admits `CSI ? n`, and `answerLiveQueryReply` takes client-supplied bytes on the relay path, so a peer could otherwise arm a two-byte `BEL n` needle and delete the first bell-then-`n` in ordinary output. #61c65151129 proved this system can eat real output when a needle outlives its budget; a length floor is cheap. Live coverage: pty-reply-echo-shapes.node-pty.test.ts writes a reply to a real bash master and feeds back what it echoes, so a shell or libc change fails the suite instead of silently disarming suppression. Registered in the shell-contracts lane. The transcript tests and the caretEcho helpers that encoded the same ESC-only assumption are corrected alongside. Suppression is display-only. This does not change what reaches the child's stdin — the reply is written to the master either way, in call order. --- .github/workflows/pr.yml | 2 + config/reliability-gates.jsonc | 11 +- .../pty-input-write-queue.test.ts | 9 +- .../pty-reply-echo-shapes.node-pty.test.ts | 122 +++++++++++++++ ...y-startup-ingress-live-query-reply.test.ts | 64 +++++++- .../pty-startup-reply-echo-shapes.test.ts | 140 ++++++++++++++++++ src/shared/pty-startup-reply-echo-shapes.ts | 74 ++++++++- 7 files changed, 409 insertions(+), 13 deletions(-) create mode 100644 src/shared/pty-reply-echo-shapes.node-pty.test.ts create mode 100644 src/shared/pty-startup-reply-echo-shapes.test.ts diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index 97c68687a81..21d43d3bee5 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -425,6 +425,7 @@ jobs: src/main/zsh-wrapper-version-mismatch.live-shell.test.ts \ src/renderer/src/components/terminal-pane/fish-color-scheme-child-stdin.node-pty.test.ts \ src/shared/fish-query-reply-child-stdin.node-pty.test.ts \ + src/shared/pty-reply-echo-shapes.node-pty.test.ts \ src/shared/startup-shell-portability.live-shell.test.ts \ src/shared/posix-command-path-lookup.test.ts @@ -470,6 +471,7 @@ jobs: --exclude=src/main/zsh-wrapper-version-mismatch.live-shell.test.ts \ --exclude=src/renderer/src/components/terminal-pane/fish-color-scheme-child-stdin.node-pty.test.ts \ --exclude=src/shared/fish-query-reply-child-stdin.node-pty.test.ts \ + --exclude=src/shared/pty-reply-echo-shapes.node-pty.test.ts \ --exclude=src/shared/startup-shell-portability.live-shell.test.ts \ --exclude=src/shared/posix-command-path-lookup.test.ts \ --exclude=tests/e2e/cross-version-wire/** \ diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc index bd7d868a998..482f9808656 100644 --- a/config/reliability-gates.jsonc +++ b/config/reliability-gates.jsonc @@ -13116,7 +13116,7 @@ "invariant": "The desktop PTY input queue retains at most 64 explicitly sourced pending terminal query replies and 4096 UTF-16 code units. Every retained reply reaches the provider as one atomic write, and the host writes each reply the moment it accepts it, so replies reach the PTY in the order they were produced with no queue that could reorder them. A reply's own echo is contained on the output side by projecting its known echo shapes; the ESC-initial verbatim shape is matched only when complete, never held as a partial, so a query torn at its own ESC is still answered. Overflow removes only the oldest query replies, never ordinary input except the documented modified-F3/CPR byte collision, and drain failures cannot clear a newer queue generation.", "oracle": "Synchronously enqueue separate 10,000-entry OSC and DA1 reply floods before the scheduled drain and assert that only the initial immediate reply and newest 64 pending replies are written, each as one provider write, before a trailing keystroke. At the host boundary, defer an OSC reply and assert that separate or legacy-coalesced DA1/CPR replies flush after it in observed query order. At the remote-runtime boundary, preserve separate writes around pending ordinary input, async validation, and viewport-claim buffering. Repeat behind 10,000 ordinary inputs and exercise the text ceiling, real xterm generation, provider-write failure, rejected yield, and clear/reuse generation fencing.", "commands": [ - "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/pty-input-write-queue.test.ts src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts src/shared/terminal-query-reply.test.ts src/shared/pty-startup-ingress-live-query-reply.test.ts", + "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/pty-input-write-queue.test.ts src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts src/shared/terminal-query-reply.test.ts src/shared/pty-startup-ingress-live-query-reply.test.ts src/shared/pty-startup-reply-echo-shapes.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/remote-runtime-pty-batching.test.ts src/renderer/src/components/terminal-pane/remote-runtime-pty-query-reply-immediate.test.ts src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-input-coalescing.test.ts", "pnpm exec playwright test tests/e2e/terminal-osc-color-queries.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1", "pnpm exec playwright test tests/e2e/terminal-typing-latency.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1", @@ -13130,6 +13130,7 @@ "src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-input-coalescing.test.ts", "src/shared/terminal-query-reply.test.ts", "src/shared/pty-startup-ingress-live-query-reply.test.ts", + "src/shared/pty-startup-reply-echo-shapes.test.ts", "tests/e2e/terminal-osc-color-queries.spec.ts", "tests/e2e/terminal-typing-latency.spec.ts", "tests/e2e/pty-input-write-queue-ssh.spec.ts" @@ -13208,13 +13209,13 @@ ], "evidenceRuns": [ { - "date": "2026-08-09", + "date": "2026-08-25", "runner": "local", "platform": "macos", - "command": "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/pty-input-write-queue.test.ts src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts src/shared/terminal-query-reply.test.ts src/shared/pty-startup-ingress-live-query-reply.test.ts", + "command": "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/pty-input-write-queue.test.ts src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts src/shared/terminal-query-reply.test.ts src/shared/pty-startup-ingress-live-query-reply.test.ts src/shared/pty-startup-reply-echo-shapes.test.ts", "result": "passed", - "durationSeconds": 2.35, - "summary": "Four files and 138 tests passed, including explicit IPC reply-source routing, the 10,000-reply count ceiling, text ceiling, 10,000-entry ordinary backlog preservation, real xterm OSC query flood, single-shot drain-failure recovery, one-reply-per-write echo containment, and clear/reuse generation fencing." + "durationSeconds": 1.05, + "summary": "Five files and 92 tests passed, including the live-pty echo-shape transcript, including explicit IPC reply-source routing, the 10,000-reply count ceiling, text ceiling, 10,000-entry ordinary backlog preservation, real xterm OSC query flood, single-shot drain-failure recovery, one-reply-per-write echo containment, and clear/reuse generation fencing." }, { "date": "2026-08-09", diff --git a/src/renderer/src/components/terminal-pane/pty-input-write-queue.test.ts b/src/renderer/src/components/terminal-pane/pty-input-write-queue.test.ts index b63ab58acec..42124f0f8c0 100644 --- a/src/renderer/src/components/terminal-pane/pty-input-write-queue.test.ts +++ b/src/renderer/src/components/terminal-pane/pty-input-write-queue.test.ts @@ -509,7 +509,14 @@ describe('pty input write queue', () => { // → ingress echo strip, and assert no `997;1n` emission at the confirm prompt. vi.useFakeTimers() const reply = mode2031SequenceFor('dark') - const caretEcho = (data: string): string => data.replaceAll('\x1b', '^[') + // ECHOCTL carets every control, not just ESC. Identical for this reply (it carries no + // other control), but modelled correctly so this does not drift from the encoder. + const caretEcho = (data: string): string => + [...data] + .map((ch) => + ch.charCodeAt(0) < 0x20 ? `^${String.fromCharCode(ch.charCodeAt(0) + 0x40)}` : ch + ) + .join('') const masterWrites: string[] = [] const emissions: PtyIngressEmission[] = [] let ingress!: PtyStartupIngress diff --git a/src/shared/pty-reply-echo-shapes.node-pty.test.ts b/src/shared/pty-reply-echo-shapes.node-pty.test.ts new file mode 100644 index 00000000000..0fe06c7cae5 --- /dev/null +++ b/src/shared/pty-reply-echo-shapes.node-pty.test.ts @@ -0,0 +1,122 @@ +/** + * Keeps the transcript in pty-startup-reply-echo-shapes.test.ts honest. + * + * That file asserts against echo bytes recorded by hand, which is exactly how the two + * shapes this PR corrected went wrong in the first place: the projection and the test both + * encoded the same guess. This one writes a reply to a real PTY master, reads what bash + * actually echoes, and feeds it to the projection — so a shell or libc change that moves + * the shape fails here instead of silently disarming suppression. + * + * Two line disciplines, because they echo differently and both are reachable: + * readline — tty raw at a prompt, readline echoes in software + * cooked — under `read`, the kernel echoes via ECHOCTL + */ +import { existsSync } from 'node:fs' +import { afterEach, describe, expect, it } from 'vitest' +import { locateEcho, replyEchoProjections } from './pty-startup-reply-echo-shapes' +import { mode2031SequenceFor } from './terminal-color-scheme-protocol' + +const BASH = '/bin/bash' +const itWithBash = process.platform !== 'win32' && existsSync(BASH) ? it : it.skip + +const COLOR_SCHEME_REPLY = mode2031SequenceFor('dark') +const OSC_COLOR_REPLY_ST = '\x1b]11;rgb:2e2e/3434/3434\x1b\\' + +type Pty = { write: (data: string) => void; kill: () => void } + +let live: Pty | null = null + +afterEach(() => { + live?.kill() + live = null +}) + +const sleep = (ms: number): Promise => new Promise((resolve) => setTimeout(resolve, ms)) + +async function waitFor(predicate: () => boolean, timeoutMs: number): Promise { + const deadline = Date.now() + timeoutMs + while (Date.now() < deadline && !predicate()) { + await sleep(25) + } +} + +/** + * Writes `reply` to the master once bash is settled, and returns what came back. + * `discipline: 'cooked'` parks bash in `read` first, which restores ICANON+ECHO. + */ +async function echoOf(reply: string, discipline: 'readline' | 'cooked'): Promise { + const { spawn } = await import('node-pty') + let output = '' + const pty = spawn(BASH, ['--norc', '--noprofile', '-i'], { + name: 'xterm-256color', + cols: 80, + rows: 24, + env: { ...process.env, PS1: 'ORCA16542> ', TERM: 'xterm-256color' } + }) + live = { write: (data) => pty.write(data), kill: () => pty.kill() } + pty.onData((data) => { + output += data + }) + + await waitFor(() => output.includes('ORCA16542> '), 10_000) + if (discipline === 'cooked') { + pty.write('read -r ORCA_LINE\r') + await sleep(400) + } + output = '' + pty.write(reply) + // No marker to wait on: the echo is all this produces, so settle instead. + await waitFor(() => output.length > 0, 5_000) + await sleep(250) + return output +} + +describe('replyEchoProjections against a real bash pty', () => { + // The reachable fix: a latched mode-2031 push echoed at a readline prompt had no + // projection at all before this, so it always painted `997;1n` on the prompt (#9993). + itWithBash( + 'matches what readline echoes for a mode-2031 reply', + async () => { + const echo = await echoOf(COLOR_SCHEME_REPLY, 'readline') + expect(echo).not.toBe('') + const match = locateEcho(replyEchoProjections(COLOR_SCHEME_REPLY, 'posix-pty'), echo) + expect(match).toMatchObject({ kind: 'complete' }) + }, + 30_000 + ) + + itWithBash( + 'matches what the kernel echoes for a mode-2031 reply', + async () => { + const echo = await echoOf(COLOR_SCHEME_REPLY, 'cooked') + expect(echo).not.toBe('') + const match = locateEcho(replyEchoProjections(COLOR_SCHEME_REPLY, 'posix-pty'), echo) + expect(match).toMatchObject({ kind: 'complete' }) + }, + 30_000 + ) + + // The ST form is what every in-tree OSC reply uses, so this is the shape that must never + // regress — the BEL form is only reachable from a foreign or older emulator. + itWithBash( + 'matches what the kernel echoes for an ST-terminated OSC reply', + async () => { + const echo = await echoOf(OSC_COLOR_REPLY_ST, 'cooked') + expect(echo).not.toBe('') + const match = locateEcho(replyEchoProjections(OSC_COLOR_REPLY_ST, 'posix-pty'), echo) + expect(match).toMatchObject({ kind: 'complete' }) + }, + 30_000 + ) + + itWithBash( + 'matches what readline echoes for an ST-terminated OSC reply', + async () => { + const echo = await echoOf(OSC_COLOR_REPLY_ST, 'readline') + expect(echo).not.toBe('') + const match = locateEcho(replyEchoProjections(OSC_COLOR_REPLY_ST, 'posix-pty'), echo) + expect(match).toMatchObject({ kind: 'complete' }) + }, + 30_000 + ) +}) diff --git a/src/shared/pty-startup-ingress-live-query-reply.test.ts b/src/shared/pty-startup-ingress-live-query-reply.test.ts index 4a8e259f377..8048ceea127 100644 --- a/src/shared/pty-startup-ingress-live-query-reply.test.ts +++ b/src/shared/pty-startup-ingress-live-query-reply.test.ts @@ -13,7 +13,15 @@ const COLOR_SCHEME_REPLY = mode2031SequenceFor('dark') const OSC_COLOR_REPLY = '\x1b]11;rgb:00/00/00\x07' const CPR_REPLY = '\x1b[6;1R' const DA1_REPLY = '\x1b[?1;2c' -const caretEcho = (reply: string): string => reply.replaceAll('\x1b', '^[') +// ECHOCTL carets every C0 control, so an OSC reply's trailing BEL prints as `^G`. This +// modelled ESC alone, which is not a shape any tty produces — see the live-pty transcript +// in pty-startup-reply-echo-shapes.test.ts. +const caretEcho = (reply: string): string => + [...reply] + .map((ch) => + ch.charCodeAt(0) < 0x20 ? `^${String.fromCharCode(ch.charCodeAt(0) + 0x40)}` : ch + ) + .join('') const readlineEcho = (reply: string): string => reply.replaceAll('\x1b]', '\x07').replaceAll('\x1b\\', '') @@ -146,6 +154,60 @@ describe('live query replies', () => { ingress.drainAndClose() }) + // The private-DSR readline needle is only 7 bytes (`BEL 997;1n`), so it is the shortest + // span suppression can ever delete. These pin that the exposure is one exact match in an + // armed window — not a standing filter over ordinary output. + describe('the short readline needle does not eat ordinary output', () => { + const armed = (chunks: readonly string[]): string => { + const emissions: PtyIngressEmission[] = [] + const ingress = new PtyStartupIngress({ + ownerBackend: 'posix-pty', + write: () => {}, + onEmission: (emission) => void emissions.push(emission) + }) + ingress.answerLiveQueryReply(COLOR_SCHEME_REPLY) + for (const chunk of chunks) { + ingress.accept(chunk) + } + ingress.drainAndClose() + return visible(emissions) + } + + it.each([ + { name: 'BEL then a bare number', text: '\x07997 files changed\n' }, + { name: 'BEL then a near-miss final byte', text: '\x07997;1m not a reply\n' }, + { name: 'the digits without the BEL', text: ' 45% ==== 997;1n bytes\n' }, + { name: 'a bell mid-word', text: 'ding\x07 done\n' }, + { name: 'ls colour output', text: '\x1b[01;34msrc\x1b[0m \x1b[01;32mrun.sh\x1b[0m\n' }, + { name: 'a stack trace carrying 997:1', text: 'Error: boom\n at f (/a/b.js:997:1)\n' } + ])('passes $name through untouched', ({ text }) => { + expect(armed([text])).toBe(text) + // Same, torn at every byte boundary: a partial hold must still release intact. + for (let split = 1; split < text.length; split += 1) { + expect(armed([text.slice(0, split), text.slice(split)])).toBe(text) + } + }) + + it('suppresses the echo once, then stops', () => { + const echo = '\x07997;1n' + expect(armed([echo])).toBe('') + // One reply written, one span consumed — a repeat is ordinary output. + expect(armed([echo, echo])).toBe(echo) + }) + + it('suppresses nothing when no reply was written', () => { + const emissions: PtyIngressEmission[] = [] + const ingress = new PtyStartupIngress({ + ownerBackend: 'posix-pty', + write: () => {}, + onEmission: (emission) => void emissions.push(emission) + }) + ingress.accept('\x07997;1n') + ingress.drainAndClose() + expect(visible(emissions)).toBe('\x07997;1n') + }) + }) + it('bounds unmatched echo projections instead of shadowing the session', () => { const { ingress, writes, emissions } = harness() for (let index = 0; index < 200; index += 1) { diff --git a/src/shared/pty-startup-reply-echo-shapes.test.ts b/src/shared/pty-startup-reply-echo-shapes.test.ts new file mode 100644 index 00000000000..541d7d74d20 --- /dev/null +++ b/src/shared/pty-startup-reply-echo-shapes.test.ts @@ -0,0 +1,140 @@ +/** + * The echo shapes here are TRANSCRIBED FROM A LIVE PTY, not derived from the spec: each + * `echo` is what `/bin/bash` actually emitted after the reply was written to the master, + * captured under node-pty at a readline prompt (tty raw, readline echoes in software) and + * under a `read` builtin (tty cooked, kernel ECHOCTL echoes). + * + * Two shapes matched nothing before this file existed: an OSC reply on a cooked tty (its + * trailing BEL prints as `^G`, and only ESC was caret-encoded) and any private DSR at a + * readline prompt (the readline projection was gated on an OSC introducer). + */ +import { describe, expect, it } from 'vitest' +import { locateEcho, replyEchoProjections } from './pty-startup-reply-echo-shapes' + +const OSC11_BEL = '\x1b]11;rgb:2e2e/3434/3434\x07' +const OSC11_ST = '\x1b]11;rgb:2e2e/3434/3434\x1b\\' +const OSC10_BEL = '\x1b]10;rgb:c6c6/c6c6/c6c6\x07' +const DSR_997 = '\x1b[?997;1n' +const DSR_996 = '\x1b[?996n' + +/** `readline` = tty raw at a bash prompt; `cooked` = kernel ECHOCTL under `read`. */ +const LIVE_ECHOES: readonly { name: string; reply: string; echo: string }[] = [ + { name: 'OSC 11 BEL / readline', reply: OSC11_BEL, echo: '\x0711;rgb:2e2e/3434/3434\x07' }, + { name: 'OSC 11 ST / readline', reply: OSC11_ST, echo: '\x0711;rgb:2e2e/3434/3434' }, + { name: 'OSC 10 BEL / readline', reply: OSC10_BEL, echo: '\x0710;rgb:c6c6/c6c6/c6c6\x07' }, + { name: 'DSR 997 / readline', reply: DSR_997, echo: '\x07997;1n' }, + { name: 'DSR 996 / readline', reply: DSR_996, echo: '\x07996n' }, + { name: 'OSC 11 BEL / cooked', reply: OSC11_BEL, echo: '^[]11;rgb:2e2e/3434/3434^G' }, + { name: 'OSC 11 ST / cooked', reply: OSC11_ST, echo: '^[]11;rgb:2e2e/3434/3434^[\\' }, + { name: 'OSC 10 BEL / cooked', reply: OSC10_BEL, echo: '^[]10;rgb:c6c6/c6c6/c6c6^G' }, + { name: 'DSR 997 / cooked', reply: DSR_997, echo: '^[[?997;1n' }, + { name: 'DSR 996 / cooked', reply: DSR_996, echo: '^[[?996n' } +] + +describe('replyEchoProjections on a POSIX pty', () => { + it.each(LIVE_ECHOES)('matches the live $name echo', ({ reply, echo }) => { + const match = locateEcho(replyEchoProjections(reply, 'posix-pty'), echo) + expect(match).toEqual({ kind: 'complete', offset: 0, length: echo.length }) + }) + + // The tty coalesces its echo with surrounding shell output, so anchoring at offset 0 + // would recognize almost no real echo. + it.each(LIVE_ECHOES)('finds the $name echo embedded in output', ({ reply, echo }) => { + const match = locateEcho(replyEchoProjections(reply, 'posix-pty'), `user@host:~$ ${echo} `) + expect(match).toEqual({ kind: 'complete', offset: 13, length: echo.length }) + }) + + // `stty -echoctl` echoes the reply verbatim. + it.each(LIVE_ECHOES.map((entry) => entry.reply))('matches the verbatim echo of %j', (reply) => { + expect(locateEcho(replyEchoProjections(reply, 'posix-pty'), reply).kind).toBe('complete') + }) + + it('caret-encodes every control, not just ESC', () => { + const [kernel] = replyEchoProjections(OSC11_BEL, 'posix-pty') + expect(kernel?.needle).toBe('^[]11;rgb:2e2e/3434/3434^G') + expect(kernel?.needle).not.toContain('\x07') + }) + + // ECHOCTL passes TAB/LF/CR through literally and renders DEL as `^?`. No reply grammar + // carries one today; this pins the encoder so a future grammar cannot silently + // over-predict. Table matches the caret notation `vis(3)` defines. + it.each([ + { name: 'TAB stays literal', input: '\t', encoded: '\t' }, + { name: 'LF stays literal', input: '\n', encoded: '\n' }, + { name: 'CR stays literal', input: '\r', encoded: '\r' }, + { name: 'NUL carets to ^@', input: '\x00', encoded: '^@' }, + { name: 'BEL carets to ^G', input: '\x07', encoded: '^G' }, + { name: 'ESC carets to ^[', input: '\x1b', encoded: '^[' }, + { name: 'DEL carets to ^?', input: '\x7f', encoded: '^?' } + ])('$name under ECHOCTL', ({ input, encoded }) => { + const [kernel] = replyEchoProjections(`\x1b[?9${input}n`, 'posix-pty') + expect(kernel?.needle).toBe(`^[[?9${encoded}n`) + }) + + // DA1 shares the `ESC [ ?` prefix but is not a cooked-echo-risk reply. Keeping this off + // the readline path here, rather than relying on the caller's predicate, means widening + // that predicate cannot silently arm a holdable `BEL 1;2c` needle. + it.each(['\x1b[?1;2c', '\x1b[?0u', '\x1b[?2026;2$y', '\x1b[?12;5R'])( + 'projects no readline needle for %j, which is not a private DSR', + (reply) => { + const needles = replyEchoProjections(reply, 'posix-pty').map( + (projection) => projection.needle + ) + expect(needles.some((needle) => needle.startsWith('\x07'))).toBe(false) + } + ) + + // The containment grammar accepts an empty parameter list and `answerLiveQueryReply` + // takes client-supplied bytes on the relay path, so a peer could otherwise arm a two-byte + // `BEL n` needle and delete the first bell-then-`n` in ordinary output. + it.each(['\x1b[?n', '\x1b[?;n', '\x1b[?5n'])( + 'projects no readline needle for %j, which would be too short to be safe', + (reply) => { + const needles = replyEchoProjections(reply, 'posix-pty').map( + (projection) => projection.needle + ) + expect(needles.some((needle) => needle.startsWith('\x07'))).toBe(false) + } + ) + + it('projects readline for a private DSR, which carries no OSC introducer', () => { + const needles = replyEchoProjections(DSR_997, 'posix-pty').map( + (projection) => projection.needle + ) + expect(needles).toContain('\x07997;1n') + }) + + // A needle starting with ESC must never be held as a partial: a read ending on a bare + // ESC is a strict prefix of it, and an expired hold would release a stolen query raw. + it('holds a partial only for needles that do not start with ESC', () => { + for (const { reply } of LIVE_ECHOES) { + for (const projection of replyEchoProjections(reply, 'posix-pty')) { + expect(projection.holdPartial).toBe(!projection.needle.startsWith('\x1b')) + } + } + }) +}) + +describe('replyEchoProjections on other backends', () => { + it('keeps ConPTY on its documented ESC-stripped form', () => { + expect(replyEchoProjections(DSR_997, 'windows-conpty')).toEqual([ + { needle: '[?997;1n', holdPartial: true } + ]) + }) + + // Documents current behaviour, and is NOT a claim that it is right: conhost's echo of a + // BEL-terminated reply has never been captured, so the needle keeps a raw BEL exactly as + // the POSIX caret form used to. Unreachable in-tree (every OSC reply Orca emits is + // ST-terminated) and deliberately not corrected blind — see the branch comment. + it('leaves a BEL literal in the ConPTY needle, which is unverified', () => { + const [conpty] = replyEchoProjections(OSC11_BEL, 'windows-conpty') + expect(conpty?.needle).toBe(']11;rgb:2e2e/3434/3434\x07') + // The ST reply is the shape #9651 was actually reported against, and it has no BEL. + const [st] = replyEchoProjections(OSC11_ST, 'windows-conpty') + expect(st?.needle).toBe(']11;rgb:2e2e/3434/3434\\') + }) + + it('suppresses nothing when the echo shape is unverified', () => { + expect(replyEchoProjections(DSR_997, 'windows-wsl')).toEqual([]) + }) +}) diff --git a/src/shared/pty-startup-reply-echo-shapes.ts b/src/shared/pty-startup-reply-echo-shapes.ts index b232880d7dc..f03610497ad 100644 --- a/src/shared/pty-startup-reply-echo-shapes.ts +++ b/src/shared/pty-startup-reply-echo-shapes.ts @@ -21,25 +21,87 @@ export type PtyStartupReplyEchoMatch = | { kind: 'partial'; offset: number } | { kind: 'none' } +/* oxlint-disable-next-line no-control-regex -- terminal reply grammars are control sequences */ +const PRIVATE_DSR_RE = new RegExp('^\\u001b\\[\\?[0-9][0-9;]*n$') + +/** + * Floor on a BEL-led needle. The containment grammar accepts an empty parameter list + * (`CSI ? n`), and `answerLiveQueryReply` takes client-supplied bytes on the relay path — + * so without this a peer could arm a two-byte `BEL n` needle that deletes the first + * bell-then-`n` in ordinary output. Below this length the match is not worth the span. + */ +const MIN_READLINE_NEEDLE_LENGTH = 4 + +/** + * ECHOCTL carets EVERY control, not just ESC. Every OSC reply Orca emits is ST-terminated + * (terminal-osc-color-reply.ts) and ST is ESC-led, so this is byte-identical to the old + * ESC-only encoding for all of them. It matters for a BEL-terminated reply, which the + * grammar admits from a foreign or older emulator: the tty prints that BEL as `^G`, where + * encoding ESC alone left a literal BEL in the needle — a string no tty produces. + * + * Shapes verified against a live pty; bash, zsh and sh echo identically here, because this + * is the kernel and not the shell. + * + * TAB/LF/CR are exempt: ECHOCTL passes them through literally, so caret-encoding them + * would over-predict. No reply grammar carries one today, which is why this is a guard + * rather than a fix. + */ +function caretEncodeControls(reply: string): string { + let out = '' + for (const ch of reply) { + const code = ch.charCodeAt(0) + if (code === 0x09 || code === 0x0a || code === 0x0d) { + out += ch + } else if (code < 0x20) { + out += `^${String.fromCharCode(code + 0x40)}` + } else { + out += code === 0x7f ? '^?' : ch + } + } + return out +} + +/** + * Readline swallows the escape introducer, rings the bell, and echoes the residue — BEL + * replaces `ESC ]` for an OSC reply and `ESC [ ?` for a private DSR. The DSR shape had no + * projection at all, so `CSI ? … n` was never matched at a readline prompt. + */ +function readlineEchoProjection(reply: string): string | null { + if (reply.includes('\x1b]')) { + return reply.replaceAll('\x1b]', '\x07').replaceAll('\x1b\\', '') + } + // Keyed on the private-DSR grammar, not a bare `ESC [ ?` prefix: DA1 (`ESC [ ? 1 ; 2 c`) + // shares that prefix and is kept off this path today only by a predicate one module away. + // Matching the final `n` here means widening that predicate cannot silently arm a needle. + if (!PRIVATE_DSR_RE.test(reply)) { + return null + } + const needle = `\x07${reply.slice(3)}` + return needle.length >= MIN_READLINE_NEEDLE_LENGTH ? needle : null +} + export function replyEchoProjections( reply: string, ownerBackend: PtyOwnerBackend ): readonly EchoProjection[] { if (ownerBackend === 'windows-conpty') { - // Why: ConPTY's projection is the documented, deterministic ESC-stripped form. + // ESC-stripped, but observed only for the ST-terminated OSC 10/11 reply that #9651 + // was reported against — which is every OSC reply Orca emits. Whether conhost leaves a + // BEL literal, carets it, or eats it is unknown, so this shares the latent defect the + // POSIX caret form had. Not corrected blind: #9500 Decision 3 forbids generalising the + // ESC-strip without evidence, and the harness that would produce it does not exist. return [{ needle: reply.replaceAll('\x1b', ''), holdPartial: true }] } if (ownerBackend !== 'posix-pty') { // wsl.exe is ConPTY-hosted but its echo shape is unverified; suppress nothing. return [] } + const readline = readlineEchoProjection(reply) return [ // The kernel's ECHOCTL caret form — the POSIX default. - { needle: reply.replaceAll('\x1b', '^['), holdPartial: true }, - // Readline rewrites OSC, and echoes it even while the kernel reports ECHO clear. - ...(reply.includes('\x1b]') - ? [{ needle: reply.replaceAll('\x1b]', '\x07').replaceAll('\x1b\\', ''), holdPartial: true }] - : []), + { needle: caretEncodeControls(reply), holdPartial: true }, + // Readline rewrites the reply, and echoes it even while the kernel reports ECHO clear. + ...(readline === null ? [] : [{ needle: readline, holdPartial: true }]), // A `stty -echoctl` tty echoes the reply verbatim. Complete-match-only, because it // starts with ESC — see `holdPartial`. We just wrote these exact bytes, and we are // the terminal, so a child emitting the identical span in the same window is not a From ef0d5931bc59c31bd7f861cc3f66233cdef4ae2e Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 26 Aug 2026 14:37:15 -0700 Subject: [PATCH 04/19] fix(source-control): budget WSL bulk git command lines by bytes, not path count (#16634) Selecting ~100 changed files in a WSL worktree and hitting Stage All did nothing: the files stayed unstaged and the operation reported a failure. Bulk stage/unstage/discard chunked pathspecs 100 at a time, a count picked against a raw argv. A WSL-routed write is not a raw argv -- it is folded into one login-shell command line that shell-quotes every pathspec, quotes the result again, and embeds it three times (one branch per guest shell), so the finished line runs ~3.4x the raw pathspec bytes. Realistic project paths blew past the 32767-character CreateProcess cap at 100 paths and wsl.exe refused to spawn, with nothing staged. Chunking now measures the finished command line through the real resolver, so the wrapper's quoting rules live in one place and native, WSL and SSH hosts each get the budget of the host that actually spawns. A pathspec too long to fit alone still ships alone rather than being dropped, and no chunk is ever emitted empty -- a pathspec-free `clean -ffdx` would have swept the whole worktree. The tracked-path listing behind that discard also fences the WSL login shell now. Its stdout was parsed NUL-delimited without a fence, so Ubuntu's interactive rc banner glued itself onto the first record: that path failed to match anything git reported and was treated as untracked, sending a tracked file to `git clean` instead of `git restore`. Not observing a path in ls-files output is not evidence the path is untracked. The Windows command-line cap and its libuv-aware length estimate move out of the WSL runner into src/shared/windows-command-line-budget.ts, shared by both callers. --- .../bulk-pathspec-command-line-budget.test.ts | 193 ++++++++++++++++++ .../git/source-control/discard-changes.ts | 57 +++--- src/main/git/source-control/git-pathspec.ts | 84 +++++++- src/main/git/source-control/staging.ts | 31 +-- .../wsl-tracked-pathspec-banner.test.ts | 109 ++++++++++ .../status-discard-and-bulk-staging.test.ts | 4 +- src/main/wsl/wsl-runner.ts | 36 +--- src/shared/windows-command-line-budget.ts | 28 +++ 8 files changed, 454 insertions(+), 88 deletions(-) create mode 100644 src/main/git/source-control/bulk-pathspec-command-line-budget.test.ts create mode 100644 src/main/git/source-control/wsl-tracked-pathspec-banner.test.ts create mode 100644 src/shared/windows-command-line-budget.ts diff --git a/src/main/git/source-control/bulk-pathspec-command-line-budget.test.ts b/src/main/git/source-control/bulk-pathspec-command-line-budget.test.ts new file mode 100644 index 00000000000..39c4135aade --- /dev/null +++ b/src/main/git/source-control/bulk-pathspec-command-line-budget.test.ts @@ -0,0 +1,193 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + commandLineLength, + MAX_COMMAND_LINE_CHARS +} from '../../../shared/windows-command-line-budget' +import { resolveGitCommandWithoutProbe } from '../command-runner/git-command-resolution' + +const gitExecFileAsync = vi.fn(async () => ({ stdout: '', stderr: '' })) + +vi.mock('../runner', () => ({ + gitExecFileAsync: (...args: unknown[]) => + (gitExecFileAsync as unknown as (...a: unknown[]) => Promise<{ stdout: string }>)(...args) +})) +vi.mock('./git-read-cache-invalidation', () => ({ invalidateGitReadCaches: vi.fn() })) +vi.mock('../../../shared/git-discard-path-safety', () => ({ + removeSafeUntrackedDiscardTarget: vi.fn(), + removeSafeUntrackedDiscardTargets: async ( + _worktreePath: string, + untrackedPaths: string[], + cleanUntracked: (paths: string[]) => Promise, + restoreTracked: () => Promise + ) => { + await restoreTracked() + if (untrackedPaths.length > 0) { + await cleanUntracked(untrackedPaths) + } + } +})) + +const WSL_DISTRO = 'Ubuntu-24.04' +const WSL_WORKTREE = `\\\\wsl$\\${WSL_DISTRO}\\home\\emilio\\projects\\orca` + +/** Windows-side length of the line `wsl.exe` is spawned with, wrapper included. */ +function finishedCommandLineLength(args: readonly string[], wslDistro?: string): number { + const resolved = resolveGitCommandWithoutProbe([...args], { + cwd: wslDistro ? WSL_WORKTREE : '/home/emilio/projects/orca', + ...(wslDistro ? { wslDistro } : {}) + }) + return commandLineLength([resolved.binary, ...resolved.args]) +} + +/** Deep nesting, a space, and non-ASCII: all three inflate the quoted line. */ +function realisticChangedPaths(count: number): string[] { + return Array.from( + { length: count }, + (_, index) => + `apps/web/src/components/dashboard/widgets/analytics/rapport trimestriel ${String(index).padStart(3, '0')}/données-générales/AnalyticsSummaryWidget${index}.tsx` + ) +} + +function capturedInvocations(): string[][] { + return gitExecFileAsync.mock.calls.map((call) => (call as unknown as [string[]])[0]) +} + +describe('bulk pathspec command-line budget', () => { + const realPlatform = process.platform + + beforeEach(() => { + gitExecFileAsync.mockReset() + gitExecFileAsync.mockImplementation(async () => ({ stdout: '', stderr: '' })) + Object.defineProperty(process, 'platform', { value: 'win32' }) + }) + + afterEach(() => { + Object.defineProperty(process, 'platform', { value: realPlatform }) + vi.resetModules() + }) + + it('keeps every WSL bulk-stage invocation inside the Windows command-line cap', async () => { + const { bulkStageFiles } = await import('./staging') + const filePaths = realisticChangedPaths(100) + + await bulkStageFiles(WSL_WORKTREE, filePaths, { wslDistro: WSL_DISTRO }) + + const invocations = capturedInvocations() + expect(invocations.length).toBeGreaterThan(0) + const lengths = invocations.map((args) => finishedCommandLineLength(args, WSL_DISTRO)) + expect(Math.max(...lengths)).toBeLessThanOrEqual(MAX_COMMAND_LINE_CHARS) + }) + + it('stages every path exactly once, in order, across the chunks', async () => { + const { bulkStageFiles } = await import('./staging') + const filePaths = realisticChangedPaths(250) + + await bulkStageFiles(WSL_WORKTREE, filePaths, { wslDistro: WSL_DISTRO }) + + const staged = capturedInvocations().flatMap((args) => args.slice(args.indexOf('--') + 1)) + expect(staged).toEqual(filePaths.map((filePath) => `:(literal)${filePath}`)) + }) + + it('keeps WSL bulk unstage inside the cap', async () => { + const { bulkUnstageFiles } = await import('./staging') + + await bulkUnstageFiles(WSL_WORKTREE, realisticChangedPaths(100), { wslDistro: WSL_DISTRO }) + + const lengths = capturedInvocations().map((args) => finishedCommandLineLength(args, WSL_DISTRO)) + expect(Math.max(...lengths)).toBeLessThanOrEqual(MAX_COMMAND_LINE_CHARS) + }) + + it('never emits a pathspec-free chunk, which would widen `clean -ffdx` to the worktree', async () => { + const { bulkStageFiles } = await import('./staging') + + await bulkStageFiles(WSL_WORKTREE, realisticChangedPaths(300), { wslDistro: WSL_DISTRO }) + + for (const args of capturedInvocations()) { + expect(args.slice(args.indexOf('--') + 1).length).toBeGreaterThan(0) + } + }) + + it('splits a WSL bulk discard of tracked paths into spawnable restores', async () => { + const filePaths = realisticChangedPaths(120) + gitExecFileAsync.mockImplementation(async () => ({ stdout: filePaths.join('\0'), stderr: '' })) + const { bulkDiscardChanges } = await import('./discard-changes') + + await bulkDiscardChanges(WSL_WORKTREE, filePaths, { wslDistro: WSL_DISTRO }) + + const restores = capturedInvocations().filter((args) => args[0] === 'restore') + expect(restores.length).toBeGreaterThan(1) + for (const args of restores) { + expect(finishedCommandLineLength(args, WSL_DISTRO)).toBeLessThanOrEqual( + MAX_COMMAND_LINE_CHARS + ) + } + }) + + it('never widens `clean -ffdx` past the paths it was given', async () => { + const filePaths = realisticChangedPaths(120) + const { bulkDiscardChanges } = await import('./discard-changes') + + // Empty ls-files output: every path is untracked, so all of them take the clean lane. + await bulkDiscardChanges(WSL_WORKTREE, filePaths, { wslDistro: WSL_DISTRO }) + + const cleans = capturedInvocations().filter((args) => args[0] === 'clean') + expect(cleans.length).toBeGreaterThan(1) + const cleaned = cleans.flatMap((args) => args.slice(args.indexOf('--') + 1)) + expect(cleaned).toEqual(filePaths.map((filePath) => `:(literal)${filePath}`)) + for (const args of cleans) { + expect(finishedCommandLineLength(args, WSL_DISTRO)).toBeLessThanOrEqual( + MAX_COMMAND_LINE_CHARS + ) + } + }) + + it('ships a single over-budget pathspec alone rather than dropping it', async () => { + const { bulkStageFiles } = await import('./staging') + const hugePath = `src/${'nested-directory/'.repeat(700)}Component.tsx` + + await bulkStageFiles(WSL_WORKTREE, [hugePath, 'src/app.tsx'], { wslDistro: WSL_DISTRO }) + + const invocations = capturedInvocations() + expect(invocations).toHaveLength(2) + expect(invocations[0]).toEqual(['add', '--', `:(literal)${hugePath}`]) + expect(invocations[1]).toEqual(['add', '--', ':(literal)src/app.tsx']) + }) + + it('packs chunks to the budget instead of splitting timidly', async () => { + const { bulkStageFiles } = await import('./staging') + + await bulkStageFiles(WSL_WORKTREE, realisticChangedPaths(100), { wslDistro: WSL_DISTRO }) + + const lengths = capturedInvocations().map((args) => finishedCommandLineLength(args, WSL_DISTRO)) + // Every chunk but the last is filled to within one pathspec of the cap. + expect(Math.min(...lengths.slice(0, -1))).toBeGreaterThan(MAX_COMMAND_LINE_CHARS * 0.9) + }) + + it('gives a native Windows git.exe the CreateProcess cap and a POSIX host a larger one', async () => { + const { bulkPathspecCommands } = await import('./git-pathspec') + // Long enough that the raw argv alone passes the Windows cap with no wrapper in sight. + const filePaths = Array.from( + { length: 100 }, + (_, index) => `packages/${'deeply-nested-module/'.repeat(18)}file-${index}.ts` + ) + + const windowsNative = bulkPathspecCommands(['add', '--'], filePaths, 'C:\\repo', {}) + expect(windowsNative.length).toBeGreaterThan(1) + for (const args of windowsNative) { + expect(finishedCommandLineLength(args)).toBeLessThanOrEqual(MAX_COMMAND_LINE_CHARS) + } + + Object.defineProperty(process, 'platform', { value: 'linux' }) + expect(bulkPathspecCommands(['add', '--'], filePaths, '/repo', {})).toHaveLength(1) + }) + + it('does not charge a native invocation for the WSL wrapper', async () => { + Object.defineProperty(process, 'platform', { value: 'darwin' }) + const { bulkStageFiles } = await import('./staging') + + await bulkStageFiles('/home/emilio/projects/orca', realisticChangedPaths(100)) + + // Same 100 paths that need several chunks under the WSL wrapper stay one native call. + expect(capturedInvocations()).toHaveLength(1) + }) +}) diff --git a/src/main/git/source-control/discard-changes.ts b/src/main/git/source-control/discard-changes.ts index 33fd2f3f84a..e06f76f0c44 100644 --- a/src/main/git/source-control/discard-changes.ts +++ b/src/main/git/source-control/discard-changes.ts @@ -7,7 +7,7 @@ import type { GitRuntimeOptions } from '../git-runtime-options' import { gitOptionsForWorktree } from '../git-runtime-options' import { gitExecFileAsync } from '../runner' import { invalidateGitReadCaches } from './git-read-cache-invalidation' -import { BULK_CHUNK_SIZE, isTrackedPathSpec, literalPathspec } from './git-pathspec' +import { bulkPathspecCommands, isTrackedPathSpec, literalPathspec } from './git-pathspec' /** * Discard working tree changes for a file. @@ -62,14 +62,15 @@ async function listTrackedPathSpecs( options: GitRuntimeOptions = {} ): Promise { const trackedPaths: string[] = [] - for (let i = 0; i < filePaths.length; i += BULK_CHUNK_SIZE) { - const chunk = filePaths.slice(i, i + BULK_CHUNK_SIZE) - const { stdout } = await gitExecFileAsync( - ['ls-files', '-z', '--', ...chunk.map((filePath) => literalPathspec(filePath, options))], - { - ...gitOptionsForWorktree(worktreePath, options) - } - ) + const commands = bulkPathspecCommands(['ls-files', '-z', '--'], filePaths, worktreePath, options) + for (const args of commands) { + const { stdout } = await gitExecFileAsync(args, { + ...gitOptionsForWorktree(worktreePath, options), + // Why: this buffers rather than streams, so it must fence -- an unfenced WSL + // login shell glues its rc banner onto the first NUL record, and a tracked + // path that fails to match is silently reclassified as untracked. + captureWslLoginShellOutput: true + }) // Why: a tracked directory can hold enough paths to exceed the JS argument limit. for (const trackedPath of stdout.split('\0')) { if (trackedPath) { @@ -85,17 +86,11 @@ async function cleanUntrackedPaths( filePaths: readonly string[], options: GitRuntimeOptions = {} ): Promise { - for (let i = 0; i < filePaths.length; i += BULK_CHUNK_SIZE) { - const chunk = filePaths.slice(i, i + BULK_CHUNK_SIZE) - if (chunk.length > 0) { - // Why: Git pathspec cleanup avoids raw recursive deletion through symlinked parents. - await gitExecFileAsync( - ['clean', '-ffdx', '--', ...chunk.map((filePath) => literalPathspec(filePath, options))], - { - ...gitOptionsForWorktree(worktreePath, options) - } - ) - } + // Why: Git pathspec cleanup avoids raw recursive deletion through symlinked parents. + // A pathspec-free `clean -ffdx` would sweep the whole worktree; the chunker emits no empty chunk. + const commands = bulkPathspecCommands(['clean', '-ffdx', '--'], filePaths, worktreePath, options) + for (const args of commands) { + await gitExecFileAsync(args, { ...gitOptionsForWorktree(worktreePath, options) }) } } @@ -133,20 +128,14 @@ export async function bulkDiscardChanges( untrackedPaths, (targetPaths) => cleanUntrackedPaths(worktreePath, targetPaths, options), async () => { - for (let i = 0; i < trackedPaths.length; i += BULK_CHUNK_SIZE) { - const chunk = trackedPaths.slice(i, i + BULK_CHUNK_SIZE) - await gitExecFileAsync( - [ - 'restore', - '--worktree', - '--source=HEAD', - '--', - ...chunk.map((filePath) => literalPathspec(filePath, options)) - ], - { - ...gitOptionsForWorktree(worktreePath, options) - } - ) + const commands = bulkPathspecCommands( + ['restore', '--worktree', '--source=HEAD', '--'], + trackedPaths, + worktreePath, + options + ) + for (const args of commands) { + await gitExecFileAsync(args, { ...gitOptionsForWorktree(worktreePath, options) }) } } ) diff --git a/src/main/git/source-control/git-pathspec.ts b/src/main/git/source-control/git-pathspec.ts index 29ce04ade99..c4fe23d697e 100644 --- a/src/main/git/source-control/git-pathspec.ts +++ b/src/main/git/source-control/git-pathspec.ts @@ -1,6 +1,20 @@ +import { + commandLineLength, + MAX_COMMAND_LINE_CHARS +} from '../../../shared/windows-command-line-budget' +import { resolveGitCommandWithoutProbe } from '../command-runner/git-command-resolution' import type { GitRuntimeOptions } from '../git-runtime-options' -export const BULK_CHUNK_SIZE = 100 +/** Ceiling on argv entries per invocation; under WSL the byte budget bites first. */ +const BULK_CHUNK_SIZE = 100 + +/** + * POSIX hosts have no CreateProcess cap: ARG_MAX is 256KB on macOS and 2MB on + * Linux, shared with the environment block. Half the macOS floor keeps a native + * or SSH-host invocation clear of E2BIG without charging it the WSL wrapper's + * quoting overhead, which is a different transport's problem. + */ +const POSIX_COMMAND_LINE_BUDGET = 128_000 function normalizeGitPathForCompare(filePath: string): string { return filePath.replace(/\\/g, '/').replace(/\/+$/, '') @@ -19,3 +33,71 @@ export function isTrackedPathSpec(filePath: string, trackedPaths: readonly strin return normalizedTracked === normalized || normalizedTracked.startsWith(`${normalized}/`) }) } + +/** + * Length of the line the OS will actually be handed, wrapper included. + * + * Why resolve rather than estimate: a WSL-routed write goes through the login + * shell, which shell-quotes every pathspec, quotes the resulting command line + * again, and embeds that three times (one branch per guest shell). The finished + * line runs ~3.4x the raw pathspec bytes, and nothing about the path list says + * so. Writes never take the direct-git lane, so this is the exact shape they + * get; a read that does take it resolves shorter, so the estimate stays safe. + */ +function finishedCommandLineLength( + args: readonly string[], + worktreePath: string, + options: GitRuntimeOptions +): number { + const resolved = resolveGitCommandWithoutProbe([...args], { + cwd: worktreePath, + ...(options.wslDistro ? { wslDistro: options.wslDistro } : {}) + }) + return commandLineLength([resolved.binary, ...resolved.args]) +} + +/** + * Split a bulk pathspec operation into invocations the host can actually spawn. + * + * Why a byte budget and not a path count: 100 was chosen against a raw argv, but + * a WSL-routed `git add` is folded into one login-shell command line, so 100 + * ordinary paths reached ~43,000 characters -- past the 32,767 CreateProcess cap + * -- and the bulk stage failed with nothing staged. Cost is measured per + * pathspec through the real resolver so the wrapper's quoting rules live in one + * place. + * + * Chunks split only between whole pathspecs, and a pathspec that alone exceeds + * the budget still ships alone rather than being dropped or truncated. If an + * invocation fails partway through, the earlier chunks stay applied: every + * operation here is idempotent and per-path, so `git status` shows the true + * state and re-running converges. + */ +export function bulkPathspecCommands( + leadingArgs: readonly string[], + filePaths: readonly string[], + worktreePath: string, + options: GitRuntimeOptions +): string[][] { + // Budget belongs to the host that spawns; the overhead measured above belongs to the transport. + const budget = process.platform === 'win32' ? MAX_COMMAND_LINE_CHARS : POSIX_COMMAND_LINE_BUDGET + const baseLength = finishedCommandLineLength(leadingArgs, worktreePath, options) + const commands: string[][] = [] + let pathspecs: string[] = [] + let length = baseLength + for (const filePath of filePaths) { + const pathspec = literalPathspec(filePath, options) + const cost = + finishedCommandLineLength([...leadingArgs, pathspec], worktreePath, options) - baseLength + if (pathspecs.length > 0 && (pathspecs.length >= BULK_CHUNK_SIZE || length + cost > budget)) { + commands.push([...leadingArgs, ...pathspecs]) + pathspecs = [] + length = baseLength + } + pathspecs.push(pathspec) + length += cost + } + if (pathspecs.length > 0) { + commands.push([...leadingArgs, ...pathspecs]) + } + return commands +} diff --git a/src/main/git/source-control/staging.ts b/src/main/git/source-control/staging.ts index 5aa8f55fa48..3aa6e16bbc9 100644 --- a/src/main/git/source-control/staging.ts +++ b/src/main/git/source-control/staging.ts @@ -2,7 +2,7 @@ import type { GitRuntimeOptions } from '../git-runtime-options' import { gitOptionsForWorktree } from '../git-runtime-options' import { gitExecFileAsync } from '../runner' import { invalidateGitReadCaches } from './git-read-cache-invalidation' -import { BULK_CHUNK_SIZE, literalPathspec } from './git-pathspec' +import { bulkPathspecCommands, literalPathspec } from './git-pathspec' /** * Stage a file. @@ -54,12 +54,8 @@ export async function bulkStageFiles( return } try { - for (let i = 0; i < filePaths.length; i += BULK_CHUNK_SIZE) { - const chunk = filePaths.slice(i, i + BULK_CHUNK_SIZE) - await gitExecFileAsync( - ['add', '--', ...chunk.map((filePath) => literalPathspec(filePath, options))], - gitOptionsForWorktree(worktreePath, options) - ) + for (const args of bulkPathspecCommands(['add', '--'], filePaths, worktreePath, options)) { + await gitExecFileAsync(args, gitOptionsForWorktree(worktreePath, options)) } } finally { invalidateGitReadCaches() @@ -79,19 +75,14 @@ export async function bulkUnstageFiles( return } try { - for (let i = 0; i < filePaths.length; i += BULK_CHUNK_SIZE) { - const chunk = filePaths.slice(i, i + BULK_CHUNK_SIZE) - await gitExecFileAsync( - [ - 'restore', - '--staged', - '--', - ...chunk.map((filePath) => literalPathspec(filePath, options)) - ], - { - ...gitOptionsForWorktree(worktreePath, options) - } - ) + const commands = bulkPathspecCommands( + ['restore', '--staged', '--'], + filePaths, + worktreePath, + options + ) + for (const args of commands) { + await gitExecFileAsync(args, { ...gitOptionsForWorktree(worktreePath, options) }) } } finally { invalidateGitReadCaches() diff --git a/src/main/git/source-control/wsl-tracked-pathspec-banner.test.ts b/src/main/git/source-control/wsl-tracked-pathspec-banner.test.ts new file mode 100644 index 00000000000..fbae736e77f --- /dev/null +++ b/src/main/git/source-control/wsl-tracked-pathspec-banner.test.ts @@ -0,0 +1,109 @@ +import { EventEmitter } from 'node:events' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { execFileMock, execFileSyncMock, spawnMock } = vi.hoisted(() => ({ + execFileMock: vi.fn(), + execFileSyncMock: vi.fn(), + spawnMock: vi.fn() +})) + +vi.mock('node:child_process', () => ({ + execFile: execFileMock, + execFileSync: execFileSyncMock, + spawn: spawnMock +})) +vi.mock('../../observability/instrumentation', () => ({ + withGitSpan: (_attributes: unknown, run: () => unknown) => run() +})) +vi.mock('../../diagnostics/main-thread-churn-probe', () => ({ recordSubprocessSpawn: vi.fn() })) +vi.mock('./git-read-cache-invalidation', () => ({ invalidateGitReadCaches: vi.fn() })) +// Stand in for the on-disk safety filter: every discard target here exists and is symlink-free. +vi.mock('../../../shared/git-discard-path-safety', () => ({ + removeSafeUntrackedDiscardTarget: vi.fn(), + removeSafeUntrackedDiscardTargets: async ( + _worktreePath: string, + untrackedPaths: string[], + cleanUntracked: (paths: string[]) => Promise, + restoreTracked: () => Promise + ) => { + await restoreTracked() + if (untrackedPaths.length > 0) { + await cleanUntracked(untrackedPaths) + } + } +})) + +import { bulkDiscardChanges } from './discard-changes' +import { resetWslGitReadEnvironmentForTests } from '../wsl-git-read-environment' + +const DISTRO = 'Ubuntu-24.04' +const WSL_WORKTREE = `\\\\wsl$\\${DISTRO}\\home\\emilio\\projects\\orca` +const TRACKED_PATHS = ['docs/architecture.md', 'src/main/git/runner.ts'] +// Stock Ubuntu writes this to *stdout* from the interactive login shell's rc. +const BANNER = 'To run a command as administrator (user "root"), use "sudo ".\n\n' + +type MockChild = EventEmitter & { stdout: EventEmitter; stderr: EventEmitter; kill: () => void } + +function createMockChild(): MockChild { + const child = new EventEmitter() as MockChild + child.stdout = new EventEmitter() + child.stderr = new EventEmitter() + child.kill = vi.fn() + return child +} + +function guestScript(args: unknown): string { + return (args as string[] | undefined)?.join(' ') ?? '' +} + +/** Wrap the payload in the caller's own fence when it asked for one; otherwise hand it over raw. */ +function loginShellStdout(script: string, payload: string): string { + const nonce = /__ORCA_WSL_CAPTURE_BEGIN_([^_]+)__/.exec(script)?.[1] + return nonce + ? `${BANNER}__ORCA_WSL_CAPTURE_BEGIN_${nonce}__${payload}__ORCA_WSL_CAPTURE_END_${nonce}__` + : `${BANNER}${payload}` +} + +function gitCommandLines(): string[] { + return execFileMock.mock.calls + .map((call) => guestScript(call[1])) + .filter((script) => !script.includes('_orca_git_path=')) +} + +describe('WSL tracked-path listing behind a login-shell banner', () => { + const realPlatform = process.platform + + beforeEach(() => { + resetWslGitReadEnvironmentForTests() + execFileMock.mockReset() + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + execFileMock.mockImplementation((_command, args, _options, callback) => { + const script = guestScript(args) + if (script.includes('_orca_git_path=')) { + // Distro exports GIT_* / XDG_CONFIG_HOME, so the direct-git read probe is rejected for good. + queueMicrotask(() => callback?.(Object.assign(new Error('probe rejected'), { code: 78 }))) + return createMockChild() + } + const payload = script.includes('ls-files') ? `${TRACKED_PATHS.join('\0')}\0` : '' + queueMicrotask(() => callback?.(null, loginShellStdout(script, payload), '')) + return createMockChild() + }) + }) + + afterEach(() => { + Object.defineProperty(process, 'platform', { configurable: true, value: realPlatform }) + resetWslGitReadEnvironmentForTests() + }) + + it('restores every tracked path instead of routing the first one to git clean', async () => { + await bulkDiscardChanges(WSL_WORKTREE, [...TRACKED_PATHS], { wslDistro: DISTRO }) + + const commandLines = gitCommandLines() + expect(commandLines.filter((line) => line.includes('clean'))).toEqual([]) + const restored = commandLines.filter((line) => line.includes('restore')) + expect(restored.length).toBeGreaterThan(0) + for (const trackedPath of TRACKED_PATHS) { + expect(restored.some((line) => line.includes(`:(literal)${trackedPath}`))).toBe(true) + } + }) +}) diff --git a/src/main/git/status-discard-and-bulk-staging.test.ts b/src/main/git/status-discard-and-bulk-staging.test.ts index b1f39f9102b..8e0b12623c4 100644 --- a/src/main/git/status-discard-and-bulk-staging.test.ts +++ b/src/main/git/status-discard-and-bulk-staging.test.ts @@ -198,7 +198,9 @@ describe('bulk git helpers', () => { ':(literal)scratch' ], { - cwd: '/repo' + cwd: '/repo', + // Why: this read buffers, so it fences the WSL login shell's rc banner off its first NUL record. + captureWslLoginShellOutput: true } ) // Why: a pathspec is tracked if git reports either the exact path or a diff --git a/src/main/wsl/wsl-runner.ts b/src/main/wsl/wsl-runner.ts index 6a0fbc051ac..30097246a86 100644 --- a/src/main/wsl/wsl-runner.ts +++ b/src/main/wsl/wsl-runner.ts @@ -1,4 +1,5 @@ import { addWslEnvKeys } from '../../shared/wsl-env' +import { commandLineLength, MAX_COMMAND_LINE_CHARS } from '../../shared/windows-command-line-budget' import { runProcess } from '../../shared/child-process/run-process' import { buildWslExecArgs } from '../../shared/wsl-login-shell-command' import { getWslGuestEnvironment, type WslGuestEnvironment } from './wsl-guest-environment' @@ -159,25 +160,6 @@ function withGuestCwd(cwd: string | undefined, argv: readonly string[]): string[ return ['sh', '-c', 'cd "$1" || exit 1; shift; exec "$@"', 'orca-wsl', cwd, ...argv] } -/** - * Argv is the default, but it has a hard ceiling that stdin does not. - * - * Windows caps a command line at 32767 characters, and the distro, `--exec`, - * the env prefix and the args all share it. A user's `orca.yaml` hook is the - * one unbounded script Orca runs -- `run-both` concatenates two of them, and a - * vendored installer is ~15KB -- so past this size the choice is between - * failing to spawn at all and accepting the stdin caveat. Degrading beats - * failing: a large script that also reads stdin was already broken, while a - * large script that does not now works where it would have died. - * - * Measured on the WHOLE command line, not on the script alone. A login PATH is - * itself a few KB and is spliced in as `PATH=...`, so a script-only threshold - * produced a perverse band: with a long enough PATH, a 7,999-char hook went to - * argv and failed to spawn while the same hook at 8,001 chars flipped to stdin - * and ran. Size decided how a hook behaved, in the wrong direction. - */ -const MAX_COMMAND_LINE_CHARS = 30_000 - /** ` -c`/`-s` for a script, otherwise the program itself. */ function guestCommandArgv(spec: WslSpec, delivery: 'argv' | 'stdin'): string[] { if (spec.script === undefined) { @@ -190,19 +172,6 @@ function guestCommandArgv(spec: WslSpec, delivery: 'argv' | 'stdin'): string[] { : [shell, '-c', spec.script, '--', ...(spec.args ?? [])] } -/** - * What `CreateProcess` will count. - * - * libuv escapes every `"` and doubles a backslash run before a quote, so a - * quote-dense script costs more than its length. Charging one extra character - * per `"` or `\\` keeps the estimate on the safe side of the cap; an earlier - * version claimed to over-count and in fact under-counted, which put a - * quote-heavy ~26KB script on argv and over the real limit. - */ -function commandLineLength(args: readonly string[]): number { - return args.reduce((total, arg) => total + arg.length + 3 + (arg.match(/["\\]/g)?.length ?? 0), 0) -} - /** Shell-free argv, with the cached environment applied when one is available. */ function buildGuestArgv( environment: WslGuestEnvironment | null, @@ -255,6 +224,9 @@ export async function runWslProcess(spec: WslSpec): Promise { // Measure what is actually spawned: `wsl.exe` and `-d --exec` are // prepended after this point and are part of the same budget. const fullLine = [resolveWslExecutablePath(), ...buildWslExecArgs(spec.distro, argvForm)] + // Argv is the default, but it has a hard ceiling that stdin does not. A user's + // `orca.yaml` hook is the one unbounded script Orca runs, so past the cap the + // choice is between failing to spawn at all and accepting the stdin caveat. const delivery: 'argv' | 'stdin' = spec.script !== undefined && commandLineLength(fullLine) > MAX_COMMAND_LINE_CHARS ? 'stdin' diff --git a/src/shared/windows-command-line-budget.ts b/src/shared/windows-command-line-budget.ts new file mode 100644 index 00000000000..77074e3f531 --- /dev/null +++ b/src/shared/windows-command-line-budget.ts @@ -0,0 +1,28 @@ +/** + * How long a command line may get before `CreateProcess` refuses it. + * + * Windows caps a command line at 32767 characters, and *everything* shares that + * one budget: the binary, `wsl.exe -d --exec`, the login-shell wrapper + * and the payload. So the number only means anything when it is measured on the + * FINISHED line. Two shipped defects came from measuring a part instead: a + * script-only threshold in the WSL runner, where a multi-KB login PATH pushed a + * legal-looking script over the real limit, and count-only chunking of bulk git + * pathspecs, where the login-shell wrapper tripled the line behind our back. + * + * The 2767-character margin absorbs what we do not model exactly (the distro + * name, libuv's requoting of the outer argv). + */ +export const MAX_COMMAND_LINE_CHARS = 30_000 + +/** + * What `CreateProcess` will count. + * + * libuv escapes every `"` and doubles a backslash run before a quote, so a + * quote-dense script costs more than its length. Charging one extra character + * per `"` or `\\` keeps the estimate on the safe side of the cap; an earlier + * version claimed to over-count and in fact under-counted, which put a + * quote-heavy ~26KB script on argv and over the real limit. + */ +export function commandLineLength(args: readonly string[]): number { + return args.reduce((total, arg) => total + arg.length + 3 + (arg.match(/["\\]/g)?.length ?? 0), 0) +} From 7d5c7aa9c3533ec3d880feb058c5e068e36df55a Mon Sep 17 00:00:00 2001 From: OrcaWin Date: Wed, 26 Aug 2026 15:04:51 -0700 Subject: [PATCH 05/19] i18n: Make skill install dialogs and errors translatable (#16682) Extract hardcoded error messages and status labels from skill installation components into the i18n system. Supports localized UI for install flows in English, Spanish, Japanese, Korean, Chinese. Co-authored-by: OrcaWin <293788423+OrcaWin@users.noreply.github.com> --- .../skills/SkillBundleInstallFlow.tsx | 12 ++- .../components/skills/SkillInstallDialog.tsx | 19 ++-- .../skills/SkillInstallManagementDialog.tsx | 19 ++-- .../components/skills/SkillShareDialog.tsx | 7 +- .../skills/skill-install-progress-state.ts | 12 +-- src/renderer/src/i18n/locales/en.json | 20 ++++- src/renderer/src/i18n/locales/es.json | 87 +++++++++++++++++++ src/renderer/src/i18n/locales/ja.json | 48 ++++++++++ src/renderer/src/i18n/locales/ko.json | 13 +++ src/renderer/src/i18n/locales/zh.json | 23 ++--- 10 files changed, 222 insertions(+), 38 deletions(-) diff --git a/src/renderer/src/components/skills/SkillBundleInstallFlow.tsx b/src/renderer/src/components/skills/SkillBundleInstallFlow.tsx index 74d3c994c14..2040b63682b 100644 --- a/src/renderer/src/components/skills/SkillBundleInstallFlow.tsx +++ b/src/renderer/src/components/skills/SkillBundleInstallFlow.tsx @@ -126,7 +126,11 @@ export function SkillBundleInstallFlow(props: { ): Promise => { const target = destination() if (!target || requestedIds.size === 0) { - setError(target ? 'Select at least one skill.' : 'Choose a workspace.') + setError( + target + ? translate('auto.components.skills.install.selectSkill', 'Select at least one skill.') + : translate('auto.components.skills.install.chooseWorkspace', 'Choose a workspace.') + ) return } setBusy(true) @@ -189,7 +193,7 @@ export function SkillBundleInstallFlow(props: { } else if (operation.status !== 'ok') { setError( operation.status === 'reconnect-required' - ? 'Reconnect your Orca account before installing.' + ? translate('auto.components.skills.install.reconnectBeforeInstalling', 'Reconnect your Orca account before installing.') : operation.message ) } else { @@ -204,7 +208,7 @@ export function SkillBundleInstallFlow(props: { } } catch (cause) { console.warn('[skills] bundle install failed:', cause) - setError('Installation failed before Orca could verify the requested bundle.') + setError(translate('auto.components.skills.install.bundleVerificationFailed', 'Installation failed before Orca could verify the requested bundle.')) } finally { installProgress.finish() setBusy(false) @@ -221,7 +225,7 @@ export function SkillBundleInstallFlow(props: { ...(environmentId === 'local' || environmentId.startsWith('ssh:') ? {} : { environmentId }) }) if (!cancelled.cancelled) { - setError('The destination had already finished this installation.') + setError(translate('auto.components.skills.install.destinationAlreadyFinished', 'The destination had already finished this installation.')) } } diff --git a/src/renderer/src/components/skills/SkillInstallDialog.tsx b/src/renderer/src/components/skills/SkillInstallDialog.tsx index e011264374f..c5498ac6b3d 100644 --- a/src/renderer/src/components/skills/SkillInstallDialog.tsx +++ b/src/renderer/src/components/skills/SkillInstallDialog.tsx @@ -91,7 +91,7 @@ export function SkillInstallDialog({ const resolveLink = useCallback(async (value: string): Promise => { const shareId = parseSkillShareId(value) if (!shareId) { - setError('Enter an Orca skill share link.') + setError(translate('auto.components.skills.install.enterShareLink', 'Enter an Orca skill share link.')) return } setBusy(true) @@ -103,14 +103,14 @@ export function SkillInstallDialog({ setError( operation.status === 'unconfigured' ? operation.message - : 'This share is unavailable. The link may be invalid, expired, or revoked.' + : translate('auto.components.skills.install.shareUnavailable', 'This share is unavailable. The link may be invalid, expired, or revoked.') ) return } setPreview({ shareId, version: operation.value.version }) } catch (cause) { console.warn('[skills] share resolution failed:', cause) - setError('This share is unavailable. The link may be invalid, expired, or revoked.') + setError(translate('auto.components.skills.install.shareUnavailable', 'This share is unavailable. The link may be invalid, expired, or revoked.')) } finally { setBusy(false) } @@ -138,7 +138,7 @@ export function SkillInstallDialog({ } const choice = workspaceChoices.find((candidate) => candidate.id === workspace) if (scope === 'workspace' && !choice) { - setError('Choose a workspace.') + setError(translate('auto.components.skills.install.chooseWorkspace', 'Choose a workspace.')) return } setBusy(true) @@ -204,7 +204,7 @@ export function SkillInstallDialog({ if (operation.status !== 'ok') { setError( operation.status === 'reconnect-required' - ? 'Reconnect your Orca account before installing.' + ? translate('auto.components.skills.install.reconnectBeforeInstalling', 'Reconnect your Orca account before installing.') : operation.message ) return @@ -215,7 +215,12 @@ export function SkillInstallDialog({ } } catch (cause) { console.warn('[skills] install failed:', cause) - setError('Installation failed before Orca could verify the requested version.') + setError( + translate( + 'auto.components.skills.install.requestedVersionVerificationFailed', + 'Installation failed before Orca could verify the requested version.' + ) + ) } finally { installProgress.finish() setBusy(false) @@ -231,7 +236,7 @@ export function SkillInstallDialog({ ...(environmentId === 'local' || environmentId.startsWith('ssh:') ? {} : { environmentId }) }) if (!cancelled.cancelled) { - setError('The destination had already finished this installation.') + setError(translate('auto.components.skills.install.destinationAlreadyFinished', 'The destination had already finished this installation.')) } } diff --git a/src/renderer/src/components/skills/SkillInstallManagementDialog.tsx b/src/renderer/src/components/skills/SkillInstallManagementDialog.tsx index 47d7518ccdd..a760af0335e 100644 --- a/src/renderer/src/components/skills/SkillInstallManagementDialog.tsx +++ b/src/renderer/src/components/skills/SkillInstallManagementDialog.tsx @@ -28,6 +28,7 @@ import { } from './skill-managed-install-groups' import { SkillInstallMachineSelect } from './SkillInstallMachineSelect' import { SkillInstallManagementStatus } from './SkillInstallManagementStatus' +import { translate } from '@/i18n/i18n' export function SkillInstallManagementDialog({ open, @@ -90,7 +91,7 @@ export function SkillInstallManagementDialog({ return } console.warn('[skills] managed install listing failed:', cause) - setError('Orca could not inspect managed installs on this machine.') + setError(translate('auto.components.skills.install.inspectManagedFailed', 'Orca could not inspect managed installs on this machine.')) } finally { if (generation === loadGeneration.current) { setBusy(false) @@ -120,7 +121,7 @@ export function SkillInstallManagementDialog({ if (operation.status !== 'ok') { setError( operation.status === 'reconnect-required' - ? 'Reconnect your Orca account to load version history.' + ? translate('auto.components.skills.install.reconnectForVersionHistory', 'Reconnect your Orca account to load version history.') : operation.message ) return @@ -132,7 +133,7 @@ export function SkillInstallManagementDialog({ return } console.warn('[skills] package history failed:', cause) - setError('Version history is unavailable for this skill.') + setError(translate('auto.components.skills.install.versionHistoryUnavailable', 'Version history is unavailable for this skill.')) } finally { if (generation === detailGeneration.current) { setBusy(false) @@ -168,7 +169,7 @@ export function SkillInstallManagementDialog({ installedNames.has(skill.name) ) if (selectedSkills.length === 0) { - setError('This version does not contain any of the installed bundle skills.') + setError(translate('auto.components.skills.install.bundleSkillsMissing', 'This version does not contain any of the installed bundle skills.')) return } const operation = await window.api.skills.installBundlePackageVersion({ @@ -193,7 +194,7 @@ export function SkillInstallManagementDialog({ if (operation.status !== 'ok') { setError( operation.status === 'reconnect-required' - ? 'Reconnect your Orca account before changing versions.' + ? translate('auto.components.skills.install.reconnectBeforeVersionChange', 'Reconnect your Orca account before changing versions.') : operation.message ) return @@ -219,7 +220,7 @@ export function SkillInstallManagementDialog({ if (operation.status !== 'ok') { setError( operation.status === 'reconnect-required' - ? 'Reconnect your Orca account before changing versions.' + ? translate('auto.components.skills.install.reconnectBeforeVersionChange', 'Reconnect your Orca account before changing versions.') : operation.message ) return @@ -233,7 +234,7 @@ export function SkillInstallManagementDialog({ } } catch (cause) { console.warn('[skills] version installation failed:', cause) - setError('Orca could not verify the requested version.') + setError(translate('auto.components.skills.install.versionVerificationFailed', 'Orca could not verify the requested version.')) } finally { installProgress.finish() setBusy(false) @@ -249,7 +250,7 @@ export function SkillInstallManagementDialog({ ...(environmentId === 'local' || environmentId.startsWith('ssh:') ? {} : { environmentId }) }) if (!cancelled.cancelled) { - setError('The destination had already finished this installation.') + setError(translate('auto.components.skills.install.destinationAlreadyFinished', 'The destination had already finished this installation.')) } } @@ -294,7 +295,7 @@ export function SkillInstallManagementDialog({ } } catch (cause) { console.warn('[skills] managed removal failed:', cause) - setError('Orca could not safely remove this skill.') + setError(translate('auto.components.skills.install.removeFailed', 'Orca could not safely remove this skill.')) } finally { setBusy(false) } diff --git a/src/renderer/src/components/skills/SkillShareDialog.tsx b/src/renderer/src/components/skills/SkillShareDialog.tsx index 145f9066095..9cf73590dfc 100644 --- a/src/renderer/src/components/skills/SkillShareDialog.tsx +++ b/src/renderer/src/components/skills/SkillShareDialog.tsx @@ -231,7 +231,12 @@ export function SkillShareDialog({ } catch { cancellationRequested.current = false setCancelling(false) - setError('Orca could not send the cancellation request. The upload may still finish.') + setError( + translate( + 'auto.components.skills.SkillShareDialog.cancelRequestFailed', + 'Orca could not send the cancellation request. The upload may still finish.' + ) + ) } } diff --git a/src/renderer/src/components/skills/skill-install-progress-state.ts b/src/renderer/src/components/skills/skill-install-progress-state.ts index d88c523ea95..c16b27f402c 100644 --- a/src/renderer/src/components/skills/skill-install-progress-state.ts +++ b/src/renderer/src/components/skills/skill-install-progress-state.ts @@ -2,11 +2,6 @@ import { useEffect, useRef, useState } from 'react' import type { SkillInstallProgress } from '../../../../shared/skill-sharing-contract' import { translate } from '@/i18n/i18n' -const INSTALL_PHASE_LABELS = { - authorizing: 'Authorizing package access…', - installing: 'Downloading, verifying, and installing…' -} as const - export function useSkillInstallProgress(): { activeOperationId: string | null phaseLabel: string | null @@ -40,7 +35,12 @@ export function useSkillInstallProgress(): { } ) : progress - ? INSTALL_PHASE_LABELS[progress.phase] + ? progress.phase === 'authorizing' + ? translate('auto.components.skills.install.authorizing', 'Authorizing package access…') + : translate( + 'auto.components.skills.install.installing', + 'Downloading, verifying, and installing…' + ) : null, begin: (operationId) => { activeOperationIdRef.current = operationId diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index ac4dae956b3..c776a97452a 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -4377,7 +4377,8 @@ "3af85f6add": "Done", "readyDescriptionV2": "Anyone with this unlisted link can inspect and install the skills.", "descriptionV2": "Review the exact files, then publish an immutable version behind an unlisted link.", - "publishBundle": "Publish bundle" + "publishBundle": "Publish bundle", + "cancelRequestFailed": "Orca could not send the cancellation request. The upload may still finish." }, "SkillCard": { "d25a1b8ae6": "Share skill", @@ -4670,6 +4671,23 @@ "showMore": "Show more" }, "install": { + "authorizing": "Authorizing package access…", + "installing": "Downloading, verifying, and installing…", + "selectSkill": "Select at least one skill.", + "chooseWorkspace": "Choose a workspace.", + "reconnectBeforeInstalling": "Reconnect your Orca account before installing.", + "bundleVerificationFailed": "Installation failed before Orca could verify the requested bundle.", + "destinationAlreadyFinished": "The destination had already finished this installation.", + "enterShareLink": "Enter an Orca skill share link.", + "shareUnavailable": "This share is unavailable. The link may be invalid, expired, or revoked.", + "requestedVersionVerificationFailed": "Installation failed before Orca could verify the requested version.", + "versionVerificationFailed": "Orca could not verify the requested version.", + "inspectManagedFailed": "Orca could not inspect managed installs on this machine.", + "reconnectForVersionHistory": "Reconnect your Orca account to load version history.", + "versionHistoryUnavailable": "Version history is unavailable for this skill.", + "bundleSkillsMissing": "This version does not contain any of the installed bundle skills.", + "reconnectBeforeVersionChange": "Reconnect your Orca account before changing versions.", + "removeFailed": "Orca could not safely remove this skill.", "agentsLabel": "Agents", "agentsCanonical": "Always installed for {{value0}}, which read {{value1}}.", "agentsNoneChosen": "No extra agents", diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index 62165ef4f48..41b300a10e8 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -3872,6 +3872,93 @@ } }, "skills": { + "SkillsPage": { + "cb142070b4": "Actualizar", + "a68dee6a32": "Buscar skills", + "f43ad6edf3": "Skills", + "ea72d6185b": "No se pudieron analizar las skills", + "dc4c3328ee": "Mostrar archivo", + "9963dff6d3": "No se encontró ninguna descripción.", + "995fde8337": "No se pudo mostrar el archivo de la skill", + "4acd6d68ec": "No se encontraron skills", + "6a62a0168c": "Sin coincidencias", + "08a321a984": "Ninguna skill coincide con la búsqueda y los filtros actuales.", + "cd7893fbc1": "Analizando skills", + "35b9a724a0": "Disponible", + "c13b82793c": "Gestionar instalaciones", + "aee7b99cc6": "Instalar desde un enlace", + "filterProvider": "Filtrar por agente", + "filterSource": "Filtrar por origen", + "allSources": "Todo", + "clearFilters": "Borrar filtros", + "closeSkills": "Cerrar skills", + "closeTooltip": "Cerrar · Esc", + "moreActions": "Más acciones", + "sharedLinks": "Enlaces compartidos", + "emptyCopy": "Las carpetas de skills analizadas están vacías. Instala un paquete compartido o actualiza después de añadir una skill.", + "retry": "Reintentar", + "remoteShareNotice": "Estas skills están en {{host}}. Abre Skills en esa máquina para compartirlas.", + "viewSwitch": "Mostrar", + "searchLinks": "Buscar enlaces", + "deleteSkills": "Eliminar skills…" + }, + "SkillShareSelectionControls": { + "01c5a15e02": "Compartir skills" + }, + "SkillRow": { + "updatedUnknown": "Sin fecha", + "pathCopied": "Ruta copiada", + "copyPath": "Copiar ruta", + "detailPath": "Ruta", + "skillActions": "Acciones para {{value0}}", + "viewDetails": "Ver detalles", + "deleteSkill": "Eliminar…" + }, + "SkillsList": { "listLabel": "Skills" }, + "sourceStatus": { + "missing": "Carpeta no encontrada", + "remoteRepo": "Repositorio remoto — sin analizar", + "unavailable": "Sin analizar" + }, + "sources": { "heading": "Carpetas de skills" }, + "sourceKind": { + "home": "Inicio", + "workspace": "Espacio de trabajo", + "bundled": "Incluida", + "plugin": "Plugin" + }, + "count": { + "skillOne": "{{count}} skill", + "skillOther": "{{count}} skills", + "sourceOne": "{{count}} origen", + "sourceOther": "{{count}} orígenes", + "fileOne": "{{count}} archivo", + "fileOther": "{{count}} archivos", + "resultOne": "{{count}} resultado", + "resultOther": "{{count}} resultados", + "selected": "{{count}} seleccionadas", + "shareOne": "Compartir {{count}} skill", + "shareOther": "Compartir {{count}} skills", + "linkOne": "{{count}} enlace", + "linkOther": "{{count}} enlaces" + }, + "filter": { + "allAgents": "Todos los agentes", + "sharedAgent": "Compartido (.agents)" + }, + "SkillsSelectionHeader": { + "exit": "Salir de la selección", + "exitTooltip": "Salir de la selección · Esc", + "title": "Seleccionar skills para compartir", + "selectAll": "Seleccionar las {{count}} aptas", + "clear": "Borrar", + "deleteTitle": "Seleccionar skills para eliminar" + }, + "SkillDetailDialog": { + "agents": "Agentes", + "updated": "Actualizado", + "copy": "Copiar" + }, "SkillFreshnessNudge": { "titleOne": "Una skill de Orca instalada está desactualizada", "titleMany": "{{value0}} skills de Orca instaladas están desactualizadas", diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index 3f2f4d4491f..d9466efd0c1 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -3872,6 +3872,54 @@ } }, "skills": { + "SkillsPage": { + "cb142070b4": "更新", + "a68dee6a32": "スキルを検索", + "f43ad6edf3": "スキル", + "ea72d6185b": "スキルをスキャンできませんでした", + "dc4c3328ee": "ファイルを表示", + "9963dff6d3": "説明が見つかりません。", + "995fde8337": "スキルファイルを表示できませんでした", + "4acd6d68ec": "スキルが見つかりません", + "6a62a0168c": "一致する項目はありません", + "08a321a984": "現在の検索条件とフィルターに一致するスキルはありません。", + "cd7893fbc1": "スキルをスキャン中", + "35b9a724a0": "利用可能", + "c13b82793c": "インストールを管理", + "aee7b99cc6": "リンクからインストール", + "filterProvider": "エージェントで絞り込み", + "filterSource": "ソースで絞り込み", + "allSources": "すべて", + "clearFilters": "フィルターをクリア", + "closeSkills": "スキルを閉じる", + "closeTooltip": "閉じる · Esc", + "moreActions": "その他の操作", + "sharedLinks": "共有リンク", + "emptyCopy": "スキャンしたスキルフォルダーは空です。共有バンドルをインストールするか、スキルを追加後に更新してください。", + "retry": "再試行", + "remoteShareNotice": "これらのスキルは {{host}} にあります。共有するには、そのマシンでスキルを開いてください。", + "viewSwitch": "表示", + "searchLinks": "リンクを検索", + "deleteSkills": "スキルを削除…" + }, + "SkillShareSelectionControls": { "01c5a15e02": "スキルを共有" }, + "SkillRow": { + "updatedUnknown": "日付なし", + "pathCopied": "パスをコピーしました", + "copyPath": "パスをコピー", + "detailPath": "パス", + "skillActions": "{{value0}} の操作", + "viewDetails": "詳細を表示", + "deleteSkill": "削除…" + }, + "SkillsList": { "listLabel": "スキル" }, + "sourceStatus": { "missing": "フォルダーが見つかりません", "remoteRepo": "リモートリポジトリ — 未スキャン", "unavailable": "未スキャン" }, + "sources": { "heading": "スキルフォルダー" }, + "sourceKind": { "home": "ホーム", "workspace": "ワークスペース", "bundled": "バンドル済み", "plugin": "プラグイン" }, + "count": { "skillOne": "{{count}} 件のスキル", "skillOther": "{{count}} 件のスキル", "sourceOne": "{{count}} 件のソース", "sourceOther": "{{count}} 件のソース", "fileOne": "{{count}} 個のファイル", "fileOther": "{{count}} 個のファイル", "resultOne": "{{count}} 件の結果", "resultOther": "{{count}} 件の結果", "selected": "{{count}} 件を選択", "shareOne": "{{count}} 件のスキルを共有", "shareOther": "{{count}} 件のスキルを共有", "linkOne": "{{count}} 件のリンク", "linkOther": "{{count}} 件のリンク" }, + "filter": { "allAgents": "すべてのエージェント", "sharedAgent": "共有 (.agents)" }, + "SkillsSelectionHeader": { "exit": "選択を終了", "exitTooltip": "選択を終了 · Esc", "title": "共有するスキルを選択", "selectAll": "対象の {{count}} 件をすべて選択", "clear": "クリア", "deleteTitle": "削除するスキルを選択" }, + "SkillDetailDialog": { "agents": "エージェント", "updated": "更新日", "copy": "コピー" }, "SkillFreshnessNudge": { "titleOne": "インストール済みの Orca スキルが古くなっています", "titleMany": "インストール済みの Orca スキル {{value0}} 件が古くなっています", diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index bf5238cbc34..f154990358d 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -3877,6 +3877,19 @@ } }, "skills": { + "SkillsPage": { + "cb142070b4": "새로 고침", "a68dee6a32": "스킬 검색", "f43ad6edf3": "스킬", "ea72d6185b": "스킬을 검색하지 못했습니다", "dc4c3328ee": "파일 표시", "9963dff6d3": "설명을 찾을 수 없습니다.", "995fde8337": "스킬 파일을 표시하지 못했습니다", "4acd6d68ec": "스킬을 찾을 수 없습니다", "6a62a0168c": "일치하는 항목 없음", "08a321a984": "현재 검색 및 필터와 일치하는 스킬이 없습니다.", "cd7893fbc1": "스킬 검색 중", "35b9a724a0": "사용 가능", "c13b82793c": "설치 관리", "aee7b99cc6": "링크에서 설치", "filterProvider": "에이전트별 필터", "filterSource": "소스별 필터", "allSources": "전체", "clearFilters": "필터 지우기", "closeSkills": "스킬 닫기", "closeTooltip": "닫기 · Esc", "moreActions": "추가 작업", "sharedLinks": "공유 링크", "emptyCopy": "검색한 스킬 폴더가 비어 있습니다. 공유 번들을 설치하거나 스킬을 추가한 후 새로 고치세요.", "retry": "다시 시도", "remoteShareNotice": "이 스킬은 {{host}}에 있습니다. 공유하려면 해당 머신에서 스킬을 여세요.", "viewSwitch": "표시", "searchLinks": "링크 검색", "deleteSkills": "스킬 삭제…" + }, + "SkillShareSelectionControls": { "01c5a15e02": "스킬 공유" }, + "SkillRow": { "updatedUnknown": "날짜 없음", "pathCopied": "경로 복사됨", "copyPath": "경로 복사", "detailPath": "경로", "skillActions": "{{value0}} 작업", "viewDetails": "세부 정보 보기", "deleteSkill": "삭제…" }, + "SkillsList": { "listLabel": "스킬" }, + "sourceStatus": { "missing": "폴더를 찾을 수 없음", "remoteRepo": "원격 리포지토리 — 검색 안 됨", "unavailable": "검색 안 됨" }, + "sources": { "heading": "스킬 폴더" }, + "sourceKind": { "home": "홈", "workspace": "워크스페이스", "bundled": "번들", "plugin": "플러그인" }, + "count": { "skillOne": "스킬 {{count}}개", "skillOther": "스킬 {{count}}개", "sourceOne": "소스 {{count}}개", "sourceOther": "소스 {{count}}개", "fileOne": "파일 {{count}}개", "fileOther": "파일 {{count}}개", "resultOne": "결과 {{count}}개", "resultOther": "결과 {{count}}개", "selected": "{{count}}개 선택됨", "shareOne": "스킬 {{count}}개 공유", "shareOther": "스킬 {{count}}개 공유", "linkOne": "링크 {{count}}개", "linkOther": "링크 {{count}}개" }, + "filter": { "allAgents": "모든 에이전트", "sharedAgent": "공유됨 (.agents)" }, + "SkillsSelectionHeader": { "exit": "선택 나가기", "exitTooltip": "선택 나가기 · Esc", "title": "공유할 스킬 선택", "selectAll": "가능한 {{count}}개 모두 선택", "clear": "지우기", "deleteTitle": "삭제할 스킬 선택" }, + "SkillDetailDialog": { "agents": "에이전트", "updated": "업데이트됨", "copy": "복사" }, "SkillFreshnessNudge": { "titleOne": "설치된 Orca 스킬이 오래되었습니다", "titleMany": "설치된 Orca 스킬 {{value0}}개가 오래되었습니다", diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index 1f41bf675dc..f68271bf57f 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -3887,6 +3887,19 @@ } }, "skills": { + "SkillsPage": { + "cb142070b4": "刷新", "a68dee6a32": "搜索技能", "f43ad6edf3": "技能", "ea72d6185b": "无法扫描技能", "dc4c3328ee": "显示文件", "9963dff6d3": "未找到说明。", "995fde8337": "无法显示技能文件", "4acd6d68ec": "未找到技能", "6a62a0168c": "无匹配项", "08a321a984": "没有技能匹配当前搜索和筛选条件。", "cd7893fbc1": "正在扫描技能", "35b9a724a0": "可用", "c13b82793c": "管理安装", "aee7b99cc6": "从链接安装", "filterProvider": "按 Agent 筛选", "filterSource": "按来源筛选", "allSources": "全部", "clearFilters": "清除筛选", "closeSkills": "关闭技能", "closeTooltip": "关闭 · Esc", "moreActions": "更多操作", "sharedLinks": "共享链接", "emptyCopy": "扫描的技能文件夹为空。请安装共享包,或添加技能后刷新。", "retry": "重试", "remoteShareNotice": "这些技能位于 {{host}}。请在该机器上打开“技能”以进行共享。", "viewSwitch": "显示", "searchLinks": "搜索链接", "deleteSkills": "删除技能…" + }, + "SkillShareSelectionControls": { "01c5a15e02": "共享技能" }, + "SkillRow": { "updatedUnknown": "无日期", "pathCopied": "路径已复制", "copyPath": "复制路径", "detailPath": "路径", "skillActions": "{{value0}} 的操作", "viewDetails": "查看详情", "deleteSkill": "删除…" }, + "SkillsList": { "listLabel": "技能" }, + "sourceStatus": { "missing": "未找到文件夹", "remoteRepo": "远程仓库 — 未扫描", "unavailable": "未扫描" }, + "sources": { "heading": "技能文件夹" }, + "sourceKind": { "home": "主目录", "workspace": "工作区", "bundled": "内置", "plugin": "插件" }, + "count": { "skillOne": "{{count}} 个技能", "skillOther": "{{count}} 个技能", "sourceOne": "{{count}} 个来源", "sourceOther": "{{count}} 个来源", "fileOne": "{{count}} 个文件", "fileOther": "{{count}} 个文件", "resultOne": "{{count}} 个结果", "resultOther": "{{count}} 个结果", "selected": "已选择 {{count}} 个", "shareOne": "共享 {{count}} 个技能", "shareOther": "共享 {{count}} 个技能", "linkOne": "{{count}} 个链接", "linkOther": "{{count}} 个链接", "deleteOne": "删除 {{count}} 个技能", "deleteOther": "删除 {{count}} 个技能", "deletedOne": "已删除 {{count}} 个技能", "deletedOther": "已删除 {{count}} 个技能", "deleteFolderOne": "{{count}} 个文件夹", "deleteFolderOther": "{{count}} 个文件夹", "deleteLinkOne": "{{count}} 个链接", "deleteLinkOther": "{{count}} 个链接" }, + "filter": { "allAgents": "所有 Agent", "sharedAgent": "共享 (.agents)" }, + "SkillsSelectionHeader": { "exit": "退出选择", "exitTooltip": "退出选择 · Esc", "title": "选择要共享的技能", "selectAll": "选择全部 {{count}} 个符合条件的项目", "clear": "清除", "deleteTitle": "选择要删除的技能" }, + "SkillDetailDialog": { "agents": "Agent", "updated": "已更新", "copy": "复制" }, "SkillFreshnessNudge": { "titleOne": "已安装的 Orca 技能已过期", "titleMany": "{{value0}} 个已安装的 Orca 技能已过期", @@ -3984,16 +3997,6 @@ "oneLocation": "1 location", "manyLocations": "{{value0}} locations" }, - "count": { - "deleteOne": "删除 {{count}} 个技能", - "deleteOther": "删除 {{count}} 个技能", - "deletedOne": "已删除 {{count}} 个技能", - "deletedOther": "已删除 {{count}} 个技能", - "deleteFolderOne": "{{count}} 个文件夹", - "deleteFolderOther": "{{count}} 个文件夹", - "deleteLinkOne": "{{count}} 个链接", - "deleteLinkOther": "{{count}} 个链接" - }, "host": { "local": "此机器", "remote": "已连接的运行时" From 0e10fc59258aa3e07e31f33d1dc0024f42389647 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Wed, 26 Aug 2026 15:09:22 -0700 Subject: [PATCH 06/19] fix(browser): retire helpers with page owners (#16564) --- config/reliability-gates.jsonc | 138 ++++++++++++- .../serve-signal-exit-diagnostic.test.ts | 102 +++++++++- src/cli/runtime/serve-update-supervisor.ts | 24 ++- ...t-browser-bridge-session-lifecycle.test.ts | 175 +++++++++++++++- src/main/browser/agent-browser-bridge.ts | 62 ++++-- src/main/browser/browser-backend.ts | 2 +- ...ffscreen-browser-backend-lifecycle.test.ts | 189 ++++++++++++++++++ src/main/browser/offscreen-browser-backend.ts | 57 +++++- src/main/index.ts | 29 ++- src/main/quit-teardown-deadline.ts | 6 +- .../startup/desktop-startup-ordering.test.ts | 34 ++++ .../startup/serve-signal-handlers.test.ts | 19 ++ src/main/startup/serve-signal-handlers.ts | 12 ++ .../window/main-window-close-lifecycle.ts | 3 +- src/shared/quit-teardown-deadline.ts | 3 + 15 files changed, 793 insertions(+), 62 deletions(-) create mode 100644 src/main/startup/serve-signal-handlers.test.ts create mode 100644 src/main/startup/serve-signal-handlers.ts create mode 100644 src/shared/quit-teardown-deadline.ts diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc index 482f9808656..0826c5dea46 100644 --- a/config/reliability-gates.jsonc +++ b/config/reliability-gates.jsonc @@ -2094,15 +2094,16 @@ "providers": ["local", "daemon", "ssh"], "coveredPlatforms": ["macos"], "coveredProviders": ["local", "daemon", "ssh"], - "coverageNotes": "Deterministic unit coverage exercises activation gating, single-instance ownership, quit policy, local/remote CLI status, headless binding persistence, local daemon identity, SSH identity transfer, the promoted renderer's agent-resume accounting, and the macOS serve update handoff from staged installer through atomic bundle replacement and target-version readiness. A macOS Electron journey covers headless promotion and persistent PTY identity. A disposable locally signed Electron canary exercised real ShipIt and a temporary LaunchAgent with the compiled production supervisor; full packaged Orca and Linux/Windows serve updates remain uncollected.", + "coverageNotes": "Deterministic unit coverage exercises activation gating, single-instance ownership, quit policy, local/remote CLI status, headless binding persistence, local daemon identity, SSH identity transfer, the promoted renderer's agent-resume accounting, and the macOS serve update handoff from staged installer through atomic bundle replacement and target-version readiness. Supervisor settlement coverage prevents a late handoff-completion failure from rearming termination after the replacement child exits. A macOS Electron journey covers headless promotion and persistent PTY identity. A disposable locally signed Electron canary exercised real ShipIt and a temporary LaunchAgent with the compiled production supervisor; full packaged Orca and Linux/Windows serve updates remain uncollected.", "motivatingLinks": [ "https://github.com/stablyai/orca/issues/8457", "https://github.com/stablyai/orca/issues/9563" ], - "invariant": "A safely promotable headless serve process is the single app owner. Desktop activation preserves its daemon-backed sessions. On macOS, a CLI-supervised serve update keeps the node-mode parent alive across ShipIt's atomic bundle swap, restarts with the original serve arguments only after the target bundle is present, and clears handoff state only after that target version reports runtime readiness. Unsupported or failed handoffs leave the current serving owner intact or recover it once without an install retry loop.", - "oracle": "Unit tests coalesce early activation, preserve daemon and SSH identity, and reproduce the update race with a staged target, old serving child, persistent CLI parent, atomic .app replacement, and replacement readiness message. They assert the parent does not exit for launchd to respawn the old app, the native updater does not launch an interactive GUI, the replacement version is verified before handoff completion, mismatches become durable failures without retries, and unsupported/preflight-failed installs do not invoke native quit or PTY cleanup. A joined lock-owner/activation/hydration contract asserts that a forced relaunch opens exactly one window and that the renderer promoted inside the serve process launches zero agent resumes, creates no replacement tab or startup command, and leaves every surviving session record untouched. The Electron journey independently verifies headless promotion retains owner/runtime/daemon/PTY identity and terminal I/O.", + "invariant": "A safely promotable headless serve process is the single app owner. Desktop activation preserves its daemon-backed sessions. On macOS, a CLI-supervised serve update keeps the node-mode parent alive across ShipIt's atomic bundle swap, restarts with the original serve arguments only after the target bundle is present, and clears handoff state only after that target version reports runtime readiness. Once a supervised child exits or fails to start, no late handoff completion may signal it or arm force-kill escalation. Unsupported or failed handoffs leave the current serving owner intact or recover it once without an install retry loop.", + "oracle": "Unit tests coalesce early activation, preserve daemon and SSH identity, and reproduce the update race with a staged target, old serving child, persistent CLI parent, atomic .app replacement, and replacement readiness message. They assert the parent does not exit for launchd to respawn the old app, the native updater does not launch an interactive GUI, the replacement version is verified before handoff completion, mismatches become durable failures without retries, and unsupported/preflight-failed installs do not invoke native quit or PTY cleanup. Force handoff completion to fail after the replacement child exits and require zero later child signals and zero escalation timers. A joined lock-owner/activation/hydration contract asserts that a forced relaunch opens exactly one window and that the renderer promoted inside the serve process launches zero agent resumes, creates no replacement tab or startup command, and leaves every surviving session record untouched. The Electron journey independently verifies headless promotion retains owner/runtime/daemon/PTY identity and terminal I/O.", "commands": [ "pnpm exec vitest run --config config/vitest.config.ts src/cli/runtime/launch.test.ts src/main/serve-update-handoff.test.ts src/main/updater.headless-serve-install.test.ts src/main/updater.test.ts src/main/updater.mac-install.test.ts src/main/window/attach-main-window-services.test.ts src/main/startup/serve-desktop-activation-wiring.test.ts", + "pnpm exec vitest run --config config/vitest.config.ts src/cli/runtime/serve-signal-exit-diagnostic.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/main/startup/serve-desktop-activation.test.ts src/main/startup/serve-desktop-activation-wiring.test.ts src/main/startup/single-instance-lock.test.ts src/main/startup/window-all-closed-quit-policy.test.ts src/cli/runtime-client.test.ts src/cli/runtime/websocket-transport.test.ts src/main/runtime/orca-runtime.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/lib/serve-desktop-promotion-session-continuity.test.ts", "pnpm exec electron-vite build --mode e2e", @@ -2112,6 +2113,7 @@ "src/main/updater.headless-serve-install.test.ts", "src/main/serve-update-handoff.test.ts", "src/cli/runtime/launch.test.ts", + "src/cli/runtime/serve-signal-exit-diagnostic.test.ts", "src/main/startup/serve-desktop-activation.test.ts", "src/main/startup/serve-desktop-activation-wiring.test.ts", "src/main/startup/single-instance-lock.test.ts", @@ -2144,6 +2146,12 @@ "a replacement version mismatch or readiness timeout is persisted and exits without an in-process retry loop" ] }, + { + "file": "src/cli/runtime/serve-signal-exit-diagnostic.test.ts", + "assertions": [ + "late update handoff failure cannot rearm termination after child exit" + ] + }, { "file": "src/main/serve-update-handoff.test.ts", "assertions": [ @@ -16222,6 +16230,130 @@ ], "demotionRule": "Demote if a live parked PTY loses its exact graph leaf, a retired PTY remains published, multi-pane identity drifts, or the routed client round trip flakes without a diagnosed cause." }, + { + "id": "agent-browser.owner-boundary-cleanup", + "title": "Headless browser helpers retire with their owning page and runtime", + "maturity": "experimental", + "protection": "partial", + "owner": "browser-runtime", + "layer": "electron-main-headless-browser-lifecycle", + "surfaces": [ + "headless offscreen browser page close", + "offscreen renderer destruction", + "app quit", + "headless serve SIGINT/SIGTERM" + ], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "remote-runtime", "ssh", "wsl"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["local"], + "coverageNotes": "The owner-boundary contract injects an isolated BrowserManager, offscreen WebContents, and AgentBrowserBridge. It proves explicit close removes the page from command routing before awaiting exact named-session retirement, runs retirement on unexpected renderer destruction, preserves unrelated pages, joins pending creation/retirement during shutdown, rejects command admission after terminal cleanup begins, retries Electron quit after a vetoed signal, preserves Windows shared-console graceful Ctrl-C, reserves the full renderer-acknowledgement and committed-teardown budgets before supervisor force-kill, bounds process-swap retirement to five seconds, and caps cleanup concurrency at four. A built-CLI macOS headless serve run used an isolated profile, folder workspace, socket directory, and port: exact PPID-1 helper/session identity disappeared within five seconds after page close and after Ctrl-C runtime shutdown, while an unrelated page survived close and reconnect. Linux cgroup and paired/SSH journeys remain gaps.", + "motivatingLinks": [ + "https://linear.app/stably/issue/STA-5400", + "https://github.com/stablyai/orca/issues/16367" + ], + "invariant": "For each Orca-owned page identity, after its owner closes or is destroyed and a five-second third-party shutdown grace expires, no live helper session named orca-tab- may remain; unrelated live page identities must survive, and runtime quit must join the bounded cleanup attempt. The serve supervisor may force-kill only after the renderer-acknowledgement deadline, committed teardown deadline, and bounded scheduling margin have all elapsed.", + "oracle": "Create two isolated offscreen pages and register their WebContents. Close one, block onPageClosed for its stable page ID, and require unregisterGuest to have already removed that page from command routing while the unrelated page remains live. Emit destroyed and require the same exact retirement. Race shutdown with creation, pending retirement, and process-swap destruction; require no replacement session or command after terminal cleanup starts, require shutdown to remain pending until retirement settles, require the swap close command to use the five-second cleanup timeout, and require at most four concurrent close commands. Deliver repeated signals and require every attempt to reach Electron quit; under Windows shared-console semantics require the child to handle Ctrl-C without an immediate child.kill. Advance the supervisor clock through the renderer-acknowledgement and committed teardown deadlines and require no SIGKILL until the bounded scheduling margin elapses. In direct built-CLI serve, inspect exact session/PID/socket identity plus RSS/fd inventory before close and after a five-second grace; reconnect between commands and stop via Ctrl-C. No process-name kill or global sweep is permitted.", + "commands": [ + "pnpm exec vitest run --config config/vitest.config.ts src/cli/runtime/serve-signal-exit-diagnostic.test.ts src/main/browser/offscreen-browser-backend-lifecycle.test.ts src/main/browser/agent-browser-bridge-session-lifecycle.test.ts src/main/browser/agent-browser-bridge-tab-routing.test.ts src/main/startup/serve-signal-handlers.test.ts src/main/startup/desktop-startup-ordering.test.ts", + "pnpm exec vitest run --config config/vitest.config.ts src/main/browser" + ], + "testFiles": [ + "src/main/browser/offscreen-browser-backend-lifecycle.test.ts", + "src/main/browser/agent-browser-bridge-session-lifecycle.test.ts", + "src/main/browser/agent-browser-bridge-tab-routing.test.ts", + "src/main/startup/serve-signal-handlers.test.ts", + "src/main/startup/desktop-startup-ordering.test.ts", + "src/cli/runtime/serve-signal-exit-diagnostic.test.ts" + ], + "assertionRefs": [ + { + "file": "src/main/browser/offscreen-browser-backend-lifecycle.test.ts", + "assertions": [ + "explicit close unregisters the page before awaiting exact owner cleanup while unrelated and same-ID replacement pages survive", + "unexpected offscreen renderer destruction invokes and joins exact helper-owner cleanup", + "backend shutdown joins every pending owner cleanup with concurrency capped at four" + ] + }, + { + "file": "src/main/browser/agent-browser-bridge-session-lifecycle.test.ts", + "assertions": [ + "owner cleanup uses a five-second close-command timeout", + "process-swap retirement uses the same five-second cleanup timeout", + "runtime shutdown includes sessions whose creation is still pending", + "runtime shutdown rejects late command and replacement-session admission", + "runtime-wide helper cleanup concurrency is capped at four" + ] + }, + { + "file": "src/main/browser/agent-browser-bridge-tab-routing.test.ts", + "assertions": [ + "closing a tab retires the exact named agent-browser session" + ] + }, + { + "file": "src/main/startup/serve-signal-handlers.test.ts", + "assertions": [ + "every repeated serve signal retries Electron quit after a possible renderer veto", + "SIGINT and SIGTERM listeners remain installed during quit draining" + ] + }, + { + "file": "src/main/startup/desktop-startup-ordering.test.ts", + "assertions": [ + "agent-browser cleanup is joined by the committed quit teardown barrier", + "repeatable serve signal handling is registered before readiness" + ] + }, + { + "file": "src/cli/runtime/serve-signal-exit-diagnostic.test.ts", + "assertions": [ + "serve supervisor force-kill grace covers renderer acknowledgement, committed teardown, and scheduling margin", + "Windows shared-console Ctrl-C is not forwarded as an immediate child termination" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-08-26", + "runner": "local", + "platform": "macos", + "result": "passed", + "command": "pnpm exec vitest run --config config/vitest.config.ts src/cli/runtime/serve-signal-exit-diagnostic.test.ts src/main/browser/offscreen-browser-backend-lifecycle.test.ts src/main/browser/agent-browser-bridge-session-lifecycle.test.ts src/main/browser/agent-browser-bridge-tab-routing.test.ts src/main/startup/serve-signal-handlers.test.ts src/main/startup/desktop-startup-ordering.test.ts", + "durationSeconds": 0.44, + "summary": "Candidate passed 63 lifecycle, bridge, routing, signal, and quit-order tests plus 1,469 browser tests. The production-reverted control was red; disabling the final swap-timeout/admission controls failed 3 of 15 bridge lifecycle tests; removing the pending-retirement join, page-routing fence, repeat-signal behavior, or Windows shared-console guard failed its exact focused oracle. Setting the supervisor scheduling margin to zero reproduced SIGKILL at the combined 30-second renderer-acknowledgement and teardown boundary. Direct built-CLI serve observed exact PPID-1 helpers at 10-11 MiB RSS and 16-18 fd rows. Closing one page removed only its helper/session within five seconds while the other page survived reconnect. With the signal fix disabled, Ctrl-C left exact helpers alive after the listener exited; the final candidate emptied session inventory and removed app/helper PIDs and port within five seconds." + } + ], + "runtimeBudget": { + "p95Seconds": 2, + "scope": "bounded offscreen lifecycle contracts and bridge cleanup ordering" + }, + "flakeHistory": { + "status": "not-started", + "evidence": "Fresh deterministic coverage has local candidate evidence only." + }, + "redGreenEvidence": { + "status": "complete", + "evidence": "The pristine current-main control failed explicit close, unexpected destruction, backend shutdown, pending creation, bounded concurrency, joined quit, and repeated-signal assertions; the candidate passed the byte-identical oracle. Disabling only process-swap cleanup timeout and terminal command admission failed 3 of 15 bridge lifecycle tests. Live candidate-with-signal-fix-disabled also retained exact helpers after Ctrl-C, while the final candidate removed them." + }, + "performanceBudget": { + "required": true, + "evidence": "Cleanup is event-driven and exact-page scoped: no polling, process-name scan, idle timer, or new wire field. Owner and residual session drains cap close subprocess concurrency at four, sequence headless ownership cleanup before the residual bridge sweep, and reuse the existing bounded app teardown deadline." + }, + "promotionCriteria": [ + "Run the same oracle on Linux headless serve with bundled agent-browser PID/session and RSS/fd inventory.", + "Exercise runtime restart, transport loss, reconnect, concurrent tabs, and paired/SSH host ownership without cross-page cleanup.", + "Collect repeated open/close/reconnect evidence showing bounded helper count and no stale PID-1-owned Orca identities." + ], + "knownGaps": [ + "No 64 GiB Linux cgroup stress was manufactured; the reported production incident remains un-soaked.", + "The real Electron headless serve oracle ran on macOS; physical Linux/SSH/WSL and paired-client journeys remain unvalidated.", + "A close-command timeout or error is treated as best-effort and does not escalate to a process kill; helper disappearance on that failure path remains unverifiable without exact PID ownership.", + "Four-way close batches remain bounded but can exceed the 20-second app quit barrier above 16 worst-case five-second retirements.", + "The patch does not add startup stale-process recovery or an idle timeout; those require an exact persisted ownership ledger and are intentionally out of scope." + ], + "demotionRule": "Demote if a closed or destroyed page leaves its exact named helper session, if quit can exit before helper retirement settles, if unrelated live pages are retired, or if cleanup regresses to a global process-name kill or timer-only sweep." + }, { "id": "terminal-session.remote-pane-layout-retry", "title": "Remote pane layouts retry identical state after reconnect", diff --git a/src/cli/runtime/serve-signal-exit-diagnostic.test.ts b/src/cli/runtime/serve-signal-exit-diagnostic.test.ts index 2379bb58854..f5d348798c3 100644 --- a/src/cli/runtime/serve-signal-exit-diagnostic.test.ts +++ b/src/cli/runtime/serve-signal-exit-diagnostic.test.ts @@ -1,8 +1,19 @@ import { EventEmitter } from 'node:events' +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' import { afterEach, describe, expect, it, vi } from 'vitest' import { serveSignalExitError } from './serve-signal-exit-diagnostic' -import { superviseForegroundServe } from './serve-update-supervisor' +import { + SERVE_CHILD_FORCE_KILL_GRACE_MS, + SERVE_CHILD_FORCE_KILL_SCHEDULING_MARGIN_MS, + superviseForegroundServe +} from './serve-update-supervisor' import { RuntimeClientError } from './types' +import { + QUIT_RENDERER_ACK_TIMEOUT_MS, + WILL_QUIT_TEARDOWN_DEADLINE_MS +} from '../../shared/quit-teardown-deadline' class FakeChildProcess extends EventEmitter { kill = vi.fn() @@ -17,7 +28,13 @@ function setPlatform(platform: NodeJS.Platform): void { function superviseUntilExit(code: number | null, signal: NodeJS.Signals | null): Promise { const child = new FakeChildProcess() - const supervised = superviseForegroundServe({ + const supervised = superviseChild(child) + child.emit('exit', code, signal) + return supervised +} + +function superviseChild(child: FakeChildProcess): Promise { + return superviseForegroundServe({ executable: '/Applications/Orca.app/Contents/MacOS/Orca', childArgs: ['--serve'], spawnOptions: {}, @@ -26,12 +43,12 @@ function superviseUntilExit(code: number | null, signal: NodeJS.Signals | null): child: child as never, expectedHandoff: null }) - child.emit('exit', code, signal) - return supervised } afterEach(() => { Object.defineProperty(process, 'platform', originalPlatform) + vi.restoreAllMocks() + vi.useRealTimers() }) describe('serveSignalExitError', () => { @@ -74,6 +91,83 @@ describe('serveSignalExitError', () => { }) describe('superviseForegroundServe signal exits', () => { + it('lets pre-commit and committed Electron quit deadlines finish before force-killing serve', async () => { + setPlatform('linux') + vi.useFakeTimers() + const child = new FakeChildProcess() + const supervised = superviseChild(child) + + expect(SERVE_CHILD_FORCE_KILL_GRACE_MS).toBe( + QUIT_RENDERER_ACK_TIMEOUT_MS + + WILL_QUIT_TEARDOWN_DEADLINE_MS + + SERVE_CHILD_FORCE_KILL_SCHEDULING_MARGIN_MS + ) + expect(SERVE_CHILD_FORCE_KILL_GRACE_MS).toBeLessThanOrEqual(35_000) + + process.emit('SIGTERM', 'SIGTERM') + expect(child.kill).toHaveBeenCalledOnce() + expect(child.kill).toHaveBeenLastCalledWith('SIGTERM') + + await vi.advanceTimersByTimeAsync(QUIT_RENDERER_ACK_TIMEOUT_MS + WILL_QUIT_TEARDOWN_DEADLINE_MS) + expect(child.kill).toHaveBeenCalledOnce() + + await vi.advanceTimersByTimeAsync(SERVE_CHILD_FORCE_KILL_SCHEDULING_MARGIN_MS) + expect(child.kill).toHaveBeenLastCalledWith('SIGKILL') + expect(child.kill).toHaveBeenCalledTimes(2) + + child.emit('exit', null, 'SIGKILL') + await expect(supervised).rejects.toThrow('Orca serve exited via SIGKILL.') + }) + + it('lets a shared-console Windows child handle Ctrl-C gracefully', async () => { + setPlatform('win32') + vi.useFakeTimers() + const child = new FakeChildProcess() + const supervised = superviseChild(child) + + process.emit('SIGINT', 'SIGINT') + expect(child.kill).not.toHaveBeenCalled() + expect(vi.getTimerCount()).toBe(1) + + child.emit('exit', 0, null) + await expect(supervised).resolves.toBe(0) + expect(vi.getTimerCount()).toBe(0) + }) + + it('does not terminate an exited child when update handoff completion fails late', async () => { + vi.useFakeTimers() + const missingParent = await mkdtemp(join(tmpdir(), 'orca-serve-missing-handoff-')) + await rm(missingParent, { recursive: true }) + const child = new FakeChildProcess() + const supervised = superviseForegroundServe({ + executable: '/Applications/Orca.app/Contents/MacOS/Orca', + childArgs: ['--serve'], + spawnOptions: {}, + spawnChild: vi.fn() as never, + handoffPath: join(missingParent, 'handoff.json'), + child: child as never, + expectedHandoff: { + schemaVersion: 1, + phase: 'install-requested', + fromVersion: '1.0.51', + targetVersion: '1.0.61', + servingPid: child.pid + } + }) + vi.spyOn(process.stderr, 'write').mockImplementation(() => true) + + child.emit('message', { + type: 'orca:serve-ready', + version: '1.0.61', + runtimeId: 'runtime-new' + }) + child.emit('exit', 0, null) + + await expect(supervised).resolves.toBe(1) + expect(child.kill).not.toHaveBeenCalled() + expect(vi.getTimerCount()).toBe(0) + }) + it('throws the macOS diagnostic when the child aborts on darwin', async () => { setPlatform('darwin') diff --git a/src/cli/runtime/serve-update-supervisor.ts b/src/cli/runtime/serve-update-supervisor.ts index d179a4354ff..a5a303dc1a2 100644 --- a/src/cli/runtime/serve-update-supervisor.ts +++ b/src/cli/runtime/serve-update-supervisor.ts @@ -6,10 +6,19 @@ import { parseServeUpdateHandoffState, type ServeUpdateHandoffState } from '../../shared/serve-update-handoff' +import { + QUIT_RENDERER_ACK_TIMEOUT_MS, + WILL_QUIT_TEARDOWN_DEADLINE_MS +} from '../../shared/quit-teardown-deadline' import { serveSignalExitError } from './serve-signal-exit-diagnostic' import { waitForMacBundleVersion } from './mac-app-update-bundle' export const SERVE_REPLACEMENT_READY_TIMEOUT_MS = 60_000 +export const SERVE_CHILD_FORCE_KILL_SCHEDULING_MARGIN_MS = 5_000 +export const SERVE_CHILD_FORCE_KILL_GRACE_MS = + QUIT_RENDERER_ACK_TIMEOUT_MS + + WILL_QUIT_TEARDOWN_DEADLINE_MS + + SERVE_CHILD_FORCE_KILL_SCHEDULING_MARGIN_MS type InstallRequestedHandoff = Extract type ServeReadiness = 'not-expected' | 'pending' | 'verified' | 'failed' @@ -111,9 +120,13 @@ function waitForForegroundChild( let readyTimer: ReturnType | null = null let readiness: ServeReadiness = expected ? 'pending' : 'not-expected' let stateWrite = Promise.resolve() + let childSettled = false const terminateChild = (): void => { + if (childSettled) { + return + } child.kill('SIGTERM') - forceKillTimer ??= setTimeout(() => child.kill('SIGKILL'), 5000) + forceKillTimer ??= setTimeout(() => child.kill('SIGKILL'), SERVE_CHILD_FORCE_KILL_GRACE_MS) } const recordReplacementFailure = (reason: string): boolean => { if (!expected || readiness !== 'pending') { @@ -140,8 +153,11 @@ function waitForForegroundChild( terminateChild() } const forwardSignal = (signal: NodeJS.Signals): void => { - child.kill(signal) - forceKillTimer ??= setTimeout(() => child.kill('SIGKILL'), 5000) + // A Windows console delivers Ctrl-C to parent and child; child.kill would terminate the child mid-teardown. + if (process.platform !== 'win32') { + child.kill(signal) + } + forceKillTimer ??= setTimeout(() => child.kill('SIGKILL'), SERVE_CHILD_FORCE_KILL_GRACE_MS) } const handleMessage = (value: unknown): void => { const message = parseServeSupervisorMessage(value) @@ -195,10 +211,12 @@ function waitForForegroundChild( }, SERVE_REPLACEMENT_READY_TIMEOUT_MS) } const handleExit = (code: number | null, signal: NodeJS.Signals | null): void => { + childSettled = true cleanup() void stateWrite.then(() => resolveWait({ code, signal, readiness })) } child.once('error', (error) => { + childSettled = true recordReplacementFailure(`Could not start the replacement process: ${String(error)}`) cleanup() child.off('exit', handleExit) diff --git a/src/main/browser/agent-browser-bridge-session-lifecycle.test.ts b/src/main/browser/agent-browser-bridge-session-lifecycle.test.ts index e24977be614..4ef3d24fa25 100644 --- a/src/main/browser/agent-browser-bridge-session-lifecycle.test.ts +++ b/src/main/browser/agent-browser-bridge-session-lifecycle.test.ts @@ -140,6 +140,43 @@ describe('AgentBrowserBridge', () => { expect(lastArgs).toContain('--cdp') }) + it('bounds owner cleanup independently from command execution timeouts', async () => { + succeedWith({ snapshot: 'initial' }) + await bridge.snapshot() + execFileMock.mockClear() + + succeedWith(null) + await bridge.onPageClosed('tab-1') + + const closeCall = execFileMock.mock.calls.find((call: unknown[]) => + (call[1] as string[]).includes('close') + ) + expect(closeCall?.[2]).toMatchObject({ timeout: 5_000 }) + }) + + it('uses the cleanup timeout when a target swap retires its session', async () => { + succeedWith({ snapshot: 'initial' }) + await bridge.snapshot() + execFileMock.mockClear() + + succeedWith(null) + await ( + bridge as unknown as { + restartSessionForTarget: ( + sessionName: string, + browserPageId: string, + webContentsId: number, + options: { recreate: boolean } + ) => Promise + } + ).restartSessionForTarget('orca-tab-tab-1', 'tab-1', 100, { recreate: false }) + + const closeCall = execFileMock.mock.calls.find((call: unknown[]) => + (call[1] as string[]).includes('close') + ) + expect(closeCall?.[2]).toMatchObject({ timeout: 5_000 }) + }) + it('waits for pending session destruction before recreating the same session', async () => { succeedWith({ snapshot: 'initial' }) await bridge.snapshot() @@ -418,22 +455,140 @@ describe('AgentBrowserBridge', () => { // ── destroyAllSessions ── - it('destroys all active sessions', async () => { + it('makes runtime-wide session destruction terminal', async () => { succeedWith({ snapshot: 'tree' }) await bridge.snapshot() - // Should have one session now - succeedWith(null) // for the 'close' call + succeedWith(null) await bridge.destroyAllSessions() + execFileMock.mockClear() - // Next command should re-create session with --cdp - succeedWith({ snapshot: 'fresh' }) - await bridge.snapshot() + await expect(bridge.snapshot()).rejects.toMatchObject({ + code: 'browser_owner_unavailable', + message: 'Browser runtime is shutting down' + }) + expect(execFileMock).not.toHaveBeenCalled() + }) - const snapshotCalls = execFileMock.mock.calls.filter((c: unknown[]) => - (c[1] as string[]).includes('snapshot') + it('bounds concurrent helper retirements during runtime shutdown', async () => { + const sessions = (bridge as unknown as { sessions: Map }).sessions + for (let index = 0; index < 6; index++) { + sessions.set(`orca-tab-tab-${index}`, { + proxy: { stop: vi.fn(async () => {}) }, + cdpEndpoint: `ws://127.0.0.1:${9200 + index}`, + initialized: true, + consecutiveTimeouts: 0, + activeInterceptPatterns: [], + activeCapture: false, + webContentsId: 100 + index, + activeProcess: null + }) + } + + let activeRetirements = 0 + let peakRetirements = 0 + const releases: (() => void)[] = [] + execFileMock.mockImplementation( + (_bin: string, _args: string[], _opts: unknown, cb: ExecFileCallback) => { + activeRetirements++ + peakRetirements = Math.max(peakRetirements, activeRetirements) + releases.push(() => { + activeRetirements-- + cb(null, JSON.stringify({ success: true, data: null }), '') + }) + return { kill: vi.fn() } + } ) - const lastSnapshotArgs = snapshotCalls.at(-1)![1] as string[] - expect(lastSnapshotArgs).toContain('--cdp') + + const shutdown = bridge.destroyAllSessions() + await vi.waitFor(() => expect(execFileMock).toHaveBeenCalledTimes(4)) + releases.splice(0).forEach((release) => release()) + await vi.waitFor(() => expect(execFileMock).toHaveBeenCalledTimes(6)) + releases.splice(0).forEach((release) => release()) + await shutdown + + expect(peakRetirements).toBe(4) + }) + + it('destroys a session that finishes creating during runtime shutdown', async () => { + const commandCalls: string[][] = [] + let releaseStaleClose: (() => void) | null = null + execFileMock.mockImplementation( + (_bin: string, args: string[], _opts: unknown, cb: ExecFileCallback) => { + commandCalls.push(args) + if (args.includes('close') && !releaseStaleClose) { + releaseStaleClose = () => { + cb(null, JSON.stringify({ success: true, data: null }), '') + } + return { kill: vi.fn() } + } + cb(null, JSON.stringify({ success: true, data: null }), '') + return { kill: vi.fn() } + } + ) + + const ensurePromise = ( + bridge as unknown as { + ensureSession: ( + sessionName: string, + browserPageId: string, + webContentsId: number + ) => Promise + } + ).ensureSession('orca-tab-tab-1', 'tab-1', 100) + await vi.waitFor(() => expect(releaseStaleClose).not.toBeNull()) + + const destroyAllPromise = bridge.destroyAllSessions() + releaseStaleClose!() + await ensurePromise + await destroyAllPromise + + const sessions = (bridge as unknown as { sessions: Map }).sessions + const proxy = CdpWsProxyMock.instances[0] as { stop: ReturnType } + expect(commandCalls.filter((args) => args.includes('close'))).toHaveLength(2) + expect(sessions.size).toBe(0) + expect(proxy.stop).toHaveBeenCalledTimes(1) + }) + + it('does not recreate a session after shutdown observes its pending retirement', async () => { + succeedWith({ snapshot: 'initial' }) + await bridge.snapshot() + execFileMock.mockClear() + + let releaseClose: (() => void) | null = null + execFileMock.mockImplementation( + (_bin: string, args: string[], _opts: unknown, cb: ExecFileCallback) => { + if (!args.includes('close')) { + throw new Error(`unexpected agent-browser args ${args.join(' ')}`) + } + releaseClose = () => cb(null, JSON.stringify({ success: true, data: null }), '') + return { kill: vi.fn() } + } + ) + + const restart = ( + bridge as unknown as { + restartSessionForTarget: ( + sessionName: string, + browserPageId: string, + webContentsId: number + ) => Promise + } + ).restartSessionForTarget('orca-tab-tab-1', 'tab-1', 100) + await vi.waitFor(() => expect(releaseClose).not.toBeNull()) + + const shutdown = bridge.destroyAllSessions() + releaseClose!() + + await expect(restart).rejects.toMatchObject({ + code: 'browser_owner_unavailable', + message: 'Browser runtime is shutting down' + }) + await shutdown + + const sessions = (bridge as unknown as { sessions: Map }).sessions + expect(sessions.size).toBe(0) + expect(CdpWsProxyMock.instances).toHaveLength(1) + expect(execFileMock).toHaveBeenCalledTimes(1) }) }) diff --git a/src/main/browser/agent-browser-bridge.ts b/src/main/browser/agent-browser-bridge.ts index f7eb19b5adf..10a64dc16ce 100644 --- a/src/main/browser/agent-browser-bridge.ts +++ b/src/main/browser/agent-browser-bridge.ts @@ -49,6 +49,7 @@ import type { } from '../../shared/runtime-types' import { assertClipboardTextWriteWithinLimitWithYield } from '../../shared/clipboard-text' import { normalizeBrowserNavigationUrl } from '../../shared/browser-url' +import { mapSettledWithConcurrency } from '../../shared/map-with-concurrency' import { iterateBrowserTextInsertionChunks } from './browser-text-insertion' import { createAgentBrowserProcessEnvironment } from './agent-browser-process-environment' @@ -57,6 +58,8 @@ const EXEC_TIMEOUT_MS = 90_000 const CONSECUTIVE_TIMEOUT_LIMIT = 3 const WAIT_PROCESS_TIMEOUT_GRACE_MS = 1_000 const STALE_SESSION_CLOSE_TIMEOUT_MS = 3_000 +const AGENT_BROWSER_CLEANUP_TIMEOUT_MS = 5_000 +const AGENT_BROWSER_CLEANUP_CONCURRENCY = 4 const EMBEDDED_NAVIGATION_TIMEOUT_MS = 30_000 export const AGENT_BROWSER_TEXT_ARGUMENT_MAX_BYTES = 8 * 1024 export const AGENT_BROWSER_CLIPBOARD_WRITE_MAX_BYTES = AGENT_BROWSER_TEXT_ARGUMENT_MAX_BYTES @@ -85,6 +88,10 @@ type ResolvedBrowserCommandTarget = { webContentsId: number } +type AgentBrowserCleanupOptions = { + closeTimeoutMs?: number +} + export type BrowserMouseModifier = 'cmd' | 'ctrl' | 'alt' | 'shift' function focusedValueSetExpression( @@ -587,6 +594,7 @@ export class AgentBrowserBridge { // Why: `agent-browser close` is async, keyed by session name — recreating before it finishes lets the old teardown close the new session. private readonly pendingSessionDestruction = new Map>() private readonly cancelledProcesses = new WeakSet() + private shutdownStarted = false constructor( private readonly browserManager: BrowserManager, @@ -678,13 +686,18 @@ export class AgentBrowserBridge { this.activeWebContentsId = nextWorktreeActiveWebContentsId } if (browserPageId) { - const sessionName = `orca-tab-${browserPageId}` - await this.destroySession(sessionName) - this.pendingInterceptRestore.delete(sessionName) + await this.onPageClosed(browserPageId) } this.options.onTabsChanged?.(owningWorktreeId) } + /** Retire a helper by its stable page identity when WebContents mapping is gone. */ + async onPageClosed(browserPageId: string): Promise { + const sessionName = `orca-tab-${browserPageId}` + await this.destroySession(sessionName) + this.pendingInterceptRestore.delete(sessionName) + } + async onProcessSwap( browserPageId: string, newWebContentsId: number, @@ -2044,12 +2057,18 @@ export class AgentBrowserBridge { // ── Session lifecycle ── - async destroyAllSessions(): Promise { - const promises: Promise[] = [] - for (const sessionName of this.sessions.keys()) { - promises.push(this.destroySession(sessionName)) - } - await Promise.allSettled(promises) + async destroyAllSessions(options?: AgentBrowserCleanupOptions): Promise { + this.shutdownStarted = true + const sessionNames = new Set([ + ...this.sessions.keys(), + ...this.pendingSessionCreation.keys(), + ...this.pendingSessionDestruction.keys() + ]) + await mapSettledWithConcurrency( + [...sessionNames], + AGENT_BROWSER_CLEANUP_CONCURRENCY, + (sessionName) => this.destroySession(sessionName, options) + ) this.pendingInterceptRestore.clear() } @@ -2073,12 +2092,14 @@ export class AgentBrowserBridge { execute: (sessionName: string, target: ResolvedBrowserCommandTarget) => Promise, options: EnqueueTargetedCommandOptions = {} ): Promise { + this.assertCommandAdmission() const target = this.resolveCommandTarget(worktreeId, browserPageId, options.requireScopedTarget) const sessionName = `orca-tab-${target.browserPageId}` if (options.ensureSession !== false) { await this.ensureSession(sessionName, target.browserPageId, target.webContentsId) } + this.assertCommandAdmission() return new Promise((resolve, reject) => { let queue = this.commandQueues.get(sessionName) @@ -2302,6 +2323,7 @@ export class AgentBrowserBridge { if (pendingDestruction) { await pendingDestruction } + this.assertCommandAdmission() if (this.sessions.has(sessionName)) { return @@ -2311,6 +2333,7 @@ export class AgentBrowserBridge { const pending = this.pendingSessionCreation.get(sessionName) if (pending) { await pending + this.assertCommandAdmission() return } @@ -2381,7 +2404,9 @@ export class AgentBrowserBridge { const destroy = (async (): Promise => { try { - await this.runAgentBrowserRaw(sessionName, ['--session', sessionName, 'close']) + await this.runAgentBrowserRaw(sessionName, ['--session', sessionName, 'close'], { + timeoutMs: AGENT_BROWSER_CLEANUP_TIMEOUT_MS + }) } catch { // Session may already be dead. } @@ -2400,7 +2425,10 @@ export class AgentBrowserBridge { } } - private async destroySession(sessionName: string): Promise { + private async destroySession( + sessionName: string, + options: AgentBrowserCleanupOptions = { closeTimeoutMs: AGENT_BROWSER_CLEANUP_TIMEOUT_MS } + ): Promise { const pendingDestruction = this.pendingSessionDestruction.get(sessionName) if (pendingDestruction) { await pendingDestruction @@ -2443,7 +2471,11 @@ export class AgentBrowserBridge { const destroy = (async (): Promise => { try { // Why: each tab has its own named session — close without --session leaves this tab's daemon running. - await this.runAgentBrowserRaw(sessionName, ['--session', sessionName, 'close']) + await this.runAgentBrowserRaw( + sessionName, + ['--session', sessionName, 'close'], + options.closeTimeoutMs === undefined ? undefined : { timeoutMs: options.closeTimeoutMs } + ) } catch { // Session may already be dead } @@ -2474,6 +2506,12 @@ export class AgentBrowserBridge { } } + private assertCommandAdmission(): void { + if (this.shutdownStarted) { + throw new BrowserError('browser_owner_unavailable', 'Browser runtime is shutting down') + } + } + private async execAgentBrowser( sessionName: string, commandArgs: string[], diff --git a/src/main/browser/browser-backend.ts b/src/main/browser/browser-backend.ts index 939684e49b3..0b3999777d1 100644 --- a/src/main/browser/browser-backend.ts +++ b/src/main/browser/browser-backend.ts @@ -20,5 +20,5 @@ export type BrowserBackend = { closeTab(browserPageId: string): Promise /** Tear down every page this backend owns (process shutdown). Optional — * renderer-hosted backends are torn down with their window. */ - destroyAll?(): void + destroyAll?(): void | Promise } diff --git a/src/main/browser/offscreen-browser-backend-lifecycle.test.ts b/src/main/browser/offscreen-browser-backend-lifecycle.test.ts index 2d6631ea7a7..52a555c9f5f 100644 --- a/src/main/browser/offscreen-browser-backend-lifecycle.test.ts +++ b/src/main/browser/offscreen-browser-backend-lifecycle.test.ts @@ -93,4 +93,193 @@ describe('OffscreenBrowserBackend lifecycle', () => { expect(vi.getTimerCount()).toBe(0) vi.useRealTimers() }) + + it('unregisters a closing page before awaiting owner retirement', async () => { + const browserManager = { + registerOffscreenGuest: vi.fn(), + unregisterGuest: vi.fn() + } + let releaseOwnerRetirement!: () => void + const ownerRetirementBlocked = new Promise((resolve) => { + releaseOwnerRetirement = resolve + }) + const onPageClosed = vi.fn(() => ownerRetirementBlocked) + const backend = new OffscreenBrowserBackend(browserManager as never, { + getAgentBrowserBridge: () => ({ onPageClosed }) + }) + + await backend.createTab({ browserPageId: 'page-1', url: 'about:blank', worktreeId: 'wt' }) + await backend.createTab({ browserPageId: 'page-2', url: 'about:blank', worktreeId: 'wt' }) + const close = backend.closeTab('page-1') + await vi.waitFor(() => expect(onPageClosed).toHaveBeenCalledWith('page-1')) + const pageWasUnregisteredBeforeRetirement = browserManager.unregisterGuest.mock.calls.some( + ([pageId]) => pageId === 'page-1' + ) + releaseOwnerRetirement() + await close + + expect(onPageClosed).toHaveBeenCalledOnce() + expect(onPageClosed).not.toHaveBeenCalledWith('page-2') + expect(backend.getWebContentsId('page-2')).toBe(2) + expect(pageWasUnregisteredBeforeRetirement).toBe(true) + }) + + it('preserves a replacement page when the old window finishes closing', async () => { + const browserManager = { + registerOffscreenGuest: vi.fn(), + unregisterGuest: vi.fn() + } + let releaseOwnerRetirement!: () => void + const ownerRetirementBlocked = new Promise((resolve) => { + releaseOwnerRetirement = resolve + }) + const onPageClosed = vi.fn(() => ownerRetirementBlocked) + const backend = new OffscreenBrowserBackend(browserManager as never, { + getAgentBrowserBridge: () => ({ onPageClosed }) + }) + + await backend.createTab({ browserPageId: 'page-1', url: 'about:blank', worktreeId: 'wt' }) + const close = backend.closeTab('page-1') + await vi.waitFor(() => expect(onPageClosed).toHaveBeenCalledWith('page-1')) + await backend.createTab({ browserPageId: 'page-1', url: 'about:blank', worktreeId: 'wt' }) + + releaseOwnerRetirement() + await close + + expect(backend.getWebContentsId('page-1')).toBe(2) + expect(browserManager.unregisterGuest).toHaveBeenCalledTimes(1) + }) + + it('retires the helper when an offscreen renderer is destroyed unexpectedly', async () => { + const browserManager = { + registerOffscreenGuest: vi.fn(), + unregisterGuest: vi.fn() + } + const onPageClosed = vi.fn(async () => {}) + const backend = new OffscreenBrowserBackend(browserManager as never, { + getAgentBrowserBridge: () => ({ onPageClosed }) + }) + + await backend.createTab({ browserPageId: 'page-1', url: 'about:blank', worktreeId: 'wt' }) + mocks.windows[0].webContents.emit('destroyed') + await vi.waitFor(() => expect(onPageClosed).toHaveBeenCalledWith('page-1')) + }) + + it('cleans every helper owner during backend shutdown', async () => { + const browserManager = { + registerOffscreenGuest: vi.fn(), + unregisterGuest: vi.fn() + } + const onPageClosed = vi.fn(async () => {}) + const backend = new OffscreenBrowserBackend(browserManager as never, { + getAgentBrowserBridge: () => ({ onPageClosed }) + }) + + await backend.createTab({ browserPageId: 'page-1', url: 'about:blank', worktreeId: 'wt' }) + await backend.createTab({ browserPageId: 'page-2', url: 'about:blank', worktreeId: 'wt' }) + await backend.destroyAll() + + expect(onPageClosed).toHaveBeenCalledTimes(2) + expect(browserManager.unregisterGuest).toHaveBeenCalledWith('page-1') + expect(browserManager.unregisterGuest).toHaveBeenCalledWith('page-2') + }) + + it('rejects a concurrent create while shutdown is draining owned pages', async () => { + const browserManager = { + registerOffscreenGuest: vi.fn(), + unregisterGuest: vi.fn() + } + let releaseOwnerRetirement!: () => void + const ownerRetirementBlocked = new Promise((resolve) => { + releaseOwnerRetirement = resolve + }) + const onPageClosed = vi.fn(() => ownerRetirementBlocked) + const backend = new OffscreenBrowserBackend(browserManager as never, { + getAgentBrowserBridge: () => ({ onPageClosed }) + }) + + await backend.createTab({ browserPageId: 'page-1', url: 'about:blank', worktreeId: 'wt' }) + const shutdown = backend.destroyAll() + await vi.waitFor(() => expect(onPageClosed).toHaveBeenCalledWith('page-1')) + + await expect( + backend.createTab({ browserPageId: 'page-2', url: 'about:blank', worktreeId: 'wt' }) + ).rejects.toThrow('Offscreen browser backend is shutting down') + + releaseOwnerRetirement() + await shutdown + expect(mocks.windows).toHaveLength(1) + expect(browserManager.unregisterGuest).toHaveBeenCalledWith('page-1') + expect(browserManager.unregisterGuest).not.toHaveBeenCalledWith('page-2') + }) + + it('joins owner retirement started by an unexpected renderer destroy', async () => { + const browserManager = { + registerOffscreenGuest: vi.fn(), + unregisterGuest: vi.fn() + } + let releaseOwnerRetirement!: () => void + const ownerRetirementBlocked = new Promise((resolve) => { + releaseOwnerRetirement = resolve + }) + const onPageClosed = vi.fn(() => ownerRetirementBlocked) + const backend = new OffscreenBrowserBackend(browserManager as never, { + getAgentBrowserBridge: () => ({ onPageClosed }) + }) + + await backend.createTab({ browserPageId: 'page-1', url: 'about:blank', worktreeId: 'wt' }) + mocks.windows[0].webContents.emit('destroyed') + await vi.waitFor(() => expect(onPageClosed).toHaveBeenCalledWith('page-1')) + + const shutdown = backend.destroyAll() + const outcome = await Promise.race([ + shutdown.then(() => 'settled'), + new Promise((resolve) => setImmediate(() => resolve('pending'))) + ]) + expect(outcome).toBe('pending') + + releaseOwnerRetirement() + await shutdown + }) + + it('bounds concurrent helper retirements during shutdown', async () => { + const browserManager = { + registerOffscreenGuest: vi.fn(), + unregisterGuest: vi.fn() + } + let activeRetirements = 0 + let peakRetirements = 0 + const releases: (() => void)[] = [] + const onPageClosed = vi.fn( + () => + new Promise((resolve) => { + activeRetirements++ + peakRetirements = Math.max(peakRetirements, activeRetirements) + releases.push(() => { + activeRetirements-- + resolve() + }) + }) + ) + const backend = new OffscreenBrowserBackend(browserManager as never, { + getAgentBrowserBridge: () => ({ onPageClosed }) + }) + + for (let index = 0; index < 6; index++) { + await backend.createTab({ + browserPageId: `page-${index}`, + url: 'about:blank', + worktreeId: 'wt' + }) + } + + const shutdown = backend.destroyAll() + await vi.waitFor(() => expect(onPageClosed).toHaveBeenCalledTimes(4)) + releases.splice(0).forEach((release) => release()) + await vi.waitFor(() => expect(onPageClosed).toHaveBeenCalledTimes(6)) + releases.splice(0).forEach((release) => release()) + await shutdown + + expect(peakRetirements).toBe(4) + }) }) diff --git a/src/main/browser/offscreen-browser-backend.ts b/src/main/browser/offscreen-browser-backend.ts index 3ca96a3748f..e17a2923aa0 100644 --- a/src/main/browser/offscreen-browser-backend.ts +++ b/src/main/browser/offscreen-browser-backend.ts @@ -2,8 +2,10 @@ import { randomUUID } from 'node:crypto' import { BrowserWindow } from 'electron' import { ORCA_BROWSER_PARTITION } from '../../shared/constants' import { ORCA_BROWSER_GUEST_WEB_PREFERENCES } from '../../shared/browser-guest-web-preferences' +import { mapSettledWithConcurrency } from '../../shared/map-with-concurrency' import type { BrowserBackend, BrowserBackendCreateTab } from './browser-backend' import type { BrowserManager } from './browser-manager' +import type { AgentBrowserBridge } from './agent-browser-bridge' import { browserSessionRegistry } from './browser-session-registry' // Why: headless orca serve has no renderer window to host a , so each @@ -16,13 +18,26 @@ import { browserSessionRegistry } from './browser-session-registry' const DEFAULT_VIEWPORT_WIDTH = 1280 const DEFAULT_VIEWPORT_HEIGHT = 800 const LOAD_TIMEOUT_MS = 30_000 +const OWNER_RETIREMENT_CONCURRENCY = 4 export class OffscreenBrowserBackend implements BrowserBackend { private readonly windowsByPageId = new Map() + // Shutdown is terminal for this backend; rejecting creates closes the race + // where destroyAll snapshots ownership and a new page appears afterward. + private shutdownStarted = false + private readonly pendingOwnerRetirements = new Set>() - constructor(private readonly browserManager: BrowserManager) {} + constructor( + private readonly browserManager: BrowserManager, + private readonly options: { + getAgentBrowserBridge?: () => Pick | null + } = {} + ) {} async createTab(params: BrowserBackendCreateTab): Promise<{ browserPageId: string }> { + if (this.shutdownStarted) { + throw new Error('Offscreen browser backend is shutting down') + } const browserPageId = params.browserPageId ?? randomUUID() if (this.windowsByPageId.has(browserPageId)) { throw new Error(`Browser page ${browserPageId} already exists`) @@ -55,6 +70,12 @@ export class OffscreenBrowserBackend implements BrowserBackend { // teardown), drop the registry entry so commands fail cleanly instead of // resolving a dead WebContents. win.webContents.once('destroyed', () => { + // Explicit close removes the page first and performs awaited cleanup; + // only an unexpected destruction still owns the bridge retirement here. + if (this.windowsByPageId.get(browserPageId) !== win) { + return + } + void this.retirePageOwner(browserPageId) this.windowsByPageId.delete(browserPageId) this.browserManager.unregisterGuest(browserPageId) }) @@ -88,8 +109,14 @@ export class OffscreenBrowserBackend implements BrowserBackend { const win = this.windowsByPageId.get(browserPageId) this.windowsByPageId.delete(browserPageId) this.browserManager.unregisterGuest(browserPageId) - if (win && !win.isDestroyed()) { - win.destroy() + try { + if (win) { + await this.retirePageOwner(browserPageId) + } + } finally { + if (win && !win.isDestroyed()) { + win.destroy() + } } } @@ -98,14 +125,24 @@ export class OffscreenBrowserBackend implements BrowserBackend { return win && !win.isDestroyed() ? win.webContents.id : null } - destroyAll(): void { - for (const [pageId, win] of this.windowsByPageId) { - this.browserManager.unregisterGuest(pageId) - if (!win.isDestroyed()) { - win.destroy() - } + async destroyAll(): Promise { + this.shutdownStarted = true + const pageIds = [...this.windowsByPageId.keys()] + await mapSettledWithConcurrency(pageIds, OWNER_RETIREMENT_CONCURRENCY, (pageId) => + this.closeTab(pageId) + ) + await Promise.all(this.pendingOwnerRetirements) + } + + private retirePageOwner(browserPageId: string): Promise { + const bridge = this.options.getAgentBrowserBridge?.() + if (!bridge) { + return Promise.resolve() } - this.windowsByPageId.clear() + const retirement = bridge.onPageClosed(browserPageId).catch(() => {}) + this.pendingOwnerRetirements.add(retirement) + void retirement.finally(() => this.pendingOwnerRetirements.delete(retirement)) + return retirement } private async loadUrl(win: BrowserWindow, url: string): Promise { diff --git a/src/main/index.ts b/src/main/index.ts index f2657430ad2..edc26214121 100644 --- a/src/main/index.ts +++ b/src/main/index.ts @@ -222,6 +222,7 @@ import { ensureWindowsUserDataAclGrant } from './startup/windows-user-data-acl' import { probeWindowsInstallDirAcl } from './startup/windows-install-dir-acl-probe' import { neutralizeLegacyTerminalShimDir } from './pty/legacy-terminal-shim-dir' import { shouldQuitWhenAllWindowsClosed } from './startup/window-all-closed-quit-policy' +import { registerServeSignalHandlers } from './startup/serve-signal-handlers' import { createServeDesktopActivationGate, settleServeDesktopActivation as settleServeDesktopActivationGate @@ -2088,15 +2089,6 @@ async function printServeReady(options: ServeOptions): Promise { notifyServeSupervisorReady(runtime.getRuntimeId()) } -function installServeSignalHandlers(): void { - const quit = (): void => { - // Why: route SIGINT/SIGTERM through Electron's normal quit so runtime metadata, daemon checkpoints, and telemetry flush. - app.quit() - } - process.once('SIGINT', quit) - process.once('SIGTERM', quit) -} - // Why: on PTY teardown drop the spinner entry explicitly, else the shared timer keeps ticking with sendSyntheticTitle no-oping forever. registerPaneKeyTeardownListener((paneKey) => { stopSyntheticTitleSpinner(paneKey) @@ -3295,7 +3287,11 @@ void app.whenReady().then(async () => { await runtime.reconcileLegacyWorkerTerminals() // Why: headless servers can't mount panes; use offscreen WebContents, gated on a real display so browser.headless.v1 stays honest. if (headlessBrowserDisplayAvailable) { - runtime.setOffscreenBrowserBackend(new OffscreenBrowserBackend(browserManager)) + runtime.setOffscreenBrowserBackend( + new OffscreenBrowserBackend(browserManager, { + getAgentBrowserBridge: () => agentBrowserBridge + }) + ) } // Why: headless servers have no renderer graph publisher; publish an explicit empty graph so status clients see a ready server. runtime.syncWindowGraph(HEADLESS_RUNTIME_WINDOW_ID, { tabs: [], leaves: [] }) @@ -3304,7 +3300,8 @@ void app.whenReady().then(async () => { throw error }) settleServeDesktopActivation() - installServeSignalHandlers() + // Why: every attempt must reach app.quit(); a page beforeunload can veto an earlier signal. + registerServeSignalHandlers(process, () => app.quit()) // Why: headless serve has no renderer to run the normal cli:install flow; do it here for macOS/Linux only (Windows-excluded: install() only mutates registry PATH, not child terminals). if (process.platform === 'darwin' || process.platform === 'linux') { try { @@ -3515,10 +3512,11 @@ app.on('will-quit', (e) => { // Why: cancels relay restart/reinstall timers and kills wsl.exe children deterministically, not via stdio-pipe teardown. wslHookRelayManager.disposeAll() const statsFlush = stats?.flushAsync() ?? Promise.resolve() - // Why: agent-browser daemon processes would otherwise linger after quit, holding ports and stale session state on disk. - runtime?.getAgentBrowserBridge()?.destroyAllSessions() - // Why: headless offscreen browser windows are main-process owned; tear them down explicitly on quit. - runtime?.getOffscreenBrowserBackend()?.destroyAll?.() + // Why: retire headless page owners first, then sweep residual helper sessions without duplicate close fanout. + const browserShutdown = (async (): Promise => { + await runtime?.getOffscreenBrowserBackend()?.destroyAll?.() + await runtime?.getAgentBrowserBridge()?.destroyAllSessions() + })() // Why (review P2-4): local SSH browser routes own loopback listeners and, on the // system-ssh path, `ssh -N -D` children that would otherwise outlive the app. const localSshRouteShutdown = import('./browser/local-ssh-browser-route') @@ -3570,6 +3568,7 @@ app.on('will-quit', (e) => { // temp+rename swap means a write cut short by the deadline leaves the old file intact. settleTeardownWithinDeadline([ { name: 'daemon', promise: daemonTeardown }, + { name: 'browser', promise: browserShutdown }, { name: 'runtime-rpc', promise: rpcStopAndClear }, { name: 'watchers', promise: watcherShutdown }, { name: 'emulator', promise: emulatorShutdown }, diff --git a/src/main/quit-teardown-deadline.ts b/src/main/quit-teardown-deadline.ts index 33f5031bdd0..6a712e0aa66 100644 --- a/src/main/quit-teardown-deadline.ts +++ b/src/main/quit-teardown-deadline.ts @@ -3,9 +3,9 @@ // socket) can leave one unsettled forever and make Force Quit the only way // out (#9447). Racing a deadline guarantees quit always completes. -// Why: generous enough for daemon checkpoint writes on a slow disk; small -// enough that a wedged teardown never needs Force Quit. -export const WILL_QUIT_TEARDOWN_DEADLINE_MS = 20_000 +import { WILL_QUIT_TEARDOWN_DEADLINE_MS } from '../shared/quit-teardown-deadline' + +export { WILL_QUIT_TEARDOWN_DEADLINE_MS } from '../shared/quit-teardown-deadline' export type NamedQuitTeardown = { name: string diff --git a/src/main/startup/desktop-startup-ordering.test.ts b/src/main/startup/desktop-startup-ordering.test.ts index 2e7453707b5..af4c7d609ca 100644 --- a/src/main/startup/desktop-startup-ordering.test.ts +++ b/src/main/startup/desktop-startup-ordering.test.ts @@ -270,6 +270,40 @@ describe('startup ordering', () => { expect(disposeIndex).toBeGreaterThan(commitIndex) }) + it('joins agent-browser cleanup before the committed quit exits', () => { + const source = readFileSync(join(process.cwd(), 'src/main/index.ts'), 'utf8') + const willQuitStart = source.indexOf("app.on('will-quit'") + const windowAllClosedStart = source.indexOf("app.on('window-all-closed'", willQuitStart) + const willQuit = source.slice(willQuitStart, windowAllClosedStart) + const cleanupStart = willQuit.indexOf('const browserShutdown') + const offscreenCleanupStart = willQuit.indexOf( + 'runtime?.getOffscreenBrowserBackend()?.destroyAll?.()' + ) + const residualCleanupStart = willQuit.indexOf( + 'runtime?.getAgentBrowserBridge()?.destroyAllSessions()' + ) + const barrierStart = willQuit.indexOf('settleTeardownWithinDeadline([') + + expect(willQuitStart).toBeGreaterThanOrEqual(0) + expect(windowAllClosedStart).toBeGreaterThan(willQuitStart) + expect(cleanupStart).toBeGreaterThanOrEqual(0) + expect(offscreenCleanupStart).toBeGreaterThan(cleanupStart) + expect(residualCleanupStart).toBeGreaterThan(offscreenCleanupStart) + expect(barrierStart).toBeGreaterThan(cleanupStart) + expect(willQuit.slice(barrierStart)).toContain("{ name: 'browser', promise: browserShutdown }") + }) + + it('registers repeatable serve signal handling before headless startup completes', () => { + const source = readFileSync(join(process.cwd(), 'src/main/index.ts'), 'utf8') + const serveStart = source.indexOf('if (serveOptions) {') + const signalHandlers = source.indexOf('registerServeSignalHandlers(process', serveStart) + const serveReady = source.indexOf('await printServeReady(serveOptions)', serveStart) + + expect(serveStart).toBeGreaterThanOrEqual(0) + expect(signalHandlers).toBeGreaterThan(serveStart) + expect(signalHandlers).toBeLessThan(serveReady) + }) + it('starts the automation scheduler before headless serve reports ready', () => { const source = readFileSync(join(process.cwd(), 'src/main/index.ts'), 'utf8') const serveStart = source.indexOf('if (serveOptions) {') diff --git a/src/main/startup/serve-signal-handlers.test.ts b/src/main/startup/serve-signal-handlers.test.ts new file mode 100644 index 00000000000..36250cd4146 --- /dev/null +++ b/src/main/startup/serve-signal-handlers.test.ts @@ -0,0 +1,19 @@ +import { EventEmitter } from 'node:events' +import { describe, expect, it, vi } from 'vitest' +import { registerServeSignalHandlers } from './serve-signal-handlers' + +describe('registerServeSignalHandlers', () => { + it('retries a vetoed quit on every delivered signal', () => { + const signalSource = new EventEmitter() + const quitApplication = vi.fn() + + registerServeSignalHandlers(signalSource, quitApplication) + signalSource.emit('SIGINT') + signalSource.emit('SIGINT') + signalSource.emit('SIGTERM') + + expect(quitApplication).toHaveBeenCalledTimes(3) + expect(signalSource.listenerCount('SIGINT')).toBe(1) + expect(signalSource.listenerCount('SIGTERM')).toBe(1) + }) +}) diff --git a/src/main/startup/serve-signal-handlers.ts b/src/main/startup/serve-signal-handlers.ts new file mode 100644 index 00000000000..3022d935ac5 --- /dev/null +++ b/src/main/startup/serve-signal-handlers.ts @@ -0,0 +1,12 @@ +type ServeSignalSource = { + on(event: 'SIGINT' | 'SIGTERM', listener: () => void): unknown +} + +export function registerServeSignalHandlers( + signalSource: ServeSignalSource, + quitApplication: () => void +): void { + // Keep both listeners installed so duplicate delivery cannot fall through to default termination. + signalSource.on('SIGINT', quitApplication) + signalSource.on('SIGTERM', quitApplication) +} diff --git a/src/main/window/main-window-close-lifecycle.ts b/src/main/window/main-window-close-lifecycle.ts index cbfeafd1357..0bf1c04ef88 100644 --- a/src/main/window/main-window-close-lifecycle.ts +++ b/src/main/window/main-window-close-lifecycle.ts @@ -1,4 +1,5 @@ import { ipcMain, Menu, Notification, type BrowserWindow } from 'electron' +import { QUIT_RENDERER_ACK_TIMEOUT_MS } from '../../shared/quit-teardown-deadline' import { translateMain } from '../i18n/main-i18n' import type { Store } from '../persistence' import { resolveWindowCloseAction } from './window-close-decision' @@ -7,7 +8,7 @@ import type { MainWindowFocusLifecycle } from './main-window-focus-lifecycle' import type { MainWindowStateLifecycle } from './main-window-state-lifecycle' import { syncTrafficLightPosition } from './main-window-visual-lifecycle' -export const WINDOW_QUIT_RENDERER_ACK_TIMEOUT_MS = 10_000 +export const WINDOW_QUIT_RENDERER_ACK_TIMEOUT_MS = QUIT_RENDERER_ACK_TIMEOUT_MS export function installMainWindowCloseLifecycle(args: { focus: MainWindowFocusLifecycle diff --git a/src/shared/quit-teardown-deadline.ts b/src/shared/quit-teardown-deadline.ts new file mode 100644 index 00000000000..b42b02b20fd --- /dev/null +++ b/src/shared/quit-teardown-deadline.ts @@ -0,0 +1,3 @@ +export const QUIT_RENDERER_ACK_TIMEOUT_MS = 10_000 +// Generous enough for slow-disk checkpoints while keeping a wedged quit bounded. +export const WILL_QUIT_TEARDOWN_DEADLINE_MS = 20_000 From ac76e0dd06b9fcf3e7bb5210de129c91ab416da0 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 26 Aug 2026 15:20:36 -0700 Subject: [PATCH 07/19] fix(runtime): cancel pending driver timers on desktop reclaim (#16337) * fix(runtime): cancel pending driver timers on desktop reclaim * fix(runtime): centralize pending driver cancellation * fix(runtime): complete pending driver cancellation extraction * test(runtime): cover desktop reclaim mutation branches --- .../mobile-subscribe-integration.test.ts | 81 +++++++++++++++++-- src/main/runtime/orca-runtime.ts | 42 ++++------ 2 files changed, 90 insertions(+), 33 deletions(-) diff --git a/src/main/runtime/mobile-subscribe-integration.test.ts b/src/main/runtime/mobile-subscribe-integration.test.ts index cee4fcfe79f..292cc13a816 100644 --- a/src/main/runtime/mobile-subscribe-integration.test.ts +++ b/src/main/runtime/mobile-subscribe-integration.test.ts @@ -840,15 +840,80 @@ describe('mobile subscribe integration', () => { await runtime.handleMobileSubscribe('pty-1', 'client-a', { cols: 45, rows: 20 }) runtime.handleMobileUnsubscribe('pty-1', 'client-a') - // No subscribers, indefinite hold — PTY stays at phone dims. - await vi.advanceTimersByTimeAsync(60_000) - expect(ptySizes.get('pty-1')).toEqual({ cols: 45, rows: 20 }) - expect(runtime.isMobileSubscriberActive('pty-1')).toBe(false) - - // Manual reclaim: PTY restored to desktop dims via the held branch. - const ok = await runtime.reclaimTerminalForDesktop('pty-1') - expect(ok).toBe(true) + await runtime.reclaimTerminalForDesktop('pty-1') expect(ptySizes.get('pty-1')).toEqual({ cols: 150, rows: 40 }) + + await vi.advanceTimersByTimeAsync(250) + expect(runtime.getDriver('pty-1')).toEqual({ kind: 'desktop' }) + }) + + it('reclaim cancels the soft-leave timer in the remote-layout branch', async () => { + const { runtime } = createRuntime() + await runtime.updateRemoteDesktopViewer('pty-1', 'sub-a', 'viewer-a', 120, 32) + await runtime.handleMobileSubscribe('pty-1', 'client-a', { cols: 45, rows: 20 }) + runtime.handleMobileUnsubscribe('pty-1', 'client-a') + + expect(await runtime.reclaimTerminalForDesktop('pty-1')).toBe(true) + await vi.advanceTimersByTimeAsync(250) + expect(runtime.getDriver('pty-1')).toEqual({ kind: 'desktop' }) + }) + + it('reclaim cancels pending restore timers on the active-subscriber branch', async () => { + const { runtime } = createRuntime() + await runtime.handleMobileSubscribe('pty-1', 'client-a', { cols: 45, rows: 20 }) + runtime.handleMobileUnsubscribe('pty-1', 'client-a') + await runtime.handleMobileSubscribe('pty-1', 'client-b', { cols: 40, rows: 18 }) + + const pendingRestore = Reflect.get(runtime, 'pendingRestoreTimers') as Map + pendingRestore.set('pty-1', { timer: setTimeout(() => {}, 60_000), clientId: 'client-b' }) + const pendingSoft = Reflect.get(runtime, 'pendingSoftLeavers') as Map + expect(pendingSoft.has('pty-1')).toBe(true) + await runtime.reclaimTerminalForDesktop('pty-1') + expect(pendingRestore.has('pty-1')).toBe(false) + expect(pendingSoft.has('pty-1')).toBe(false) + }) + + it('reclaim cancels pending restore timers on the orphan-driver branch', async () => { + const { runtime } = createRuntime() + await runtime.handleMobileSubscribe('pty-1', 'client-a', { cols: 45, rows: 20 }) + runtime.handleMobileUnsubscribe('pty-1', 'client-a') + ;(Reflect.get(runtime, 'terminalFitOverrides') as Map).delete('pty-1') + + const pendingRestore = Reflect.get(runtime, 'pendingRestoreTimers') as Map + const pendingSoft = Reflect.get(runtime, 'pendingSoftLeavers') as Map + await runtime.reclaimTerminalForDesktop('pty-1') + expect(pendingRestore.has('pty-1')).toBe(false) + expect(pendingSoft.has('pty-1')).toBe(false) + }) + + it('reclaim cancels pending restore timers when no driver lock remains', async () => { + const { runtime } = createRuntime() + await runtime.handleMobileSubscribe('pty-1', 'client-a', { cols: 45, rows: 20 }) + runtime.handleMobileUnsubscribe('pty-1', 'client-a') + ;(Reflect.get(runtime, 'terminalFitOverrides') as Map).delete('pty-1') + ;(Reflect.get(runtime, 'currentDriver') as Map).set('pty-1', { + kind: 'idle' + }) + + const pendingRestore = Reflect.get(runtime, 'pendingRestoreTimers') as Map + const pendingSoft = Reflect.get(runtime, 'pendingSoftLeavers') as Map + expect(await runtime.reclaimTerminalForDesktop('pty-1')).toBe(false) + expect(pendingRestore.has('pty-1')).toBe(false) + expect(pendingSoft.has('pty-1')).toBe(false) + }) + + it('reclaim revokes soft-leave grace admission for mobile input', async () => { + const { runtime } = createRuntime() + await runtime.handleMobileSubscribe('pty-1', 'client-a', { cols: 45, rows: 20 }) + runtime.handleMobileUnsubscribe('pty-1', 'client-a') + + const claim = runtime.beginMobileInputFloor('pty-1', 'client-a') + expect(claim).not.toBeNull() + claim?.rollback() + + await runtime.reclaimTerminalForDesktop('pty-1') + expect(runtime.getDriver('pty-1')).toEqual({ kind: 'desktop' }) + expect(runtime.beginMobileInputFloor('pty-1', 'client-a')).toBeNull() }) it('reclaimTerminalForDesktop prefers fresh desktop geometry for a held PTY', async () => { diff --git a/src/main/runtime/orca-runtime.ts b/src/main/runtime/orca-runtime.ts index a50cef258e6..db9be9e7aa0 100644 --- a/src/main/runtime/orca-runtime.ts +++ b/src/main/runtime/orca-runtime.ts @@ -15821,16 +15821,7 @@ export class OrcaRuntimeService { this.layouts.delete(ptyId) this.layoutQueues.delete(ptyId) this.freshSubscribeGuard.delete(ptyId) - const pendingRestore = this.pendingRestoreTimers.get(ptyId) - if (pendingRestore) { - clearTimeout(pendingRestore.timer) - this.pendingRestoreTimers.delete(ptyId) - } - const pendingSoft = this.pendingSoftLeavers.get(ptyId) - if (pendingSoft) { - clearTimeout(pendingSoft.timer) - this.pendingSoftLeavers.delete(ptyId) - } + this.cancelPendingDriverMutations(ptyId) // Why: a cold restore can respawn under the same session id within the // delayed-Enter window; the armed Enter would inject \r into the // replacement and stamp rows it never received. @@ -16487,6 +16478,7 @@ export class OrcaRuntimeService { // frame. Returns `true` whenever there was a lock to reclaim, `false` only // when there was nothing to reclaim. async reclaimTerminalForDesktop(ptyId: string): Promise { + this.cancelPendingDriverMutations(ptyId) if (this.isMobileSubscriberActive(ptyId)) { this.setMobileDisplayMode(ptyId, 'desktop') await this.applyMobileDisplayMode(ptyId) @@ -16506,16 +16498,6 @@ export class OrcaRuntimeService { } const heldOverride = this.terminalFitOverrides.get(ptyId) if (heldOverride && this.hasRemoteDesktopLayoutState(ptyId)) { - const pending = this.pendingRestoreTimers.get(ptyId) - if (pending) { - clearTimeout(pending.timer) - this.pendingRestoreTimers.delete(ptyId) - } - const softLeaver = this.pendingSoftLeavers.get(ptyId) - if (softLeaver) { - clearTimeout(softLeaver.timer) - this.pendingSoftLeavers.delete(ptyId) - } // Why: applyRemoteDesktopLayout no-ops while the driver still reads mobile. this.setDriver(ptyId, { kind: 'idle' }) // Why: best-effort, like the local held branch below. A host whose resize @@ -16528,11 +16510,6 @@ export class OrcaRuntimeService { return true } if (heldOverride) { - const pending = this.pendingRestoreTimers.get(ptyId) - if (pending) { - clearTimeout(pending.timer) - this.pendingRestoreTimers.delete(ptyId) - } // Why: with no subscribers, resolveDesktopRestoreTarget can fall through // to current PTY size — which is at phone dims (wrong). Prefer a fresh // desktop renderer measurement when one exists; otherwise use the @@ -16557,6 +16534,21 @@ export class OrcaRuntimeService { return false } + // Why: teardown and desktop reclaim supersede delayed mobile mutations, + // revoking soft-leave grace admission for input floors. + private cancelPendingDriverMutations(ptyId: string): void { + const pendingRestore = this.pendingRestoreTimers.get(ptyId) + if (pendingRestore) { + clearTimeout(pendingRestore.timer) + this.pendingRestoreTimers.delete(ptyId) + } + const pendingSoft = this.pendingSoftLeavers.get(ptyId) + if (pendingSoft) { + clearTimeout(pendingSoft.timer) + this.pendingSoftLeavers.delete(ptyId) + } + } + // Why: the shared "banner must be gone now" step for an explicit desktop // take-back. Releases the presence lock (driver → desktop) and, if the // best-effort resize left a fit-override held (resize didn't converge), From 81f89a705c5bf334b7ce1086db62fbeafccf1d47 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 26 Aug 2026 15:23:07 -0700 Subject: [PATCH 08/19] fix(terminal): bound WebGL context-loss retries on tab reveal (#16338) * fix(terminal): retry bounded WebGL recovery on tab reveal * test(terminal): cover reveal repaint and pruned diagnostics * fix(terminal): make WebGL diagnostics pure and cover reveal refusal * fix(terminal): correct WebGL retry comments --- .../lib/pane-manager/pane-manager-types.ts | 3 + .../pane-manager/pane-rendering-control.ts | 12 ++-- .../pane-rendering-diagnostics.ts | 2 + .../pane-manager/pane-reveal-repaint.test.ts | 15 +++++ .../pane-terminal-gpu-acceleration.ts | 2 + .../pane-webgl-context-loss-policy.ts | 38 ++++++++++++ .../pane-webgl-context-recovery.test.ts | 59 ++++++++++++++++++- .../lib/pane-manager/pane-webgl-reattach.ts | 14 ++++- .../lib/pane-manager/pane-webgl-renderer.ts | 9 +-- 9 files changed, 143 insertions(+), 11 deletions(-) create mode 100644 src/renderer/src/lib/pane-manager/pane-webgl-context-loss-policy.ts diff --git a/src/renderer/src/lib/pane-manager/pane-manager-types.ts b/src/renderer/src/lib/pane-manager/pane-manager-types.ts index 62c69836841..49ceabb75f1 100644 --- a/src/renderer/src/lib/pane-manager/pane-manager-types.ts +++ b/src/renderer/src/lib/pane-manager/pane-manager-types.ts @@ -114,6 +114,7 @@ export type PaneRenderingDiagnostics = { gpuRenderingEnabled: boolean webglAttachmentDeferred: boolean webglDisabledAfterContextLoss: boolean + webglContextLossesInWindow?: number webglAttachFailedSinceRecovery: boolean hasComplexScriptOutput: boolean terminalWebglAutoDecision: TerminalWebglAutoDecision @@ -142,6 +143,8 @@ export type ManagedPaneInternal = { gpuRenderingEnabled: boolean webglAttachmentDeferred: boolean webglDisabledAfterContextLoss: boolean + // Shared history bounds context-loss retries across resume and settled reveal. + webglContextLossTimestamps?: number[] // Hidden retained renderers rebuild at the resume boundary, never behind the hidden surface. webglRebuildDeferred?: boolean // Why per-pane: one pane's failed WebGL attach must not strand every other diff --git a/src/renderer/src/lib/pane-manager/pane-rendering-control.ts b/src/renderer/src/lib/pane-manager/pane-rendering-control.ts index ed56a1d65ec..6709a7a1f98 100644 --- a/src/renderer/src/lib/pane-manager/pane-rendering-control.ts +++ b/src/renderer/src/lib/pane-manager/pane-rendering-control.ts @@ -10,7 +10,11 @@ import { presentPaneViewport, resetWebglTextureAtlas } from './pane-webgl-renderer' -import { rebuildAttachedWebgl, reattachWebglIfNeeded } from './pane-webgl-reattach' +import { + clearPaneWebglContextLossForRetry, + rebuildAttachedWebgl, + reattachWebglIfNeeded +} from './pane-webgl-reattach' import { releaseHiddenWebglRetention, tryRetainHiddenPanesWebgl @@ -81,13 +85,11 @@ export function resumePaneRendering( releaseHiddenWebglRetention(retentionOwner) } for (const pane of panes) { - // Why: resume (worktree foreground, window wake) is the WebGL retry - // boundary — Chromium may have restored the GPU process since a context - // loss, and bounding retries to resume events cannot loop on live loss. clearTerminalWebglAttachBackoff(pane) const rebuildDeferred = pane.webglRebuildDeferred === true pane.webglAttachmentDeferred = false - pane.webglDisabledAfterContextLoss = false + // Reveal can retry before the next resume, so both paths share the bounded loss policy. + clearPaneWebglContextLossForRetry(pane) pane.webglRebuildDeferred = false if (pane.webglAddon && isPaneWebglContextLost(pane)) { disposeWebgl(pane) diff --git a/src/renderer/src/lib/pane-manager/pane-rendering-diagnostics.ts b/src/renderer/src/lib/pane-manager/pane-rendering-diagnostics.ts index 15faad3136c..4d3d32a40cc 100644 --- a/src/renderer/src/lib/pane-manager/pane-rendering-diagnostics.ts +++ b/src/renderer/src/lib/pane-manager/pane-rendering-diagnostics.ts @@ -1,5 +1,6 @@ import type { ManagedPaneInternal, PaneRenderingDiagnostics } from './pane-manager-types' import { getTerminalWebglAutoDecision } from './terminal-webgl-auto-policy' +import { countPaneWebglContextLosses } from './pane-webgl-context-loss-policy' export function collectPaneRenderingDiagnostics( panes: Map @@ -10,6 +11,7 @@ export function collectPaneRenderingDiagnostics( gpuRenderingEnabled: pane.gpuRenderingEnabled, webglAttachmentDeferred: pane.webglAttachmentDeferred, webglDisabledAfterContextLoss: pane.webglDisabledAfterContextLoss, + webglContextLossesInWindow: countPaneWebglContextLosses(pane), webglAttachFailedSinceRecovery: pane.webglAttachFailedSinceRecovery === true, hasComplexScriptOutput: pane.hasComplexScriptOutput, terminalWebglAutoDecision: getTerminalWebglAutoDecision(), diff --git a/src/renderer/src/lib/pane-manager/pane-reveal-repaint.test.ts b/src/renderer/src/lib/pane-manager/pane-reveal-repaint.test.ts index 8604a0fd6dc..50c5784dd30 100644 --- a/src/renderer/src/lib/pane-manager/pane-reveal-repaint.test.ts +++ b/src/renderer/src/lib/pane-manager/pane-reveal-repaint.test.ts @@ -235,5 +235,20 @@ describe('schedulePaneRevealRepaint', () => { expect(pane.webglAddon).not.toBeNull() expect(pane.terminal.refresh).toHaveBeenCalled() }) + + it('reattaches a pane that lost WebGL while hidden when its tab is revealed', () => { + const pane = createPane() + pane.webglDisabledAfterContextLoss = true + pane.webglContextLossTimestamps = [Date.now()] + + schedulePaneRevealPresent(() => [pane]) + flushFrame() + flushFrame() + + expect(pane.webglDisabledAfterContextLoss).toBe(false) + expect(pane.webglAddon).not.toBeNull() + expect(pane.terminal.refresh).toHaveBeenCalledTimes(2) + expect(pane.terminal.refresh).toHaveBeenNthCalledWith(2, 0, 23) + }) }) }) diff --git a/src/renderer/src/lib/pane-manager/pane-terminal-gpu-acceleration.ts b/src/renderer/src/lib/pane-manager/pane-terminal-gpu-acceleration.ts index 86b474196f7..099e3a40c82 100644 --- a/src/renderer/src/lib/pane-manager/pane-terminal-gpu-acceleration.ts +++ b/src/renderer/src/lib/pane-manager/pane-terminal-gpu-acceleration.ts @@ -6,6 +6,7 @@ import { shouldUseTerminalWebgl } from './pane-webgl-renderer' import { safeFit } from './pane-tree-ops' +import { resetPaneWebglContextLosses } from './pane-webgl-context-loss-policy' export function applyTerminalGpuAcceleration( panes: Iterable, @@ -26,6 +27,7 @@ export function applyTerminalGpuAcceleration( // renderer; context-loss and attach-failure latches from the old mode // should not pin DOM. pane.webglDisabledAfterContextLoss = false + resetPaneWebglContextLosses(pane) pane.webglAttachFailedSinceRecovery = false } if (!shouldUseTerminalWebgl(pane)) { diff --git a/src/renderer/src/lib/pane-manager/pane-webgl-context-loss-policy.ts b/src/renderer/src/lib/pane-manager/pane-webgl-context-loss-policy.ts new file mode 100644 index 00000000000..be81b824a40 --- /dev/null +++ b/src/renderer/src/lib/pane-manager/pane-webgl-context-loss-policy.ts @@ -0,0 +1,38 @@ +import type { ManagedPaneInternal } from './pane-manager-types' + +/** Retry a pane after a few transient losses, but stop retrying a persistently unstable context. */ +export const WEBGL_CONTEXT_LOSS_RETRY_LIMIT = 3 +export const WEBGL_CONTEXT_LOSS_RETRY_WINDOW_MS = 60_000 + +function recentContextLosses(pane: ManagedPaneInternal, now: number): number[] { + const cutoff = now - WEBGL_CONTEXT_LOSS_RETRY_WINDOW_MS + return (pane.webglContextLossTimestamps ?? []).filter((timestamp) => timestamp > cutoff) +} + +export function prunePaneWebglContextLosses(pane: ManagedPaneInternal, now = Date.now()): number { + const losses = recentContextLosses(pane, now) + pane.webglContextLossTimestamps = losses + return losses.length +} + +export function countPaneWebglContextLosses(pane: ManagedPaneInternal, now = Date.now()): number { + return recentContextLosses(pane, now).length +} + +export function recordPaneWebglContextLoss(pane: ManagedPaneInternal, now = Date.now()): number { + const losses = recentContextLosses(pane, now) + losses.push(now) + pane.webglContextLossTimestamps = losses + return losses.length +} + +export function canRetryPaneWebglAfterContextLoss( + pane: ManagedPaneInternal, + now = Date.now() +): boolean { + return prunePaneWebglContextLosses(pane, now) < WEBGL_CONTEXT_LOSS_RETRY_LIMIT +} + +export function resetPaneWebglContextLosses(pane: ManagedPaneInternal): void { + pane.webglContextLossTimestamps = undefined +} diff --git a/src/renderer/src/lib/pane-manager/pane-webgl-context-recovery.test.ts b/src/renderer/src/lib/pane-manager/pane-webgl-context-recovery.test.ts index 128057e463d..2897c1627de 100644 --- a/src/renderer/src/lib/pane-manager/pane-webgl-context-recovery.test.ts +++ b/src/renderer/src/lib/pane-manager/pane-webgl-context-recovery.test.ts @@ -2,6 +2,8 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { setTerminalWebglDiagnosticRecorder } from '../../../../shared/terminal-webgl-diagnostics' import type { ManagedPaneInternal } from './pane-manager-types' import { resumePaneRendering, suspendPaneRendering } from './pane-rendering-control' +import { collectPaneRenderingDiagnostics } from './pane-rendering-diagnostics' +import { schedulePaneRevealPresent } from './pane-reveal-repaint' import { attachWebgl, resetTerminalWebglSuggestion } from './pane-webgl-renderer' import { rebuildAttachedWebgl } from './pane-webgl-reattach' @@ -168,6 +170,44 @@ describe('terminal WebGL context recovery', () => { expect(pane.webglAddon).toBeNull() }) + it('stops resume retries after repeated losses in the retry window', () => { + const pane = createPane() + + attachWebgl(pane) + fireContextLoss(pane) + resumePaneRendering([pane]) + fireContextLoss(pane) + resumePaneRendering([pane]) + fireContextLoss(pane) + + expect(pane.webglContextLossTimestamps).toHaveLength(3) + expect(pane.webglDisabledAfterContextLoss).toBe(true) + expect(pane.webglAddon).toBeNull() + + resumePaneRendering([pane]) + + expect(pane.webglDisabledAfterContextLoss).toBe(true) + expect(pane.webglAddon).toBeNull() + expect(pane.terminal.loadAddon).toHaveBeenCalledTimes(3) + }) + + it('stops reveal retries after repeated losses in the retry window', () => { + const pane = createPane() + + attachWebgl(pane) + fireContextLoss(pane) + resumePaneRendering([pane]) + fireContextLoss(pane) + resumePaneRendering([pane]) + fireContextLoss(pane) + + schedulePaneRevealPresent(() => [pane]) + + expect(pane.webglDisabledAfterContextLoss).toBe(true) + expect(pane.webglAddon).toBeNull() + expect(pane.terminal.loadAddon).toHaveBeenCalledTimes(3) + }) + // Why exact keys: a GPU death loses every pane's context at once and the // crash ring coalesces the repeats, so the population has to survive on the // payload. It must use the same names the fit-retry crumb uses, or one ring @@ -186,8 +226,25 @@ describe('terminal WebGL context recovery', () => { expect(recorded).toEqual([ { kind: 'webgl-context-loss', - detail: { paneId: 1, livePanes: expect.any(Number), livePaneManagers: expect.any(Number) } + detail: { + paneId: 1, + lossesInWindow: 1, + livePanes: expect.any(Number), + livePaneManagers: expect.any(Number) + } } ]) }) + + it('reports only recent context losses in rendering diagnostics', () => { + const now = 1_700_000_000_000 + vi.spyOn(Date, 'now').mockReturnValue(now) + const pane = createPane() + pane.webglContextLossTimestamps = [now - 60_001, now - 1_000] + + const [diagnostics] = collectPaneRenderingDiagnostics(new Map([[pane.id, pane]])) + + expect(diagnostics.webglContextLossesInWindow).toBe(1) + expect(pane.webglContextLossTimestamps).toEqual([now - 60_001, now - 1_000]) + }) }) diff --git a/src/renderer/src/lib/pane-manager/pane-webgl-reattach.ts b/src/renderer/src/lib/pane-manager/pane-webgl-reattach.ts index 98c5c10e079..1bb79d6f6e0 100644 --- a/src/renderer/src/lib/pane-manager/pane-webgl-reattach.ts +++ b/src/renderer/src/lib/pane-manager/pane-webgl-reattach.ts @@ -1,8 +1,20 @@ import type { ManagedPaneInternal } from './pane-manager-types' import { attachWebgl, clearTerminalWebglAttachBackoff, disposeWebgl } from './pane-webgl-renderer' +import { canRetryPaneWebglAfterContextLoss } from './pane-webgl-context-loss-policy' + +export function clearPaneWebglContextLossForRetry(pane: ManagedPaneInternal): boolean { + if (!pane.webglDisabledAfterContextLoss) { + return true + } + if (!canRetryPaneWebglAfterContextLoss(pane)) { + return false + } + pane.webglDisabledAfterContextLoss = false + return true +} export function reattachWebglIfNeeded(pane: ManagedPaneInternal): void { - if (pane.gpuRenderingEnabled && !pane.webglAddon && !pane.webglDisabledAfterContextLoss) { + if (pane.gpuRenderingEnabled && !pane.webglAddon && clearPaneWebglContextLossForRetry(pane)) { attachWebgl(pane) } } diff --git a/src/renderer/src/lib/pane-manager/pane-webgl-renderer.ts b/src/renderer/src/lib/pane-manager/pane-webgl-renderer.ts index 463bc76e53a..e15becd0cff 100644 --- a/src/renderer/src/lib/pane-manager/pane-webgl-renderer.ts +++ b/src/renderer/src/lib/pane-manager/pane-webgl-renderer.ts @@ -14,6 +14,7 @@ import { import { safeFit, safeFitAndThen } from './pane-fit' import { setPaneFitWebglAttachHook } from './pane-fit-webgl-attach-signal' import { repairPaneWebglCanvasDprMismatch } from './terminal-canvas-dpr-repair' +import { recordPaneWebglContextLoss } from './pane-webgl-context-loss-policy' export const ENABLE_WEBGL_RENDERER = true let suggestedRendererType: 'dom' | undefined @@ -340,15 +341,15 @@ export function attachWebgl(pane: ManagedPaneInternal): void { // once, and the crash-report ring coalesces repeats, so the count has to // be in the payload rather than in the number of crumbs. const census = getLivePaneCensus() + const lossesInWindow = recordPaneWebglContextLoss(pane) recordTerminalWebglDiagnostic('webgl-context-loss', { paneId: pane.id, + lossesInWindow, livePanes: census.panes, livePaneManagers: census.managers }) - // Why: Chromium starts reclaiming terminal contexts under pressure. - // Recreating WebGL for this pane can loop context loss and leave xterm - // visually blank, so keep the pane on the DOM renderer until the next - // rendering resume (worktree foreground / window wake) retries it. + // Why: context loss switches this pane to DOM until the next resume or + // settled reveal; the bounded loss window refuses unstable retries. pane.webglDisabledAfterContextLoss = true disposeWebgl(pane, { refreshDimensions: true }) }) From 614d2d4a28b828a1d9c949d74fe5a69d187a8069 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 26 Aug 2026 15:33:11 -0700 Subject: [PATCH 09/19] fix(cmd+j): always enable See more for soft preview hints (#16661) The leading preview section now shows an actionable 'See more' button even when all rows fit within the hard cap, letting users expand and browse more tabs without scrolling past the worktrees section. --- .../src/components/WorktreeJumpPalette.tsx | 12 +++------ ...jump-palette-interleaved-sections.test.tsx | 27 ++++++++++++++++--- 2 files changed, 28 insertions(+), 11 deletions(-) diff --git a/src/renderer/src/components/WorktreeJumpPalette.tsx b/src/renderer/src/components/WorktreeJumpPalette.tsx index 3c3a7fce32d..2cbe63a9576 100644 --- a/src/renderer/src/components/WorktreeJumpPalette.tsx +++ b/src/renderer/src/components/WorktreeJumpPalette.tsx @@ -2303,14 +2303,10 @@ function WorktreeJumpPaletteContent({ pushLeadingHeader() appendPaletteListEntries(entries, multiPrimaryLayout.leadingPreview as PaletteItem[]) // Soft more for the leading section (rows resuming below + hard-cap tail). - // Why: only actionable when rows are actually hidden — with the whole - // section already rendered below, expanding would just reshuffle rows. - pushOverflowHint( - leadingHintId, - multiPrimaryLayout.leadingMoreCount, - multiPrimaryLayout.leadingHardOverflowCount > 0 - ? () => handleExpandSection(leadingSectionKey) - : undefined + // Why: reveals the next batch into the leading preview so the user can + // keep browsing tabs without having to scroll past the worktrees section. + pushOverflowHint(leadingHintId, multiPrimaryLayout.leadingMoreCount, () => + handleExpandSection(leadingSectionKey) ) pushTrailingHeader() // Floor first, then remaining leading rows, then trailing rest — same order diff --git a/src/renderer/src/components/worktree-jump-palette-interleaved-sections.test.tsx b/src/renderer/src/components/worktree-jump-palette-interleaved-sections.test.tsx index 3f2c3025407..62055b5027b 100644 --- a/src/renderer/src/components/worktree-jump-palette-interleaved-sections.test.tsx +++ b/src/renderer/src/components/worktree-jump-palette-interleaved-sections.test.tsx @@ -456,7 +456,7 @@ describe('WorktreeJumpPalette interleaved primary sections', () => { expect(testContainer.textContent).toContain('10 more') }) - it('leaves the soft preview hint non-actionable when no rows are hidden', async () => { + it('expands soft preview when clicking See more even when all rows fit within the hard cap', async () => { await renderPalette(perfTabsPaletteProps(30)) await act(async () => { @@ -464,12 +464,33 @@ describe('WorktreeJumpPalette interleaved primary sections', () => { }) await flushEffects() - // All 30 tabs render (6 preview + 24 remainder), so expanding would only reorder rows. + // 6 preview tabs, 24 more follow below the worktrees section expect(testContainer.textContent).toContain('24 more') const seeMoreBtn = Array.from(testContainer.querySelectorAll('button')).find((btn) => btn.textContent?.includes('See more') ) - expect(seeMoreBtn).toBeUndefined() + expect(seeMoreBtn).toBeDefined() + + // Click See more: preview expands to 6 + 20 = 26 tabs, leaving 4 more + await act(async () => { + seeMoreBtn?.click() + }) + await flushEffects() + + expect(testContainer.textContent).toContain('4 more') + + const seeMoreBtn2 = Array.from(testContainer.querySelectorAll('button')).find((btn) => + btn.textContent?.includes('See more') + ) + expect(seeMoreBtn2).toBeDefined() + + // Click See more again: preview expands to 26 + 20 = 46 tabs (fits all 30), hint disappears + await act(async () => { + seeMoreBtn2?.click() + }) + await flushEffects() + + expect(testContainer.textContent).not.toContain('more') }) it('resets expanded section caps when query changes', async () => { From 015f904fca143c630291910d4096688b39e46ebb Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 26 Aug 2026 15:42:54 -0700 Subject: [PATCH 10/19] fix(codex): stop re-scanning all Codex session history on every launch (#16251) (#16593) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(codex): stop re-scanning all Codex session history on every launch (#16251) A launch deleted the backfill completion marker, and a marker could never be written while a Codex pane was open, so every launch re-derived "needs full scan" and walked the entire .codex/sessions tree — on Windows with a large history that read as a hung window. - v4 marker keeps a durable full-history baseline plus a bounded set of pending dates. v3 is read as a baseline, so upgrades pay no full scan. - A launch now marks dates pending instead of deleting the marker, and a full pass certifies the baseline even while a pane is still running; the live pane's own date just stays pending. - Pending dates are persisted, so an abnormal exit or a cross-midnight pane recovers a bounded window instead of a full walk. - A date-limited pass can only extend an existing baseline, never create one, so it can no longer certify history it never looked at. - Marker and index-heal target roots compare through normalizeRuntimePathForComparison, so Windows spellings of one directory stop invalidating each other. - Both append-only ledgers stream instead of readFileSync + whole-file JSON.parse, keeping the main thread responsive on large histories. * fix(codex): keep the backfill marker's full-scan demand durable Review follow-ups on the v4 backfill marker: - markCodexSessionBackfillMarkerPending no longer erases a persisted needsFullScan; the demand survives until a generation-current full walk retires it, and the function now reports it so the launch path folds it into its own in-memory flag (as @rumoii's #16252 does). - A full pass settles the whole pending set instead of subtracting the empty set, so a date a full walk provably covered stops forcing an extra bounded pass on every startup. - isCodexSessionBackfillDate does a real calendar check, so a corrupted marker cannot carry 2026/99/99. No age or future bound: the same guard gates rollout publication and a clock-skewed directory holds real sessions. - 'scans only the current date once a baseline exists' now has a second date directory, so it fails on a full walk instead of passing either way. --- ...untime-home-real-home-lane-routing.test.ts | 24 +- .../codex-accounts/runtime-home-service.ts | 28 +- ...untime-home-session-migration-pass.test.ts | 149 ++++++++ .../codex/codex-session-backfill-audit.ts | 55 +-- src/main/codex/codex-session-backfill-date.ts | 20 +- .../codex/codex-session-backfill-fs-mocks.ts | 117 ++++++ .../codex-session-backfill-marker.test.ts | 250 +++++++++++++ .../codex/codex-session-backfill-marker.ts | 279 +++++++++++--- .../codex-session-backfill-scan-dates.test.ts | 102 +++++ .../codex-session-backfill-scan-dates.ts | 112 ++++++ .../codex/codex-session-backfill-types.ts | 10 +- src/main/codex/codex-session-backfill.test.ts | 349 +++++++++--------- src/main/codex/codex-session-backfill.ts | 59 ++- .../codex-session-index-heal-state.test.ts | 154 ++++++++ .../codex/codex-session-index-heal-state.ts | 52 +-- src/main/codex/codex-session-index-heal.ts | 2 +- src/main/codex/codex-session-ledger-stream.ts | 55 +++ .../codex-session-migration-scheduler.test.ts | 87 ++++- .../codex-session-migration-scheduler.ts | 65 ++-- src/main/index.ts | 3 +- 20 files changed, 1565 insertions(+), 407 deletions(-) create mode 100644 src/main/codex-accounts/runtime-home-session-migration-pass.test.ts create mode 100644 src/main/codex/codex-session-backfill-fs-mocks.ts create mode 100644 src/main/codex/codex-session-backfill-marker.test.ts create mode 100644 src/main/codex/codex-session-backfill-scan-dates.test.ts create mode 100644 src/main/codex/codex-session-backfill-scan-dates.ts create mode 100644 src/main/codex/codex-session-index-heal-state.test.ts create mode 100644 src/main/codex/codex-session-ledger-stream.ts diff --git a/src/main/codex-accounts/runtime-home-real-home-lane-routing.test.ts b/src/main/codex-accounts/runtime-home-real-home-lane-routing.test.ts index 9bd55bba6e4..c8c3c407c0c 100644 --- a/src/main/codex-accounts/runtime-home-real-home-lane-routing.test.ts +++ b/src/main/codex-accounts/runtime-home-real-home-lane-routing.test.ts @@ -18,6 +18,18 @@ import { testState, writePaneRegistry } from './runtime-home-service-test-harness' +import { hasCompletedCodexSessionBackfillMarker } from '../codex/codex-session-backfill-marker' +import { getCodexSessionBackfillDate } from '../codex/codex-session-backfill-scan-dates' + +function expectBaselineKeptWithLaunchDatePending(markerPath: string): void { + const marker = JSON.parse(readFileSync(markerPath, 'utf-8')) as { + pendingScanDates?: unknown + } + expect(marker.pendingScanDates).toEqual([getCodexSessionBackfillDate()]) + expect( + hasCompletedCodexSessionBackfillMarker(markerPath, join(getSystemCodexHomePath(), 'sessions')) + ).toBe(true) +} vi.mock('electron', () => ({ app: { @@ -54,7 +66,9 @@ describe('CodexRuntimeHomeService', () => { const { CodexRuntimeHomeService } = await import('./runtime-home-service') const service = new CodexRuntimeHomeService(store as never) expect(service.prepareForCodexLaunch()).toBe(getRuntimeCodexHomePath()) - expect(existsSync(markerPath)).toBe(false) + expect( + hasCompletedCodexSessionBackfillMarker(markerPath, join(getSystemCodexHomePath(), 'sessions')) + ).toBe(false) service.finishHostSystemDefaultSessionMigrationPass() expect(service.beginHostSystemDefaultSessionMigrationLaunch(getRuntimeCodexHomePath())).toBe( true @@ -82,7 +96,7 @@ describe('CodexRuntimeHomeService', () => { expect(service.beginHostSystemDefaultSessionMigrationLaunch(getRuntimeCodexHomePath())).toBe( false ) - expect(existsSync(markerPath)).toBe(false) + expectBaselineKeptWithLaunchDatePending(markerPath) service.prepareForCodexLaunch() expect(service.beginHostSystemDefaultSessionMigrationLaunch(getRuntimeCodexHomePath())).toBe( false @@ -101,7 +115,7 @@ describe('CodexRuntimeHomeService', () => { expect(service.beginHostSystemDefaultSessionMigrationLaunch(null, { reattached: true })).toBe( false ) - expect(existsSync(markerPath)).toBe(false) + expectBaselineKeptWithLaunchDatePending(markerPath) store.updateSettings({ codexSessionSourceHome: { host: join(testState.fakeHomeDir, 'moved-history'), wsl: {} } }) @@ -140,7 +154,9 @@ describe('CodexRuntimeHomeService', () => { mkdirSync(join(testState.userDataDir, 'codex-session-backfill'), { recursive: true }) writeFileSync(markerPath, '{}\n', 'utf-8') expect(service.prepareForCodexLaunch()).toBe(getRuntimeCodexHomePath()) - expect(existsSync(markerPath)).toBe(false) + expect( + hasCompletedCodexSessionBackfillMarker(markerPath, join(getSystemCodexHomePath(), 'sessions')) + ).toBe(false) expect(service.beginHostSystemDefaultSessionMigrationLaunch(getRuntimeCodexHomePath())).toBe( true ) diff --git a/src/main/codex-accounts/runtime-home-service.ts b/src/main/codex-accounts/runtime-home-service.ts index 3049a2a70b6..1d6f80dbafc 100644 --- a/src/main/codex-accounts/runtime-home-service.ts +++ b/src/main/codex-accounts/runtime-home-service.ts @@ -71,8 +71,10 @@ import { getDefaultWslDistro, getWslHome } from '../wsl' import { hasCustomCodexHomeOverrideForLaunch } from '../codex/codex-real-home-path' import { hasCompletedCodexSessionBackfillMarker, - invalidateCodexSessionBackfillMarker + markCodexSessionBackfillMarkerPending } from '../codex/codex-session-backfill-marker' +import { getCodexSessionBackfillDate } from '../codex/codex-session-backfill-scan-dates' +import type { CodexSessionBackfillDate } from '../codex/codex-session-backfill-types' import { resolveCodexSessionBackfillPaths } from '../codex/codex-session-backfill' import { ManagedCodexHomeTemporarilyUnavailableError, @@ -305,18 +307,30 @@ export class CodexRuntimeHomeService { ) } - prepareHostSystemDefaultSessionMigrationPass(): boolean { + prepareHostSystemDefaultSessionMigrationPass( + scanDates: readonly CodexSessionBackfillDate[] = [] + ): boolean { const paths = resolveCodexSessionBackfillPaths( resolveHostCodexSessionSourceHome(this.store.getSettings()) ) + const target = normalizeRuntimePathForComparison(paths.systemSessionsRoot) if ( this.hostSystemDefaultSessionMigrationPending && - this.pendingHostSystemDefaultSessionMigrationTarget !== paths.systemSessionsRoot + this.pendingHostSystemDefaultSessionMigrationTarget !== target ) { this.pendingHostSystemDefaultSessionMigrationNeedsFullScan = true - this.pendingHostSystemDefaultSessionMigrationTarget = paths.systemSessionsRoot + this.pendingHostSystemDefaultSessionMigrationTarget = target } - invalidateCodexSessionBackfillMarker(paths.markerPath) + // Why: the launch creates rollouts for these dates; record them durably so a + // force-quit recovers a bounded window instead of re-walking all history. + const markerOwesFullScan = markCodexSessionBackfillMarkerPending( + paths.markerPath, + paths.systemSessionsRoot, + scanDates.length > 0 ? scanDates : [getCodexSessionBackfillDate()] + ) + // Why: the marker is the only place an overflowed pending window survives a + // restart, so its demand has to reach this pass rather than die in the file. + this.pendingHostSystemDefaultSessionMigrationNeedsFullScan ||= markerOwesFullScan return this.pendingHostSystemDefaultSessionMigrationNeedsFullScan } @@ -526,7 +540,9 @@ export class CodexRuntimeHomeService { ) this.pendingHostSystemDefaultSessionMigrationNeedsFullScan = !hasCompletedCodexSessionBackfillMarker(paths.markerPath, paths.systemSessionsRoot) - this.pendingHostSystemDefaultSessionMigrationTarget = paths.systemSessionsRoot + this.pendingHostSystemDefaultSessionMigrationTarget = normalizeRuntimePathForComparison( + paths.systemSessionsRoot + ) this.hostSystemDefaultSessionMigrationPending = true } return this.prepareHostSystemDefaultSessionMigrationPass() diff --git a/src/main/codex-accounts/runtime-home-session-migration-pass.test.ts b/src/main/codex-accounts/runtime-home-session-migration-pass.test.ts new file mode 100644 index 00000000000..640c35a8e60 --- /dev/null +++ b/src/main/codex-accounts/runtime-home-session-migration-pass.test.ts @@ -0,0 +1,149 @@ +import { beforeEach, afterEach, describe, expect, it, vi } from 'vitest' +import { mkdirSync, readFileSync, writeFileSync } from 'node:fs' +import type * as NodeOs from 'node:os' +import { join } from 'node:path' +import { createSettings } from './runtime-home-settings-test-fixtures' +import { + createStore, + getRuntimeCodexHomePath, + setupRuntimeHomeTest, + teardownRuntimeHomeTest, + testState +} from './runtime-home-service-test-harness' +import { getCodexSessionBackfillDate } from '../codex/codex-session-backfill-scan-dates' +import type { CodexSessionBackfillDate } from '../codex/codex-session-backfill-types' + +vi.mock('electron', () => ({ + app: { + getPath: () => testState.userDataDir + } +})) + +vi.mock('node:os', async () => { + const actual = await vi.importActual('node:os') + return { + ...actual, + homedir: () => testState.fakeHomeDir + } +}) + +const CUSTOM_HISTORY_HOME = 'C:\\Users\\Me\\.codex' + +function getMarkerPath(): string { + return join(testState.userDataDir, 'codex-session-backfill', 'backfill-complete.json') +} + +function writeBaselineMarker(systemCodexHomePath: string, needsFullScan = false): void { + mkdirSync(join(testState.userDataDir, 'codex-session-backfill'), { recursive: true }) + writeFileSync( + getMarkerPath(), + `${JSON.stringify({ + version: 4, + systemSessionsRoot: join(systemCodexHomePath, 'sessions'), + coverage: 'full', + baselineScannedFiles: 5, + pendingScanDates: [], + needsFullScan, + summary: { scannedFiles: 5 } + })}\n`, + 'utf-8' + ) +} + +function readPendingScanDates(): unknown { + return (JSON.parse(readFileSync(getMarkerPath(), 'utf-8')) as { pendingScanDates?: unknown }) + .pendingScanDates +} + +describe('host system default session migration pass preparation', () => { + beforeEach(() => { + setupRuntimeHomeTest() + }) + + afterEach(() => { + teardownRuntimeHomeTest() + }) + + it('records the launch date and keeps the baseline instead of deleting it', async () => { + writeBaselineMarker(CUSTOM_HISTORY_HOME) + const store = createStore( + createSettings({ codexSessionSourceHome: { host: CUSTOM_HISTORY_HOME, wsl: {} } }) + ) + const { CodexRuntimeHomeService } = await import('./runtime-home-service') + const service = new CodexRuntimeHomeService(store as never) + + expect(service.prepareHostSystemDefaultSessionMigrationPass()).toBe(false) + + expect(readPendingScanDates()).toEqual([getCodexSessionBackfillDate()]) + expect(JSON.parse(readFileSync(getMarkerPath(), 'utf-8'))).toMatchObject({ coverage: 'full' }) + }) + + it('persists every date a scheduled pass reports so a force-quit stays bounded', async () => { + writeBaselineMarker(CUSTOM_HISTORY_HOME) + const store = createStore( + createSettings({ codexSessionSourceHome: { host: CUSTOM_HISTORY_HOME, wsl: {} } }) + ) + const { CodexRuntimeHomeService } = await import('./runtime-home-service') + const service = new CodexRuntimeHomeService(store as never) + const spannedDates: CodexSessionBackfillDate[] = [ + ['2026', '08', '05'], + ['2026', '08', '06'] + ] + + service.prepareHostSystemDefaultSessionMigrationPass(spannedDates) + + expect(readPendingScanDates()).toEqual(spannedDates) + }) + + it('does not demand a full scan when the same history home is spelled differently', async () => { + writeBaselineMarker(CUSTOM_HISTORY_HOME) + const store = createStore( + createSettings({ codexSessionSourceHome: { host: CUSTOM_HISTORY_HOME, wsl: {} } }) + ) + const { CodexRuntimeHomeService } = await import('./runtime-home-service') + const service = new CodexRuntimeHomeService(store as never) + expect(service.beginHostSystemDefaultSessionMigrationLaunch(getRuntimeCodexHomePath())).toBe( + false + ) + + store.updateSettings({ + codexSessionSourceHome: { host: 'c:/users/me/.codex', wsl: {} } + }) + + expect(service.prepareHostSystemDefaultSessionMigrationPass()).toBe(false) + }) + + it('carries a full-scan demand persisted by an earlier launch into this pass', async () => { + writeBaselineMarker(CUSTOM_HISTORY_HOME, true) + const store = createStore( + createSettings({ codexSessionSourceHome: { host: CUSTOM_HISTORY_HOME, wsl: {} } }) + ) + const { CodexRuntimeHomeService } = await import('./runtime-home-service') + const service = new CodexRuntimeHomeService(store as never) + + expect(service.prepareHostSystemDefaultSessionMigrationPass()).toBe(true) + + // Recording this launch must not erase the demand the marker still carries. + expect(JSON.parse(readFileSync(getMarkerPath(), 'utf-8'))).toMatchObject({ + needsFullScan: true + }) + }) + + it('still demands a full scan when the history home really moves', async () => { + writeBaselineMarker(CUSTOM_HISTORY_HOME) + const store = createStore( + createSettings({ codexSessionSourceHome: { host: CUSTOM_HISTORY_HOME, wsl: {} } }) + ) + const { CodexRuntimeHomeService } = await import('./runtime-home-service') + const service = new CodexRuntimeHomeService(store as never) + expect(service.beginHostSystemDefaultSessionMigrationLaunch(getRuntimeCodexHomePath())).toBe( + false + ) + + store.updateSettings({ + codexSessionSourceHome: { host: 'C:\\Users\\Me\\moved-codex', wsl: {} } + }) + + expect(service.prepareHostSystemDefaultSessionMigrationPass()).toBe(true) + }) +}) diff --git a/src/main/codex/codex-session-backfill-audit.ts b/src/main/codex/codex-session-backfill-audit.ts index 1bec593821a..8f84966f0cb 100644 --- a/src/main/codex/codex-session-backfill-audit.ts +++ b/src/main/codex/codex-session-backfill-audit.ts @@ -1,9 +1,9 @@ import { createHash, randomUUID } from 'node:crypto' -import { createReadStream, type Stats } from 'node:fs' +import type { Stats } from 'node:fs' import { appendFile, mkdir } from 'node:fs/promises' import { dirname } from 'node:path' -import { createInterface } from 'node:readline' import { normalizeRuntimePathForComparison } from '../../shared/cross-platform-path' +import { streamCodexSessionLedgerRecords } from './codex-session-ledger-stream' import type { CodexSessionBackfillSummary } from './codex-session-backfill-types' export type CodexSessionBackfillAuditWriter = (record: Record) => Promise @@ -68,38 +68,23 @@ export async function readCodexSessionBackfillAuditCoverage( diagnosticEventIds: new Set(), hasRunSummary: false } - const input = createReadStream(auditLogPath, { encoding: 'utf-8' }) - const lines = createInterface({ input, crlfDelay: Infinity }) - try { - for await (const raw of lines) { - try { - const parsed: unknown = JSON.parse(raw) - if (!parsed || typeof parsed !== 'object' || Array.isArray(parsed)) { - continue - } - const record = parsed as Record - coverage.hasRunSummary ||= record.action === 'run-summary' - if ( - typeof record.action === 'string' && - HEAL_AUDIT_ACTIONS.has(record.action) && - typeof record.fileEventId === 'string' - ) { - coverage.fileEventIds.add(record.fileEventId) - } - if ( - typeof record.action === 'string' && - DIAGNOSTIC_AUDIT_ACTIONS.has(record.action) && - typeof record.diagnosticEventId === 'string' - ) { - coverage.diagnosticEventIds.add(record.diagnosticEventId) - } - } catch { - // Torn audit tails are quarantined by the writer's leading newline. - } + for await (const record of streamCodexSessionLedgerRecords(auditLogPath, { + throwOnReadFailure: true + })) { + coverage.hasRunSummary ||= record.action === 'run-summary' + if ( + typeof record.action === 'string' && + HEAL_AUDIT_ACTIONS.has(record.action) && + typeof record.fileEventId === 'string' + ) { + coverage.fileEventIds.add(record.fileEventId) } - } catch (error) { - if (!isNotFoundError(error)) { - throw error + if ( + typeof record.action === 'string' && + DIAGNOSTIC_AUDIT_ACTIONS.has(record.action) && + typeof record.diagnosticEventId === 'string' + ) { + coverage.diagnosticEventIds.add(record.diagnosticEventId) } } return coverage @@ -192,7 +177,3 @@ export async function recordExistingCodexSessionForHeal( ...(fileEventId ? { fileEventId } : {}) }) } - -function isNotFoundError(error: unknown): boolean { - return (error as NodeJS.ErrnoException | null)?.code === 'ENOENT' -} diff --git a/src/main/codex/codex-session-backfill-date.ts b/src/main/codex/codex-session-backfill-date.ts index 8c2157d9d61..9cd39e6ddd8 100644 --- a/src/main/codex/codex-session-backfill-date.ts +++ b/src/main/codex/codex-session-backfill-date.ts @@ -1,30 +1,18 @@ import { join, relative, sep } from 'node:path' +import { isCodexSessionBackfillDate } from './codex-session-backfill-scan-dates' import { listCodexSessionJsonlFilesIncrementally } from './codex-session-file-listing' import type { CodexSessionBackfillDate, CodexSessionBackfillOptions } from './codex-session-backfill-types' -export function getCodexSessionBackfillDate(date = new Date()): CodexSessionBackfillDate { - return [ - String(date.getUTCFullYear()).padStart(4, '0'), - String(date.getUTCMonth() + 1).padStart(2, '0'), - String(date.getUTCDate()).padStart(2, '0') - ] -} - export function isCodexSessionRolloutPath(sessionsRoot: string, filePath: string): boolean { const pathParts = relative(sessionsRoot, filePath).split(sep) if (pathParts.length !== 4) { return false } const [year, month, day, fileName] = pathParts - return ( - /^\d{4}$/.test(year) && - /^\d{2}$/.test(month) && - /^\d{2}$/.test(day) && - /^rollout-.+\.jsonl$/.test(fileName) - ) + return isCodexSessionBackfillDate([year, month, day]) && /^rollout-.+\.jsonl$/.test(fileName) } export async function* listCodexSessionBackfillFilesForDates( @@ -54,9 +42,7 @@ function resolveCodexSessionBackfillDateRoots( return [sessionsRoot] } return scanDates - .filter( - ([year, month, day]) => /^\d{4}$/.test(year) && /^\d{2}$/.test(month) && /^\d{2}$/.test(day) - ) + .filter(isCodexSessionBackfillDate) .map(([year, month, day]) => join(sessionsRoot, year, month, day)) } diff --git a/src/main/codex/codex-session-backfill-fs-mocks.ts b/src/main/codex/codex-session-backfill-fs-mocks.ts new file mode 100644 index 00000000000..40f352cc337 --- /dev/null +++ b/src/main/codex/codex-session-backfill-fs-mocks.ts @@ -0,0 +1,117 @@ +import type * as NodeFs from 'node:fs' +import type * as NodeFsPromises from 'node:fs/promises' + +// Fault-injection doubles for the backfill tests. Kept out of the spec files so +// several suites can drive the same failure modes from one switchboard. + +export const fsMockState = { + failLink: false, + failLinkTransiently: false, + failLinkPermission: false, + raceTargetIntoExistence: false, + failMarkerRm: false, + failMarkerReplacement: false, + failAuditMkdirOnce: false, + failAuditWrites: false, + failMkdirPath: null as string | null, + failDirectoryPath: null as string | null, + failLstatPath: null as string | null +} + +export function resetCodexSessionBackfillFsMocks(): void { + fsMockState.failLink = false + fsMockState.failLinkTransiently = false + fsMockState.failLinkPermission = false + fsMockState.raceTargetIntoExistence = false + fsMockState.failMarkerRm = false + fsMockState.failMarkerReplacement = false + fsMockState.failAuditMkdirOnce = false + fsMockState.failAuditWrites = false + fsMockState.failMkdirPath = null + fsMockState.failDirectoryPath = null + fsMockState.failLstatPath = null +} + +function errnoError(message: string, code: string): NodeJS.ErrnoException { + const error = new Error(message) as NodeJS.ErrnoException + error.code = code + return error +} + +function isMarkerPath(value: unknown): boolean { + return ( + String(value).includes('codex-session-backfill') && + String(value).endsWith('backfill-complete.json') + ) +} + +export function createNodeFsMock(actual: typeof NodeFs): typeof NodeFs { + return { + ...actual, + existsSync: (...args: Parameters) => + args[0] === fsMockState.failLstatPath ? false : actual.existsSync(...args), + rmSync: (...args: Parameters) => { + if (fsMockState.failMarkerRm && isMarkerPath(args[0])) { + throw errnoError('EACCES: marker removal failed', 'EACCES') + } + return actual.rmSync(...args) + }, + renameSync: (...args: Parameters) => { + if (fsMockState.failMarkerReplacement && isMarkerPath(args[1])) { + throw errnoError('EACCES: marker replacement failed', 'EACCES') + } + return actual.renameSync(...args) + } + } +} + +export function createNodeFsPromisesMock(actual: typeof NodeFsPromises): typeof NodeFsPromises { + return { + ...actual, + mkdir: (...args: Parameters) => { + if (args[0] === fsMockState.failMkdirPath) { + throw errnoError('EACCES: target directory inaccessible', 'EACCES') + } + if (fsMockState.failAuditMkdirOnce && String(args[0]).includes('codex-session-backfill')) { + fsMockState.failAuditMkdirOnce = false + throw errnoError('EACCES: transient audit directory failure', 'EACCES') + } + return actual.mkdir(...args) + }, + appendFile: (...args: Parameters) => { + if (fsMockState.failAuditWrites && String(args[0]).includes('codex-session-backfill')) { + throw errnoError('ENOSPC: audit write failed', 'ENOSPC') + } + return actual.appendFile(...args) + }, + lstat: (...args: Parameters) => { + if (args[0] === fsMockState.failLstatPath) { + throw errnoError('EACCES: path inaccessible', 'EACCES') + } + return actual.lstat(...args) + }, + link: async (...args: Parameters) => { + if (fsMockState.raceTargetIntoExistence && String(args[0]).includes('codex-runtime-home')) { + fsMockState.raceTargetIntoExistence = false + await actual.writeFile(args[1], 'concurrent target\n', 'utf-8') + throw errnoError('EEXIST: concurrent target', 'EEXIST') + } + if (fsMockState.failLink && String(args[0]).includes('codex-runtime-home')) { + throw errnoError('EXDEV: cross-device link', 'EXDEV') + } + if (fsMockState.failLinkTransiently && String(args[0]).includes('codex-runtime-home')) { + throw errnoError('EIO: transient hardlink failure', 'EIO') + } + if (fsMockState.failLinkPermission && String(args[0]).includes('codex-runtime-home')) { + throw errnoError('EACCES: hardlink permission denied', 'EACCES') + } + return actual.link(...args) + }, + opendir: (...args: Parameters) => { + if (args[0] === fsMockState.failDirectoryPath) { + throw errnoError('EACCES: directory unreadable', 'EACCES') + } + return actual.opendir(...args) + } + } as typeof NodeFsPromises +} diff --git a/src/main/codex/codex-session-backfill-marker.test.ts b/src/main/codex/codex-session-backfill-marker.test.ts new file mode 100644 index 00000000000..5540bd53a4e --- /dev/null +++ b/src/main/codex/codex-session-backfill-marker.test.ts @@ -0,0 +1,250 @@ +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { + captureCodexSessionBackfillMarkerGeneration, + hasCompletedCodexSessionBackfillMarker, + markCodexSessionBackfillMarkerPending, + readCodexSessionBackfillBaseline, + writeCodexSessionBackfillMarker +} from './codex-session-backfill-marker' +import { + getCodexSessionBackfillDate, + getCodexSessionBackfillDatesBetween +} from './codex-session-backfill-scan-dates' +import type { + CodexSessionBackfillDate, + CodexSessionBackfillSummary +} from './codex-session-backfill-types' + +const WINDOWS_ROOT = 'C:\\Users\\Me\\.codex\\sessions' +const TODAY = getCodexSessionBackfillDate() +const LAUNCH_DATE: CodexSessionBackfillDate = ['2026', '08', '05'] + +let stateDir: string +let markerPath: string + +function createSummary(scannedFiles = 3): CodexSessionBackfillSummary { + return { + stopped: false, + scannedFiles, + linkedFiles: scannedFiles, + copiedFiles: 0, + skippedExistingFiles: 0, + skippedUnexpectedFiles: 0, + skippedSymlinkFiles: 0, + skippedUnsupportedFilesystemFiles: 0, + failedDirectories: 0, + failedFiles: 0, + failedHealAuditRecords: 0 + } +} + +function writeFullBaseline( + root: string, + options: { coveredScanDates?: readonly CodexSessionBackfillDate[]; retain?: boolean } = {} +): void { + writeCodexSessionBackfillMarker( + markerPath, + root, + createSummary(), + captureCodexSessionBackfillMarkerGeneration(), + { + coverage: 'full', + coveredScanDates: options.coveredScanDates ?? [], + retainPendingScanDates: options.retain + } + ) +} + +/** More dates than MAX_PENDING_SCAN_DATES, so the marker gives up on bounding. */ +function overflowingDates(): CodexSessionBackfillDate[] { + return getCodexSessionBackfillDatesBetween( + new Date(Date.UTC(2026, 6, 1)), + new Date(Date.UTC(2026, 7, 9)) + ) +} + +function readMarker(): Record { + return JSON.parse(readFileSync(markerPath, 'utf-8')) as Record +} + +beforeEach(() => { + stateDir = mkdtempSync(join(tmpdir(), 'orca-codex-marker-')) + markerPath = join(stateDir, 'backfill-complete.json') +}) + +afterEach(() => { + rmSync(stateDir, { recursive: true, force: true }) +}) + +describe('codex session backfill marker', () => { + it('treats Windows spellings of one directory as the same target', () => { + writeFullBaseline(WINDOWS_ROOT) + + for (const alias of ['C:/Users/Me/.codex/sessions', 'c:\\users\\me\\.codex\\sessions']) { + expect(hasCompletedCodexSessionBackfillMarker(markerPath, alias)).toBe(true) + } + expect( + hasCompletedCodexSessionBackfillMarker(markerPath, 'C:\\Users\\Me\\other-codex\\sessions') + ).toBe(false) + }) + + it('reads a legacy v3 marker as a certified baseline with nothing pending', () => { + writeFileSync( + markerPath, + `${JSON.stringify({ + version: 3, + systemSessionsRoot: WINDOWS_ROOT, + completedAt: Date.now(), + summary: { scannedFiles: 12 } + })}\n`, + 'utf-8' + ) + + expect(readCodexSessionBackfillBaseline(markerPath, 'c:/users/me/.codex/sessions')).toEqual({ + pendingScanDates: [] + }) + }) + + it('keeps honoring the v3 empty-source guard', () => { + writeFileSync( + markerPath, + `${JSON.stringify({ + version: 3, + systemSessionsRoot: WINDOWS_ROOT, + summary: { scannedFiles: 0 } + })}\n`, + 'utf-8' + ) + + expect(readCodexSessionBackfillBaseline(markerPath, WINDOWS_ROOT)).toBeNull() + }) + + it('does not treat a bounded pass that scanned nothing as an empty source', () => { + writeFullBaseline(WINDOWS_ROOT) + writeCodexSessionBackfillMarker( + markerPath, + WINDOWS_ROOT, + createSummary(0), + captureCodexSessionBackfillMarkerGeneration(), + { coverage: 'bounded', coveredScanDates: [LAUNCH_DATE] } + ) + + expect(readMarker()).toMatchObject({ baselineScannedFiles: 3 }) + expect(hasCompletedCodexSessionBackfillMarker(markerPath, WINDOWS_ROOT)).toBe(true) + }) + + it('refuses to let a bounded pass invent a baseline it never verified', () => { + writeCodexSessionBackfillMarker( + markerPath, + WINDOWS_ROOT, + createSummary(), + captureCodexSessionBackfillMarkerGeneration(), + { coverage: 'bounded', coveredScanDates: [LAUNCH_DATE] } + ) + + expect(hasCompletedCodexSessionBackfillMarker(markerPath, WINDOWS_ROOT)).toBe(false) + }) + + it('records a launch date without destroying the historical baseline', () => { + writeFullBaseline(WINDOWS_ROOT) + + markCodexSessionBackfillMarkerPending(markerPath, 'C:/Users/Me/.codex/sessions', [LAUNCH_DATE]) + + expect(readMarker()).toMatchObject({ coverage: 'full', pendingScanDates: [LAUNCH_DATE] }) + expect(readCodexSessionBackfillBaseline(markerPath, WINDOWS_ROOT)).toEqual({ + pendingScanDates: [LAUNCH_DATE] + }) + }) + + it('leaves a marker for a different history untouched', () => { + writeFullBaseline(WINDOWS_ROOT) + + markCodexSessionBackfillMarkerPending(markerPath, 'D:\\other\\.codex\\sessions', [LAUNCH_DATE]) + + expect(readMarker()).toMatchObject({ systemSessionsRoot: WINDOWS_ROOT, pendingScanDates: [] }) + }) + + it('keeps a racing launch date pending when an older pass publishes', () => { + writeFullBaseline(WINDOWS_ROOT) + const staleGeneration = captureCodexSessionBackfillMarkerGeneration() + markCodexSessionBackfillMarkerPending(markerPath, WINDOWS_ROOT, [LAUNCH_DATE]) + + writeCodexSessionBackfillMarker(markerPath, WINDOWS_ROOT, createSummary(), staleGeneration, { + coverage: 'bounded', + coveredScanDates: [LAUNCH_DATE] + }) + + expect(readMarker()).toMatchObject({ pendingScanDates: [LAUNCH_DATE] }) + }) + + it('keeps the pane date pending while the pane is still live', () => { + writeFullBaseline(WINDOWS_ROOT, { coveredScanDates: [LAUNCH_DATE], retain: true }) + + expect(readMarker()).toMatchObject({ launchActive: true, pendingScanDates: [LAUNCH_DATE] }) + }) + + it('settles every pending date once a full walk covers them', () => { + writeFullBaseline(WINDOWS_ROOT) + markCodexSessionBackfillMarkerPending(markerPath, WINDOWS_ROOT, [LAUNCH_DATE]) + + writeFullBaseline(WINDOWS_ROOT) + + // The walk looked at every date, so nothing is left to revisit. + expect(readMarker()).toMatchObject({ pendingScanDates: [] }) + }) + + it('demands a full walk once the pending window outgrows its bound', () => { + writeFullBaseline(WINDOWS_ROOT) + + expect( + markCodexSessionBackfillMarkerPending(markerPath, WINDOWS_ROOT, overflowingDates()) + ).toBe(true) + + expect(readMarker()).toMatchObject({ needsFullScan: true, pendingScanDates: [] }) + expect(hasCompletedCodexSessionBackfillMarker(markerPath, WINDOWS_ROOT)).toBe(false) + }) + + it('keeps owing that full walk when the next launch records its own date', () => { + writeFullBaseline(WINDOWS_ROOT) + markCodexSessionBackfillMarkerPending(markerPath, WINDOWS_ROOT, overflowingDates()) + + expect(markCodexSessionBackfillMarkerPending(markerPath, WINDOWS_ROOT, [TODAY])).toBe(true) + + expect(readMarker()).toMatchObject({ needsFullScan: true, pendingScanDates: [] }) + expect(hasCompletedCodexSessionBackfillMarker(markerPath, WINDOWS_ROOT)).toBe(false) + }) + + it('keeps the full-walk demand when a later launch overtook the walk', () => { + writeFullBaseline(WINDOWS_ROOT) + const staleGeneration = captureCodexSessionBackfillMarkerGeneration() + markCodexSessionBackfillMarkerPending(markerPath, WINDOWS_ROOT, overflowingDates()) + + writeCodexSessionBackfillMarker(markerPath, WINDOWS_ROOT, createSummary(), staleGeneration, { + coverage: 'full', + coveredScanDates: [] + }) + + // The walk may have passed those dates before their rollouts existed. + expect(readMarker()).toMatchObject({ needsFullScan: true }) + }) + + it('clears the full-walk demand only once a full walk certifies the tree', () => { + writeFullBaseline(WINDOWS_ROOT) + markCodexSessionBackfillMarkerPending(markerPath, WINDOWS_ROOT, overflowingDates()) + + writeFullBaseline(WINDOWS_ROOT) + + expect(readMarker()).toMatchObject({ needsFullScan: false, pendingScanDates: [] }) + expect(hasCompletedCodexSessionBackfillMarker(markerPath, WINDOWS_ROOT)).toBe(true) + }) + + it('persists a launch date even before any baseline exists', () => { + markCodexSessionBackfillMarkerPending(markerPath, WINDOWS_ROOT, [TODAY]) + + expect(readMarker()).toMatchObject({ coverage: 'bounded', pendingScanDates: [TODAY] }) + expect(hasCompletedCodexSessionBackfillMarker(markerPath, WINDOWS_ROOT)).toBe(false) + }) +}) diff --git a/src/main/codex/codex-session-backfill-marker.ts b/src/main/codex/codex-session-backfill-marker.ts index 8c2c9fb9c50..b802a666980 100644 --- a/src/main/codex/codex-session-backfill-marker.ts +++ b/src/main/codex/codex-session-backfill-marker.ts @@ -1,91 +1,264 @@ import { mkdirSync, readFileSync, rmSync } from 'node:fs' import { dirname } from 'node:path' +import { normalizeRuntimePathForComparison } from '../../shared/cross-platform-path' import { writeFileAtomically } from '../codex-accounts/fs-utils' -import type { CodexSessionBackfillSummary } from './codex-session-backfill-types' +import { + expandCodexSessionBackfillDatesThroughToday, + getCodexSessionBackfillDate, + mergeCodexSessionBackfillDates, + parseCodexSessionBackfillDates, + subtractCodexSessionBackfillDates +} from './codex-session-backfill-scan-dates' +import type { + CodexSessionBackfillDate, + CodexSessionBackfillSummary +} from './codex-session-backfill-types' // Why: bump to re-run the backfill for every host after a layout or semantics // change; the run itself stays skip-existing so re-runs never overwrite. -const CODEX_SESSION_BACKFILL_MARKER_VERSION = 3 +const CODEX_SESSION_BACKFILL_MARKER_VERSION = 4 +// Why: v3 was only ever written after a certified full-tree walk, so it reads +// as a v4 baseline and existing installs never pay one more full scan. +const MARKER_BASELINE_VERSIONS: ReadonlySet = new Set([3, 4]) +// Why: bounds abnormal-exit recovery; past this many dates a full walk is the +// cheaper certainty. +const MAX_PENDING_SCAN_DATES = 31 + let markerInvalidationGeneration = 0 +export type CodexSessionBackfillBaseline = { + /** Dates whose managed rollouts may not be published yet, widened through today. */ + pendingScanDates: readonly CodexSessionBackfillDate[] +} + export function captureCodexSessionBackfillMarkerGeneration(): number { return markerInvalidationGeneration } +/** + * Reads the certified full-history baseline for this target, if one exists. + * + * Null means no usable baseline, which is the only state that justifies a + * full-tree walk. A baseline with pending dates still needs a bounded pass. + */ +export function readCodexSessionBackfillBaseline( + markerPath: string, + systemSessionsRoot: string, + today: CodexSessionBackfillDate = getCodexSessionBackfillDate() +): CodexSessionBackfillBaseline | null { + const record = readMarkerRecord(markerPath) + if (!record?.hasBaseline || !matchesTargetRoot(record, systemSessionsRoot)) { + return null + } + // Why: an empty source can become populated after an early migration run; + // let the incremental async walk verify it without blocking the main thread. + if (record.baselineScannedFiles === 0 || record.needsFullScan) { + return null + } + // Why: only a pane that was still running when the pass ended can have written + // dates nobody recorded — a pane held open across midnight, then force-quit. + const pendingScanDates = record.launchActive + ? expandCodexSessionBackfillDatesThroughToday( + record.pendingScanDates, + today, + MAX_PENDING_SCAN_DATES + ) + : record.pendingScanDates + return pendingScanDates ? { pendingScanDates } : null +} + export function hasCompletedCodexSessionBackfillMarker( markerPath: string, systemSessionsRoot: string ): boolean { - try { - const parsed: unknown = JSON.parse(readFileSync(markerPath, 'utf-8')) - if (!parsed || typeof parsed !== 'object' || Array.isArray(parsed)) { - return false - } - const marker = parsed as { - version?: unknown - systemSessionsRoot?: unknown - summary?: { scannedFiles?: unknown } - } - // Why: changing the configured real Codex home must backfill the new - // target instead of honoring a marker written for a different history. - const markerMatchesTarget = - marker.version === CODEX_SESSION_BACKFILL_MARKER_VERSION && - marker.systemSessionsRoot === systemSessionsRoot - if (!markerMatchesTarget) { - return false - } - // Why: an empty source can become populated after an early migration run; - // let the incremental async walk verify it without blocking the main thread. - return marker.summary?.scannedFiles !== 0 - } catch { - return false - } + return readCodexSessionBackfillBaseline(markerPath, systemSessionsRoot) !== null } export function writeCodexSessionBackfillMarker( markerPath: string, systemSessionsRoot: string, summary: CodexSessionBackfillSummary, - expectedGeneration: number + expectedGeneration: number, + options: { + /** A full pass certifies the whole tree; a bounded one may only extend that. */ + coverage: 'full' | 'bounded' + coveredScanDates: readonly CodexSessionBackfillDate[] + /** A live Codex pane keeps writing into its own date, so it stays pending. */ + retainPendingScanDates?: boolean + } ): void { - // Why: a launch can invalidate this pass before its delayed replacement begins. - if (expectedGeneration !== markerInvalidationGeneration) { + const record = readMarkerRecord(markerPath) + const current = record && matchesTargetRoot(record, systemSessionsRoot) ? record : null + // Why: a date-limited pass cannot certify the dates it never looked at, so + // without an existing baseline it must publish nothing at all. + if (options.coverage !== 'full' && !current?.hasBaseline) { return } - mkdirSync(dirname(markerPath), { recursive: true }) - writeFileAtomically( - markerPath, - `${JSON.stringify( - { - version: CODEX_SESSION_BACKFILL_MARKER_VERSION, - systemSessionsRoot, - completedAt: Date.now(), - summary - }, - null, - 2 - )}\n` - ) + // Why: a launch that began during this pass can have written rollouts the + // walk had already gone past, so its dates must survive as pending. + const generationCurrent = expectedGeneration === markerInvalidationGeneration + const clearCovered = generationCurrent && options.retainPendingScanDates !== true + const currentPendingScanDates = current?.pendingScanDates ?? [] + // Why: a full walk speaks for every date, so it settles the whole pending set + // rather than only the dates a bounded pass was asked to look at. + const coveredScanDates = + options.coverage === 'full' ? currentPendingScanDates : options.coveredScanDates + const pendingScanDates = clearCovered + ? subtractCodexSessionBackfillDates(currentPendingScanDates, coveredScanDates) + : mergeCodexSessionBackfillDates(currentPendingScanDates, options.coveredScanDates) + writeMarkerRecord(markerPath, { + ...current?.raw, + version: CODEX_SESSION_BACKFILL_MARKER_VERSION, + systemSessionsRoot, + // Why: only reached with a baseline in hand, so this records that a baseline + // exists, not the scope of this particular pass. + coverage: 'full', + completedAt: Date.now(), + launchActive: options.retainPendingScanDates === true, + // Why: only a full walk can speak for the whole tree, so a bounded pass + // that legitimately scanned nothing must not read back as an empty source. + baselineScannedFiles: + options.coverage === 'full' ? summary.scannedFiles : current?.baselineScannedFiles, + summary, + // Why: only a full walk that no later launch overtook can retire the demand. + ...describePendingScanDates( + pendingScanDates, + current?.needsFullScan === true && !(generationCurrent && options.coverage === 'full') + ) + }) } -export function invalidateCodexSessionBackfillMarker(markerPath: string): void { +/** + * Records dates a launch may still be writing into, keeping the baseline. + * + * Why not delete: the marker is the only record that the full history was ever + * published. Dropping it makes every launch re-walk the whole sessions tree. + * + * Returns whether a full walk is still owed for this target, so the caller can + * fold that demand into its own in-memory state instead of losing it. + */ +export function markCodexSessionBackfillMarkerPending( + markerPath: string, + systemSessionsRoot: string, + scanDates: readonly CodexSessionBackfillDate[] +): boolean { + // Why: an older in-flight pass must not clear dates recorded after it started. markerInvalidationGeneration += 1 + const record = readMarkerRecord(markerPath) + // Why: a marker for a different history says nothing about this target, so + // only a full walk can certify it. + if (record && !matchesTargetRoot(record, systemSessionsRoot)) { + return true + } + const pendingScanDates = mergeCodexSessionBackfillDates(record?.pendingScanDates, scanDates) + const pending = describePendingScanDates(pendingScanDates, record?.needsFullScan === true) + if (pendingScanDates.length === record?.pendingScanDates.length) { + return record.needsFullScan + } try { - // Why: a managed-lane system-default launch can create new source - // rollouts, so a prior one-time marker must not suppress the next opt-in. - rmSync(markerPath, { force: true }) + writeMarkerRecord(markerPath, { + version: CODEX_SESSION_BACKFILL_MARKER_VERSION, + systemSessionsRoot, + coverage: 'bounded', + // Why: defaults only — an existing record keeps its own version and + // coverage, which is what preserves a v3 marker's implicit baseline. + ...record?.raw, + ...pending + }) } catch (error) { - console.warn('[codex-session-backfill] Failed to invalidate completion marker:', error) + console.warn('[codex-session-backfill] Failed to record pending scan dates:', error) try { - writeFileAtomically( - markerPath, - `${JSON.stringify({ version: 0, invalidatedAt: Date.now() })}\n` - ) + // Why: fail closed — a full rescan costs one slow pass, while a silently + // unrecorded launch date hides its rollouts forever. + rmSync(markerPath, { force: true }) } catch (fallbackError) { throw new AggregateError( [error, fallbackError], - 'Failed to invalidate Codex session backfill marker' + 'Failed to record pending Codex session backfill scan dates' ) } + return true + } + return pending.needsFullScan +} + +type CodexSessionBackfillMarkerRecord = { + raw: Record + systemSessionsRoot: string + hasBaseline: boolean + pendingScanDates: CodexSessionBackfillDate[] + needsFullScan: boolean + launchActive: boolean + /** Files the last full-tree walk saw; a bounded pass must not overwrite it. */ + baselineScannedFiles: number | undefined +} + +function readMarkerRecord(markerPath: string): CodexSessionBackfillMarkerRecord | null { + try { + const parsed: unknown = JSON.parse(readFileSync(markerPath, 'utf-8')) + if (!parsed || typeof parsed !== 'object' || Array.isArray(parsed)) { + return null + } + const marker = parsed as { + version?: unknown + systemSessionsRoot?: unknown + coverage?: unknown + pendingScanDates?: unknown + needsFullScan?: unknown + launchActive?: unknown + baselineScannedFiles?: unknown + summary?: { scannedFiles?: unknown } + } + if ( + typeof marker.version !== 'number' || + !MARKER_BASELINE_VERSIONS.has(marker.version) || + typeof marker.systemSessionsRoot !== 'string' + ) { + return null + } + const baselineScannedFiles = marker.baselineScannedFiles ?? marker.summary?.scannedFiles + return { + raw: parsed as Record, + systemSessionsRoot: marker.systemSessionsRoot, + // v3 predates `coverage` and was only written after a full-tree walk. + hasBaseline: marker.version === 3 || marker.coverage === 'full', + pendingScanDates: parseCodexSessionBackfillDates(marker.pendingScanDates), + needsFullScan: marker.needsFullScan === true, + launchActive: marker.launchActive === true, + // v3 wrote only one summary, and it was always a full-tree one. + baselineScannedFiles: + typeof baselineScannedFiles === 'number' ? baselineScannedFiles : undefined + } + } catch { + return null } } + +/** Why: changing the configured real Codex home must backfill the new target + * instead of honoring a marker written for a different history — and Windows + * spells one directory several ways, so compare normalized. */ +function matchesTargetRoot( + record: CodexSessionBackfillMarkerRecord, + systemSessionsRoot: string +): boolean { + return ( + normalizeRuntimePathForComparison(record.systemSessionsRoot) === + normalizeRuntimePathForComparison(systemSessionsRoot) + ) +} + +/** Why: an unmet full-scan demand outlives the launch that raised it — only a + * full walk may clear it, or the overflowed dates are never revisited. */ +function describePendingScanDates( + pendingScanDates: readonly CodexSessionBackfillDate[], + stillOwesFullScan: boolean +): { pendingScanDates: readonly CodexSessionBackfillDate[]; needsFullScan: boolean } { + return pendingScanDates.length > MAX_PENDING_SCAN_DATES || stillOwesFullScan + ? { pendingScanDates: [], needsFullScan: true } + : { pendingScanDates, needsFullScan: false } +} + +function writeMarkerRecord(markerPath: string, record: Record): void { + mkdirSync(dirname(markerPath), { recursive: true }) + writeFileAtomically(markerPath, `${JSON.stringify(record, null, 2)}\n`) +} diff --git a/src/main/codex/codex-session-backfill-scan-dates.test.ts b/src/main/codex/codex-session-backfill-scan-dates.test.ts new file mode 100644 index 00000000000..f9bf4530733 --- /dev/null +++ b/src/main/codex/codex-session-backfill-scan-dates.test.ts @@ -0,0 +1,102 @@ +import { describe, expect, it } from 'vitest' +import { + compareCodexSessionBackfillDates, + expandCodexSessionBackfillDatesThroughToday, + getCodexSessionBackfillDate, + getCodexSessionBackfillDatesBetween, + isCodexSessionBackfillDate, + mergeCodexSessionBackfillDates, + parseCodexSessionBackfillDates, + subtractCodexSessionBackfillDates +} from './codex-session-backfill-scan-dates' + +describe('codex session backfill scan dates', () => { + it('reads UTC parts so a local evening never lands on the wrong directory', () => { + expect(getCodexSessionBackfillDate(new Date('2026-08-05T23:59:59Z'))).toEqual([ + '2026', + '08', + '05' + ]) + expect(getCodexSessionBackfillDate(new Date('2026-01-02T00:00:00Z'))).toEqual([ + '2026', + '01', + '02' + ]) + }) + + it('rejects anything that is not a zero-padded YYYY/MM/DD triple', () => { + expect(isCodexSessionBackfillDate(['2026', '08', '05'])).toBe(true) + expect(isCodexSessionBackfillDate(['2026', '8', '05'])).toBe(false) + expect(isCodexSessionBackfillDate(['2026', '08'])).toBe(false) + expect(isCodexSessionBackfillDate('2026-08-05')).toBe(false) + }) + + it('rejects dates the calendar never produced', () => { + expect(isCodexSessionBackfillDate(['2026', '99', '99'])).toBe(false) + expect(isCodexSessionBackfillDate(['2026', '02', '30'])).toBe(false) + expect(isCodexSessionBackfillDate(['2025', '02', '29'])).toBe(false) + expect(isCodexSessionBackfillDate(['2026', '00', '10'])).toBe(false) + expect(isCodexSessionBackfillDate(['2024', '02', '29'])).toBe(true) + expect(isCodexSessionBackfillDate(['2026', '12', '31'])).toBe(true) + }) + + it('merges and subtracts date sets by identity, not by reference', () => { + const merged = mergeCodexSessionBackfillDates( + [ + ['2026', '08', '06'], + ['2026', '08', '05'] + ], + [['2026', '08', '06']], + undefined + ) + + expect(merged).toEqual([ + ['2026', '08', '05'], + ['2026', '08', '06'] + ]) + expect(subtractCodexSessionBackfillDates(merged, [['2026', '08', '05']])).toEqual([ + ['2026', '08', '06'] + ]) + expect(compareCodexSessionBackfillDates(merged[0], merged[1])).toBeLessThan(0) + }) + + it('discards unparseable persisted dates instead of scanning bogus roots', () => { + expect(parseCodexSessionBackfillDates([['2026', '08', '05'], 'nope', ['2026'], null])).toEqual([ + ['2026', '08', '05'] + ]) + expect(parseCodexSessionBackfillDates('not an array')).toEqual([]) + }) + + it('walks every date a launch could have spanned, including across a month end', () => { + expect( + getCodexSessionBackfillDatesBetween( + new Date('2026-07-31T23:00:00Z'), + new Date('2026-08-02T01:00:00Z') + ) + ).toEqual([ + ['2026', '07', '31'], + ['2026', '08', '01'], + ['2026', '08', '02'] + ]) + }) + + it('widens a pending set into the contiguous window that ends today', () => { + expect( + expandCodexSessionBackfillDatesThroughToday([['2026', '08', '05']], ['2026', '08', '07'], 31) + ).toEqual([ + ['2026', '08', '05'], + ['2026', '08', '06'], + ['2026', '08', '07'] + ]) + }) + + it('leaves an empty pending set empty rather than inventing today', () => { + expect(expandCodexSessionBackfillDatesThroughToday([], ['2026', '08', '07'], 31)).toEqual([]) + }) + + it('gives up on a window wider than the bound so a full walk can recertify', () => { + expect( + expandCodexSessionBackfillDatesThroughToday([['2026', '01', '01']], ['2026', '08', '07'], 31) + ).toBeNull() + }) +}) diff --git a/src/main/codex/codex-session-backfill-scan-dates.ts b/src/main/codex/codex-session-backfill-scan-dates.ts new file mode 100644 index 00000000000..ca4e1d311f7 --- /dev/null +++ b/src/main/codex/codex-session-backfill-scan-dates.ts @@ -0,0 +1,112 @@ +import type { CodexSessionBackfillDate } from './codex-session-backfill-types' + +// Set algebra over the UTC date directories that make up a bounded backfill +// pass. Kept apart from the walk itself so the durable marker, the scheduler, +// and the pass all agree on identity, ordering, and range expansion. + +export function getCodexSessionBackfillDate(date = new Date()): CodexSessionBackfillDate { + return [ + String(date.getUTCFullYear()).padStart(4, '0'), + String(date.getUTCMonth() + 1).padStart(2, '0'), + String(date.getUTCDate()).padStart(2, '0') + ] +} + +export function isCodexSessionBackfillDate(value: unknown): value is CodexSessionBackfillDate { + if (!Array.isArray(value) || value.length !== 3) { + return false + } + const key = value.join('-') + // Why: shape alone accepts directories the calendar never produces (2026/99/99 + // from a corrupted marker, 2025/02/29); a UTC round-trip rejects them for free. + // Deliberately no age or future bound — this also gates which managed rollouts + // get published, and a clock-skewed future directory holds real sessions. + return ( + /^\d{4}-\d{2}-\d{2}$/.test(key) && + toCodexSessionBackfillDateKey(getCodexSessionBackfillDate(toUtcDate(value))) === key + ) +} + +export function toCodexSessionBackfillDateKey(date: CodexSessionBackfillDate): string { + return date.join('-') +} + +export function compareCodexSessionBackfillDates( + left: CodexSessionBackfillDate, + right: CodexSessionBackfillDate +): number { + return toCodexSessionBackfillDateKey(left).localeCompare(toCodexSessionBackfillDateKey(right)) +} + +/** Deduplicated ascending union; invalid entries are dropped. */ +export function mergeCodexSessionBackfillDates( + ...groups: readonly (readonly CodexSessionBackfillDate[] | undefined)[] +): CodexSessionBackfillDate[] { + const merged = new Map() + for (const group of groups) { + for (const date of group ?? []) { + if (isCodexSessionBackfillDate(date)) { + merged.set(toCodexSessionBackfillDateKey(date), date) + } + } + } + return [...merged.values()].sort(compareCodexSessionBackfillDates) +} + +export function subtractCodexSessionBackfillDates( + dates: readonly CodexSessionBackfillDate[], + removed: readonly CodexSessionBackfillDate[] +): CodexSessionBackfillDate[] { + const removedKeys = new Set(removed.map(toCodexSessionBackfillDateKey)) + return dates.filter((date) => !removedKeys.has(toCodexSessionBackfillDateKey(date))) +} + +/** Reads persisted marker dates; anything unrecognized is discarded. */ +export function parseCodexSessionBackfillDates(value: unknown): CodexSessionBackfillDate[] { + return Array.isArray(value) + ? mergeCodexSessionBackfillDates(value.filter(isCodexSessionBackfillDate)) + : [] +} + +export function getCodexSessionBackfillDatesBetween( + startedAt: Date, + finishedAt: Date +): CodexSessionBackfillDate[] { + const dates: CodexSessionBackfillDate[] = [] + const cursor = toUtcMidnight(startedAt) + const last = toUtcMidnight(finishedAt) + while (cursor <= last) { + dates.push(getCodexSessionBackfillDate(cursor)) + cursor.setUTCDate(cursor.getUTCDate() + 1) + } + return dates +} + +/** + * Widens a pending set into the contiguous range that ends at `today`. + * + * Why: an abnormal exit or a pane held open across midnight leaves activity on + * dates nobody got to record. The gap between the oldest pending date and today + * is the smallest window that provably contains them. Returns null once that + * window outgrows `maxDates`, where a full walk is the cheaper certainty. + */ +export function expandCodexSessionBackfillDatesThroughToday( + dates: readonly CodexSessionBackfillDate[], + today: CodexSessionBackfillDate, + maxDates: number +): CodexSessionBackfillDate[] | null { + if (dates.length === 0) { + return [] + } + const bounds = mergeCodexSessionBackfillDates(dates, [today]) + const range = getCodexSessionBackfillDatesBetween(toUtcDate(bounds[0]), toUtcDate(bounds.at(-1)!)) + return range.length > maxDates ? null : range +} + +function toUtcDate([year, month, day]: readonly string[]): Date { + return new Date(Date.UTC(Number(year), Number(month) - 1, Number(day))) +} + +function toUtcMidnight(date: Date): Date { + return new Date(Date.UTC(date.getUTCFullYear(), date.getUTCMonth(), date.getUTCDate())) +} diff --git a/src/main/codex/codex-session-backfill-types.ts b/src/main/codex/codex-session-backfill-types.ts index 92891beaaff..bf830419a0b 100644 --- a/src/main/codex/codex-session-backfill-types.ts +++ b/src/main/codex/codex-session-backfill-types.ts @@ -26,12 +26,12 @@ export type CodexSessionBackfillOptions = CodexSessionBridgeIncrementalOptions & shouldStop?: () => boolean /** Limits a launch-triggered pass to the date directories that can contain its rollouts. */ scanDates?: readonly CodexSessionBackfillDate[] - /** A scheduled launch pass must not be suppressed by a marker it just invalidated. */ + /** Forces recertification of the whole tree, ignoring any existing baseline. */ + fullScanRequired?: boolean + /** A scheduled launch pass runs even when the baseline reports nothing pending. */ ignoreCompletionMarker?: boolean - /** Active launch passes defer global completion until their final exit scan. */ - writeCompletionMarker?: boolean - /** Final launch scans can extend a previously certified full-tree baseline. */ - writeBoundedCompletionMarker?: boolean + /** A live Codex pane keeps writing into its own date, so it stays pending. */ + retainPendingScanDates?: boolean /** Rechecks launch scheduling state immediately before marker publication. */ canWriteCompletionMarker?: () => boolean } diff --git a/src/main/codex/codex-session-backfill.test.ts b/src/main/codex/codex-session-backfill.test.ts index 418ef0d8381..a3dff66c6f0 100644 --- a/src/main/codex/codex-session-backfill.test.ts +++ b/src/main/codex/codex-session-backfill.test.ts @@ -21,129 +21,16 @@ const { homedirMock } = vi.hoisted(() => ({ homedirMock: vi.fn<() => string>() })) -const { fsMockState } = vi.hoisted(() => ({ - fsMockState: { - failLink: false, - failLinkTransiently: false, - failLinkPermission: false, - raceTargetIntoExistence: false, - failMarkerRm: false, - failMarkerReplacement: false, - failAuditMkdirOnce: false, - failAuditWrites: false, - failMkdirPath: null as string | null, - failDirectoryPath: null as string | null, - failLstatPath: null as string | null - } -})) - vi.mock('node:fs', async () => { - const actual = await vi.importActual('node:fs') - return { - ...actual, - existsSync: (...args: Parameters) => { - if (args[0] === fsMockState.failLstatPath) { - return false - } - return actual.existsSync(...args) - }, - rmSync: (...args: Parameters) => { - if ( - fsMockState.failMarkerRm && - String(args[0]).includes('codex-session-backfill') && - String(args[0]).endsWith('backfill-complete.json') - ) { - const error = new Error('EACCES: marker removal failed') as NodeJS.ErrnoException - error.code = 'EACCES' - throw error - } - return actual.rmSync(...args) - }, - renameSync: (...args: Parameters) => { - if ( - fsMockState.failMarkerReplacement && - String(args[1]).includes('codex-session-backfill') && - String(args[1]).endsWith('backfill-complete.json') - ) { - const error = new Error('EACCES: marker replacement failed') as NodeJS.ErrnoException - error.code = 'EACCES' - throw error - } - return actual.renameSync(...args) - } - } + const mocks = await import('./codex-session-backfill-fs-mocks') + return mocks.createNodeFsMock(await vi.importActual('node:fs')) }) vi.mock('node:fs/promises', async () => { - const actual = await vi.importActual('node:fs/promises') - return { - ...actual, - mkdir: (...args: Parameters) => { - if (args[0] === fsMockState.failMkdirPath) { - const error = new Error('EACCES: target directory inaccessible') as NodeJS.ErrnoException - error.code = 'EACCES' - throw error - } - if (fsMockState.failAuditMkdirOnce && String(args[0]).includes('codex-session-backfill')) { - fsMockState.failAuditMkdirOnce = false - const error = new Error( - 'EACCES: transient audit directory failure' - ) as NodeJS.ErrnoException - error.code = 'EACCES' - throw error - } - return actual.mkdir(...args) - }, - appendFile: (...args: Parameters) => { - if (fsMockState.failAuditWrites && String(args[0]).includes('codex-session-backfill')) { - const error = new Error('ENOSPC: audit write failed') as NodeJS.ErrnoException - error.code = 'ENOSPC' - throw error - } - return actual.appendFile(...args) - }, - lstat: (...args: Parameters) => { - if (args[0] === fsMockState.failLstatPath) { - const error = new Error('EACCES: path inaccessible') as NodeJS.ErrnoException - error.code = 'EACCES' - throw error - } - return actual.lstat(...args) - }, - link: async (...args: Parameters) => { - if (fsMockState.raceTargetIntoExistence && String(args[0]).includes('codex-runtime-home')) { - fsMockState.raceTargetIntoExistence = false - await actual.writeFile(args[1], 'concurrent target\n', 'utf-8') - const error = new Error('EEXIST: concurrent target') as NodeJS.ErrnoException - error.code = 'EEXIST' - throw error - } - if (fsMockState.failLink && String(args[0]).includes('codex-runtime-home')) { - const error = new Error('EXDEV: cross-device link') as NodeJS.ErrnoException - error.code = 'EXDEV' - throw error - } - if (fsMockState.failLinkTransiently && String(args[0]).includes('codex-runtime-home')) { - const error = new Error('EIO: transient hardlink failure') as NodeJS.ErrnoException - error.code = 'EIO' - throw error - } - if (fsMockState.failLinkPermission && String(args[0]).includes('codex-runtime-home')) { - const error = new Error('EACCES: hardlink permission denied') as NodeJS.ErrnoException - error.code = 'EACCES' - throw error - } - return actual.link(...args) - }, - opendir: (...args: Parameters) => { - if (args[0] === fsMockState.failDirectoryPath) { - const error = new Error('EACCES: directory unreadable') as NodeJS.ErrnoException - error.code = 'EACCES' - throw error - } - return actual.opendir(...args) - } - } + const mocks = await import('./codex-session-backfill-fs-mocks') + return mocks.createNodeFsPromisesMock( + await vi.importActual('node:fs/promises') + ) }) vi.mock('node:os', async () => { @@ -159,7 +46,15 @@ import { resolveCodexSessionBackfillPaths, startCodexSessionBackfillInBackground } from './codex-session-backfill' -import { invalidateCodexSessionBackfillMarker } from './codex-session-backfill-marker' +import { + markCodexSessionBackfillMarkerPending, + readCodexSessionBackfillBaseline +} from './codex-session-backfill-marker' +import { fsMockState, resetCodexSessionBackfillFsMocks } from './codex-session-backfill-fs-mocks' +import { getCodexSessionBackfillDate } from './codex-session-backfill-scan-dates' +import type { CodexSessionBackfillDate } from './codex-session-backfill-types' + +const FIXTURE_LAUNCH_DATE: CodexSessionBackfillDate = ['2026', '05', '26'] let fakeHomeDir: string let userDataDir: string @@ -212,18 +107,21 @@ function readAuditActions(): string[] { return readBackfillAuditRecords().map((record) => record.action) } +/** Stands in for a Codex pane launch: records its date without dropping the baseline. */ +function markLaunchPending(...scanDates: CodexSessionBackfillDate[]): void { + markCodexSessionBackfillMarkerPending( + getMarkerPath(), + getSystemSessionsRoot(), + scanDates.length > 0 ? scanDates : [FIXTURE_LAUNCH_DATE] + ) +} + +function readMarker(): Record { + return JSON.parse(readFileSync(getMarkerPath(), 'utf-8')) as Record +} + beforeEach(() => { - fsMockState.failLink = false - fsMockState.failLinkTransiently = false - fsMockState.failLinkPermission = false - fsMockState.raceTargetIntoExistence = false - fsMockState.failMarkerRm = false - fsMockState.failMarkerReplacement = false - fsMockState.failAuditMkdirOnce = false - fsMockState.failAuditWrites = false - fsMockState.failMkdirPath = null - fsMockState.failDirectoryPath = null - fsMockState.failLstatPath = null + resetCodexSessionBackfillFsMocks() fakeHomeDir = mkdtempSync(join(tmpdir(), 'orca-codex-backfill-home-')) userDataDir = mkdtempSync(join(tmpdir(), 'orca-codex-backfill-user-data-')) previousUserDataPath = process.env.ORCA_USER_DATA_PATH @@ -557,18 +455,48 @@ describe('startCodexSessionBackfillInBackground', () => { expect(existsSync(getMarkerPath())).toBe(false) }) - it('defers completion while a launch lease is active', async () => { - writeManagedSession(join('2026', '05', '26', 'rollout-a.jsonl'), '{"id":"a"}\n') + it('certifies the historical baseline while a launch lease is still active', async () => { + const today = getCodexSessionBackfillDate() + writeManagedSession(join('2026', '05', '26', 'rollout-history.jsonl'), 'history\n') + writeManagedSession(join(...today, 'rollout-live.jsonl'), 'live\n') + markLaunchPending(today) const active = await startCodexSessionBackfillInBackground({ - writeCompletionMarker: false + ignoreCompletionMarker: true, + retainPendingScanDates: true + }) + + expect(active).toMatchObject({ scannedFiles: 2, linkedFiles: 2 }) + // The baseline is certified despite the open pane; only its date stays pending. + expect(readMarker()).toMatchObject({ + version: 4, + coverage: 'full', + launchActive: true, + pendingScanDates: [today] + }) + expect(readCodexSessionBackfillBaseline(getMarkerPath(), getSystemSessionsRoot())).toEqual({ + pendingScanDates: [today] }) - expect(active).toMatchObject({ linkedFiles: 1 }) - expect(existsSync(getMarkerPath())).toBe(false) const completed = await startCodexSessionBackfillInBackground() - expect(completed).toMatchObject({ skippedExistingFiles: 1 }) - expect(existsSync(getMarkerPath())).toBe(true) + expect(completed).toMatchObject({ scannedFiles: 1, skippedExistingFiles: 1 }) + expect(readMarker()).toMatchObject({ pendingScanDates: [], launchActive: false }) + expect(await startCodexSessionBackfillInBackground()).toBeNull() + }) + + it('scans only the current date once a baseline exists', async () => { + writeManagedSession(join('2026', '05', '26', 'rollout-a.jsonl'), '{"id":"a"}\n') + await startCodexSessionBackfillInBackground() + // A second date directory: a full walk would report two scanned files. + writeManagedSession(join('2026', '06', '02', 'rollout-other.jsonl'), '{"id":"other"}\n') + markLaunchPending() + + const scanned = await startCodexSessionBackfillInBackground({ + ignoreCompletionMarker: true + }) + + // No scanDates were requested, so only the recorded pending date is walked. + expect(scanned).toMatchObject({ scannedFiles: 1, skippedExistingFiles: 1 }) }) it('rechecks launch state before publishing completion', async () => { @@ -582,54 +510,93 @@ describe('startCodexSessionBackfillInBackground', () => { expect(existsSync(getMarkerPath())).toBe(false) }) - it('keeps an invalidated active pass from recreating the completion marker', async () => { + it('keeps a racing launch date pending without discarding the baseline', async () => { writeManagedSession(join('2026', '05', '26', 'rollout-a.jsonl'), '{"id":"a"}\n') - let invalidated = false + let raceStarted = false const raced = await startCodexSessionBackfillInBackground({ shouldStop: () => { - if (!invalidated) { - invalidated = true - invalidateCodexSessionBackfillMarker(getMarkerPath()) + if (!raceStarted) { + raceStarted = true + markLaunchPending(['2026', '08', '05']) } return false } }) expect(raced).toMatchObject({ linkedFiles: 1, stopped: false }) - expect(existsSync(getMarkerPath())).toBe(false) + // The full walk still certifies history, but the racing launch's date stays pending. + expect(readMarker()).toMatchObject({ + version: 4, + coverage: 'full', + pendingScanDates: [['2026', '08', '05']] + }) const racedAudit = readFileSync(getAuditLogPath(), 'utf-8') + writeManagedSession(join('2026', '08', '05', 'rollout-raced.jsonl'), '{"id":"raced"}\n') const recovered = await startCodexSessionBackfillInBackground() - expect(recovered).toMatchObject({ skippedExistingFiles: 1, failedHealAuditRecords: 0 }) - expect(existsSync(getMarkerPath())).toBe(true) - expect(readFileSync(getAuditLogPath(), 'utf-8')).toBe(racedAudit) + expect(recovered).toMatchObject({ scannedFiles: 1, linkedFiles: 1 }) + expect(readMarker()).toMatchObject({ pendingScanDates: [] }) + expect(readFileSync(getAuditLogPath(), 'utf-8')).not.toBe(racedAudit) }) - it('replaces a stale marker when direct removal fails', async () => { + it('falls back to a full rescan when the pending rewrite fails', async () => { writeManagedSession(join('2026', '05', '26', 'rollout-a.jsonl'), '{"id":"a"}\n') await startCodexSessionBackfillInBackground() - fsMockState.failMarkerRm = true + fsMockState.failMarkerReplacement = true + const warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {}) - invalidateCodexSessionBackfillMarker(getMarkerPath()) + markLaunchPending(['2026', '08', '05']) - expect(JSON.parse(readFileSync(getMarkerPath(), 'utf-8'))).toMatchObject({ version: 0 }) + // Fail closed: no marker at all beats a marker missing the launch's date. + expect(existsSync(getMarkerPath())).toBe(false) + fsMockState.failMarkerReplacement = false const recovered = await startCodexSessionBackfillInBackground() expect(recovered).toMatchObject({ skippedExistingFiles: 1 }) - expect(JSON.parse(readFileSync(getMarkerPath(), 'utf-8'))).toMatchObject({ version: 3 }) + expect(readMarker()).toMatchObject({ version: 4, coverage: 'full' }) + expect(warnSpy).toHaveBeenCalled() + warnSpy.mockRestore() }) - it('fails launch preparation when a stale marker cannot be invalidated', async () => { + it('fails launch preparation when the baseline can neither be updated nor cleared', async () => { writeManagedSession(join('2026', '05', '26', 'rollout-a.jsonl'), '{"id":"a"}\n') await startCodexSessionBackfillInBackground() fsMockState.failMarkerRm = true fsMockState.failMarkerReplacement = true + const warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {}) - expect(() => invalidateCodexSessionBackfillMarker(getMarkerPath())).toThrow( - 'Failed to invalidate Codex session backfill marker' + expect(() => markLaunchPending(['2026', '08', '05'])).toThrow( + 'Failed to record pending Codex session backfill scan dates' ) - expect(JSON.parse(readFileSync(getMarkerPath(), 'utf-8'))).toMatchObject({ version: 3 }) + expect(readMarker()).toMatchObject({ version: 4 }) + warnSpy.mockRestore() + }) + + it('reads a legacy v3 marker as a certified baseline', async () => { + writeManagedSession(join('2026', '05', '26', 'rollout-a.jsonl'), '{"id":"a"}\n') + mkdirSync(dirname(getMarkerPath()), { recursive: true }) + writeFileSync( + getMarkerPath(), + `${JSON.stringify({ + version: 3, + systemSessionsRoot: getSystemSessionsRoot(), + completedAt: Date.now(), + summary: { scannedFiles: 1 } + })}\n`, + 'utf-8' + ) + + // No full walk on upgrade: the v3 baseline is honored as-is. + expect(await startCodexSessionBackfillInBackground()).toBeNull() + expect(existsSync(join(getSystemSessionsRoot(), '2026', '05', '26', 'rollout-a.jsonl'))).toBe( + false + ) + + markLaunchPending() + const bounded = await startCodexSessionBackfillInBackground() + expect(bounded).toMatchObject({ scannedFiles: 1, linkedFiles: 1 }) + expect(readMarker()).toMatchObject({ version: 4, coverage: 'full' }) }) it('writes a completion marker and skips the walk on later runs', async () => { @@ -638,7 +605,7 @@ describe('startCodexSessionBackfillInBackground', () => { const first = await startCodexSessionBackfillInBackground() expect(first).toMatchObject({ linkedFiles: 1, failedFiles: 0 }) expect(existsSync(getMarkerPath())).toBe(true) - expect(JSON.parse(readFileSync(getMarkerPath(), 'utf-8'))).toMatchObject({ version: 3 }) + expect(readMarker()).toMatchObject({ version: 4, coverage: 'full' }) // An ordinary call remains a no-op; only a launch-scheduled pass bypasses the marker. writeManagedSession(join('2026', '07', '01', 'rollout-later.jsonl'), '{"id":"later"}\n') @@ -655,11 +622,7 @@ describe('startCodexSessionBackfillInBackground', () => { expect(scheduled).toMatchObject({ scannedFiles: 1, linkedFiles: 1 }) }) - it('does not let a bounded pass certify older unscanned history', async () => { - writeManagedSession(join('2026', '05', '26', 'rollout-baseline.jsonl'), 'baseline\n') - await startCodexSessionBackfillInBackground() - - invalidateCodexSessionBackfillMarker(getMarkerPath()) + it('does not let a bounded pass certify history no baseline ever covered', async () => { const missedRelativePath = join('2026', '06', '01', 'rollout-missed.jsonl') writeManagedSession(missedRelativePath, 'missed\n') writeManagedSession(join('2026', '08', '05', 'rollout-launch.jsonl'), 'launch\n') @@ -672,26 +635,61 @@ describe('startCodexSessionBackfillInBackground', () => { expect(existsSync(getMarkerPath())).toBe(false) const recovered = await startCodexSessionBackfillInBackground() - expect(recovered).toMatchObject({ scannedFiles: 3, linkedFiles: 1 }) + expect(recovered).toMatchObject({ scannedFiles: 2, linkedFiles: 1 }) expect(existsSync(join(getSystemSessionsRoot(), missedRelativePath))).toBe(true) - expect(existsSync(getMarkerPath())).toBe(true) + expect(readMarker()).toMatchObject({ version: 4, coverage: 'full' }) }) - it('lets an explicit bounded final pass restore a certified baseline', async () => { + it('lets a bounded launch pass extend a certified baseline', async () => { writeManagedSession(join('2026', '05', '26', 'rollout-baseline.jsonl'), 'baseline\n') await startCodexSessionBackfillInBackground() - invalidateCodexSessionBackfillMarker(getMarkerPath()) + markLaunchPending(['2026', '08', '05']) writeManagedSession(join('2026', '08', '05', 'rollout-launch.jsonl'), 'launch\n') const bounded = await startCodexSessionBackfillInBackground({ ignoreCompletionMarker: true, - scanDates: [['2026', '08', '05']], - writeBoundedCompletionMarker: true + scanDates: [['2026', '08', '05']] }) expect(bounded).toMatchObject({ scannedFiles: 1, linkedFiles: 1 }) - expect(existsSync(getMarkerPath())).toBe(true) + expect(readMarker()).toMatchObject({ coverage: 'full', pendingScanDates: [] }) + expect(await startCodexSessionBackfillInBackground()).toBeNull() + }) + + it('recovers a bounded window after an abnormal exit instead of a full walk', async () => { + writeManagedSession(join('2026', '05', '26', 'rollout-baseline.jsonl'), 'baseline\n') + await startCodexSessionBackfillInBackground() + + // A launch records its date, then the app dies before any pass runs. + markLaunchPending(['2026', '08', '05']) + writeManagedSession(join('2026', '06', '01', 'rollout-untouched.jsonl'), 'untouched\n') + writeManagedSession(join('2026', '08', '05', 'rollout-launch.jsonl'), 'launch\n') + + const recovered = await startCodexSessionBackfillInBackground() + + expect(recovered).toMatchObject({ scannedFiles: 1, linkedFiles: 1 }) + expect( + existsSync(join(getSystemSessionsRoot(), '2026', '08', '05', 'rollout-launch.jsonl')) + ).toBe(true) + }) + + it('widens recovery across midnight when a pane was still live', async () => { + writeManagedSession(join('2026', '05', '26', 'rollout-baseline.jsonl'), 'baseline\n') + await startCodexSessionBackfillInBackground() + const launchDate = getCodexSessionBackfillDate(new Date(Date.now() - 2 * 24 * 60 * 60 * 1000)) + await startCodexSessionBackfillInBackground({ + scanDates: [launchDate], + ignoreCompletionMarker: true, + retainPendingScanDates: true + }) + + const baseline = readCodexSessionBackfillBaseline(getMarkerPath(), getSystemSessionsRoot()) + + // The pane could have written on every date from its launch through today. + expect(baseline?.pendingScanDates).toHaveLength(3) + expect(baseline?.pendingScanDates.at(0)).toEqual(launchDate) + expect(baseline?.pendingScanDates.at(-1)).toEqual(getCodexSessionBackfillDate()) }) it('records a new heal event when a linked rollout grows in place', async () => { @@ -700,7 +698,7 @@ describe('startCodexSessionBackfillInBackground', () => { await startCodexSessionBackfillInBackground() const firstRecord = readBackfillAuditRecords().find((record) => record.action === 'hardlink') - invalidateCodexSessionBackfillMarker(getMarkerPath()) + markLaunchPending() appendFileSync(managedPath, '{"event":"later"}\n', 'utf-8') await startCodexSessionBackfillInBackground() @@ -720,7 +718,7 @@ describe('startCodexSessionBackfillInBackground', () => { const firstAudit = readFileSync(getAuditLogPath(), 'utf-8') for (let pass = 0; pass < 2; pass += 1) { - invalidateCodexSessionBackfillMarker(getMarkerPath()) + markLaunchPending() const repeated = await startCodexSessionBackfillInBackground() expect(repeated).toMatchObject({ linkedFiles: 0, @@ -745,14 +743,15 @@ describe('startCodexSessionBackfillInBackground', () => { writeManagedSession(firstRelativePath, '{"id":"a"}\n') await startCodexSessionBackfillInBackground() - invalidateCodexSessionBackfillMarker(getMarkerPath()) + markLaunchPending() writeManagedSession(secondRelativePath, '{"id":"b"}\n') fsMockState.failAuditWrites = true const interrupted = await startCodexSessionBackfillInBackground() expect(interrupted).toMatchObject({ linkedFiles: 1, failedHealAuditRecords: 1 }) - expect(existsSync(getMarkerPath())).toBe(false) + // The failed pass certifies nothing, so its date stays queued for the retry. + expect(readMarker()).toMatchObject({ pendingScanDates: [FIXTURE_LAUNCH_DATE] }) expect( readBackfillAuditRecords().filter((record) => ['hardlink', 'copy', 'existing'].includes(record.action) @@ -810,7 +809,7 @@ describe('startCodexSessionBackfillInBackground', () => { rmSync(getMarkerPath(), { recursive: true }) const resumed = await startCodexSessionBackfillInBackground() expect(resumed).toMatchObject({ skippedExistingFiles: 1, failedHealAuditRecords: 0 }) - expect(JSON.parse(readFileSync(getMarkerPath(), 'utf-8'))).toMatchObject({ version: 3 }) + expect(readMarker()).toMatchObject({ version: 4 }) expect(warnSpy).toHaveBeenCalled() warnSpy.mockRestore() }) @@ -841,7 +840,7 @@ describe('startCodexSessionBackfillInBackground', () => { expect(existsSync(getMarkerPath())).toBe(true) const firstAudit = readFileSync(getAuditLogPath(), 'utf-8') - invalidateCodexSessionBackfillMarker(getMarkerPath()) + markLaunchPending() const repeated = await startCodexSessionBackfillInBackground() expect(repeated).toMatchObject({ skippedUnsupportedFilesystemFiles: 1, failedFiles: 0 }) diff --git a/src/main/codex/codex-session-backfill.ts b/src/main/codex/codex-session-backfill.ts index 720469cf2b0..b6a8b5a07f6 100644 --- a/src/main/codex/codex-session-backfill.ts +++ b/src/main/codex/codex-session-backfill.ts @@ -17,10 +17,16 @@ import { } from './codex-session-backfill-date' import { captureCodexSessionBackfillMarkerGeneration, - hasCompletedCodexSessionBackfillMarker, - writeCodexSessionBackfillMarker as writeBackfillMarker + readCodexSessionBackfillBaseline, + writeCodexSessionBackfillMarker as writeBackfillMarker, + type CodexSessionBackfillBaseline } from './codex-session-backfill-marker' +import { + getCodexSessionBackfillDate, + mergeCodexSessionBackfillDates +} from './codex-session-backfill-scan-dates' import type { + CodexSessionBackfillDate, CodexSessionBackfillOptions, CodexSessionBackfillPaths, CodexSessionBackfillSummary @@ -88,30 +94,61 @@ async function runCodexSessionBackfillOncePerHost( ): Promise { const paths = resolveCodexSessionBackfillPaths(systemCodexHomePathOverride) const markerGeneration = captureCodexSessionBackfillMarkerGeneration() - if ( - !options.ignoreCompletionMarker && - hasCompletedCodexSessionBackfillMarker(paths.markerPath, paths.systemSessionsRoot) - ) { + const baseline = readCodexSessionBackfillBaseline(paths.markerPath, paths.systemSessionsRoot) + const scanPlan = resolveCodexSessionBackfillScanPlan(baseline, options) + if (!scanPlan) { return null } - const summary = await backfillManagedCodexSessionsIntoSystemHome(paths, options) - // Why: file or heal-queue failures leave the marker unset so the next + const summary = await backfillManagedCodexSessionsIntoSystemHome(paths, { + ...options, + scanDates: scanPlan.scanDates + }) + // Why: file or heal-queue failures leave the pass uncertified so the next // startup retries; skip-existing keeps those retries cheap. if ( !summary.stopped && options.shouldStop?.() !== true && - options.writeCompletionMarker !== false && options.canWriteCompletionMarker?.() !== false && - (options.scanDates === undefined || options.writeBoundedCompletionMarker === true) && summary.failedFiles === 0 && summary.failedDirectories === 0 && summary.failedHealAuditRecords === 0 ) { - writeBackfillMarker(paths.markerPath, paths.systemSessionsRoot, summary, markerGeneration) + writeBackfillMarker(paths.markerPath, paths.systemSessionsRoot, summary, markerGeneration, { + coverage: scanPlan.scanDates ? 'bounded' : 'full', + coveredScanDates: scanPlan.scanDates ?? [], + retainPendingScanDates: options.retainPendingScanDates === true + }) } return summary } +/** + * Decides how much of the sessions tree this pass must walk. + * + * Null means the baseline already covers everything and there is nothing + * pending; an absent `scanDates` means a full walk, which is only ever needed + * when no certified baseline exists (or the caller demands recertification). + */ +function resolveCodexSessionBackfillScanPlan( + baseline: CodexSessionBackfillBaseline | null, + options: CodexSessionBackfillOptions +): { scanDates?: readonly CodexSessionBackfillDate[] } | null { + const requestedScanDates = options.scanDates?.length ? options.scanDates : undefined + if (options.fullScanRequired) { + return {} + } + if (!baseline) { + return { scanDates: requestedScanDates } + } + const scanDates = mergeCodexSessionBackfillDates(baseline.pendingScanDates, requestedScanDates) + if (scanDates.length > 0) { + return { scanDates } + } + // Why: a launch-scheduled pass exists to publish rollouts the running pane is + // creating right now, so with a baseline in hand the current date is enough. + return options.ignoreCompletionMarker ? { scanDates: [getCodexSessionBackfillDate()] } : null +} + /** * Backfills managed-home session rollout files into the real Codex home. * diff --git a/src/main/codex/codex-session-index-heal-state.test.ts b/src/main/codex/codex-session-index-heal-state.test.ts new file mode 100644 index 00000000000..92270911dfe --- /dev/null +++ b/src/main/codex/codex-session-index-heal-state.test.ts @@ -0,0 +1,154 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { appendFileSync, mkdirSync, mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { + CODEX_SESSION_INDEX_HEAL_VERSION, + appendHealLedgerRecord, + collectPendingHealThreads, + isHealMarkerCurrent, + writeHealMarker, + type CodexSessionIndexHealPaths +} from './codex-session-index-heal-state' + +const WINDOWS_SESSIONS_ROOT = 'C:\\Users\\Me\\.codex\\sessions' +const THREAD_ID = '019f0000-1111-7222-8333-000000000001' + +let tempRoots: string[] = [] + +afterEach(() => { + for (const root of tempRoots) { + rmSync(root, { recursive: true, force: true }) + } + tempRoots = [] +}) + +function createPaths(systemSessionsRoot = WINDOWS_SESSIONS_ROOT): CodexSessionIndexHealPaths { + const stateDir = mkdtempSync(join(tmpdir(), 'orca-codex-heal-state-')) + tempRoots.push(stateDir) + return { + auditLogPath: join(stateDir, 'audit.jsonl'), + systemSessionsRoot, + healLedgerPath: join(stateDir, 'index-heal-ledger.jsonl'), + healMarkerPath: join(stateDir, 'index-heal-complete.json') + } +} + +function appendAuditRecord(paths: CodexSessionIndexHealPaths, target: string, recordId: string) { + // Mirrors the real writer's leading newline, which quarantines a torn tail. + appendFileSync( + paths.auditLogPath, + `\n${JSON.stringify({ action: 'hardlink', source: '/managed/x.jsonl', target, recordId })}\n` + ) +} + +function windowsRolloutTarget(root: string): string { + return `${root}\\2026\\07\\01\\rollout-2026-07-01T10-00-00-${THREAD_ID}.jsonl` +} + +describe('codex session index heal state', () => { + it('keeps the heal marker current across Windows spellings of one target', () => { + const paths = createPaths() + writeHealMarker(paths, 42, { healedThreads: 1, missingThreads: 0, failedThreads: 0 }) + + for (const alias of ['C:/Users/Me/.codex/sessions', 'c:\\users\\me\\.codex\\sessions']) { + expect(isHealMarkerCurrent({ ...paths, systemSessionsRoot: alias }, 42)).toBe(true) + } + }) + + it('still re-heals when the real target path actually changes', () => { + const paths = createPaths() + writeHealMarker(paths, 42, { healedThreads: 1, missingThreads: 0, failedThreads: 0 }) + + expect( + isHealMarkerCurrent( + { ...paths, systemSessionsRoot: 'C:\\Users\\Me\\moved-codex\\sessions' }, + 42 + ) + ).toBe(false) + }) + + it('treats a heal record as processed regardless of how its root was spelled', async () => { + const paths = createPaths() + appendAuditRecord(paths, windowsRolloutTarget(WINDOWS_SESSIONS_ROOT), 'audit-1') + appendHealLedgerRecord( + { ...paths, systemSessionsRoot: 'c:/users/me/.codex/sessions' }, + THREAD_ID, + 'healed', + 'audit-1' + ) + + expect(await collectPendingHealThreads(paths)).toEqual([]) + }) + + it('re-heals a thread whose record belongs to a different real home', async () => { + const paths = createPaths() + appendAuditRecord(paths, windowsRolloutTarget(WINDOWS_SESSIONS_ROOT), 'audit-1') + appendHealLedgerRecord( + { ...paths, systemSessionsRoot: 'C:\\Users\\Me\\moved-codex\\sessions' }, + THREAD_ID, + 'healed', + 'audit-1' + ) + + expect(await collectPendingHealThreads(paths)).toEqual([ + expect.objectContaining({ threadId: THREAD_ID, auditRecordId: 'audit-1' }) + ]) + }) + + it('re-queues a thread when a later publication event supersedes a healed one', async () => { + const paths = createPaths() + appendAuditRecord(paths, windowsRolloutTarget(WINDOWS_SESSIONS_ROOT), 'audit-1') + appendHealLedgerRecord(paths, THREAD_ID, 'healed', 'audit-1') + appendAuditRecord(paths, windowsRolloutTarget(WINDOWS_SESSIONS_ROOT), 'audit-2') + + expect(await collectPendingHealThreads(paths)).toEqual([ + expect.objectContaining({ threadId: THREAD_ID, auditRecordId: 'audit-2' }) + ]) + }) + + it('leaves the main thread free while walking a large audit ledger', async () => { + const paths = createPaths() + for (let index = 0; index < 20_000; index += 1) { + appendAuditRecord( + paths, + `${WINDOWS_SESSIONS_ROOT}\\2026\\07\\01\\rollout-2026-07-01T10-00-00-019f0000-1111-7222-8333-${String(index).padStart(12, '0')}.jsonl`, + `audit-${index}` + ) + } + let ticks = 0 + const ticker = setInterval(() => { + ticks += 1 + }, 1) + + const pending = await collectPendingHealThreads(paths) + clearInterval(ticker) + + expect(pending).toHaveLength(20_000) + // A blocking readFileSync + whole-file JSON.parse would starve every timer. + expect(ticks).toBeGreaterThan(0) + }) + + it('refuses to treat an unreadable audit ledger as an empty work queue', async () => { + const paths = createPaths() + // A directory in the audit's place surfaces EISDIR rather than ENOENT. + mkdirSync(paths.auditLogPath, { recursive: true }) + + await expect(collectPendingHealThreads(paths)).rejects.toThrow() + }) + + it('treats a missing audit ledger as no pending work', async () => { + await expect(collectPendingHealThreads(createPaths())).resolves.toEqual([]) + }) + + it('skips torn ledger lines instead of failing the pass', async () => { + const paths = createPaths() + appendFileSync(paths.auditLogPath, '{"action":"hardlink","target":"/x/rollout') + appendAuditRecord(paths, windowsRolloutTarget(WINDOWS_SESSIONS_ROOT), 'audit-1') + + expect(await collectPendingHealThreads(paths)).toEqual([ + expect.objectContaining({ threadId: THREAD_ID }) + ]) + expect(CODEX_SESSION_INDEX_HEAL_VERSION).toBe(3) + }) +}) diff --git a/src/main/codex/codex-session-index-heal-state.ts b/src/main/codex/codex-session-index-heal-state.ts index ec7b0caf9e0..0a5fd0499c1 100644 --- a/src/main/codex/codex-session-index-heal-state.ts +++ b/src/main/codex/codex-session-index-heal-state.ts @@ -5,6 +5,7 @@ import { normalizeRuntimePathForComparison } from '../../shared/cross-platform-path' import { writeFileAtomically } from '../codex-accounts/fs-utils' +import { streamCodexSessionLedgerRecords } from './codex-session-ledger-stream' // State files for the session index heal: which backfilled rollouts exist // (the backfill audit ledger), which thread ids this pass already processed @@ -50,10 +51,14 @@ export type HealMarkerSummary = { * Diffs the backfill audit ledger against the heal ledger: every hardlinked or * copied rollout whose thread id has not been processed yet, most recent first. */ -export function collectPendingHealThreads(paths: CodexSessionIndexHealPaths): PendingHealThread[] { - const processed = readProcessedHealThreads(paths) +export async function collectPendingHealThreads( + paths: CodexSessionIndexHealPaths +): Promise { + const processed = await readProcessedHealThreads(paths) const pendingByThreadId = new Map() - for (const line of readJsonlLines(paths.auditLogPath, true)) { + for await (const line of streamCodexSessionLedgerRecords(paths.auditLogPath, { + throwOnReadFailure: true + })) { if (line.action !== 'hardlink' && line.action !== 'copy' && line.action !== 'existing') { continue } @@ -94,18 +99,18 @@ function lastPathSegment(filePath: string): string { return filePath.split(/[\\/]/).at(-1) ?? '' } -function readProcessedHealThreads(paths: CodexSessionIndexHealPaths): { +async function readProcessedHealThreads(paths: CodexSessionIndexHealPaths): Promise<{ healedAuditRecords: Set legacyHealedThreadIds: Set missingAuditRecords: Set legacyMissingThreadIds: Set -} { +}> { const healedAuditRecords = new Set() const legacyHealedThreadIds = new Set() const missingAuditRecords = new Set() const legacyMissingThreadIds = new Set() const expectedRoot = normalizeRuntimePathForComparison(paths.systemSessionsRoot) - for (const line of readJsonlLines(paths.healLedgerPath)) { + for await (const line of streamCodexSessionLedgerRecords(paths.healLedgerPath)) { if ( line.v === CODEX_SESSION_INDEX_HEAL_VERSION && typeof line.threadId === 'string' && @@ -165,35 +170,6 @@ export function appendHealLedgerRecord( } } -function readJsonlLines(filePath: string, throwOnReadFailure = false): Record[] { - let contents: string - try { - contents = readFileSync(filePath, 'utf-8') - } catch (error) { - if (throwOnReadFailure && !isNotFoundError(error)) { - // Why: the audit is the heal work queue. Treating EACCES/EIO as empty - // would write a completion marker that permanently skips every session. - throw error - } - return [] - } - const lines: Record[] = [] - for (const raw of contents.split('\n')) { - if (!raw.trim()) { - continue - } - try { - const parsed: unknown = JSON.parse(raw) - if (parsed && typeof parsed === 'object' && !Array.isArray(parsed)) { - lines.push(parsed as Record) - } - } catch { - // Skip torn/corrupt lines; both ledgers are append-only diagnostics. - } - } - return lines -} - export function readAuditLogSize(auditLogPath: string): number { try { return statSync(auditLogPath).size @@ -225,9 +201,13 @@ export function isHealMarkerCurrent( unsupportedAt?: unknown retryableFailureAt?: unknown } + // Why: one Windows directory has several spellings (drive case, separators), + // so a raw compare re-drives the whole heal for what is the same target. if ( marker.version !== CODEX_SESSION_INDEX_HEAL_VERSION || - marker.systemSessionsRoot !== paths.systemSessionsRoot + typeof marker.systemSessionsRoot !== 'string' || + normalizeRuntimePathForComparison(marker.systemSessionsRoot) !== + normalizeRuntimePathForComparison(paths.systemSessionsRoot) ) { return false } diff --git a/src/main/codex/codex-session-index-heal.ts b/src/main/codex/codex-session-index-heal.ts index baaba13358b..645d96a76b4 100644 --- a/src/main/codex/codex-session-index-heal.ts +++ b/src/main/codex/codex-session-index-heal.ts @@ -118,7 +118,7 @@ export async function runCodexSessionIndexHeal( } } - const pending = collectPendingHealThreads(paths) + const pending = await collectPendingHealThreads(paths) const summary: CodexSessionIndexHealSummary = { outcome: 'completed', pendingThreads: pending.length, diff --git a/src/main/codex/codex-session-ledger-stream.ts b/src/main/codex/codex-session-ledger-stream.ts new file mode 100644 index 00000000000..b9c817bfa21 --- /dev/null +++ b/src/main/codex/codex-session-ledger-stream.ts @@ -0,0 +1,55 @@ +import { createReadStream } from 'node:fs' +import { createInterface } from 'node:readline' + +/** + * Streams an append-only JSONL ledger one record at a time. + * + * Why: the backfill audit holds one line per published session file and neither + * ledger is ever compacted, so a large Codex history makes a `readFileSync` plus + * whole-file `JSON.parse` a multi-megabyte block of the Electron main thread. + * Reading chunk by chunk keeps the window responsive while the pass runs. + */ +export async function* streamCodexSessionLedgerRecords( + filePath: string, + options: { throwOnReadFailure?: boolean } = {} +): AsyncGenerator> { + const lines = createInterface({ + input: createReadStream(filePath, { encoding: 'utf-8' }), + crlfDelay: Infinity + }) + try { + for await (const raw of lines) { + const record = parseLedgerRecord(raw) + if (record) { + yield record + } + } + } catch (error) { + if (options.throwOnReadFailure && !isNotFoundError(error)) { + // Why: the audit is the heal work queue. Treating EACCES/EIO as empty + // would write a completion marker that permanently skips every session. + throw error + } + } finally { + lines.close() + } +} + +function parseLedgerRecord(raw: string): Record | null { + if (!raw.trim()) { + return null + } + try { + const parsed: unknown = JSON.parse(raw) + // Torn tails are quarantined by the writer's leading newline; skip them. + return parsed && typeof parsed === 'object' && !Array.isArray(parsed) + ? (parsed as Record) + : null + } catch { + return null + } +} + +function isNotFoundError(error: unknown): boolean { + return (error as NodeJS.ErrnoException | null)?.code === 'ENOENT' +} diff --git a/src/main/codex/codex-session-migration-scheduler.test.ts b/src/main/codex/codex-session-migration-scheduler.test.ts index b495afb8e0e..d57041bc4e5 100644 --- a/src/main/codex/codex-session-migration-scheduler.test.ts +++ b/src/main/codex/codex-session-migration-scheduler.test.ts @@ -94,6 +94,57 @@ describe('createCodexSessionMigrationScheduler', () => { ) }) + it('publishes the baseline while a Codex pane is still open', async () => { + vi.setSystemTime(new Date('2026-08-05T10:00:00Z')) + const prepareScheduledRun = vi.fn() + const startBackfill = vi.fn().mockResolvedValue({ stopped: false }) + const scheduler = createCodexSessionMigrationScheduler({ + isEligible: () => true, + isQuitting: () => false, + resolveSystemCodexHomePathOverride: () => undefined, + prepareScheduledRun, + startBackfill, + startIndexHeal: vi.fn().mockResolvedValue(null), + initialDelayMs: 1_000 + }) + + scheduler.beginLaunch('pty-1') + await vi.advanceTimersByTimeAsync(1_000) + await vi.waitFor(() => expect(startBackfill).toHaveBeenCalledOnce()) + + const options = startBackfill.mock.calls[0]?.[0] + // The open pane only holds its own date pending; publication is not blocked. + expect(options?.retainPendingScanDates).toBe(true) + expect(options?.canWriteCompletionMarker?.()).toBe(true) + // Preparation is handed the dates so it can persist them before the walk. + expect(prepareScheduledRun).toHaveBeenCalledWith([['2026', '08', '05']]) + }) + + it('hands preparation every date a cross-midnight launch spanned', async () => { + vi.setSystemTime(new Date('2026-08-05T23:59:59Z')) + const prepareScheduledRun = vi.fn() + const scheduler = createCodexSessionMigrationScheduler({ + isEligible: () => true, + isQuitting: () => false, + resolveSystemCodexHomePathOverride: () => undefined, + prepareScheduledRun, + startBackfill: vi.fn().mockResolvedValue({ stopped: false }), + startIndexHeal: vi.fn().mockResolvedValue(null), + initialDelayMs: 1_000 + }) + + scheduler.beginLaunch('pty-1') + vi.setSystemTime(new Date('2026-08-07T01:00:00Z')) + scheduler.finishLaunch('pty-1') + await vi.advanceTimersByTimeAsync(1_000) + + expect(prepareScheduledRun).toHaveBeenLastCalledWith([ + ['2026', '08', '05'], + ['2026', '08', '06'], + ['2026', '08', '07'] + ]) + }) + it('keeps launch passes full when no completed baseline can cover older failures', async () => { const startBackfill = vi.fn().mockResolvedValue({ stopped: false }) const scheduler = createCodexSessionMigrationScheduler({ @@ -375,7 +426,7 @@ describe('createCodexSessionMigrationScheduler', () => { await vi.advanceTimersByTimeAsync(1_000) await vi.waitFor(() => expect(startBackfill).toHaveBeenCalledOnce()) expect(startBackfill).toHaveBeenLastCalledWith( - expect.objectContaining({ writeCompletionMarker: false }), + expect.objectContaining({ retainPendingScanDates: true }), undefined ) expect(finishScheduledRun).not.toHaveBeenCalled() @@ -392,8 +443,8 @@ describe('createCodexSessionMigrationScheduler', () => { ['2026', '08', '07'] ], ignoreCompletionMarker: true, - writeCompletionMarker: true, - writeBoundedCompletionMarker: true + retainPendingScanDates: false, + fullScanRequired: false }), undefined ) @@ -421,7 +472,7 @@ describe('createCodexSessionMigrationScheduler', () => { await vi.advanceTimersByTimeAsync(1_000) await vi.waitFor(() => expect(startBackfill).toHaveBeenCalledOnce()) expect(startBackfill).toHaveBeenLastCalledWith( - expect.objectContaining({ scanDates: undefined, writeBoundedCompletionMarker: false }), + expect.objectContaining({ scanDates: undefined, fullScanRequired: true }), undefined ) @@ -429,7 +480,7 @@ describe('createCodexSessionMigrationScheduler', () => { await vi.advanceTimersByTimeAsync(1_000) await vi.waitFor(() => expect(startBackfill).toHaveBeenCalledTimes(2)) expect(startBackfill).toHaveBeenLastCalledWith( - expect.objectContaining({ scanDates: undefined, writeBoundedCompletionMarker: false }), + expect.objectContaining({ scanDates: undefined, fullScanRequired: true }), undefined ) await vi.waitFor(() => expect(finishScheduledRun).toHaveBeenCalledOnce()) @@ -485,7 +536,7 @@ describe('createCodexSessionMigrationScheduler', () => { await vi.advanceTimersByTimeAsync(1_000) expect(startBackfill).toHaveBeenCalledWith( - expect.objectContaining({ scanDates: undefined, writeBoundedCompletionMarker: false }), + expect.objectContaining({ scanDates: undefined, fullScanRequired: true }), undefined ) await vi.waitFor(() => expect(finishScheduledRun).toHaveBeenCalledOnce()) @@ -510,8 +561,8 @@ describe('createCodexSessionMigrationScheduler', () => { expect(startBackfill).toHaveBeenCalledWith( expect.objectContaining({ scanDates: expect.any(Array), - writeCompletionMarker: false, - writeBoundedCompletionMarker: false + retainPendingScanDates: true, + fullScanRequired: false }), undefined ) @@ -536,8 +587,8 @@ describe('createCodexSessionMigrationScheduler', () => { expect(startBackfill).toHaveBeenCalledWith( expect.objectContaining({ scanDates: expect.any(Array), - writeCompletionMarker: false, - writeBoundedCompletionMarker: false + retainPendingScanDates: true, + fullScanRequired: false }), undefined ) @@ -560,14 +611,14 @@ describe('createCodexSessionMigrationScheduler', () => { await vi.advanceTimersByTimeAsync(1_000) expect(startBackfill).toHaveBeenCalledWith( - expect.objectContaining({ writeCompletionMarker: false }), + expect.objectContaining({ retainPendingScanDates: true }), undefined ) scheduler.finishLaunch('stable-pty', 4) await vi.advanceTimersByTimeAsync(1_000) expect(startBackfill).toHaveBeenLastCalledWith( - expect.objectContaining({ writeCompletionMarker: true }), + expect.objectContaining({ retainPendingScanDates: false }), undefined ) }) @@ -588,14 +639,14 @@ describe('createCodexSessionMigrationScheduler', () => { scheduler.beginLaunch('stable-pty', false, new Date(), 2) await vi.advanceTimersByTimeAsync(61_000) expect(startBackfill).toHaveBeenLastCalledWith( - expect.objectContaining({ writeCompletionMarker: false }), + expect.objectContaining({ retainPendingScanDates: true }), undefined ) scheduler.finishLaunch('stable-pty', 3) await vi.advanceTimersByTimeAsync(1_000) expect(startBackfill).toHaveBeenLastCalledWith( - expect.objectContaining({ writeCompletionMarker: true }), + expect.objectContaining({ retainPendingScanDates: false }), undefined ) }) @@ -623,7 +674,7 @@ describe('createCodexSessionMigrationScheduler', () => { await vi.advanceTimersByTimeAsync(1_000) expect(startBackfill).toHaveBeenCalledTimes(2) expect(startBackfill).toHaveBeenLastCalledWith( - expect.objectContaining({ writeCompletionMarker: true }), + expect.objectContaining({ retainPendingScanDates: false }), undefined ) }) @@ -646,7 +697,7 @@ describe('createCodexSessionMigrationScheduler', () => { scheduler.beginLaunch('stable-pty', false, new Date(), 5) await vi.advanceTimersByTimeAsync(1_000) expect(startBackfill).toHaveBeenLastCalledWith( - expect.objectContaining({ writeCompletionMarker: false }), + expect.objectContaining({ retainPendingScanDates: true }), undefined ) @@ -654,7 +705,7 @@ describe('createCodexSessionMigrationScheduler', () => { await vi.advanceTimersByTimeAsync(1_000) expect(startBackfill).toHaveBeenCalledTimes(2) expect(startBackfill).toHaveBeenLastCalledWith( - expect.objectContaining({ writeCompletionMarker: true }), + expect.objectContaining({ retainPendingScanDates: false }), undefined ) }) @@ -675,7 +726,7 @@ describe('createCodexSessionMigrationScheduler', () => { await vi.advanceTimersByTimeAsync(1_000) expect(startBackfill).toHaveBeenCalledWith( - expect.objectContaining({ writeCompletionMarker: false }), + expect.objectContaining({ retainPendingScanDates: true }), undefined ) }) diff --git a/src/main/codex/codex-session-migration-scheduler.ts b/src/main/codex/codex-session-migration-scheduler.ts index 9f08d83c5ea..d0f696031ae 100644 --- a/src/main/codex/codex-session-migration-scheduler.ts +++ b/src/main/codex/codex-session-migration-scheduler.ts @@ -1,4 +1,9 @@ -import { getCodexSessionBackfillDate } from './codex-session-backfill-date' +import { + compareCodexSessionBackfillDates, + getCodexSessionBackfillDate, + getCodexSessionBackfillDatesBetween, + toCodexSessionBackfillDateKey +} from './codex-session-backfill-scan-dates' import { CodexSessionMigrationIgnoredLaunches } from './codex-session-migration-ignored-launches' import { CodexSessionMigrationRecentExits } from './codex-session-migration-recent-exits' import type { @@ -29,7 +34,7 @@ export function createCodexSessionMigrationScheduler(args: { isEligible: () => boolean isQuitting: () => boolean resolveSystemCodexHomePathOverride: () => string | undefined - prepareScheduledRun?: () => boolean | void + prepareScheduledRun?: (scanDates: readonly CodexSessionBackfillDate[]) => boolean | void finishScheduledRun?: () => void startBackfill: MigrationRun startIndexHeal: MigrationRun @@ -63,7 +68,7 @@ export function createCodexSessionMigrationScheduler(args: { pendingScheduledRunGeneration ?? requestedGeneration ) for (const scanDate of requestedScanDates) { - pendingScanDates.set(scanDate.join('-'), scanDate) + pendingScanDates.set(toCodexSessionBackfillDateKey(scanDate), scanDate) } pendingFullScan ||= requestedFullScan } @@ -77,17 +82,16 @@ export function createCodexSessionMigrationScheduler(args: { } const isScheduledRun = pendingScheduledRunGeneration !== null const activeScheduledRunGeneration = pendingScheduledRunGeneration + const runScanDates = [...pendingScanDates.values()].sort(compareCodexSessionBackfillDates) let preparationNeedsFullScan = false if (isScheduledRun) { pendingScheduledRunGeneration = null - // Why: an older active pass can rewrite the marker after launch invalidates it. - preparationNeedsFullScan = args.prepareScheduledRun?.() === true + // Why: preparation persists these dates so an abnormal exit still yields a + // bounded recovery window instead of another full-tree walk. + preparationNeedsFullScan = args.prepareScheduledRun?.(runScanDates) === true } const fullScanRequired = pendingFullScan || preparationNeedsFullScan - const scanDates = - !fullScanRequired && pendingScanDates.size > 0 - ? [...pendingScanDates.values()].sort(compareBackfillDates) - : undefined + const scanDates = !fullScanRequired && runScanDates.length > 0 ? runScanDates : undefined pendingScanDates.clear() pendingFullScan = false activeRunStopObserved = false @@ -105,12 +109,12 @@ export function createCodexSessionMigrationScheduler(args: { { shouldStop, scanDates, + fullScanRequired, ignoreCompletionMarker: isScheduledRun, - writeCompletionMarker: activeLaunches.size === 0, - writeBoundedCompletionMarker: - isScheduledRun && activeLaunches.size === 0 && !fullScanRequired, + // Why: a live pane keeps appending to its own date directory, so that + // date stays pending — but the historical baseline is still certified. + retainPendingScanDates: activeLaunches.size > 0, canWriteCompletionMarker: () => - activeLaunches.size === 0 && scheduledTimer === null && pendingScheduledRunGeneration === null && (!isScheduledRun || activeScheduledRunGeneration === scheduledRunGeneration) @@ -144,7 +148,7 @@ export function createCodexSessionMigrationScheduler(args: { ) pendingFullScan ||= fullScanRequired for (const scanDate of scanDates ?? []) { - pendingScanDates.set(scanDate.join('-'), scanDate) + pendingScanDates.set(toCodexSessionBackfillDateKey(scanDate), scanDate) } } if ( @@ -168,9 +172,9 @@ export function createCodexSessionMigrationScheduler(args: { scheduledTimer = null if (generation !== undefined) { const currentDate = getCodexSessionBackfillDate() - scheduledScanDates.set(currentDate.join('-'), currentDate) + scheduledScanDates.set(toCodexSessionBackfillDateKey(currentDate), currentDate) } - const scanDates = [...scheduledScanDates.values()].sort(compareBackfillDates) + const scanDates = [...scheduledScanDates.values()].sort(compareCodexSessionBackfillDates) scheduledScanDates.clear() const fullScanRequired = scheduledFullScan scheduledFullScan = false @@ -186,7 +190,7 @@ export function createCodexSessionMigrationScheduler(args: { scheduledRunGeneration += 1 scheduledFullScan ||= fullScanRequired const launchDate = getCodexSessionBackfillDate() - scheduledScanDates.set(launchDate.join('-'), launchDate) + scheduledScanDates.set(toCodexSessionBackfillDateKey(launchDate), launchDate) armScheduledRun(scheduledRunGeneration) } @@ -235,7 +239,7 @@ export function createCodexSessionMigrationScheduler(args: { return } for (const scanDate of getCodexSessionBackfillDatesBetween(startedAt, new Date())) { - scheduledScanDates.set(scanDate.join('-'), scanDate) + scheduledScanDates.set(toCodexSessionBackfillDateKey(scanDate), scanDate) } scheduleRun() }, @@ -249,31 +253,6 @@ export function createCodexSessionMigrationScheduler(args: { } } -function getCodexSessionBackfillDatesBetween( - startedAt: Date, - finishedAt: Date -): CodexSessionBackfillDate[] { - const dates: CodexSessionBackfillDate[] = [] - const cursor = new Date( - Date.UTC(startedAt.getUTCFullYear(), startedAt.getUTCMonth(), startedAt.getUTCDate()) - ) - const last = new Date( - Date.UTC(finishedAt.getUTCFullYear(), finishedAt.getUTCMonth(), finishedAt.getUTCDate()) - ) - while (cursor <= last) { - dates.push(getCodexSessionBackfillDate(cursor)) - cursor.setUTCDate(cursor.getUTCDate() + 1) - } - return dates -} - -function compareBackfillDates( - left: CodexSessionBackfillDate, - right: CodexSessionBackfillDate -): number { - return left.join('-').localeCompare(right.join('-')) -} - function isStoppedMigrationResult(result: unknown): boolean { return Boolean(result && typeof result === 'object' && 'stopped' in result && result.stopped) } diff --git a/src/main/index.ts b/src/main/index.ts index edc26214121..df30a05638e 100644 --- a/src/main/index.ts +++ b/src/main/index.ts @@ -2586,7 +2586,8 @@ void app.whenReady().then(async () => { isQuitting: () => isQuitting, resolveSystemCodexHomePathOverride: () => resolveHostCodexSessionSourceHome(store!.getSettings()), - prepareScheduledRun: () => codexRuntimeHome?.prepareHostSystemDefaultSessionMigrationPass(), + prepareScheduledRun: (scanDates) => + codexRuntimeHome?.prepareHostSystemDefaultSessionMigrationPass(scanDates), finishScheduledRun: () => codexRuntimeHome?.finishHostSystemDefaultSessionMigrationPass(), startBackfill: startCodexSessionBackfillInBackground, startIndexHeal: startCodexSessionIndexHealInBackground From 64c992cd56ac2ed2414a71f8c8e3816fe3ba22a7 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 26 Aug 2026 15:43:02 -0700 Subject: [PATCH 11/19] fix(memory): report the Windows number that predicts paging, not just resident pages (#16211) (#16589) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(memory): report Windows commit charge, not just working set (#16211) On Windows the per-process figure was working set — resident pages only. An agent whose pages Windows has trimmed to the pagefile shrinks its working set while still holding the commit that pushes the host into paging, so Resource Manager and `orca diagnostics memory` understated an owned tree by 10-40x (9 codex.exe: 1.4 GB working set, 13.4 GB private) and could not warn before the host was already thrashing. Add committed private bytes as a second, separately-labelled quantity rather than redefining the existing one: - CIM sweep gains one property (PageFileUsage, UInt32 KB); the typeperf fallback gains one counter (\Process(*)\Private Bytes). Both ride the sweep that already runs. - MemorySnapshot gains optional `privateMemory` per app/worktree/session plus `processCommitMetric` and `totalPrivateMemory`. Rule 1 additive optional fields: old clients ignore them, and absence reads as "not measured", never as zero — Unix hosts and older hosts send nothing. - `totalMemory` and `processMemoryMetric` keep their exact meaning, so the "shared pages may repeat" copy stays true; the working-set copy now also says paged-out memory is not counted. - Resource Manager shows "Σ Private" beside "Σ WS", and tints the badge yellow/red once tracked commit passes 60/80% of physical RAM — the same thresholds `usageTextColorClass` already uses for host usage. Tint and tooltip only; no toast, and the badge number is unchanged. The parsers move to windows-process-sample-parsing.ts and the Windows sweep tests to their own file to stay under max-lines. Not migrating the collector to windows-process-table.ts: the native snapshot exposes no commit figure and no CPU times, and truncates WorkingSetSize through a DWORD. Documented in the enumeration reference. * fix(memory): derive the typeperf field cap from the counter list The fallback parser's 8192-field cap was sized for three `\Process(*)` counters. Adding `Private Bytes` cut the parsable process count from ~2730 to ~2047, and overrun is a blackout (`parseTypeperfCsvLine` returns `[]`, so the whole sweep reports nothing) rather than a truncation. The counter list now lives beside the decoder that reads those names back out of the PDH header, and the cap is derived from it. Also collapses the four spellings of "omit privateMemory when unmeasured" in collector.ts onto one `commitField` helper, drops the unread parameter and the never-rendered `columnLabel` from `getResourceCommitMetricCopy`, folds `getCommitPressurePercent` into the only function that called it, and reverts unrelated Prettier churn in the Windows enumeration doc. The commit tint's doc comment no longer claims to predict host paging: it measures Orca's own share of physical RAM. Host commit charge / commit limit stays a follow-up (#16211). --- .../remote-shared-control-retirement-probe.ts | 4 + docs/reference/windows-process-enumeration.md | 7 + src/cli/index-memory-diagnostics.test.ts | 107 +++ src/cli/workspace-format.ts | 22 + .../memory/collector-windows-sweep.test.ts | 620 ++++++++++++++++++ src/main/memory/collector.test.ts | 430 +----------- src/main/memory/collector.ts | 95 ++- .../windows-process-resource-collector.ts | 213 +----- .../windows-process-sample-parsing.test.ts | 140 ++++ .../memory/windows-process-sample-parsing.ts | 240 +++++++ .../status-bar/ResourceUsageStatusSegment.tsx | 80 ++- .../resource-memory-metric-copy.test.ts | 53 +- .../status-bar/resource-memory-metric-copy.ts | 39 +- src/renderer/src/i18n/locales/en.json | 3 +- src/renderer/src/i18n/locales/es.json | 3 +- src/renderer/src/i18n/locales/ja.json | 3 +- src/renderer/src/i18n/locales/ko.json | 3 +- src/renderer/src/i18n/locales/zh.json | 3 +- src/shared/process-stats-types.ts | 31 + 19 files changed, 1436 insertions(+), 660 deletions(-) create mode 100644 src/main/memory/collector-windows-sweep.test.ts create mode 100644 src/main/memory/windows-process-sample-parsing.test.ts create mode 100644 src/main/memory/windows-process-sample-parsing.ts diff --git a/config/scripts/remote-shared-control-retirement-probe.ts b/config/scripts/remote-shared-control-retirement-probe.ts index 629ba1fddde..8c38b312a1d 100644 --- a/config/scripts/remote-shared-control-retirement-probe.ts +++ b/config/scripts/remote-shared-control-retirement-probe.ts @@ -190,8 +190,10 @@ function summarizeMemory(snapshot: MemorySnapshot): { app: MemorySnapshot['app'] host: MemorySnapshot['host'] processMemoryMetric: MemorySnapshot['processMemoryMetric'] + processCommitMetric: MemorySnapshot['processCommitMetric'] totalCpu: number totalMemory: number + totalPrivateMemory: MemorySnapshot['totalPrivateMemory'] worktreeCount: number sessionCount: number worktreeMemory: number @@ -208,8 +210,10 @@ function summarizeMemory(snapshot: MemorySnapshot): { app: snapshot.app, host: snapshot.host, processMemoryMetric: snapshot.processMemoryMetric, + processCommitMetric: snapshot.processCommitMetric, totalCpu: snapshot.totalCpu, totalMemory: snapshot.totalMemory, + totalPrivateMemory: snapshot.totalPrivateMemory, worktreeCount: snapshot.worktrees.length, sessionCount: snapshot.worktrees.reduce( (total, worktree) => total + worktree.sessions.length, diff --git a/docs/reference/windows-process-enumeration.md b/docs/reference/windows-process-enumeration.md index f79f70b4096..236d6adbf47 100644 --- a/docs/reference/windows-process-enumeration.md +++ b/docs/reference/windows-process-enumeration.md @@ -177,6 +177,13 @@ time to prove a PID has not been recycled — daemon identity, managed-hook ownership, and CPU accounting in the memory collector — still reads it through its own query. Those callers are not migrated. +Committed private bytes have no equivalent either, and the one memory value the +snapshot does carry is unusable for the sizes Orca now sees: `process.cc` stores +`pmc.WorkingSetSize` into a `DWORD`, so anything above 4 GB wraps. That is the +second reason `windows-process-resource-collector.ts` still runs its own +`Get-CimInstance` sweep — it needs `PageFileUsage` (commit) and the CPU-time +counters in the same pass. Migrating it to the native table would cost both. + Start time is a proxy for identity, not identity. The durable answer for the process trees Orca itself spawns is an inherited handle: a job object names the tree Orca created, so no start-time comparison is needed. Those readers should diff --git a/src/cli/index-memory-diagnostics.test.ts b/src/cli/index-memory-diagnostics.test.ts index d7cdcf2491b..9896c40b0aa 100644 --- a/src/cli/index-memory-diagnostics.test.ts +++ b/src/cli/index-memory-diagnostics.test.ts @@ -113,5 +113,112 @@ describe('orca cli worktree awareness', () => { expect(output).toContain('hostAvailable: 2.0 MB (free-memory)') expect(output).toContain('app: 1.0 MB') expect(output).toContain('- feature 1.0 MB 2.5% 1 session') + // Back-compat: a host that never heard of committed bytes prints none. + expect(output).not.toContain('totalPrivateMemory') + expect(output).not.toContain('processCommitMetric') + }) + + it('reports committed private bytes alongside the resident figure', async () => { + queueFixtures( + callMock, + okFixture('req_memory', { + app: { + cpu: 1.25, + memory: 1024 * 1024, + privateMemory: 2 * 1024 * 1024, + main: { cpu: 0.5, memory: 512 * 1024, privateMemory: 1024 * 1024 }, + renderer: { cpu: 0.5, memory: 384 * 1024, privateMemory: 768 * 1024 }, + other: { cpu: 0.25, memory: 128 * 1024, privateMemory: 256 * 1024 }, + history: [1024 * 1024] + }, + worktrees: [ + { + worktreeId: 'repo::/tmp/repo/feature', + worktreeName: 'feature', + repoId: 'repo', + repoName: 'Orca', + cpu: 2.5, + memory: 1024 * 1024, + privateMemory: 14 * 1024 * 1024, + sessions: [ + { + sessionId: 'pty-1', + paneKey: null, + pid: 123, + cpu: 2.5, + memory: 1024 * 1024, + privateMemory: 14 * 1024 * 1024 + } + ], + history: [1024 * 1024] + } + ], + host: { + totalMemory: 8 * 1024 * 1024, + freeMemory: 2 * 1024 * 1024, + availableMemory: 2 * 1024 * 1024, + availableMemorySource: 'free-memory', + usedMemory: 6 * 1024 * 1024, + memoryUsagePercent: 75, + cpuCoreCount: 8, + loadAverage1m: 1.25 + }, + processMemoryMetric: 'working-set', + processCommitMetric: 'private-bytes', + totalCpu: 3.75, + totalMemory: 2 * 1024 * 1024, + totalPrivateMemory: 16 * 1024 * 1024, + collectedAt: 1000 + }) + ) + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['diagnostics', 'memory'], '/tmp/repo') + + const output = logSpy.mock.calls.flat().join('\n') + expect(output).toContain('totalMemory: 2.0 MB') + expect(output).toContain('processMemoryMetric: summed working set; shared pages may repeat') + expect(output).toContain('totalPrivateMemory: 16 MB') + expect(output).toContain( + 'processCommitMetric: summed private bytes; committed memory, counted whether resident or paged out' + ) + expect(output).toContain('- feature 1.0 MB 14 MB committed 2.5% 1 session') + }) + + it('passes committed bytes through --json untouched', async () => { + const snapshot = { + app: { + cpu: 0, + memory: 1024, + privateMemory: 4096, + main: { cpu: 0, memory: 1024, privateMemory: 4096 }, + renderer: { cpu: 0, memory: 0, privateMemory: 0 }, + other: { cpu: 0, memory: 0, privateMemory: 0 }, + history: [] + }, + worktrees: [], + host: { + totalMemory: 16 * 1024, + freeMemory: 1024, + availableMemory: 1024, + availableMemorySource: 'free-memory', + usedMemory: 15 * 1024, + memoryUsagePercent: 93.75, + cpuCoreCount: 8, + loadAverage1m: 0 + }, + processMemoryMetric: 'working-set', + processCommitMetric: 'private-bytes', + totalCpu: 0, + totalMemory: 1024, + totalPrivateMemory: 4096, + collectedAt: 1000 + } + queueFixtures(callMock, okFixture('req_memory', snapshot)) + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['diagnostics', 'memory', '--json'], '/tmp/repo') + + expect(JSON.parse(logSpy.mock.calls.flat().join('\n')).result).toEqual(snapshot) }) }) diff --git a/src/cli/workspace-format.ts b/src/cli/workspace-format.ts index 008c2cf4c21..8cdfce86b74 100644 --- a/src/cli/workspace-format.ts +++ b/src/cli/workspace-format.ts @@ -16,6 +16,7 @@ export function formatMemorySnapshot(snapshot: MemorySnapshot): string { `collectedAt: ${new Date(snapshot.collectedAt).toISOString()}`, `totalMemory: ${formatByteCount(snapshot.totalMemory)}`, `processMemoryMetric: ${formatProcessMemoryMetric(snapshot.processMemoryMetric)}`, + ...formatCommitLines(snapshot), `totalCpu: ${formatCpu(snapshot.totalCpu)}`, [ `hostUsed: ${formatByteCount(snapshot.host.usedMemory)}`, @@ -51,11 +52,32 @@ function formatWorktreeMemoryLine(worktree: WorktreeMemory): string { return [ `- ${worktree.worktreeName}`, `${formatByteCount(worktree.memory)}`, + ...(worktree.privateMemory === undefined + ? [] + : [`${formatByteCount(worktree.privateMemory)} committed`]), `${formatCpu(worktree.cpu)}`, `${worktree.sessions.length} session${worktree.sessions.length === 1 ? '' : 's'}` ].join(' ') } +// Why omitted rather than zeroed: a host that predates the field, or cannot read +// commit at all, must not be printed as agents committing nothing. +function formatCommitLines(snapshot: MemorySnapshot): string[] { + if (typeof snapshot.totalPrivateMemory !== 'number') { + return [] + } + return [ + `totalPrivateMemory: ${formatByteCount(snapshot.totalPrivateMemory)}`, + `processCommitMetric: ${formatProcessCommitMetric(snapshot.processCommitMetric)}` + ] +} + +function formatProcessCommitMetric(metric: MemorySnapshot['processCommitMetric']): string { + return metric === 'private-bytes' + ? 'summed private bytes; committed memory, counted whether resident or paged out' + : `summed ${metric ?? 'unknown'}` +} + function formatCpu(cpu: number): string { return `${cpu.toFixed(1)}%` } diff --git a/src/main/memory/collector-windows-sweep.test.ts b/src/main/memory/collector-windows-sweep.test.ts new file mode 100644 index 00000000000..de180fcfb57 --- /dev/null +++ b/src/main/memory/collector-windows-sweep.test.ts @@ -0,0 +1,620 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import os from 'node:os' +import type { MemorySnapshotStore } from './collector' +import { setAppEnvironment } from '../../shared/app-environment' + +type AppMetricFixture = { + pid: number + type: string + cpu: { percentCPUUsage: number } + memory: { workingSetSize: number } +} + +const { appMetricsMock, runProcessMock, execMock, listRegisteredPtysMock } = vi.hoisted(() => ({ + appMetricsMock: vi.fn<() => AppMetricFixture[]>(() => []), + runProcessMock: vi.fn(), + execMock: vi.fn(), + listRegisteredPtysMock: vi.fn() +})) + +vi.mock('child_process', () => ({ + exec: (cmd: string, opts: unknown, cb: (err: Error | null, out: { stdout: string }) => void) => + execMock(cmd, opts, cb) +})) + +// Why mock the chokepoint for the Windows sweep: maxBuffer, timeout and the +// hidden console are its contract now, so the assertions below are about which +// query runs, not how a process is started. +vi.mock('../../shared/child-process/run-process', () => ({ + runProcess: (spec: { program: string; args?: string[] }) => runProcessMock(spec) +})) + +vi.mock('./pty-registry', () => ({ + listRegisteredPtys: listRegisteredPtysMock +})) + +function appEnvironment() { + return { + getPath: () => process.cwd(), + getAppPath: () => process.cwd(), + getVersion: () => '0.0.0-test', + isPackaged: () => false, + onWillQuit: () => {}, + exit: () => {}, + getAppMetrics: appMetricsMock + } +} + +async function loadCollector() { + vi.resetModules() + const { setAppEnvironment: setResetAppEnvironment } = await import('../../shared/app-environment') + setResetAppEnvironment(appEnvironment()) + return await import('./collector') +} + +const emptyStore = { + getWorktreeMeta: () => undefined, + getRepo: () => undefined +} satisfies MemorySnapshotStore + +describe('collectMemorySnapshot on Windows', () => { + beforeEach(() => { + setAppEnvironment(appEnvironment()) + vi.restoreAllMocks() + appMetricsMock.mockReset() + appMetricsMock.mockReturnValue([]) + runProcessMock.mockReset() + execMock.mockReset() + listRegisteredPtysMock.mockReset() + listRegisteredPtysMock.mockReturnValue([]) + }) + + function mockPsResponse(stdout: string) { + execMock.mockImplementation((_cmd, _opts, cb) => cb(null, { stdout, stderr: '' })) + runProcessMock.mockImplementation((spec: { program: string }) => + Promise.resolve({ + code: 0, + signal: null, + stdout: + spec.program === 'typeperf.exe' + ? psFixtureToTypeperfOutput(stdout) + : psFixtureToWindowsProcessOutput(stdout), + stderr: '', + timedOut: false + }) + ) + } + + function psFixtureToWindowsProcessOutput(stdout: string): string { + return stdout + .split('\n') + .map((line) => line.trim()) + .filter(Boolean) + .map((line) => { + const [pid, ppid, _cpu, rssKb] = line.split(/\s+/, 4) + const memory = Number.parseInt(rssKb ?? '', 10) + return [ + pid ?? '', + ppid ?? '', + Number.isFinite(memory) && memory > 0 ? memory * 1024 : 0, + '0', + '0', + '1' + ].join('\t') + }) + .join('\r\n') + } + + function psFixtureToTypeperfOutput(stdout: string): string { + const rows = stdout + .split('\n') + .map((line) => line.trim()) + .filter(Boolean) + .map((line, index) => { + const [pid, ppid, _cpu, rssKb] = line.split(/\s+/, 4) + const memoryKb = Number.parseInt(rssKb ?? '', 10) + return { + instance: `fixture${index}`, + pid: pid ?? '', + ppid: ppid ?? '', + memory: Number.isFinite(memoryKb) && memoryKb > 0 ? memoryKb * 1024 : 0 + } + }) + const counterColumns = (counter: string): string[] => + rows.map((row) => `"\\\\HOST\\Process(${row.instance})\\${counter}"`) + const valueColumns = (field: 'pid' | 'ppid' | 'memory'): string[] => + rows.map((row) => `"${row[field]}"`) + + return [ + [ + '"(PDH-CSV 4.0)"', + ...counterColumns('ID Process'), + ...counterColumns('Creating Process ID'), + ...counterColumns('Working Set') + ].join(','), + ['"time"', ...valueColumns('pid'), ...valueColumns('ppid'), ...valueColumns('memory')].join( + ',' + ) + ].join('\r\n') + } + + it('uses one CIM process for Windows memory and CPU sampling', async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + mockPsResponse('10 1 0 1024') + const { collectMemorySnapshot } = await loadCollector() + + await collectMemorySnapshot(emptyStore) + + expect(execMock).not.toHaveBeenCalled() + expect(runProcessMock).toHaveBeenCalledTimes(1) + const spec = runProcessMock.mock.calls[0][0] + expect(spec.program).toBe('powershell.exe') + expect(spec.args.join(' ')).toContain('Get-CimInstance Win32_Process') + expect(spec.args.join(' ')).toContain('KernelModeTime') + expect(spec.args.join(' ')).toContain('UserModeTime') + expect(spec.args.join(' ')).toContain('CreationDate') + expect(spec).toMatchObject({ maxOutputBytes: 10 * 1024 * 1024, timeoutMs: 5_000 }) + }) + + it('attributes Windows process CPU from cumulative time deltas between sweeps', async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + vi.spyOn(performance, 'now').mockReturnValueOnce(1_000).mockReturnValueOnce(3_000) + const cpuOutputs = [ + '10\t1\t1048576\t10000000\t0\t638830000000000000', + '10\t1\t1048576\t30000000\t0\t638830000000000000' + ] + runProcessMock.mockImplementation(() => + Promise.resolve({ + code: 0, + signal: null, + stdout: cpuOutputs.shift() ?? '', + stderr: '', + timedOut: false + }) + ) + listRegisteredPtysMock.mockReturnValue([ + { + ptyId: 'windows-cpu-pty', + worktreeId: 'repo-1::C:\\repo', + sessionId: 'session-1', + paneKey: 'pane-1', + pid: 10 + } + ]) + const { collectMemorySnapshot } = await loadCollector() + + const first = await collectMemorySnapshot(emptyStore) + const second = await collectMemorySnapshot(emptyStore) + + expect(first.worktrees[0].sessions[0].cpu).toBe(0) + expect(second.worktrees[0].sessions[0].cpu).toBe(100) + expect(runProcessMock.mock.calls.map(([spec]) => spec.program)).toEqual([ + 'powershell.exe', + 'powershell.exe' + ]) + }) + + it('does not attribute prior CPU time after Windows reuses a process id', async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + vi.spyOn(performance, 'now').mockReturnValueOnce(1_000).mockReturnValueOnce(3_000) + const cpuOutputs = [ + '10\t1\t1048576\t10000000\t0\t638830000000000000', + '10\t1\t1048576\t30000000\t0\t638830000000000001' + ] + runProcessMock.mockImplementation(() => + Promise.resolve({ + code: 0, + signal: null, + stdout: cpuOutputs.shift() ?? '', + stderr: '', + timedOut: false + }) + ) + listRegisteredPtysMock.mockReturnValue([ + { + ptyId: 'reused-pid-pty', + worktreeId: 'repo-1::C:\\repo', + sessionId: 'session-1', + paneKey: 'pane-1', + pid: 10 + } + ]) + const { collectMemorySnapshot } = await loadCollector() + + await collectMemorySnapshot(emptyStore) + const second = await collectMemorySnapshot(emptyStore) + + expect(second.worktrees[0].sessions[0].cpu).toBe(0) + }) + + it('supports cumulative CPU counters above JavaScript safe integers', async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + vi.spyOn(performance, 'now').mockReturnValueOnce(1_000).mockReturnValueOnce(3_000) + const cpuOutputs = [ + '10\t1\t1048576\t90071992547409920\t0\t638830000000000000', + '10\t1\t1048576\t90071992567409920\t0\t638830000000000000' + ] + runProcessMock.mockImplementation(() => + Promise.resolve({ + code: 0, + signal: null, + stdout: cpuOutputs.shift() ?? '', + stderr: '', + timedOut: false + }) + ) + listRegisteredPtysMock.mockReturnValue([ + { + ptyId: 'large-counter-pty', + worktreeId: 'repo-1::C:\\repo', + sessionId: 'session-1', + paneKey: 'pane-1', + pid: 10 + } + ]) + const { collectMemorySnapshot } = await loadCollector() + + await collectMemorySnapshot(emptyStore) + const second = await collectMemorySnapshot(emptyStore) + + expect(second.worktrees[0].sessions[0].cpu).toBe(100) + }) + + it('keeps the older CPU baseline when forced snapshots are too close together', async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + vi.spyOn(performance, 'now') + .mockReturnValueOnce(1_000) + .mockReturnValueOnce(1_100) + .mockReturnValueOnce(3_000) + const cpuOutputs = [ + '10\t1\t1048576\t0\t0\t638830000000000000', + '10\t1\t1048576\t1000000\t0\t638830000000000000', + '10\t1\t1048576\t20000000\t0\t638830000000000000' + ] + runProcessMock.mockImplementation(() => + Promise.resolve({ + code: 0, + signal: null, + stdout: cpuOutputs.shift() ?? '', + stderr: '', + timedOut: false + }) + ) + listRegisteredPtysMock.mockReturnValue([ + { + ptyId: 'short-sample-pty', + worktreeId: 'repo-1::C:\\repo', + sessionId: 'session-1', + paneKey: 'pane-1', + pid: 10 + } + ]) + const { collectMemorySnapshot } = await loadCollector() + + await collectMemorySnapshot(emptyStore) + const tooSoon = await collectMemorySnapshot(emptyStore) + const normalPoll = await collectMemorySnapshot(emptyStore) + + expect(tooSoon.worktrees[0].sessions[0].cpu).toBe(0) + expect(normalPoll.worktrees[0].sessions[0].cpu).toBe(100) + }) + + it('caps impossible Windows CPU deltas at the host core capacity', async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + vi.spyOn(os, 'cpus').mockReturnValue([{}, {}] as ReturnType) + vi.spyOn(performance, 'now').mockReturnValueOnce(1_000).mockReturnValueOnce(3_000) + const cpuOutputs = [ + '10\t1\t1048576\t0\t0\t638830000000000000', + '10\t1\t1048576\t1000000000\t0\t638830000000000000' + ] + runProcessMock.mockImplementation(() => + Promise.resolve({ + code: 0, + signal: null, + stdout: cpuOutputs.shift() ?? '', + stderr: '', + timedOut: false + }) + ) + listRegisteredPtysMock.mockReturnValue([ + { + ptyId: 'impossible-cpu-pty', + worktreeId: 'repo-1::C:\\repo', + sessionId: 'session-1', + paneKey: 'pane-1', + pid: 10 + } + ]) + const { collectMemorySnapshot } = await loadCollector() + + await collectMemorySnapshot(emptyStore) + const capped = await collectMemorySnapshot(emptyStore) + + expect(capped.worktrees[0].sessions[0].cpu).toBe(200) + }) + + it('warms CPU sampling again after Resource Manager was closed', async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + vi.spyOn(performance, 'now').mockReturnValueOnce(1_000).mockReturnValueOnce(12_000) + const cpuOutputs = [ + '10\t1\t1048576\t0\t0\t638830000000000000', + '10\t1\t1048576\t100000000\t0\t638830000000000000' + ] + runProcessMock.mockImplementation(() => + Promise.resolve({ + code: 0, + signal: null, + stdout: cpuOutputs.shift() ?? '', + stderr: '', + timedOut: false + }) + ) + listRegisteredPtysMock.mockReturnValue([ + { + ptyId: 'stale-counter-pty', + worktreeId: 'repo-1::C:\\repo', + sessionId: 'session-1', + paneKey: 'pane-1', + pid: 10 + } + ]) + const { collectMemorySnapshot } = await loadCollector() + + await collectMemorySnapshot(emptyStore) + const reopened = await collectMemorySnapshot(emptyStore) + + expect(reopened.worktrees[0].sessions[0].cpu).toBe(0) + }) + + it('preserves Windows process memory when CPU counters are unavailable', async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + runProcessMock.mockImplementation(() => + Promise.resolve({ + code: 0, + signal: null, + stdout: '10\t1\t1048576\t\t\t638830000000000000', + stderr: '', + timedOut: false + }) + ) + listRegisteredPtysMock.mockReturnValue([ + { + ptyId: 'cpu-failure-pty', + worktreeId: 'repo-1::C:\\repo', + sessionId: 'session-1', + paneKey: 'pane-1', + pid: 10 + } + ]) + const { collectMemorySnapshot } = await loadCollector() + + const snapshot = await collectMemorySnapshot(emptyStore) + + expect(snapshot.worktrees[0].sessions[0]).toMatchObject({ cpu: 0, memory: 1024 * 1024 }) + }) + + it('uses Typeperf during the CIM retry cooldown', async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + runProcessMock.mockImplementation((spec: { program: string }) => + Promise.resolve( + spec.program === 'powershell.exe' + ? { code: 1, signal: null, stdout: '', stderr: 'CIM unavailable', timedOut: false } + : { + code: 0, + signal: null, + stdout: psFixtureToTypeperfOutput('10 1 0 1024'), + stderr: '', + timedOut: false + } + ) + ) + listRegisteredPtysMock.mockReturnValue([ + { + ptyId: 'cim-pty', + worktreeId: null, + sessionId: null, + paneKey: null, + pid: 10 + } + ]) + const { collectMemorySnapshot } = await loadCollector() + + const first = await collectMemorySnapshot(emptyStore) + const second = await collectMemorySnapshot(emptyStore) + + expect(runProcessMock).toHaveBeenCalledTimes(3) + expect(runProcessMock.mock.calls.map(([spec]) => spec.program)).toEqual([ + 'powershell.exe', + 'typeperf.exe', + 'typeperf.exe' + ]) + expect(runProcessMock.mock.calls[1][0]).toMatchObject({ timeoutMs: 5_000 }) + expect(first.worktrees[0].memory).toBe(1048576) + expect(second.worktrees[0].memory).toBe(1048576) + }) + + it('retries CIM after fallback and warms CPU sampling before restoring deltas', async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + vi.spyOn(performance, 'now') + .mockReturnValueOnce(1_000) + .mockReturnValueOnce(2_000) + .mockReturnValueOnce(31_001) + .mockReturnValueOnce(32_000) + .mockReturnValueOnce(34_000) + const cimOutputs = [ + '10\t1\t1048576\t10000000\t0\t638830000000000000', + '10\t1\t1048576\t30000000\t0\t638830000000000000' + ] + let cimCalls = 0 + runProcessMock.mockImplementation((spec: { program: string }) => { + if (spec.program === 'typeperf.exe') { + return Promise.resolve({ + code: 0, + signal: null, + stdout: psFixtureToTypeperfOutput('10 1 0 1024'), + stderr: '', + timedOut: false + }) + } + cimCalls += 1 + return Promise.resolve( + cimCalls === 1 + ? { code: 1, signal: null, stdout: '', stderr: 'transient CIM failure', timedOut: false } + : { + code: 0, + signal: null, + stdout: cimOutputs.shift() ?? '', + stderr: '', + timedOut: false + } + ) + }) + listRegisteredPtysMock.mockReturnValue([ + { + ptyId: 'recovering-cim-pty', + worktreeId: 'repo-1::C:\\repo', + sessionId: 'session-1', + paneKey: 'pane-1', + pid: 10 + } + ]) + const { collectMemorySnapshot } = await loadCollector() + + await collectMemorySnapshot(emptyStore) + await collectMemorySnapshot(emptyStore) + const warming = await collectMemorySnapshot(emptyStore) + const recovered = await collectMemorySnapshot(emptyStore) + + expect(runProcessMock.mock.calls.map(([spec]) => spec.program)).toEqual([ + 'powershell.exe', + 'typeperf.exe', + 'typeperf.exe', + 'powershell.exe', + 'powershell.exe' + ]) + expect(warming.worktrees[0].sessions[0].cpu).toBe(0) + expect(recovered.worktrees[0].sessions[0].cpu).toBe(100) + }) + + it('sums committed private bytes across the whole PTY subtree on Windows', async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + // Working set stays small while commit is 10-40x larger — the reported shape. + const rows = [ + '10\t1\t52428800\t0\t0\t638830000000000000\t1048576', + '11\t10\t104857600\t0\t0\t638830000000000000\t2097152', + '12\t11\t52428800\t0\t0\t638830000000000000\t524288', + '900\t1\t20971520\t0\t0\t638830000000000000\t262144' + ].join('\r\n') + runProcessMock.mockResolvedValue({ + code: 0, + signal: null, + stdout: rows, + stderr: '', + timedOut: false + }) + appMetricsMock.mockReturnValue([ + { pid: 900, type: 'Browser', cpu: { percentCPUUsage: 0 }, memory: { workingSetSize: 0 } } + ]) + listRegisteredPtysMock.mockReturnValue([ + { + ptyId: 'pty-1', + worktreeId: 'repo-1::C:\\repo', + sessionId: 'session-1', + paneKey: 'pane-1', + pid: 10 + } + ]) + const { collectMemorySnapshot } = await loadCollector() + + const snapshot = await collectMemorySnapshot(emptyStore) + + const committedKb = 1048576 + 2097152 + 524288 + expect(snapshot.worktrees[0].sessions[0].privateMemory).toBe(committedKb * 1024) + expect(snapshot.worktrees[0].privateMemory).toBe(committedKb * 1024) + expect(snapshot.app.privateMemory).toBe(262144 * 1024) + expect(snapshot.totalPrivateMemory).toBe((committedKb + 262144) * 1024) + expect(snapshot.processCommitMetric).toBe('private-bytes') + // The resident figure keeps its old meaning rather than being redefined. + expect(snapshot.processMemoryMetric).toBe('working-set') + expect(snapshot.worktrees[0].memory).toBe(52428800 + 104857600 + 52428800) + }) + + it('omits the commit metric entirely when the Windows sweep cannot report it', async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + runProcessMock.mockResolvedValue({ + code: 0, + signal: null, + stdout: '10\t1\t52428800\t0\t0\t638830000000000000', + stderr: '', + timedOut: false + }) + listRegisteredPtysMock.mockReturnValue([ + { + ptyId: 'pty-1', + worktreeId: 'repo-1::C:\\repo', + sessionId: 'session-1', + paneKey: 'pane-1', + pid: 10 + } + ]) + const { collectMemorySnapshot } = await loadCollector() + + const snapshot = await collectMemorySnapshot(emptyStore) + + // Why not zero: a host that cannot measure commit must be distinguishable + // from agents that hold none. + expect(snapshot.processCommitMetric).toBeUndefined() + expect(snapshot.totalPrivateMemory).toBeUndefined() + expect(snapshot.worktrees[0].privateMemory).toBeUndefined() + expect(snapshot.worktrees[0].sessions[0].privateMemory).toBeUndefined() + expect(snapshot.app.privateMemory).toBeUndefined() + expect(snapshot.totalMemory).toBe(52428800) + }) + + it('carries no commit metric on Unix, where ps has no committed-bytes column', async () => { + vi.spyOn(os, 'platform').mockReturnValue('darwin') + mockPsResponse('10 1 0 1024') + listRegisteredPtysMock.mockReturnValue([ + { + ptyId: 'pty-1', + worktreeId: 'repo-1::/repo', + sessionId: 'session-1', + paneKey: 'pane-1', + pid: 10 + } + ]) + const { collectMemorySnapshot } = await loadCollector() + + const snapshot = await collectMemorySnapshot(emptyStore) + + expect(snapshot.processMemoryMetric).toBe('rss') + expect(snapshot.processCommitMetric).toBeUndefined() + expect(snapshot.totalPrivateMemory).toBeUndefined() + expect(snapshot.worktrees[0].sessions[0].privateMemory).toBeUndefined() + }) + + it("attributes a shared ancestor's commit to one PTY only", async () => { + vi.spyOn(os, 'platform').mockReturnValue('win32') + runProcessMock.mockResolvedValue({ + code: 0, + signal: null, + stdout: [ + '10\t1\t1024\t0\t0\t638830000000000000\t1024', + '11\t10\t1024\t0\t0\t638830000000000000\t2048' + ].join('\r\n'), + stderr: '', + timedOut: false + }) + listRegisteredPtysMock.mockReturnValue([ + { ptyId: 'a', worktreeId: 'repo::C:\\a', sessionId: 's-a', paneKey: null, pid: 10 }, + { ptyId: 'b', worktreeId: 'repo::C:\\b', sessionId: 's-b', paneKey: null, pid: 11 } + ]) + const { collectMemorySnapshot } = await loadCollector() + + const snapshot = await collectMemorySnapshot(emptyStore) + + expect(snapshot.worktrees[0].sessions[0].privateMemory).toBe((1024 + 2048) * 1024) + expect(snapshot.worktrees[1].sessions[0].privateMemory).toBe(0) + expect(snapshot.totalPrivateMemory).toBe((1024 + 2048) * 1024) + }) +}) diff --git a/src/main/memory/collector.test.ts b/src/main/memory/collector.test.ts index 22e55825923..b1fe8095171 100644 --- a/src/main/memory/collector.test.ts +++ b/src/main/memory/collector.test.ts @@ -52,11 +52,6 @@ async function loadCollector() { return await import('./collector') } -async function loadWindowsProcessResourceCollector() { - vi.resetModules() - return await import('./windows-process-resource-collector') -} - const emptyStore = { getWorktreeMeta: () => undefined, getRepo: () => undefined @@ -125,68 +120,6 @@ describe('parsePsOutput', () => { }) }) -describe('parseWindowsProcessOutput', () => { - it('parses tab-delimited CIM process rows', async () => { - const { parseWindowsProcessOutput } = await loadWindowsProcessResourceCollector() - - expect(parseWindowsProcessOutput('100\t1\t2048\r\n200\t100\t1024')).toEqual([ - { pid: 100, ppid: 1, cpu: 0, memory: 2048 }, - { pid: 200, ppid: 100, cpu: 0, memory: 1024 } - ]) - }) - - it('skips malformed rows and clamps invalid memory to zero', async () => { - const { parseWindowsProcessOutput } = await loadWindowsProcessResourceCollector() - - expect( - parseWindowsProcessOutput( - [ - 'garbage', - 'abc\t1\t100', - '10\txyz\t100', - '0\t0\t100', - '-5\t0\t100', - '30\t-1\t100', - '20\t1\t-50' - ].join('\n') - ) - ).toEqual([{ pid: 20, ppid: 1, cpu: 0, memory: 0 }]) - }) - - it('preserves empty CIM field positions instead of shifting CPU ticks into memory', async () => { - const { parseWindowsProcessOutput } = await loadWindowsProcessResourceCollector() - - expect(parseWindowsProcessOutput('100\t1\t\t200\t300\t638830000000000000')).toEqual([ - { pid: 100, ppid: 1, cpu: 0, memory: 0 } - ]) - }) -}) - -describe('parseTypeperfProcessOutput', () => { - it('joins PID, parent PID, and working-set counters by process instance', async () => { - const { parseTypeperfProcessOutput } = await loadWindowsProcessResourceCollector() - const stdout = [ - '"(PDH-CSV 4.0)","\\\\HOST\\Process(node)\\ID Process","\\\\HOST\\Process(node#1)\\ID Process","\\\\HOST\\Process(node)\\Creating Process ID","\\\\HOST\\Process(node#1)\\Creating Process ID","\\\\HOST\\Process(node)\\Working Set","\\\\HOST\\Process(node#1)\\Working Set"', - '"07/15/2026 01:44:54.514","100.000000","200.000000","1.000000","100.000000","2048.000000","4096.000000"' - ].join('\r\n') - - expect(parseTypeperfProcessOutput(stdout)).toEqual([ - { pid: 100, ppid: 1, cpu: 0, memory: 2048 }, - { pid: 200, ppid: 100, cpu: 0, memory: 4096 } - ]) - }) - - it('ignores aggregate and incomplete rows and clamps invalid memory', async () => { - const { parseTypeperfProcessOutput } = await loadWindowsProcessResourceCollector() - const stdout = [ - '"(PDH-CSV 4.0)","\\\\HOST\\Process(_Total)\\ID Process","\\\\HOST\\Process(cmd)\\ID Process","\\\\HOST\\Process(orphan)\\ID Process","\\\\HOST\\Process(_Total)\\Creating Process ID","\\\\HOST\\Process(cmd)\\Creating Process ID","\\\\HOST\\Process(_Total)\\Working Set","\\\\HOST\\Process(cmd)\\Working Set"', - '"time","0.000000","100.000000","200.000000","0.000000","1.000000","999999.000000","-1.000000"' - ].join('\r\n') - - expect(parseTypeperfProcessOutput(stdout)).toEqual([{ pid: 100, ppid: 1, cpu: 0, memory: 0 }]) - }) -}) - describe('collectSubtree', () => { function makeIndex(rows: { pid: number; ppid: number }[]) { const byPid = new Map() @@ -200,7 +133,7 @@ describe('collectSubtree', () => { childrenOf.set(r.ppid, [r.pid]) } } - return { byPid, childrenOf } + return { byPid, childrenOf, hasPrivateMemory: false } } it('walks every descendant of the root inclusive', async () => { @@ -240,7 +173,8 @@ describe('collectSubtree', () => { // those as "walked" but do not fabricate a row for them. const index = { byPid: new Map([[1, { pid: 1, ppid: 0, cpu: 0, memory: 0 }]]), - childrenOf: new Map([[1, [2]]]) + childrenOf: new Map([[1, [2]]]), + hasPrivateMemory: false } expect(collectSubtree(index, 1)).toEqual([1]) @@ -336,364 +270,6 @@ describe('collectMemorySnapshot', () => { expect(execMock).toHaveBeenCalledTimes(count) } - it('uses one CIM process for Windows memory and CPU sampling', async () => { - vi.spyOn(os, 'platform').mockReturnValue('win32') - mockPsResponse('10 1 0 1024') - const { collectMemorySnapshot } = await loadCollector() - - await collectMemorySnapshot(emptyStore) - - expect(execMock).not.toHaveBeenCalled() - expect(runProcessMock).toHaveBeenCalledTimes(1) - const spec = runProcessMock.mock.calls[0][0] - expect(spec.program).toBe('powershell.exe') - expect(spec.args.join(' ')).toContain('Get-CimInstance Win32_Process') - expect(spec.args.join(' ')).toContain('KernelModeTime') - expect(spec.args.join(' ')).toContain('UserModeTime') - expect(spec.args.join(' ')).toContain('CreationDate') - expect(spec).toMatchObject({ maxOutputBytes: 10 * 1024 * 1024, timeoutMs: 5_000 }) - }) - - it('attributes Windows process CPU from cumulative time deltas between sweeps', async () => { - vi.spyOn(os, 'platform').mockReturnValue('win32') - vi.spyOn(performance, 'now').mockReturnValueOnce(1_000).mockReturnValueOnce(3_000) - const cpuOutputs = [ - '10\t1\t1048576\t10000000\t0\t638830000000000000', - '10\t1\t1048576\t30000000\t0\t638830000000000000' - ] - runProcessMock.mockImplementation(() => - Promise.resolve({ - code: 0, - signal: null, - stdout: cpuOutputs.shift() ?? '', - stderr: '', - timedOut: false - }) - ) - listRegisteredPtysMock.mockReturnValue([ - { - ptyId: 'windows-cpu-pty', - worktreeId: 'repo-1::C:\\repo', - sessionId: 'session-1', - paneKey: 'pane-1', - pid: 10 - } - ]) - const { collectMemorySnapshot } = await loadCollector() - - const first = await collectMemorySnapshot(emptyStore) - const second = await collectMemorySnapshot(emptyStore) - - expect(first.worktrees[0].sessions[0].cpu).toBe(0) - expect(second.worktrees[0].sessions[0].cpu).toBe(100) - expect(runProcessMock.mock.calls.map(([spec]) => spec.program)).toEqual([ - 'powershell.exe', - 'powershell.exe' - ]) - }) - - it('does not attribute prior CPU time after Windows reuses a process id', async () => { - vi.spyOn(os, 'platform').mockReturnValue('win32') - vi.spyOn(performance, 'now').mockReturnValueOnce(1_000).mockReturnValueOnce(3_000) - const cpuOutputs = [ - '10\t1\t1048576\t10000000\t0\t638830000000000000', - '10\t1\t1048576\t30000000\t0\t638830000000000001' - ] - runProcessMock.mockImplementation(() => - Promise.resolve({ - code: 0, - signal: null, - stdout: cpuOutputs.shift() ?? '', - stderr: '', - timedOut: false - }) - ) - listRegisteredPtysMock.mockReturnValue([ - { - ptyId: 'reused-pid-pty', - worktreeId: 'repo-1::C:\\repo', - sessionId: 'session-1', - paneKey: 'pane-1', - pid: 10 - } - ]) - const { collectMemorySnapshot } = await loadCollector() - - await collectMemorySnapshot(emptyStore) - const second = await collectMemorySnapshot(emptyStore) - - expect(second.worktrees[0].sessions[0].cpu).toBe(0) - }) - - it('supports cumulative CPU counters above JavaScript safe integers', async () => { - vi.spyOn(os, 'platform').mockReturnValue('win32') - vi.spyOn(performance, 'now').mockReturnValueOnce(1_000).mockReturnValueOnce(3_000) - const cpuOutputs = [ - '10\t1\t1048576\t90071992547409920\t0\t638830000000000000', - '10\t1\t1048576\t90071992567409920\t0\t638830000000000000' - ] - runProcessMock.mockImplementation(() => - Promise.resolve({ - code: 0, - signal: null, - stdout: cpuOutputs.shift() ?? '', - stderr: '', - timedOut: false - }) - ) - listRegisteredPtysMock.mockReturnValue([ - { - ptyId: 'large-counter-pty', - worktreeId: 'repo-1::C:\\repo', - sessionId: 'session-1', - paneKey: 'pane-1', - pid: 10 - } - ]) - const { collectMemorySnapshot } = await loadCollector() - - await collectMemorySnapshot(emptyStore) - const second = await collectMemorySnapshot(emptyStore) - - expect(second.worktrees[0].sessions[0].cpu).toBe(100) - }) - - it('keeps the older CPU baseline when forced snapshots are too close together', async () => { - vi.spyOn(os, 'platform').mockReturnValue('win32') - vi.spyOn(performance, 'now') - .mockReturnValueOnce(1_000) - .mockReturnValueOnce(1_100) - .mockReturnValueOnce(3_000) - const cpuOutputs = [ - '10\t1\t1048576\t0\t0\t638830000000000000', - '10\t1\t1048576\t1000000\t0\t638830000000000000', - '10\t1\t1048576\t20000000\t0\t638830000000000000' - ] - runProcessMock.mockImplementation(() => - Promise.resolve({ - code: 0, - signal: null, - stdout: cpuOutputs.shift() ?? '', - stderr: '', - timedOut: false - }) - ) - listRegisteredPtysMock.mockReturnValue([ - { - ptyId: 'short-sample-pty', - worktreeId: 'repo-1::C:\\repo', - sessionId: 'session-1', - paneKey: 'pane-1', - pid: 10 - } - ]) - const { collectMemorySnapshot } = await loadCollector() - - await collectMemorySnapshot(emptyStore) - const tooSoon = await collectMemorySnapshot(emptyStore) - const normalPoll = await collectMemorySnapshot(emptyStore) - - expect(tooSoon.worktrees[0].sessions[0].cpu).toBe(0) - expect(normalPoll.worktrees[0].sessions[0].cpu).toBe(100) - }) - - it('caps impossible Windows CPU deltas at the host core capacity', async () => { - vi.spyOn(os, 'platform').mockReturnValue('win32') - vi.spyOn(os, 'cpus').mockReturnValue([{}, {}] as ReturnType) - vi.spyOn(performance, 'now').mockReturnValueOnce(1_000).mockReturnValueOnce(3_000) - const cpuOutputs = [ - '10\t1\t1048576\t0\t0\t638830000000000000', - '10\t1\t1048576\t1000000000\t0\t638830000000000000' - ] - runProcessMock.mockImplementation(() => - Promise.resolve({ - code: 0, - signal: null, - stdout: cpuOutputs.shift() ?? '', - stderr: '', - timedOut: false - }) - ) - listRegisteredPtysMock.mockReturnValue([ - { - ptyId: 'impossible-cpu-pty', - worktreeId: 'repo-1::C:\\repo', - sessionId: 'session-1', - paneKey: 'pane-1', - pid: 10 - } - ]) - const { collectMemorySnapshot } = await loadCollector() - - await collectMemorySnapshot(emptyStore) - const capped = await collectMemorySnapshot(emptyStore) - - expect(capped.worktrees[0].sessions[0].cpu).toBe(200) - }) - - it('warms CPU sampling again after Resource Manager was closed', async () => { - vi.spyOn(os, 'platform').mockReturnValue('win32') - vi.spyOn(performance, 'now').mockReturnValueOnce(1_000).mockReturnValueOnce(12_000) - const cpuOutputs = [ - '10\t1\t1048576\t0\t0\t638830000000000000', - '10\t1\t1048576\t100000000\t0\t638830000000000000' - ] - runProcessMock.mockImplementation(() => - Promise.resolve({ - code: 0, - signal: null, - stdout: cpuOutputs.shift() ?? '', - stderr: '', - timedOut: false - }) - ) - listRegisteredPtysMock.mockReturnValue([ - { - ptyId: 'stale-counter-pty', - worktreeId: 'repo-1::C:\\repo', - sessionId: 'session-1', - paneKey: 'pane-1', - pid: 10 - } - ]) - const { collectMemorySnapshot } = await loadCollector() - - await collectMemorySnapshot(emptyStore) - const reopened = await collectMemorySnapshot(emptyStore) - - expect(reopened.worktrees[0].sessions[0].cpu).toBe(0) - }) - - it('preserves Windows process memory when CPU counters are unavailable', async () => { - vi.spyOn(os, 'platform').mockReturnValue('win32') - runProcessMock.mockImplementation(() => - Promise.resolve({ - code: 0, - signal: null, - stdout: '10\t1\t1048576\t\t\t638830000000000000', - stderr: '', - timedOut: false - }) - ) - listRegisteredPtysMock.mockReturnValue([ - { - ptyId: 'cpu-failure-pty', - worktreeId: 'repo-1::C:\\repo', - sessionId: 'session-1', - paneKey: 'pane-1', - pid: 10 - } - ]) - const { collectMemorySnapshot } = await loadCollector() - - const snapshot = await collectMemorySnapshot(emptyStore) - - expect(snapshot.worktrees[0].sessions[0]).toMatchObject({ cpu: 0, memory: 1024 * 1024 }) - }) - - it('uses Typeperf during the CIM retry cooldown', async () => { - vi.spyOn(os, 'platform').mockReturnValue('win32') - runProcessMock.mockImplementation((spec: { program: string }) => - Promise.resolve( - spec.program === 'powershell.exe' - ? { code: 1, signal: null, stdout: '', stderr: 'CIM unavailable', timedOut: false } - : { - code: 0, - signal: null, - stdout: psFixtureToTypeperfOutput('10 1 0 1024'), - stderr: '', - timedOut: false - } - ) - ) - listRegisteredPtysMock.mockReturnValue([ - { - ptyId: 'cim-pty', - worktreeId: null, - sessionId: null, - paneKey: null, - pid: 10 - } - ]) - const { collectMemorySnapshot } = await loadCollector() - - const first = await collectMemorySnapshot(emptyStore) - const second = await collectMemorySnapshot(emptyStore) - - expect(runProcessMock).toHaveBeenCalledTimes(3) - expect(runProcessMock.mock.calls.map(([spec]) => spec.program)).toEqual([ - 'powershell.exe', - 'typeperf.exe', - 'typeperf.exe' - ]) - expect(runProcessMock.mock.calls[1][0]).toMatchObject({ timeoutMs: 5_000 }) - expect(first.worktrees[0].memory).toBe(1048576) - expect(second.worktrees[0].memory).toBe(1048576) - }) - - it('retries CIM after fallback and warms CPU sampling before restoring deltas', async () => { - vi.spyOn(os, 'platform').mockReturnValue('win32') - vi.spyOn(performance, 'now') - .mockReturnValueOnce(1_000) - .mockReturnValueOnce(2_000) - .mockReturnValueOnce(31_001) - .mockReturnValueOnce(32_000) - .mockReturnValueOnce(34_000) - const cimOutputs = [ - '10\t1\t1048576\t10000000\t0\t638830000000000000', - '10\t1\t1048576\t30000000\t0\t638830000000000000' - ] - let cimCalls = 0 - runProcessMock.mockImplementation((spec: { program: string }) => { - if (spec.program === 'typeperf.exe') { - return Promise.resolve({ - code: 0, - signal: null, - stdout: psFixtureToTypeperfOutput('10 1 0 1024'), - stderr: '', - timedOut: false - }) - } - cimCalls += 1 - return Promise.resolve( - cimCalls === 1 - ? { code: 1, signal: null, stdout: '', stderr: 'transient CIM failure', timedOut: false } - : { - code: 0, - signal: null, - stdout: cimOutputs.shift() ?? '', - stderr: '', - timedOut: false - } - ) - }) - listRegisteredPtysMock.mockReturnValue([ - { - ptyId: 'recovering-cim-pty', - worktreeId: 'repo-1::C:\\repo', - sessionId: 'session-1', - paneKey: 'pane-1', - pid: 10 - } - ]) - const { collectMemorySnapshot } = await loadCollector() - - await collectMemorySnapshot(emptyStore) - await collectMemorySnapshot(emptyStore) - const warming = await collectMemorySnapshot(emptyStore) - const recovered = await collectMemorySnapshot(emptyStore) - - expect(runProcessMock.mock.calls.map(([spec]) => spec.program)).toEqual([ - 'powershell.exe', - 'typeperf.exe', - 'typeperf.exe', - 'powershell.exe', - 'powershell.exe' - ]) - expect(warming.worktrees[0].sessions[0].cpu).toBe(0) - expect(recovered.worktrees[0].sessions[0].cpu).toBe(100) - }) - it('coalesces concurrent callers onto a single in-flight sweep', async () => { // Why: the collector exists in part to prevent a burst of renderer // polls from spawning overlapping `ps` children. If a regression ever diff --git a/src/main/memory/collector.ts b/src/main/memory/collector.ts index b48a6bb75ed..c0e77513004 100644 --- a/src/main/memory/collector.ts +++ b/src/main/memory/collector.ts @@ -30,7 +30,9 @@ import { getAppEnvironment, type AppEnvironment } from '../../shared/app-environ import type { AppMemory, MemorySnapshot, + ProcessCommitMetric, SessionMemory, + UsageValues, WorktreeMemory } from '../../shared/process-stats-types' import type { Store } from '../persistence' @@ -81,12 +83,43 @@ type ProcRow = { cpu: number /** Resident memory in bytes. */ memory: number + /** Committed bytes, resident or paged out. Absent when the host cannot report it. */ + privateMemory?: number } /** Indexed view of a single host process sweep. */ type ProcIndex = { byPid: Map childrenOf: Map + /** + * Whether this sweep reported committed bytes at all. Data-driven rather than + * platform-driven: the Windows typeperf fallback can be missing the counter, + * and reporting a 0 sum then would read as "agents commit nothing". + */ + hasPrivateMemory: boolean +} + +const PROCESS_COMMIT_METRIC: ProcessCommitMetric = 'private-bytes' + +/** + * The one rule for every committed-bytes key: present only when the sweep could + * measure it, because a 0 would read as "these processes commit nothing". + */ +function commitField(hasPrivateMemory: boolean, privateMemory: number): { privateMemory?: number } { + return hasPrivateMemory ? { privateMemory: clampNumber(privateMemory) } : {} +} + +/** The snapshot-level pair, which names the unit alongside the total. */ +function snapshotCommitFields( + hasPrivateMemory: boolean, + totalPrivateMemory: number +): Pick { + return hasPrivateMemory + ? { + processCommitMetric: PROCESS_COMMIT_METRIC, + totalPrivateMemory: clampNumber(totalPrivateMemory) + } + : {} } function clampNumber(value: unknown): number { @@ -155,9 +188,11 @@ async function enumerateProcesses(): Promise { const byPid = new Map() const childrenOf = new Map() + let hasPrivateMemory = false for (const row of rows) { byPid.set(row.pid, row) + hasPrivateMemory ||= row.privateMemory !== undefined const siblings = childrenOf.get(row.ppid) if (siblings) { siblings.push(row.pid) @@ -166,7 +201,7 @@ async function enumerateProcesses(): Promise { } } - return { byPid, childrenOf } + return { byPid, childrenOf, hasPrivateMemory } } async function enumerateUnix(): Promise { @@ -261,13 +296,16 @@ function electronMetricMemoryBytes( } function bucketElectronMetrics(processIndex: ProcIndex): AppBucketsRaw { - const main = { cpu: 0, memory: 0 } - const renderer = { cpu: 0, memory: 0 } - const other = { cpu: 0, memory: 0 } + const main = { cpu: 0, memory: 0, privateMemory: 0 } + const renderer = { cpu: 0, memory: 0, privateMemory: 0 } + const other = { cpu: 0, memory: 0, privateMemory: 0 } for (const proc of getAppEnvironment().getAppMetrics()) { const cpu = clampNumber(proc.cpu?.percentCPUUsage) const memoryBytes = electronMetricMemoryBytes(proc, processIndex) + // Why the host row rather than Electron's own metric: getAppMetrics has no + // commit figure for helper processes, and the sweep already indexed them. + const privateBytes = clampNumber(processIndex.byPid.get(proc.pid)?.privateMemory) // Why: lowercase once so future Electron versions emitting different // casing ('browser' vs 'Browser') still bucket correctly. @@ -281,14 +319,24 @@ function bucketElectronMetrics(processIndex: ProcIndex): AppBucketsRaw { target.cpu += cpu target.memory += memoryBytes + target.privateMemory += privateBytes } + const usage = (bucket: typeof main): UsageValues => ({ + cpu: bucket.cpu, + memory: bucket.memory, + ...commitField(processIndex.hasPrivateMemory, bucket.privateMemory) + }) + return { - main, - renderer, - other, - cpu: main.cpu + renderer.cpu + other.cpu, - memory: main.memory + renderer.memory + other.memory + main: usage(main), + renderer: usage(renderer), + other: usage(other), + ...usage({ + cpu: main.cpu + renderer.cpu + other.cpu, + memory: main.memory + renderer.memory + other.memory, + privateMemory: main.privateMemory + renderer.privateMemory + other.privateMemory + }) } } @@ -301,6 +349,7 @@ type WorktreeBucket = { repoName: string cpu: number memory: number + privateMemory: number sessions: SessionMemory[] } @@ -334,7 +383,16 @@ function makeEmptyBucket( repoId: string, repoName: string ): WorktreeBucket { - return { worktreeId, worktreeName, repoId, repoName, cpu: 0, memory: 0, sessions: [] } + return { + worktreeId, + worktreeName, + repoId, + repoName, + cpu: 0, + memory: 0, + privateMemory: 0, + sessions: [] + } } // ─── Main collection path ─────────────────────────────────────────── @@ -361,6 +419,7 @@ async function runSnapshot(store: MemorySnapshotStore): Promise for (const pty of ptys) { let sessionCpu = 0 let sessionMemory = 0 + let sessionPrivateMemory = 0 if (pty.pid != null) { for (const pid of collectSubtree(processIndex, pty.pid)) { @@ -374,6 +433,9 @@ async function runSnapshot(store: MemorySnapshotStore): Promise claimed.add(pid) sessionCpu += row.cpu sessionMemory += row.memory + // Why the whole subtree: an agent's committed bytes live in the + // children it spawned (codex.exe, MCP servers), not in the shell. + sessionPrivateMemory += clampNumber(row.privateMemory) } } @@ -382,7 +444,8 @@ async function runSnapshot(store: MemorySnapshotStore): Promise paneKey: pty.paneKey, pid: pty.pid ?? 0, cpu: clampNumber(sessionCpu), - memory: clampNumber(sessionMemory) + memory: clampNumber(sessionMemory), + ...commitField(processIndex.hasPrivateMemory, sessionPrivateMemory) } let bucket: WorktreeBucket @@ -401,6 +464,7 @@ async function runSnapshot(store: MemorySnapshotStore): Promise bucket.cpu += session.cpu bucket.memory += session.memory + bucket.privateMemory += clampNumber(session.privateMemory) bucket.sessions.push(session) } @@ -419,16 +483,19 @@ async function runSnapshot(store: MemorySnapshotStore): Promise } sweepStaleHistory(now) - const worktrees: WorktreeMemory[] = bucketList.map((b) => ({ + const worktrees: WorktreeMemory[] = bucketList.map(({ privateMemory, ...b }) => ({ ...b, + ...commitField(processIndex.hasPrivateMemory, privateMemory), history: readHistory(b.worktreeId) })) let sessionCpuTotal = 0 let sessionMemoryTotal = 0 + let sessionPrivateTotal = 0 for (const wt of worktrees) { sessionCpuTotal += wt.cpu sessionMemoryTotal += wt.memory + sessionPrivateTotal += clampNumber(wt.privateMemory) } return { @@ -436,6 +503,10 @@ async function runSnapshot(store: MemorySnapshotStore): Promise worktrees, host, processMemoryMetric: getProcessMemoryMetric(), + ...snapshotCommitFields( + processIndex.hasPrivateMemory, + clampNumber(appBuckets.privateMemory) + sessionPrivateTotal + ), totalCpu: appBuckets.cpu + sessionCpuTotal, totalMemory: appBuckets.memory + sessionMemoryTotal, collectedAt: now diff --git a/src/main/memory/windows-process-resource-collector.ts b/src/main/memory/windows-process-resource-collector.ts index 5c929ebf376..bc7e85a79f8 100644 --- a/src/main/memory/windows-process-resource-collector.ts +++ b/src/main/memory/windows-process-resource-collector.ts @@ -2,50 +2,24 @@ import { runProcess } from '../../shared/child-process/run-process' import os from 'node:os' import { performance } from 'node:perf_hooks' import { - iterateProcessOutputLines, - PROCESS_OUTPUT_FIELD_SCAN_MAX_CHARS -} from '../../shared/process-output-field-scanner' + parseTypeperfProcessOutput, + parseWindowsProcessSample, + TYPEPERF_COUNTERS, + type ParsedWindowsProcessSample, + type WindowsProcessResourceRow +} from './windows-process-sample-parsing' + +export type { WindowsProcessResourceRow } from './windows-process-sample-parsing' const PROCESS_QUERY_TIMEOUT_MS = 5_000 const PROCESS_QUERY_MAX_BUFFER = 10 * 1024 * 1024 -const TYPEPERF_COUNTERS = [ - '\\Process(*)\\ID Process', - '\\Process(*)\\Creating Process ID', - '\\Process(*)\\Working Set' -] as const -const TYPEPERF_MAX_FIELDS = 8_192 -const TYPEPERF_MAX_LINE_CHARS = 1024 * 1024 const CPU_MIN_SAMPLE_MS = 250 const CPU_STALE_AFTER_MS = 10_000 const HUNDRED_NS_TICKS_PER_MS = 10_000 const CIM_RETRY_AFTER_MS = 30_000 -export type WindowsProcessResourceRow = { - pid: number - ppid: number - /** Percent of one core (may exceed 100 on multi-core). */ - cpu: number - /** Resident memory in bytes. */ - memory: number -} - -type WindowsCpuTimes = { - cpuTicks: bigint - startTimeId: string -} - -type WindowsProcessSample = { +type WindowsProcessSample = ParsedWindowsProcessSample & { sampledAtMs: number - rows: WindowsProcessResourceRow[] - cpuByPid: Map -} - -type ParsedWindowsProcessSample = Omit - -type TypeperfProcessFields = { - pid?: number - ppid?: number - memory?: number } let processBackend: 'cim' | 'typeperf' = 'cim' @@ -114,14 +88,16 @@ function applyWindowsCpuSample(sample: WindowsProcessSample): WindowsProcessReso } async function enumerateWindowsWithCim(): Promise { + // PageFileUsage rides along on the sweep that already runs: it is the commit + // charge the working set stops showing once Windows starts trimming pages. const args = [ '-NoLogo', '-NoProfile', '-NonInteractive', '-Command', "$ErrorActionPreference = 'Stop'; $ProgressPreference = 'SilentlyContinue'; " + - 'Get-CimInstance Win32_Process -Property ProcessId,ParentProcessId,WorkingSetSize,KernelModeTime,UserModeTime,CreationDate | ' + - 'ForEach-Object { try { [string]::Join([char]9, @($_.ProcessId, $_.ParentProcessId, $_.WorkingSetSize, [string]$_.KernelModeTime, [string]$_.UserModeTime, $_.CreationDate.ToUniversalTime().Ticks)) } catch {} }' + 'Get-CimInstance Win32_Process -Property ProcessId,ParentProcessId,WorkingSetSize,KernelModeTime,UserModeTime,CreationDate,PageFileUsage | ' + + 'ForEach-Object { try { [string]::Join([char]9, @($_.ProcessId, $_.ParentProcessId, $_.WorkingSetSize, [string]$_.KernelModeTime, [string]$_.UserModeTime, $_.CreationDate.ToUniversalTime().Ticks, $_.PageFileUsage)) } catch {} }' ] try { const stdout = await execFileText('powershell.exe', args) @@ -164,169 +140,6 @@ async function execFileText(file: string, args: string[]): Promise { return result.stdout } -function parseWindowsProcessSample(stdout: string): ParsedWindowsProcessSample { - const rows: WindowsProcessResourceRow[] = [] - const cpuByPid = new Map() - for (const line of iterateProcessOutputLines(stdout)) { - const fields = parseCimTabFields(line) - if (fields.length < 3) { - continue - } - const pid = Number.parseInt(fields[0], 10) - const ppid = Number.parseInt(fields[1], 10) - const memory = Number.parseInt(fields[2], 10) - if (!Number.isSafeInteger(pid) || pid <= 0 || !Number.isSafeInteger(ppid) || ppid < 0) { - continue - } - rows.push({ - pid, - ppid, - cpu: 0, - memory: Number.isFinite(memory) && memory > 0 ? memory : 0 - }) - - const kernelTicks = parseUnsignedBigInt(fields[3]) - const userTicks = parseUnsignedBigInt(fields[4]) - const startTimeId = fields[5] ?? '' - if ( - kernelTicks !== null && - userTicks !== null && - /^\d+$/.test(startTimeId) && - !/^0+$/.test(startTimeId) - ) { - cpuByPid.set(pid, { cpuTicks: kernelTicks + userTicks, startTimeId }) - } - } - return { rows, cpuByPid } -} - -function parseCimTabFields(line: string): string[] { - // Why: CIM serializes null properties as empty tab fields; collapsing - // whitespace would shift CPU counters into the working-set column. - if (line.length > PROCESS_OUTPUT_FIELD_SCAN_MAX_CHARS) { - return [] - } - return line.split('\t', 6).map((field) => field.trim()) -} - -/** Parse tab-delimited PowerShell CIM process rows without deriving CPU deltas. */ -export function parseWindowsProcessOutput(stdout: string): WindowsProcessResourceRow[] { - return parseWindowsProcessSample(stdout).rows -} - -/** Parse one CSV sample from Windows Typeperf. */ -export function parseTypeperfProcessOutput(stdout: string): WindowsProcessResourceRow[] { - let headers: string[] | null = null - let values: string[] | null = null - - for (const line of iterateProcessOutputLines(stdout)) { - if (!line || line.length > TYPEPERF_MAX_LINE_CHARS) { - continue - } - const fields = parseTypeperfCsvLine(line) - if (!headers && fields[0]?.startsWith('(PDH-CSV')) { - headers = fields - continue - } - if (headers && fields.length === headers.length) { - values = fields - break - } - } - - if (!headers || !values) { - return [] - } - - const byInstance = new Map() - for (let index = 1; index < headers.length; index += 1) { - const path = parseTypeperfCounterPath(headers[index]) - if (!path || path.instance === '_Total') { - continue - } - const value = Number.parseFloat(values[index]) - if (!Number.isFinite(value)) { - continue - } - const row = byInstance.get(path.instance) ?? {} - if (path.counter === 'ID Process') { - row.pid = Math.trunc(value) - } else if (path.counter === 'Creating Process ID') { - row.ppid = Math.trunc(value) - } else if (path.counter === 'Working Set') { - row.memory = value - } - byInstance.set(path.instance, row) - } - - const rows: WindowsProcessResourceRow[] = [] - for (const row of byInstance.values()) { - if (row.pid === undefined || row.pid <= 0 || row.ppid === undefined || row.ppid < 0) { - continue - } - rows.push({ - pid: row.pid, - ppid: row.ppid, - cpu: 0, - memory: row.memory !== undefined && row.memory > 0 ? row.memory : 0 - }) - } - return rows -} - -function parseTypeperfCounterPath(path: string): { instance: string; counter: string } | null { - const processStart = path.lastIndexOf('\\Process(') - const counterStart = path.lastIndexOf(')\\') - if (processStart === -1 || counterStart <= processStart + 9) { - return null - } - return { - instance: path.slice(processStart + 9, counterStart), - counter: path.slice(counterStart + 2) - } -} - -function parseTypeperfCsvLine(line: string): string[] { - const fields: string[] = [] - let value = '' - let quoted = false - - for (let index = 0; index < line.length; index += 1) { - const char = line[index] - if (char === '"') { - if (quoted && line[index + 1] === '"') { - value += '"' - index += 1 - } else { - quoted = !quoted - } - continue - } - if (char === ',' && !quoted) { - fields.push(value) - value = '' - if (fields.length >= TYPEPERF_MAX_FIELDS) { - return [] - } - continue - } - value += char - } - fields.push(value) - return fields -} - -function parseUnsignedBigInt(value: string | undefined): bigint | null { - if (!value || !/^\d+$/.test(value)) { - return null - } - try { - return BigInt(value) - } catch { - return null - } -} - function nonNegativeNumber(value: unknown): number { return typeof value === 'number' && Number.isFinite(value) ? Math.max(0, value) : 0 } diff --git a/src/main/memory/windows-process-sample-parsing.test.ts b/src/main/memory/windows-process-sample-parsing.test.ts new file mode 100644 index 00000000000..1601d183b80 --- /dev/null +++ b/src/main/memory/windows-process-sample-parsing.test.ts @@ -0,0 +1,140 @@ +import { describe, expect, it, vi } from 'vitest' + +async function loadWindowsProcessSampleParsing() { + vi.resetModules() + return await import('./windows-process-sample-parsing') +} + +describe('parseWindowsProcessOutput', () => { + it('parses tab-delimited CIM process rows', async () => { + const { parseWindowsProcessOutput } = await loadWindowsProcessSampleParsing() + + expect(parseWindowsProcessOutput('100\t1\t2048\r\n200\t100\t1024')).toEqual([ + { pid: 100, ppid: 1, cpu: 0, memory: 2048 }, + { pid: 200, ppid: 100, cpu: 0, memory: 1024 } + ]) + }) + + it('skips malformed rows and clamps invalid memory to zero', async () => { + const { parseWindowsProcessOutput } = await loadWindowsProcessSampleParsing() + + expect( + parseWindowsProcessOutput( + [ + 'garbage', + 'abc\t1\t100', + '10\txyz\t100', + '0\t0\t100', + '-5\t0\t100', + '30\t-1\t100', + '20\t1\t-50' + ].join('\n') + ) + ).toEqual([{ pid: 20, ppid: 1, cpu: 0, memory: 0 }]) + }) + + it('reads PageFileUsage kilobytes into committed private bytes', async () => { + const { parseWindowsProcessOutput } = await loadWindowsProcessSampleParsing() + + // 5,600,000 KB of commit behind a 96 MB working set is the reported shape. + expect( + parseWindowsProcessOutput('100\t1\t100663296\t0\t0\t638830000000000000\t5600000') + ).toEqual([{ pid: 100, ppid: 1, cpu: 0, memory: 100663296, privateMemory: 5600000 * 1024 }]) + }) + + it('leaves committed bytes absent when the host omits PageFileUsage', async () => { + const { parseWindowsProcessOutput } = await loadWindowsProcessSampleParsing() + + const [row] = parseWindowsProcessOutput('100\t1\t2048\t0\t0\t638830000000000000') + + expect(row.privateMemory).toBeUndefined() + expect(parseWindowsProcessOutput('100\t1\t2048\t0\t0\t1\t')[0].privateMemory).toBeUndefined() + }) + + it('keeps a zero PageFileUsage distinct from an unreported one', async () => { + const { parseWindowsProcessOutput } = await loadWindowsProcessSampleParsing() + + expect(parseWindowsProcessOutput('4\t0\t2048\t0\t0\t1\t0')[0].privateMemory).toBe(0) + }) + + it('preserves empty CIM field positions instead of shifting CPU ticks into memory', async () => { + const { parseWindowsProcessOutput } = await loadWindowsProcessSampleParsing() + + expect(parseWindowsProcessOutput('100\t1\t\t200\t300\t638830000000000000')).toEqual([ + { pid: 100, ppid: 1, cpu: 0, memory: 0 } + ]) + }) +}) + +describe('parseTypeperfProcessOutput', () => { + it('joins PID, parent PID, and working-set counters by process instance', async () => { + const { parseTypeperfProcessOutput } = await loadWindowsProcessSampleParsing() + const stdout = [ + '"(PDH-CSV 4.0)","\\\\HOST\\Process(node)\\ID Process","\\\\HOST\\Process(node#1)\\ID Process","\\\\HOST\\Process(node)\\Creating Process ID","\\\\HOST\\Process(node#1)\\Creating Process ID","\\\\HOST\\Process(node)\\Working Set","\\\\HOST\\Process(node#1)\\Working Set"', + '"07/15/2026 01:44:54.514","100.000000","200.000000","1.000000","100.000000","2048.000000","4096.000000"' + ].join('\r\n') + + expect(parseTypeperfProcessOutput(stdout)).toEqual([ + { pid: 100, ppid: 1, cpu: 0, memory: 2048 }, + { pid: 200, ppid: 100, cpu: 0, memory: 4096 } + ]) + }) + + it('joins the Private Bytes counter onto the same process instance', async () => { + const { parseTypeperfProcessOutput } = await loadWindowsProcessSampleParsing() + const stdout = [ + '"(PDH-CSV 4.0)","\\\\HOST\\Process(codex)\\ID Process","\\\\HOST\\Process(codex)\\Creating Process ID","\\\\HOST\\Process(codex)\\Working Set","\\\\HOST\\Process(codex)\\Private Bytes"', + '"07/15/2026 01:44:54.514","100.000000","1.000000","100663296.000000","5734400000.000000"' + ].join('\r\n') + + expect(parseTypeperfProcessOutput(stdout)).toEqual([ + { pid: 100, ppid: 1, cpu: 0, memory: 100663296, privateMemory: 5734400000 } + ]) + }) + + it('leaves committed bytes absent when the Private Bytes counter is missing', async () => { + const { parseTypeperfProcessOutput } = await loadWindowsProcessSampleParsing() + const stdout = [ + '"(PDH-CSV 4.0)","\\\\HOST\\Process(codex)\\ID Process","\\\\HOST\\Process(codex)\\Creating Process ID","\\\\HOST\\Process(codex)\\Working Set"', + '"time","100.000000","1.000000","2048.000000"' + ].join('\r\n') + + expect(parseTypeperfProcessOutput(stdout)[0].privateMemory).toBeUndefined() + }) + + it('still parses a busy host once a fourth counter widens every sample line', async () => { + const { parseTypeperfProcessOutput } = await loadWindowsProcessSampleParsing() + // 2100 processes x 4 counters overruns a fixed 8192-field cap; the reported + // MCP fan-out host runs well past 2048 processes. + const instanceCount = 2100 + const headers = ['"(PDH-CSV 4.0)"'] + const values = ['"time"'] + for (let index = 0; index < instanceCount; index += 1) { + for (const counter of ['ID Process', 'Creating Process ID', 'Working Set', 'Private Bytes']) { + headers.push(`"\\\\HOST\\Process(node#${index})\\${counter}"`) + } + values.push(`"${1000 + index}"`, '"1"', '"2048"', '"4096"') + } + const stdout = [headers.join(','), values.join(',')].join('\r\n') + + const rows = parseTypeperfProcessOutput(stdout) + expect(rows).toHaveLength(instanceCount) + expect(rows[instanceCount - 1]).toEqual({ + pid: 1000 + instanceCount - 1, + ppid: 1, + cpu: 0, + memory: 2048, + privateMemory: 4096 + }) + }) + + it('ignores aggregate and incomplete rows and clamps invalid memory', async () => { + const { parseTypeperfProcessOutput } = await loadWindowsProcessSampleParsing() + const stdout = [ + '"(PDH-CSV 4.0)","\\\\HOST\\Process(_Total)\\ID Process","\\\\HOST\\Process(cmd)\\ID Process","\\\\HOST\\Process(orphan)\\ID Process","\\\\HOST\\Process(_Total)\\Creating Process ID","\\\\HOST\\Process(cmd)\\Creating Process ID","\\\\HOST\\Process(_Total)\\Working Set","\\\\HOST\\Process(cmd)\\Working Set"', + '"time","0.000000","100.000000","200.000000","0.000000","1.000000","999999.000000","-1.000000"' + ].join('\r\n') + + expect(parseTypeperfProcessOutput(stdout)).toEqual([{ pid: 100, ppid: 1, cpu: 0, memory: 0 }]) + }) +}) diff --git a/src/main/memory/windows-process-sample-parsing.ts b/src/main/memory/windows-process-sample-parsing.ts new file mode 100644 index 00000000000..b29c109dbdc --- /dev/null +++ b/src/main/memory/windows-process-sample-parsing.ts @@ -0,0 +1,240 @@ +/** + * Text parsers for the two Windows process-table formats the memory collector + * reads: tab-delimited `Get-CimInstance Win32_Process` rows and one Typeperf + * CSV sample. Kept apart from the backend/CPU-delta orchestration so each side + * stays readable on its own. + */ + +import { + iterateProcessOutputLines, + PROCESS_OUTPUT_FIELD_SCAN_MAX_CHARS +} from '../../shared/process-output-field-scanner' + +/** + * Counter paths typeperf is asked for, kept beside the decoder that reads their + * names back out of the PDH header. + */ +export const TYPEPERF_COUNTERS = [ + '\\Process(*)\\ID Process', + '\\Process(*)\\Creating Process ID', + '\\Process(*)\\Working Set', + '\\Process(*)\\Private Bytes' +] as const + +const TYPEPERF_MAX_INSTANCES = 4_096 +// Why derived: PDH emits one field per counter per instance plus a timestamp, so +// a fixed cap silently shrinks the parsable process count each time a counter is +// added. The 1 MB line cap bounds memory independently. +const TYPEPERF_MAX_FIELDS = 1 + TYPEPERF_COUNTERS.length * TYPEPERF_MAX_INSTANCES +const TYPEPERF_MAX_LINE_CHARS = 1024 * 1024 + +export type WindowsProcessResourceRow = { + pid: number + ppid: number + /** Percent of one core (may exceed 100 on multi-core). */ + cpu: number + /** Resident memory in bytes. */ + memory: number + /** Committed private bytes, resident or paged out. Absent when the host did not report it. */ + privateMemory?: number +} + +export type WindowsCpuTimes = { + cpuTicks: bigint + startTimeId: string +} + +export type ParsedWindowsProcessSample = { + rows: WindowsProcessResourceRow[] + cpuByPid: Map +} + +type TypeperfProcessFields = { + pid?: number + ppid?: number + memory?: number + privateMemory?: number +} + +export function parseWindowsProcessSample(stdout: string): ParsedWindowsProcessSample { + const rows: WindowsProcessResourceRow[] = [] + const cpuByPid = new Map() + for (const line of iterateProcessOutputLines(stdout)) { + const fields = parseCimTabFields(line) + if (fields.length < 3) { + continue + } + const pid = Number.parseInt(fields[0], 10) + const ppid = Number.parseInt(fields[1], 10) + const memory = Number.parseInt(fields[2], 10) + if (!Number.isSafeInteger(pid) || pid <= 0 || !Number.isSafeInteger(ppid) || ppid < 0) { + continue + } + const privateMemory = parseCimPageFileBytes(fields[6]) + rows.push({ + pid, + ppid, + cpu: 0, + memory: Number.isFinite(memory) && memory > 0 ? memory : 0, + ...(privateMemory === null ? {} : { privateMemory }) + }) + + const kernelTicks = parseUnsignedBigInt(fields[3]) + const userTicks = parseUnsignedBigInt(fields[4]) + const startTimeId = fields[5] ?? '' + if ( + kernelTicks !== null && + userTicks !== null && + /^\d+$/.test(startTimeId) && + !/^0+$/.test(startTimeId) + ) { + cpuByPid.set(pid, { cpuTicks: kernelTicks + userTicks, startTimeId }) + } + } + return { rows, cpuByPid } +} + +function parseCimTabFields(line: string): string[] { + // Why: CIM serializes null properties as empty tab fields; collapsing + // whitespace would shift CPU counters into the working-set column. + if (line.length > PROCESS_OUTPUT_FIELD_SCAN_MAX_CHARS) { + return [] + } + return line.split('\t', 7).map((field) => field.trim()) +} + +/** + * Win32_Process.PageFileUsage is a UInt32 of KILOBYTES. null (not 0) when the + * property is missing, because a host that cannot report commit must not be + * indistinguishable from a process holding none. + */ +function parseCimPageFileBytes(field: string | undefined): number | null { + if (!field) { + return null + } + const kb = Number.parseInt(field, 10) + return Number.isSafeInteger(kb) && kb >= 0 ? kb * 1024 : null +} + +/** Parse tab-delimited PowerShell CIM process rows without deriving CPU deltas. */ +export function parseWindowsProcessOutput(stdout: string): WindowsProcessResourceRow[] { + return parseWindowsProcessSample(stdout).rows +} + +/** Parse one CSV sample from Windows Typeperf. */ +export function parseTypeperfProcessOutput(stdout: string): WindowsProcessResourceRow[] { + let headers: string[] | null = null + let values: string[] | null = null + + for (const line of iterateProcessOutputLines(stdout)) { + if (!line || line.length > TYPEPERF_MAX_LINE_CHARS) { + continue + } + const fields = parseTypeperfCsvLine(line) + if (!headers && fields[0]?.startsWith('(PDH-CSV')) { + headers = fields + continue + } + if (headers && fields.length === headers.length) { + values = fields + break + } + } + + if (!headers || !values) { + return [] + } + + const byInstance = new Map() + for (let index = 1; index < headers.length; index += 1) { + const path = parseTypeperfCounterPath(headers[index]) + if (!path || path.instance === '_Total') { + continue + } + const value = Number.parseFloat(values[index]) + if (!Number.isFinite(value)) { + continue + } + const row = byInstance.get(path.instance) ?? {} + if (path.counter === 'ID Process') { + row.pid = Math.trunc(value) + } else if (path.counter === 'Creating Process ID') { + row.ppid = Math.trunc(value) + } else if (path.counter === 'Working Set') { + row.memory = value + } else if (path.counter === 'Private Bytes') { + row.privateMemory = value + } + byInstance.set(path.instance, row) + } + + const rows: WindowsProcessResourceRow[] = [] + for (const row of byInstance.values()) { + if (row.pid === undefined || row.pid <= 0 || row.ppid === undefined || row.ppid < 0) { + continue + } + rows.push({ + pid: row.pid, + ppid: row.ppid, + cpu: 0, + memory: row.memory !== undefined && row.memory > 0 ? row.memory : 0, + ...(row.privateMemory !== undefined && row.privateMemory >= 0 + ? { privateMemory: row.privateMemory } + : {}) + }) + } + return rows +} + +function parseTypeperfCounterPath(path: string): { instance: string; counter: string } | null { + const processStart = path.lastIndexOf('\\Process(') + const counterStart = path.lastIndexOf(')\\') + if (processStart === -1 || counterStart <= processStart + 9) { + return null + } + return { + instance: path.slice(processStart + 9, counterStart), + counter: path.slice(counterStart + 2) + } +} + +function parseTypeperfCsvLine(line: string): string[] { + const fields: string[] = [] + let value = '' + let quoted = false + + for (let index = 0; index < line.length; index += 1) { + const char = line[index] + if (char === '"') { + if (quoted && line[index + 1] === '"') { + value += '"' + index += 1 + } else { + quoted = !quoted + } + continue + } + if (char === ',' && !quoted) { + fields.push(value) + value = '' + if (fields.length >= TYPEPERF_MAX_FIELDS) { + return [] + } + continue + } + value += char + } + fields.push(value) + return fields +} + +function parseUnsignedBigInt(value: string | undefined): bigint | null { + if (!value || !/^\d+$/.test(value)) { + return null + } + try { + return BigInt(value) + } catch { + return null + } +} diff --git a/src/renderer/src/components/status-bar/ResourceUsageStatusSegment.tsx b/src/renderer/src/components/status-bar/ResourceUsageStatusSegment.tsx index d93f7a69825..a4d57276d1c 100644 --- a/src/renderer/src/components/status-bar/ResourceUsageStatusSegment.tsx +++ b/src/renderer/src/components/status-bar/ResourceUsageStatusSegment.tsx @@ -67,7 +67,11 @@ import { getResourceManagerAriaLabel, getResourceManagerTooltipLines } from './resource-manager-terminal-copy' -import { getResourceMemoryMetricCopy } from './resource-memory-metric-copy' +import { + getCommitPressureToneClass, + getResourceCommitMetricCopy, + getResourceMemoryMetricCopy +} from './resource-memory-metric-copy' import { requiresKillConfirmation } from './resource-session-kill-confirmation' import { resolveResourceManagerWorktreeTarget } from './resource-manager-worktree-target' import { @@ -960,15 +964,29 @@ export function ResourceUsageStatusSegment({ const memoryMetricCopy = getResourceMemoryMetricCopy( resourceSnapshot?.processMemoryMetric ?? 'rss' ) - const { totalMemory, totalCpu, memBadgeLabel } = useMemo(() => { - const memory = resourceSnapshot?.totalMemory ?? 0 - const cpu = resourceSnapshot?.totalCpu ?? 0 - return { - totalMemory: memory, - totalCpu: cpu, - memBadgeLabel: resourceSnapshot ? formatMemory(memory) : '—' - } - }, [resourceSnapshot]) + // Why null-not-zero: a host that cannot read commit (every Unix host, and any + // host older than the field) must render nothing here, never "0 B committed". + const commitMetricCopy = resourceSnapshot?.processCommitMetric + ? getResourceCommitMetricCopy() + : null + const { totalMemory, totalCpu, memBadgeLabel, totalPrivateMemory, commitToneClass } = + useMemo(() => { + const memory = resourceSnapshot?.totalMemory ?? 0 + const cpu = resourceSnapshot?.totalCpu ?? 0 + const privateMemory = resourceSnapshot?.totalPrivateMemory + return { + totalMemory: memory, + totalCpu: cpu, + memBadgeLabel: resourceSnapshot ? formatMemory(memory) : '—', + totalPrivateMemory: privateMemory, + commitToneClass: getCommitPressureToneClass({ + privateMemory, + hostTotalMemory: resourceSnapshot?.host.totalMemory ?? 0 + }) + } + }, [resourceSnapshot]) + const commitBadgeLabel = + commitMetricCopy && totalPrivateMemory !== undefined ? formatMemory(totalPrivateMemory) : null // Why: memorySnapshotError null means "succeeded" OR "never fetched"; a sessions failure before any snapshot still counts as daemon-unreachable. const daemonUnreachable = sessionsError && (memorySnapshotError !== null || snapshot === null) @@ -976,7 +994,14 @@ export function ResourceUsageStatusSegment({ const sessionsOnlyError = sessionsError && memorySnapshotError === null const resourceManagerTooltipLines = getResourceManagerTooltipLines({ memoryLabel: resourceSnapshot - ? `${memBadgeLabel} · ${memoryMetricCopy.summaryLabel}` + ? [ + `${memBadgeLabel} · ${memoryMetricCopy.summaryLabel}`, + commitBadgeLabel && commitMetricCopy + ? `${commitBadgeLabel} ${commitMetricCopy.summaryLabel}` + : null + ] + .filter(Boolean) + .join(' · ') : memBadgeLabel, sessionCount: triggerSessionCount, spaceScanReady @@ -1162,7 +1187,14 @@ export function ResourceUsageStatusSegment({ {!iconOnly && ( <> - + {/* Tint only: the number stays the resident sum it has always been, + and the tooltip names the commit figure that raised the tone. */} + {memBadgeLabel} · @@ -1353,6 +1385,30 @@ export function ResourceUsageStatusSegment({ {memoryMetricCopy.description} + {commitBadgeLabel && commitMetricCopy && ( + <> + · + + + + {commitBadgeLabel}{' '} + + {commitMetricCopy.summaryLabel} + + + + + {commitMetricCopy.description} + + + + )} {orphanCount > 0 && ( diff --git a/src/renderer/src/components/status-bar/resource-memory-metric-copy.test.ts b/src/renderer/src/components/status-bar/resource-memory-metric-copy.test.ts index c4d65bdb151..b743916e5b6 100644 --- a/src/renderer/src/components/status-bar/resource-memory-metric-copy.test.ts +++ b/src/renderer/src/components/status-bar/resource-memory-metric-copy.test.ts @@ -4,7 +4,11 @@ vi.mock('@/i18n/i18n', () => ({ translate: (_key: string, fallback: string) => fallback })) -import { getResourceMemoryMetricCopy } from './resource-memory-metric-copy' +import { + getCommitPressureToneClass, + getResourceCommitMetricCopy, + getResourceMemoryMetricCopy +} from './resource-memory-metric-copy' describe('resource memory metric copy', () => { it('discloses that Unix RSS sums can repeat shared and aliased pages', () => { @@ -16,11 +20,54 @@ describe('resource memory metric copy', () => { }) }) - it('uses working-set terminology for Windows snapshots', () => { + it('says working set counts only resident pages, so paged-out memory is missing', () => { expect(getResourceMemoryMetricCopy('working-set')).toEqual({ columnLabel: 'WS', summaryLabel: 'Σ WS', - description: 'Summed working set (WS). Shared pages can appear in more than one process.' + description: + 'Summed working set (WS): pages resident in RAM right now. Shared pages can appear in more than one process, and memory Windows has paged out is not counted here.' + }) + }) + + it('labels committed bytes as a separate quantity, not a corrected working set', () => { + expect(getResourceCommitMetricCopy()).toEqual({ + summaryLabel: 'Σ Private', + description: + 'Summed private bytes: memory these processes have committed, counted whether it is resident or paged out. This is what the host charges against its commit limit, so it keeps rising while the working set above shrinks under paging.' }) }) }) + +describe('commit pressure tone', () => { + const hostTotalMemory = 16 * 1024 ** 3 + + it('stays silent while tracked commit is a modest share of RAM', () => { + expect(getCommitPressureToneClass({ privateMemory: 4 * 1024 ** 3, hostTotalMemory })).toBeNull() + }) + + it('warns at the same 60/80 thresholds the host usage bars already use', () => { + expect(getCommitPressureToneClass({ privateMemory: 10 * 1024 ** 3, hostTotalMemory })).toBe( + 'text-yellow-500' + ) + // The reported host: 13.4 GB committed by agents on 16 GB of RAM. + expect(getCommitPressureToneClass({ privateMemory: 13.4 * 1024 ** 3, hostTotalMemory })).toBe( + 'text-red-500' + ) + }) + + it('stays silent for a snapshot that carries no commit figure at all', () => { + expect(getCommitPressureToneClass({ privateMemory: undefined, hostTotalMemory })).toBeNull() + }) + + it('stays silent when the host total is unknown, rather than dividing by zero', () => { + expect( + getCommitPressureToneClass({ privateMemory: 8 * 1024 ** 3, hostTotalMemory: 0 }) + ).toBeNull() + }) + + it('keeps warning above 100% of RAM rather than capping the share', () => { + expect(getCommitPressureToneClass({ privateMemory: 32 * 1024 ** 3, hostTotalMemory })).toBe( + 'text-red-500' + ) + }) +}) diff --git a/src/renderer/src/components/status-bar/resource-memory-metric-copy.ts b/src/renderer/src/components/status-bar/resource-memory-metric-copy.ts index 50448bc4ae6..33105696d7a 100644 --- a/src/renderer/src/components/status-bar/resource-memory-metric-copy.ts +++ b/src/renderer/src/components/status-bar/resource-memory-metric-copy.ts @@ -1,5 +1,6 @@ import type { ProcessMemoryMetric } from '../../../../shared/process-stats-types' import { translate } from '@/i18n/i18n' +import { usageTextColorClass } from './usage-roster-formatting' export type ResourceMemoryMetricCopy = { columnLabel: string @@ -14,7 +15,7 @@ export function getResourceMemoryMetricCopy(metric: ProcessMemoryMetric): Resour summaryLabel: 'Σ WS', description: translate( 'auto.components.status.bar.resource.memory.metric.workingSetDescription', - 'Summed working set (WS). Shared pages can appear in more than one process.' + 'Summed working set (WS): pages resident in RAM right now. Shared pages can appear in more than one process, and memory Windows has paged out is not counted here.' ) } } @@ -27,3 +28,39 @@ export function getResourceMemoryMetricCopy(metric: ProcessMemoryMetric): Resour ) } } + +/** No column of its own yet, so no `columnLabel`: the commit figure is a summary + tooltip. */ +export function getResourceCommitMetricCopy(): Omit { + return { + summaryLabel: 'Σ Private', + description: translate( + 'auto.components.status.bar.resource.memory.metric.privateBytesDescription', + 'Summed private bytes: memory these processes have committed, counted whether it is resident or paged out. This is what the host charges against its commit limit, so it keeps rising while the working set above shrinks under paging.' + ) + } +} + +/** + * Warning tint once *Orca's own* tracked commit grows large against physical + * RAM, on the same 60/80 bands as the host usage bars. Deliberately not a + * host-wide paging predictor: that needs the host's commit charge and commit + * limit, which this snapshot does not carry (#16211). + * + * Null both when the share is unremarkable and when the snapshot has no commit + * figure at all — silence is the honest answer for an unmeasured host. + */ +export function getCommitPressureToneClass(args: { + privateMemory: number | undefined + hostTotalMemory: number +}): string | null { + const { privateMemory, hostTotalMemory } = args + if (typeof privateMemory !== 'number' || !Number.isFinite(privateMemory)) { + return null + } + if (!Number.isFinite(hostTotalMemory) || hostTotalMemory <= 0) { + return null + } + // Uncapped on purpose: commit past 100% of RAM is the loudest case, not an error. + const tone = usageTextColorClass((privateMemory / hostTotalMemory) * 100) + return tone === 'text-foreground' ? null : tone +} diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index c776a97452a..2432d4c5952 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -3842,7 +3842,8 @@ "resource": { "memory": { "metric": { - "workingSetDescription": "Summed working set (WS). Shared pages can appear in more than one process.", + "workingSetDescription": "Summed working set (WS): pages resident in RAM right now. Shared pages can appear in more than one process, and memory Windows has paged out is not counted here.", + "privateBytesDescription": "Summed private bytes: memory these processes have committed, counted whether it is resident or paged out. This is what the host charges against its commit limit, so it keeps rising while the working set above shrinks under paging.", "rssDescription": "Summed resident set size (RSS). Shared or aliased pages can appear in more than one process." } }, diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index 41b300a10e8..5142116a66b 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -3501,7 +3501,8 @@ "resource": { "memory": { "metric": { - "workingSetDescription": "Suma del conjunto de trabajo (WS). Las páginas compartidas pueden aparecer en más de un proceso.", + "workingSetDescription": "Suma del conjunto de trabajo (WS): las páginas residentes en RAM en este momento. Las páginas compartidas pueden aparecer en más de un proceso, y la memoria que Windows ha paginado a disco no se cuenta aquí.", + "privateBytesDescription": "Suma de bytes privados: la memoria que estos procesos han confirmado, contada tanto si está residente como si está paginada a disco. Es lo que el host imputa a su límite de confirmación, así que sigue subiendo mientras el conjunto de trabajo de arriba se reduce por la paginación.", "rssDescription": "Suma del tamaño del conjunto residente (RSS). Las páginas compartidas o con alias pueden aparecer en más de un proceso." } }, diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index d9466efd0c1..3b9ebdd2393 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -3501,7 +3501,8 @@ "resource": { "memory": { "metric": { - "workingSetDescription": "合計ワーキングセット(WS)。共有ページが複数のプロセスに表示されることがあります。", + "workingSetDescription": "合計ワーキングセット(WS): 現在 RAM に常駐しているページです。共有ページが複数のプロセスに表示されることがあり、Windows がページアウトしたメモリはここには含まれません。", + "privateBytesDescription": "合計プライベートバイト: これらのプロセスがコミットしたメモリで、常駐中かページアウト済みかを問わず計上されます。ホストがコミット制限に対して計上する値であるため、ページングによって上のワーキングセットが縮小しても増え続けます。", "rssDescription": "合計レジデントセットサイズ(RSS)。共有またはエイリアスされたページが複数のプロセスに表示されることがあります。" } }, diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index f154990358d..e5aefb9291a 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -3506,7 +3506,8 @@ "resource": { "memory": { "metric": { - "workingSetDescription": "합산된 워킹 세트(WS). 공유 페이지가 둘 이상의 프로세스에 나타날 수 있습니다.", + "workingSetDescription": "합산된 워킹 세트(WS): 지금 RAM에 상주 중인 페이지입니다. 공유 페이지가 둘 이상의 프로세스에 나타날 수 있으며, Windows가 페이지 아웃한 메모리는 여기에 포함되지 않습니다.", + "privateBytesDescription": "합산된 프라이빗 바이트: 이 프로세스들이 커밋한 메모리로, 상주 여부와 관계없이 계산됩니다. 호스트가 커밋 한도에 반영하는 값이므로, 페이징으로 위의 워킹 세트가 줄어드는 동안에도 계속 늘어납니다.", "rssDescription": "합산된 레지던트 세트 크기(RSS). 공유 또는 별칭된 페이지가 둘 이상의 프로세스에 나타날 수 있습니다." } }, diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index f68271bf57f..6b5cae8fc24 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -3516,7 +3516,8 @@ "resource": { "memory": { "metric": { - "workingSetDescription": "工作集 (WS) 的总和。共享页可能会出现在多个进程中。", + "workingSetDescription": "工作集 (WS) 的总和:当前驻留在 RAM 中的页。共享页可能会出现在多个进程中,被 Windows 换出到页面文件的内存不计入其中。", + "privateBytesDescription": "专用字节的总和:这些进程已提交的内存,无论驻留还是已换出都会计入。这是主机计入提交限制的数值,因此在分页导致上方工作集缩小时它仍会继续上升。", "rssDescription": "驻留集大小 (RSS) 的总和。共享页或别名映射页可能会出现在多个进程中。" } }, diff --git a/src/shared/process-stats-types.ts b/src/shared/process-stats-types.ts index ec0187864ab..47933d0978d 100644 --- a/src/shared/process-stats-types.ts +++ b/src/shared/process-stats-types.ts @@ -14,10 +14,29 @@ export type StatsSummary = { export type UsageValues = { cpu: number memory: number + /** + * Committed bytes (see `ProcessCommitMetric`), a second quantity alongside + * `memory` — never a substitute for it. Absent means the host cannot report + * it; absent must never be read as zero. + */ + privateMemory?: number } export type ProcessMemoryMetric = 'rss' | 'working-set' +/** + * Unit of every `privateMemory` field in a snapshot. + * + * `private-bytes` is the Windows private commit charge + * (`Win32_Process.PageFileUsage` / `\Process(*)\Private Bytes`): memory a + * process has committed whether or not it is currently resident. Working set + * counts only resident pages, so an agent whose pages have been trimmed to the + * pagefile shrinks its working set while still holding the commit that pushes + * the host into paging. Unix has no equivalent, so snapshots from those hosts + * carry no commit metric at all. + */ +export type ProcessCommitMetric = 'private-bytes' + export type HostAvailableMemorySource = 'memory-pressure' | 'proc-meminfo' | 'free-memory' /** The top-level cpu/memory are the sum of main + renderer + other. */ @@ -66,9 +85,21 @@ export type MemorySnapshot = { host: HostMemory /** Per-process byte metric used by app, session, worktree, history, and totalMemory values. */ processMemoryMetric: ProcessMemoryMetric + /** + * Names the unit of every `privateMemory` field below. Absent when this sweep + * produced none — an older host, or any host whose process table cannot + * report committed bytes. Readers must treat absence as unknown, not zero. + */ + processCommitMetric?: ProcessCommitMetric /** Sum of app + all tracked worktree sessions. Percent of a single core, so may exceed 100 on multi-core machines. */ totalCpu: number /** Sum of per-process samples. Shared pages may repeat, so this can exceed host.totalMemory. */ totalMemory: number + /** + * Sum of app + all tracked worktree `privateMemory`. Present exactly when + * `processCommitMetric` is. Committed bytes are not bounded by physical RAM, + * so exceeding host.totalMemory is the signal, not a bug. + */ + totalPrivateMemory?: number collectedAt: number } From 96565fe370010d67491a0dce5bc5597ffb212a4f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 26 Aug 2026 15:44:11 -0700 Subject: [PATCH 12/19] perf(source-control): stop re-running every git read on each file selection (#15036) (#16600) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(source-control): stop blocking main on four sync git-dir probes per status poll detectConflictOperation ran four existsSync calls against the git dir on every status poll. On a `\\wsl.localhost\...` worktree each one is a 9p round trip, and being synchronous they landed on the Electron main thread back to back. Replace them with concurrent fs/promises access probes: same "any failure reads as absent" semantics existsSync had, one wave instead of four serialized blocking calls. The outer try/catch went with them -- neither resolveGitDir nor the probes can throw now, so it was unreachable. Part of #15036 (source-control latency). * perf(wsl): let git reads take the shell-free route from a cwd-derived distro shouldAttemptWslDirectGit required options.wslDistro, so a `\\wsl.localhost\...` worktree without a resolved WSL project runtime never qualified -- even though the distro is right there in the cwd and wslDistroForCommand already knew how to read it. Every `git show` behind a diff therefore ran through the user's login shell, executing their rc once per blob read. Three changes: - Derive the distro from the cwd when no override was supplied. This is the fix; the routing decision now depends on where the repo actually lives. - Wait, bounded, for a cold read-environment probe instead of resolving without it. The probe is one wsl.exe call shared per distro, so the wait is paid at most once, and past WSL_GIT_READ_ENVIRONMENT_WAIT_MS the shell route runs exactly as before. It returns null rather than a settled promise when there is nothing to wait for, so a non-WSL git call is not pushed into a later microtask. - Opt the blob reads into preferWslDirectGit via gitReadOptionsForWorktree (renamed from gitStatusReadOptionsForWorktree; it was never status-specific). Belt-and- braces only: `show`, `config --get-regexp`, `ls-files` and `rev-parse` were all already matched by isWslDirectGitReadCommand, so this changes no routing today -- it just stops the diff path depending on a heuristic it knows the answer to. git-blob-read also gains a `failed` flag distinguishing "git ran and reported the path absent" (exit 128) from "the read never got an answer"; nothing consumes it yet, the settled diff cache does. Part of #15036 (source-control latency). * perf(source-control): give diff reads a settled cache keyed on stamped git state gitDiffReadDedupe coalesces only while a read is in flight, so every file selection re-ran the whole read: a `git config --file .gitmodules` spawn, one or two `git show` spawns, and a working-tree stat+read. On a WSL/UNC worktree each git spawn is a wsl.exe invocation, which is the ">3s Loading diff..." in #15036. Correctness first -- a stale diff is worse than a slow one. The cache never expires on a clock and there is no TTL to tune. Instead: - worktree-diff-stamp.ts takes a subprocess-free stamp of exactly the inputs a file diff is built from: HEAD (by resolved tip *content*, so a commit is visible even though HEAD's own bytes never move), `.git/index` (mtime+size), `.gitmodules` (submodule routing), and the working-tree file. A linked worktree's commondir and the packed-refs/reftable fallback are handled; an unborn branch is caught by recording "no loose ref" rather than only the packed stamps. - The stamp is captured BEFORE the read and stored with the result. Anything that moves during or after the read leaves the stored stamp behind, so the next lookup misses. That, not a freshness window, is why a stale diff cannot be served. - A store is refused unless the stamp was taken a full mtime bucket (2s, FAT's granularity) after its newest component. Below that, a second write inside the same bucket would be invisible -- git's own racy-index rule. - `null` stamp means "cannot prove" and never caches: a folder workspace, a repo whose layout cannot be read, or a filesystem reporting no usable mtime. - Submodule routes and reads that failed rather than proved absence are not reusable. A wsl.exe hiccup produces the same empty left side a new file does, and pinning that would persist a wrong diff. - invalidateGitReadCaches clears it and bumps a generation, so a read that started pre-mutation cannot store its result post-mutation. `ino` is deliberately optional in the working-tree component: Windows reports 0 for it on the redirector behind `\\wsl.localhost`, and requiring an unstable 0 to match would make the cache silently never hit on the exact host it exists for. Cache counters are exposed for the same reason -- a miss storm and a cold start otherwise look identical. Also drops gitDiffReadDedupe.clear() from getStatus. A status poll is a read; all it did was destroy a live coalescing entry so a concurrent identical request started duplicate git work. Mutations still invalidate through the shared point. Memory is bounded by retained characters, not entry count -- one diff result can legitimately hold megabytes. Fixes the source-control half of #15036. * perf(source-control): reuse BoundedMap and stop the WSL probe wait from outliving its answer Review follow-ups on the settled-diff-cache work: - SettledDiffCache now sits on the shared BoundedMap instead of hand-rolling the same Map + character ledger + evict-oldest loop. - pendingWslDirectGitReadEnvironment returns null once the probe has settled either way, so a distro whose direct route was disabled no longer pays for a 1.5s timer and two microtask hops on every git read. - That wait now honours the read's abort signal and goes through withTimeout, so an aborted read is not held for the full bound and a probe rejection can never surface as a read failure. - The settled-cache generation fence is taken before the stamp read, so a mutation that lands entirely inside the stamp's stats can no longer store an entry whose stamp is torn across it. - The cache counters are folded into the main-thread churn probe report, which is what tells a permanently-cold cache apart from a cold start in the field. * fix(source-control): tell WSL clock skew apart from a genuinely fresh write The racy-write margin compares two clocks: capturedAtMs is this host's, while the component mtimes come from whatever wrote the files. On a \\wsl.localhost worktree the guest sets them, so a guest running ahead pushes every recently-touched file past the margin and the cache refuses to store — for as long as the skew lasts, on exactly the platform this cache exists for. Nothing was wrong with the refusal; it was invisible. racyWrites alone cannot distinguish "the repo was just edited" from "the clocks disagree and this will never resolve on its own", so a permanently cold cache looked like a cold start. isDiffStampClockSkewed flags the one thing no local write can produce — an mtime in this host's future — and the cache counts those separately as clockSkewedWrites. A nonzero count is the signal that the cache is off for a reason idling will not fix. Found by review of #16600; behavior is unchanged, only observability. --- .../main-thread-churn-probe.test.ts | 32 +- .../diagnostics/main-thread-churn-probe.ts | 17 +- .../command-runner/git-command-resolution.ts | 68 +++- src/main/git/command-runner/git-exec-file.ts | 11 +- .../git/command-runner/git-stream-stdout.ts | 5 + src/main/git/git-runtime-options.ts | 8 +- src/main/git/runner-wsl-direct-read.test.ts | 136 ++++++- .../git/settled-diff-cache-bounds.test.ts | 154 ++++++++ .../effective-upstream-status-probe.ts | 6 +- src/main/git/source-control/file-diff.ts | 113 ++++-- src/main/git/source-control/git-blob-read.ts | 38 +- .../source-control/git-conflict-operation.ts | 29 +- .../git-read-cache-invalidation.ts | 5 + .../git/source-control/settled-diff-cache.ts | 155 ++++++++ .../status-branch-line-total-input.ts | 4 +- .../git/source-control/status-line-stats.ts | 4 +- src/main/git/source-control/status-read.ts | 8 +- .../git/source-control/worktree-diff-stamp.ts | 226 ++++++++++++ .../git/status-conflict-operations.test.ts | 82 +++-- .../git/status-diff-settled-cache.test.ts | 331 ++++++++++++++++++ src/main/git/status-diff.test.ts | 41 ++- src/main/git/status-submodule.test.ts | 11 +- src/main/git/status-test-harness.ts | 9 +- src/main/git/status.test.ts | 43 +-- src/main/git/wsl-git-read-environment.ts | 37 +- src/main/index.ts | 5 +- 26 files changed, 1425 insertions(+), 153 deletions(-) create mode 100644 src/main/git/settled-diff-cache-bounds.test.ts create mode 100644 src/main/git/source-control/settled-diff-cache.ts create mode 100644 src/main/git/source-control/worktree-diff-stamp.ts create mode 100644 src/main/git/status-diff-settled-cache.test.ts diff --git a/src/main/diagnostics/main-thread-churn-probe.test.ts b/src/main/diagnostics/main-thread-churn-probe.test.ts index 62ede5c99c8..bb87989d56d 100644 --- a/src/main/diagnostics/main-thread-churn-probe.test.ts +++ b/src/main/diagnostics/main-thread-churn-probe.test.ts @@ -1,10 +1,19 @@ import { afterEach, describe, expect, it, vi } from 'vitest' + +const { writeStartupDiagnosticLineMock } = vi.hoisted(() => ({ + writeStartupDiagnosticLineMock: vi.fn() +})) +vi.mock('../startup/startup-diagnostics', () => ({ + writeStartupDiagnosticLine: writeStartupDiagnosticLineMock +})) + import { MAIN_THREAD_DIAGNOSTICS_ENV, classifySubprocessCommand, drainSubprocessSpawnStats, isMainThreadDiagnosticsEnabled, - recordSubprocessSpawn + recordSubprocessSpawn, + startMainThreadChurnProbe } from './main-thread-churn-probe' afterEach(() => { @@ -87,3 +96,24 @@ describe('recordSubprocessSpawn', () => { expect(drainSubprocessSpawnStats()).toEqual({}) }) }) + +describe('startMainThreadChurnProbe', () => { + // Counters nothing reads are counters nobody can act on: the probe report is what makes + // "the diff cache never hit" visible outside a test run. + it('folds caller-contributed counters into the report line', async () => { + vi.stubEnv(MAIN_THREAD_DIAGNOSTICS_ENV, '1') + writeStartupDiagnosticLineMock.mockClear() + vi.useFakeTimers() + try { + startMainThreadChurnProbe({ extraStats: () => ({ diffCache: { hits: 3, misses: 1 } }) }) + await vi.advanceTimersByTimeAsync(5_100) + } finally { + vi.useRealTimers() + } + + const line = String(writeStartupDiagnosticLineMock.mock.calls.at(-1)?.[0] ?? '') + expect(JSON.parse(line.replace('[main-thread] ', ''))).toMatchObject({ + diffCache: { hits: 3, misses: 1 } + }) + }) +}) diff --git a/src/main/diagnostics/main-thread-churn-probe.ts b/src/main/diagnostics/main-thread-churn-probe.ts index 1d70999bc14..af6877e408e 100644 --- a/src/main/diagnostics/main-thread-churn-probe.ts +++ b/src/main/diagnostics/main-thread-churn-probe.ts @@ -136,14 +136,20 @@ export function writeMainThreadDiagnosticMarker(marker: string): void { ) } +export type MainThreadChurnProbeOptions = { + /** Extra counters folded into each report line, sampled once per window. */ + extraStats?: () => Record +} + /** * Long-running main-process jank probe for benchmarks and field diagnosis of * issue #7576. Every 5s emits one `[main-thread] {json}` stderr line with the - * window's worst event-loop stall, stall counts over 50/250ms, and drained - * subprocess spawn stats. Unlike the startup stall probe this never stops: - * the churn it measures (git status polling, updater retries) is steady-state. + * window's worst event-loop stall, stall counts over 50/250ms, drained subprocess + * spawn stats, and any counters the caller contributes. Unlike the startup stall + * probe this never stops: the churn it measures (git status polling, updater + * retries) is steady-state. */ -export function startMainThreadChurnProbe(): void { +export function startMainThreadChurnProbe(options: MainThreadChurnProbeOptions = {}): void { if (!isMainThreadDiagnosticsEnabled()) { return } @@ -176,7 +182,8 @@ export function startMainThreadChurnProbe(): void { gapsOver50Ms, gapsOver250Ms, spawnCount: Object.values(spawns).reduce((sum, s) => sum + s.count, 0), - spawns + spawns, + ...options.extraStats?.() } windowMaxGapMs = 0 gapsOver50Ms = 0 diff --git a/src/main/git/command-runner/git-command-resolution.ts b/src/main/git/command-runner/git-command-resolution.ts index 9c53658c159..ddde988b933 100644 --- a/src/main/git/command-runner/git-command-resolution.ts +++ b/src/main/git/command-runner/git-command-resolution.ts @@ -1,10 +1,14 @@ +import { waitForPromiseWithSignal } from '../../../shared/abort-signal-reason' +import { withTimeout } from '../../../shared/promise-timeout-fallback' import { parseWslPath } from '../../wsl' import { isWslDirectGitReadCommand } from '../wsl-direct-git-read-commands' import { disableWslGitReadEnvironment, getWslGitReadEnvironment, invalidateWslGitReadEnvironment, - peekWslGitReadEnvironment + isWslGitReadEnvironmentSettled, + peekWslGitReadEnvironment, + WSL_GIT_READ_ENVIRONMENT_WAIT_MS } from '../wsl-git-read-environment' import { usesHostGitForWslLinkedWorktree } from '../wsl-linked-worktree-git-routing' import { resolveCommand, type ResolvedCommand } from './wsl-command-resolution' @@ -27,23 +31,39 @@ export function resolveGitCommand( // Why: WSL Git resolves a Windows-authored linked-worktree pointer relative to cwd. return { binary: 'git', args, cwd: options.cwd, wsl: null, wslMode: null } } - if (!forceLoginShell && shouldAttemptWslDirectGit(args, options)) { - const distro = wslDistroForCommand(options.cwd, options.wslDistro) - const environment = distro ? peekWslGitReadEnvironment(distro) : undefined + const distro = directWslGitReadDistro(args, options, forceLoginShell) + if (distro) { + const environment = peekWslGitReadEnvironment(distro) if (environment) { - return resolveCommand('git', args, options.cwd, options.wslDistro, { + return resolveCommand('git', args, options.cwd, distro, { wslGitReadEnvironment: environment, env: options.env, terminationBarrier: options.terminationBarrier }) } - if (distro) { - void getWslGitReadEnvironment(distro) - } + void getWslGitReadEnvironment(distro) } return resolveGitCommandWithoutProbe(args, options, captureLoginShellOutput) } +/** + * The distro this command could run shell-free in, or null when it can't. + * + * Why cwd counts: a `\\wsl.localhost\\...` worktree names its distro in + * the path, so requiring the caller to have resolved a WSL project runtime first + * left every diff read on the login shell — one rc run per `git show`. + */ +function directWslGitReadDistro( + args: string[], + options: GitExecOptions, + forceLoginShell: boolean +): string | null { + if (forceLoginShell || !shouldAttemptWslDirectGit(args, options)) { + return null + } + return wslDistroForCommand(options.cwd, options.wslDistro) +} + function shouldAttemptWslDirectGit(args: string[], options: GitExecOptions): boolean { return Boolean( process.platform === 'win32' && @@ -55,8 +75,36 @@ function shouldAttemptWslDirectGit(args: string[], options: GitExecOptions): boo !Object.entries(options.env ?? {}).some( ([key, value]) => key.startsWith('GIT_') && key !== 'GIT_OPTIONAL_LOCKS' && value !== process.env[key] - ) && - options.wslDistro + ) + ) +} + +/** + * Give a read the chance to take the shell-free route instead of silently + * falling back while the probe is still resolving. + * + * The probe is one `wsl.exe` call, shared per distro and reused by every later + * command, so waiting costs at most once. Bounded because a cold or wedged + * distro must not hold a read behind it — past the bound the login-shell route + * runs exactly as it did before. + */ +export function pendingWslDirectGitReadEnvironment( + args: string[], + options: GitExecOptions +): Promise | null { + // Why null rather than a resolved promise: every non-WSL git call goes through here, and + // awaiting even an already-settled promise would push the spawn into a later microtask. + // A settled probe — including a distro whose direct route was permanently disabled — has + // nothing left to wait for, and neither does a read that is already aborted. + const distro = directWslGitReadDistro(args, options, false) + if (!distro || isWslGitReadEnvironmentSettled(distro) || options.signal?.aborted) { + return null + } + // withTimeout also absorbs rejection, so a probe failure can never become a read failure. + return withTimeout( + waitForPromiseWithSignal(getWslGitReadEnvironment(distro), options.signal), + WSL_GIT_READ_ENVIRONMENT_WAIT_MS, + null ) } diff --git a/src/main/git/command-runner/git-exec-file.ts b/src/main/git/command-runner/git-exec-file.ts index c2815c4bb4c..d9762030813 100644 --- a/src/main/git/command-runner/git-exec-file.ts +++ b/src/main/git/command-runner/git-exec-file.ts @@ -13,6 +13,7 @@ import { resolveCommand, type ResolvedCommand } from './wsl-command-resolution' import type { GitExecOptions } from './git-exec-options' import { execFileCapture, execFileCaptureToTermination } from './exec-file-capture' import { + pendingWslDirectGitReadEnvironment, directWslGitExitCode, disableDirectWslGitAfterSuccessfulFallback, invalidateMissingDirectWslGit, @@ -39,6 +40,10 @@ async function gitExecFileAsyncUnlocked( signal: options.signal }) } + const readEnvironmentReady = pendingWslDirectGitReadEnvironment(args, options) + if (readEnvironmentReady) { + await readEnvironmentReady + } let resolved = resolveGitCommand(args, options, false, options.captureWslLoginShellOutput) const environmentReady = prepareWindowsHostGitEnvironment( resolved, @@ -127,11 +132,15 @@ export function gitExecFileAsync( */ export async function gitExecFileAsyncBuffer( args: string[], - options: { cwd: string; maxBuffer?: number; wslDistro?: string } + options: { cwd: string; maxBuffer?: number; wslDistro?: string; preferWslDirectGit?: boolean } ): Promise<{ stdout: Buffer }> { if (isWslLinkedWorktreeGitRoutingCandidate(options.cwd, options.wslDistro)) { await prepareWslLinkedWorktreeGitRouting(options.cwd, options.wslDistro) } + const readEnvironmentReady = pendingWslDirectGitReadEnvironment(args, options) + if (readEnvironmentReady) { + await readEnvironmentReady + } // `git show` is a read, so this normally runs with no shell at all. The fence // still matters for the login-shell fallback: these are raw blob bytes going // straight to the diff/blob viewer, where a banner becomes file content. diff --git a/src/main/git/command-runner/git-stream-stdout.ts b/src/main/git/command-runner/git-stream-stdout.ts index 318be527c41..7f33308d747 100644 --- a/src/main/git/command-runner/git-stream-stdout.ts +++ b/src/main/git/command-runner/git-stream-stdout.ts @@ -11,6 +11,7 @@ import { killSpawnedCommandTree } from './spawned-command-tree-kill' import type { ResolvedCommand } from './wsl-command-resolution' import { DEFAULT_GIT_MAX_BUFFER, type GitExecOptions } from './git-exec-options' import { + pendingWslDirectGitReadEnvironment, directWslGitExitCode, disableDirectWslGitAfterSuccessfulFallback, invalidateMissingDirectWslGit, @@ -64,6 +65,10 @@ export async function gitStreamStdout( ...(options.preferWslDirectGit ? { preferWslDirectGit: true } : {}), ...(options.signal ? { signal: options.signal } : {}) } + const readEnvironmentReady = pendingWslDirectGitReadEnvironment(args, gitOptions) + if (readEnvironmentReady) { + await readEnvironmentReady + } let resolved = resolveGitCommand(args, gitOptions) const environmentReady = prepareWindowsHostGitEnvironment( resolved, diff --git a/src/main/git/git-runtime-options.ts b/src/main/git/git-runtime-options.ts index caa52bd02b4..edc3bd4b30c 100644 --- a/src/main/git/git-runtime-options.ts +++ b/src/main/git/git-runtime-options.ts @@ -14,7 +14,13 @@ export function gitOptionsForWorktree( } } -export function gitStatusReadOptionsForWorktree( +/** + * Options for a git invocation that only reads. Opting in explicitly keeps the + * shell-free WSL route from depending on `wsl-direct-git-read-commands` + * classifying the argv, which is a heuristic these call sites already know the + * answer to. + */ +export function gitReadOptionsForWorktree( cwd: string, options: GitRuntimeOptions = {} ): { diff --git a/src/main/git/runner-wsl-direct-read.test.ts b/src/main/git/runner-wsl-direct-read.test.ts index aa5bd5dbc0d..9cc43223c76 100644 --- a/src/main/git/runner-wsl-direct-read.test.ts +++ b/src/main/git/runner-wsl-direct-read.test.ts @@ -17,11 +17,14 @@ vi.mock('../observability/instrumentation', () => ({ })) vi.mock('../diagnostics/main-thread-churn-probe', () => ({ recordSubprocessSpawn: vi.fn() })) +import { pendingWslDirectGitReadEnvironment } from './command-runner/git-command-resolution' import { gitExecFileAsync, gitSpawn, gitStreamStdout } from './runner' import { + disableWslGitReadEnvironment, getWslGitReadEnvironment, resetWslGitReadEnvironmentForTests, - seedWslGitReadEnvironmentForTests + seedWslGitReadEnvironmentForTests, + WSL_GIT_READ_ENVIRONMENT_WAIT_MS } from './wsl-git-read-environment' import { prepareWslLinkedWorktreeGitRouting, @@ -349,7 +352,9 @@ describe('WSL direct Git reads', () => { }) }) - it('leaves override-less WSL UNC routing on its existing non-login shell', async () => { + // A UNC worktree names its distro in the path, so a read there needs no resolved WSL project + // runtime to skip the shell -- requiring one is what kept every diff read on the login shell. + it('takes the direct read route for an override-less WSL UNC cwd', async () => { await withPlatform('win32', async () => { seedWslGitReadEnvironmentForTests(DISTRO, LOGIN_ENVIRONMENT) succeedExecFile() @@ -359,7 +364,132 @@ describe('WSL direct Git reads', () => { preferWslDirectGit: true }) - expect(execFileMock.mock.calls[0]?.[1]?.slice(3, 5)).toEqual(['bash', '-c']) + const resolved = execFileMock.mock.calls[0]?.[1] ?? [] + expect(resolved).toContain('--exec') + expect(resolved).toContain(`PATH=${LOGIN_ENVIRONMENT.path}`) + expect(resolved).not.toContain('bash') + }) + }) + + // The classifier already recognizes `show`, so the blob reads behind every diff take the + // direct route from the cwd alone -- no caller-supplied distro, no explicit opt-in. + it('takes the direct read route for an unclassified-caller blob read on a UNC cwd', async () => { + await withPlatform('win32', async () => { + seedWslGitReadEnvironmentForTests(DISTRO, LOGIN_ENVIRONMENT) + succeedExecFile() + + await gitExecFileAsync(['show', ':src/file.ts'], { + cwd: String.raw`\\wsl.localhost\Ubuntu\repo` + }) + + expect(execFileMock.mock.calls[0]?.[1]).toContain('--exec') + }) + }) + + // Kicking the probe off and resolving without it left the reads issued before it + // answered on the login shell -- the slow route, chosen by nothing but timing. + it('waits for a cold probe so the first read already skips the shell', async () => { + await withPlatform('win32', async () => { + execFileMock.mockImplementation((_command, args, _options, callback) => { + const child = createMockChild() + if (String(args).includes('_orca_git_path')) { + setTimeout( + () => callback?.(null, fencedProbeStdout(args, LOGIN_ENVIRONMENT_FIELDS), ''), + 0 + ) + } else { + queueMicrotask(() => callback?.(null, 'ok', '')) + } + return child + }) + + await gitExecFileAsync(['show', ':src/file.ts'], { + cwd: String.raw`\\wsl.localhost\Ubuntu\repo` + }) + + expect(execFileMock.mock.calls.at(-1)?.[1]).toContain('--exec') + expect(execFileMock.mock.calls.at(-1)?.[1]).toContain(`PATH=${LOGIN_ENVIRONMENT.path}`) + }) + }) + + it('stops waiting for a wedged probe and reads through the shell route', async () => { + await withPlatform('win32', async () => { + vi.useFakeTimers() + try { + execFileMock.mockImplementation((_command, args, _options, callback) => { + const child = createMockChild() + // The probe never answers; only the git command itself does. + if (!String(args).includes('_orca_git_path')) { + queueMicrotask(() => callback?.(null, 'ok', '')) + } + return child + }) + + const pending = gitExecFileAsync(['show', ':src/file.ts'], { + cwd: String.raw`\\wsl.localhost\Ubuntu\repo` + }) + await vi.advanceTimersByTimeAsync(WSL_GIT_READ_ENVIRONMENT_WAIT_MS) + await pending + + // Exactly the routing this read had before the wait existed: an unanswered + // probe must cost the bound and nothing else. + const resolved = execFileMock.mock.calls.at(-1)?.[1] ?? [] + expect(resolved.slice(3, 5)).toEqual(['bash', '-c']) + expect(resolved).not.toContain('/usr/bin/env') + } finally { + vi.useRealTimers() + } + }) + }) + + // Once the route is disabled the answer is already known, so paying for a timer and two + // extra microtask hops on every later read is pure overhead on the host this PR targets. + it('stops deferring reads once the direct route is disabled for the distro', async () => { + await withPlatform('win32', async () => { + seedWslGitReadEnvironmentForTests(DISTRO, LOGIN_ENVIRONMENT) + disableWslGitReadEnvironment(DISTRO) + + expect( + pendingWslDirectGitReadEnvironment(['show', ':src/file.ts'], { + cwd: String.raw`\\wsl.localhost\Ubuntu\repo` + }) + ).toBeNull() + }) + }) + + it('never waits on the probe for an already-aborted read', async () => { + await withPlatform('win32', async () => { + const controller = new AbortController() + controller.abort() + + expect( + pendingWslDirectGitReadEnvironment(['show', ':src/file.ts'], { + cwd: String.raw`\\wsl.localhost\Ubuntu\repo`, + signal: controller.signal + }) + ).toBeNull() + expect(execFileMock).not.toHaveBeenCalled() + }) + }) + + it('stops waiting for a cold probe as soon as the read aborts', async () => { + await withPlatform('win32', async () => { + vi.useFakeTimers() + try { + // The probe never answers, so only the abort can end the wait. + execFileMock.mockImplementation(() => createMockChild()) + const controller = new AbortController() + + const pending = pendingWslDirectGitReadEnvironment(['show', ':src/file.ts'], { + cwd: String.raw`\\wsl.localhost\Ubuntu\repo`, + signal: controller.signal + }) + controller.abort() + + await expect(pending).resolves.toBeNull() + } finally { + vi.useRealTimers() + } }) }) diff --git a/src/main/git/settled-diff-cache-bounds.test.ts b/src/main/git/settled-diff-cache-bounds.test.ts new file mode 100644 index 00000000000..8d25bba40de --- /dev/null +++ b/src/main/git/settled-diff-cache-bounds.test.ts @@ -0,0 +1,154 @@ +import { describe, expect, it } from 'vitest' +import type { GitDiffResult } from '../../shared/git-diff-compare-types' +import { + MAX_SETTLED_DIFF_CACHE_ENTRIES, + MAX_SETTLED_DIFF_CACHE_RESULT_CHARACTERS, + MAX_SETTLED_DIFF_CACHE_TOTAL_CHARACTERS, + SettledDiffCache +} from './source-control/settled-diff-cache' +import type { WorktreeDiffStamp } from './source-control/worktree-diff-stamp' + +function settledStamp(value: string): WorktreeDiffStamp { + // Old enough that the racy-write margin is satisfied. + return { value, newestMtimeMs: Date.now() - 60_000, capturedAtMs: Date.now() } +} + +function diffOfSize(characters: number): GitDiffResult { + return { + kind: 'text', + originalContent: '', + modifiedContent: 'x'.repeat(characters), + originalIsBinary: false, + modifiedIsBinary: false + } +} + +describe('SettledDiffCache bounds', () => { + it('evicts the least recently used entry past the entry cap', () => { + const cache = new SettledDiffCache() + for (let index = 0; index <= MAX_SETTLED_DIFF_CACHE_ENTRIES; index += 1) { + cache.set(`key-${index}`, settledStamp(`stamp-${index}`), diffOfSize(1), cache.beginRead()) + } + + expect(cache.stats().entries).toBe(MAX_SETTLED_DIFF_CACHE_ENTRIES) + expect(cache.get('key-0', settledStamp('stamp-0'))).toBeUndefined() + expect( + cache.get( + `key-${MAX_SETTLED_DIFF_CACHE_ENTRIES}`, + settledStamp(`stamp-${MAX_SETTLED_DIFF_CACHE_ENTRIES}`) + ) + ).toBeDefined() + }) + + it('keeps a re-read entry alive by refreshing its LRU position', () => { + const cache = new SettledDiffCache() + cache.set('hot', settledStamp('hot-stamp'), diffOfSize(1), cache.beginRead()) + for (let index = 0; index < MAX_SETTLED_DIFF_CACHE_ENTRIES; index += 1) { + cache.get('hot', settledStamp('hot-stamp')) + cache.set(`cold-${index}`, settledStamp(`cold-${index}`), diffOfSize(1), cache.beginRead()) + } + + expect(cache.get('hot', settledStamp('hot-stamp'))).toBeDefined() + }) + + // One entry can legitimately hold megabytes, so an entry count alone bounds nothing. + it('holds total retained content under the character budget', () => { + const cache = new SettledDiffCache() + const chunk = MAX_SETTLED_DIFF_CACHE_RESULT_CHARACTERS + const needed = Math.ceil(MAX_SETTLED_DIFF_CACHE_TOTAL_CHARACTERS / chunk) + 2 + + for (let index = 0; index < needed; index += 1) { + cache.set( + `key-${index}`, + settledStamp(`stamp-${index}`), + diffOfSize(chunk), + cache.beginRead() + ) + } + + expect(cache.stats().retainedCharacters).toBeLessThanOrEqual( + MAX_SETTLED_DIFF_CACHE_TOTAL_CHARACTERS + ) + }) + + it('declines a single result larger than the per-entry cap', () => { + const cache = new SettledDiffCache() + + cache.set( + 'huge', + settledStamp('huge-stamp'), + diffOfSize(MAX_SETTLED_DIFF_CACHE_RESULT_CHARACTERS + 1), + cache.beginRead() + ) + + expect(cache.stats().entries).toBe(0) + expect(cache.stats().retainedCharacters).toBe(0) + }) + + it('releases retained characters when a key is overwritten', () => { + const cache = new SettledDiffCache() + cache.set('key', settledStamp('first'), diffOfSize(1_000), cache.beginRead()) + cache.set('key', settledStamp('second'), diffOfSize(10), cache.beginRead()) + + expect(cache.stats().entries).toBe(1) + expect(cache.stats().retainedCharacters).toBe(10) + expect(cache.get('key', settledStamp('first'))).toBeUndefined() + expect(cache.get('key', settledStamp('second'))).toBeDefined() + }) + + it('drops everything and refuses in-flight stores after a clear', () => { + const cache = new SettledDiffCache() + const readGeneration = cache.beginRead() + cache.clear() + cache.set('key', settledStamp('stamp'), diffOfSize(1), readGeneration) + + expect(cache.stats().entries).toBe(0) + expect(cache.stats().invalidatedDuringRead).toBe(1) + }) +}) + +// Why (#15036 review): the margin compares this host's clock to mtimes the WSL guest +// wrote. A guest running ahead refuses every store for as long as the skew lasts, and +// without its own counter that is indistinguishable from a repo nobody has touched. +describe('settled diff cache clock skew', () => { + const result: GitDiffResult = diffOfSize(10) + + it('counts a future-dated mtime separately from an ordinary racy write', () => { + const cache = new SettledDiffCache() + const now = Date.now() + + // Honestly fresh: written just now, margin not yet satisfied. + cache.set( + 'fresh', + { value: 'a', newestMtimeMs: now, capturedAtMs: now }, + result, + cache.beginRead() + ) + // Skewed: mtime in this host's future, which no local write can produce. + cache.set( + 'skewed', + { value: 'b', newestMtimeMs: now + 60_000, capturedAtMs: now }, + result, + cache.beginRead() + ) + + const stats = cache.stats() + expect(stats.racyWrites).toBe(2) + expect(stats.clockSkewedWrites).toBe(1) + expect(stats.stores).toBe(0) + }) + + it('does not flag a stamp whose components were all absent', () => { + const cache = new SettledDiffCache() + const now = Date.now() + + cache.set( + 'absent', + { value: 'c', newestMtimeMs: Number.NEGATIVE_INFINITY, capturedAtMs: now }, + result, + cache.beginRead() + ) + + expect(cache.stats().clockSkewedWrites).toBe(0) + }) +}) diff --git a/src/main/git/source-control/effective-upstream-status-probe.ts b/src/main/git/source-control/effective-upstream-status-probe.ts index bea3dc93d79..1fd15cda486 100644 --- a/src/main/git/source-control/effective-upstream-status-probe.ts +++ b/src/main/git/source-control/effective-upstream-status-probe.ts @@ -6,7 +6,7 @@ import { } from '../../../shared/git-effective-upstream' import { createGitConfigSnapshotRunner } from '../../../shared/git-config-snapshot-runner' import type { GitRuntimeOptions } from '../git-runtime-options' -import { gitStatusReadOptionsForWorktree } from '../git-runtime-options' +import { gitReadOptionsForWorktree } from '../git-runtime-options' import { gitExecFileAsync } from '../runner' import { MAX_EFFECTIVE_UPSTREAM_NEGATIVE_CACHE_ENTRIES, @@ -90,7 +90,7 @@ async function probeOrRevalidateEffectiveUpstreamStatus( } else if (cached) { try { const status = await getGitUpstreamStatusForUpstreamName( - (args) => gitExecFileAsync(args, gitStatusReadOptionsForWorktree(worktreePath, options)), + (args) => gitExecFileAsync(args, gitReadOptionsForWorktree(worktreePath, options)), cached.upstreamName ) return { status, probedSameNameOriginRef: false } @@ -127,7 +127,7 @@ async function probeEffectiveUpstreamStatus( ): Promise<{ status: GitUpstreamStatus; probedSameNameOriginRef: boolean }> { let probedSameNameOriginRef = false const snapshotRunner = createGitConfigSnapshotRunner((args) => - gitExecFileAsync(args, gitStatusReadOptionsForWorktree(worktreePath, options)) + gitExecFileAsync(args, gitReadOptionsForWorktree(worktreePath, options)) ) const status = await getEffectiveGitUpstreamStatus((args) => { if (args[0] === 'rev-parse' && args.includes(`refs/remotes/origin/${branchName}`)) { diff --git a/src/main/git/source-control/file-diff.ts b/src/main/git/source-control/file-diff.ts index a8fe78a6270..5595f0f6804 100644 --- a/src/main/git/source-control/file-diff.ts +++ b/src/main/git/source-control/file-diff.ts @@ -3,7 +3,8 @@ import type { GitDiffResult } from '../../../shared/git-diff-compare-types' import { stableInFlightKey } from '../../../shared/in-flight-promise-dedupe' import type { GitRuntimeOptions } from '../git-runtime-options' import { gitRuntimeOptionsKey } from './git-runtime-options-cache-key' -import { gitDiffReadDedupe } from './git-read-cache-invalidation' +import { gitDiffReadDedupe, settledDiffCache } from './git-read-cache-invalidation' +import { readWorktreeDiffStamp } from './worktree-diff-stamp' import { buildDiffResult } from './diff-result' import { readGitBlobAtIndexPath, @@ -33,27 +34,74 @@ export async function getDiff( compareAgainstHead = false, options: GitRuntimeOptions = {} ): Promise { - // Why: register the dedupe synchronously (before any await) so concurrent identical reads coalesce. - return gitDiffReadDedupe.run( - stableInFlightKey([ - 'diff', + const readKey = stableInFlightKey([ + 'diff', + worktreePath, + filePath, + staged, + compareAgainstHead, + ...gitRuntimeOptionsKey(options) + ]) + // Why: register the dedupe synchronously (before any await) so concurrent identical reads + // coalesce — including on the settled-cache lookup, which is itself I/O. + return gitDiffReadDedupe.run(readKey, () => + loadDiffThroughSettledCache( + readKey, worktreePath, filePath, staged, compareAgainstHead, - ...gitRuntimeOptionsKey(options) - ]), - () => loadDiff(worktreePath, filePath, staged, compareAgainstHead, options) + options + ) ) } +/** + * Serve a settled diff when the git state it was built from is provably + * unchanged, otherwise read and — only if the read proved everything it touched + * — record it under the stamp taken *before* the read. + * + * Stamping first is what makes staleness impossible: anything that moves during + * or after the read leaves the stored stamp behind, so the next lookup misses. + */ +async function loadDiffThroughSettledCache( + readKey: string, + worktreePath: string, + filePath: string, + staged: boolean, + compareAgainstHead: boolean, + options: GitRuntimeOptions +): Promise { + // Why before the stamp read: the stamp is itself several awaited stats, and a mutation that + // lands entirely inside that window would otherwise leave the fence covering only the git read. + const readGeneration = settledDiffCache.beginRead() + // A staged diff compares HEAD to the index, so the working tree is not one of its inputs. + const stamp = await readWorktreeDiffStamp(worktreePath, filePath, !staged) + const cached = settledDiffCache.get(readKey, stamp) + if (cached) { + return cached + } + const loaded = await loadDiff(worktreePath, filePath, staged, compareAgainstHead, options) + if (loaded.reusable) { + settledDiffCache.set(readKey, stamp, loaded.result, readGeneration) + } + return loaded.result +} + +/** + * `reusable` is false when the result cannot be proven to describe the stamped + * state: a submodule route, whose inputs live in another repo and are stamped by + * that repo's own read, or a blob read that failed rather than proving absence. + */ +type LoadedDiff = { result: GitDiffResult; reusable: boolean } + async function loadDiff( worktreePath: string, filePath: string, staged: boolean, compareAgainstHead: boolean, options: GitRuntimeOptions -): Promise { +): Promise { // Why: gitlink paths can't be read as blobs, so route submodule diffs explicitly (root → pointer, inner → recurse). const submodulePaths = await listSubmodulePaths(worktreePath, options) if (submodulePaths.length > 0) { @@ -63,13 +111,15 @@ async function loadDiff( const submoduleWorktreePath = resolveSubmoduleWorktreePath(worktreePath, matchedSubmodule) const normalizedFilePath = filePath.replace(/\\/g, '/').replace(/\/+$/, '') if (normalizedFilePath === matchedSubmodule) { - return buildSubmodulePointerDiff( - worktreePath, - matchedSubmodule, - staged, - compareAgainstHead, - options, - submoduleWorktreePath + return notReusable( + await buildSubmodulePointerDiff( + worktreePath, + matchedSubmodule, + staged, + compareAgainstHead, + options, + submoduleWorktreePath + ) ) } const innerPath = normalizedFilePath.slice(matchedSubmodule.length + 1) @@ -82,15 +132,20 @@ async function loadDiff( : await readWorkingSubmoduleHead(submoduleWorktreePath, options) // Why: a moved gitlink with a clean submodule worktree means the change is committed — diff the two commits. if (fromOid && toOid && fromOid !== toOid) { - return buildSubmoduleInnerCommitRangeDiff( - submoduleWorktreePath, - innerPath, - fromOid, - toOid, - options + return notReusable( + await buildSubmoduleInnerCommitRangeDiff( + submoduleWorktreePath, + innerPath, + fromOid, + toOid, + options + ) ) } - return getDiff(submoduleWorktreePath, innerPath, staged, compareAgainstHead, options) + // The inner read stamps and caches against the submodule's own repo state. + return notReusable( + await getDiff(submoduleWorktreePath, innerPath, staged, compareAgainstHead, options) + ) } } @@ -99,6 +154,7 @@ async function loadDiff( let originalIsBinary = false let modifiedIsBinary = false let modifiedDeleted = false + let readFailed = false try { if (staged) { @@ -113,6 +169,7 @@ async function loadDiff( modifiedContent = rightBlob.content modifiedIsBinary = rightBlob.isBinary modifiedDeleted = !rightBlob.exists + readFailed = leftBlob.failed === true || rightBlob.failed === true } else { // The left chain (index→HEAD) is sequential within itself, but the working // tree read is a plain fs read that does not depend on it. @@ -127,9 +184,11 @@ async function loadDiff( modifiedContent = workingTreeBlob.content modifiedIsBinary = workingTreeBlob.isBinary modifiedDeleted = !workingTreeBlob.exists + readFailed = leftBlob.failed === true || workingTreeBlob.failed === true } } catch { // Fallback + readFailed = true } const result = buildDiffResult( @@ -141,7 +200,11 @@ async function loadDiff( ) // Why: mark a proven deletion so previewers don't mistake a read failure's empty side for one. if (result.kind === 'binary' && modifiedDeleted) { - return { ...result, modifiedDeleted: true } + return { result: { ...result, modifiedDeleted: true }, reusable: !readFailed } } - return result + return { result, reusable: !readFailed } +} + +function notReusable(result: GitDiffResult): LoadedDiff { + return { result, reusable: false } } diff --git a/src/main/git/source-control/git-blob-read.ts b/src/main/git/source-control/git-blob-read.ts index e9e9182939c..f25a80eedbb 100644 --- a/src/main/git/source-control/git-blob-read.ts +++ b/src/main/git/source-control/git-blob-read.ts @@ -2,7 +2,7 @@ import { readFile, stat } from 'node:fs/promises' import * as path from 'node:path' import { isBinaryBuffer } from '../../../shared/binary-buffer' import type { GitRuntimeOptions } from '../git-runtime-options' -import { gitOptionsForWorktree } from '../git-runtime-options' +import { gitReadOptionsForWorktree } from '../git-runtime-options' import { gitExecFileAsyncBuffer } from '../runner' import { isMaxBufferOverflowError } from '../max-buffer-overflow' import { MAX_GIT_SHOW_BYTES } from './git-show-max-bytes' @@ -12,6 +12,21 @@ export type GitBlobReadResult = { content: string isBinary: boolean exists: boolean + /** + * The read did not complete: the blob is neither known-present nor proven + * absent. Callers must not persist a diff built on one, because the empty side + * it produces is indistinguishable from a genuinely new file. + */ + failed?: boolean +} + +/** + * Tell "Git ran and said the path is not there" apart from "the read never got + * an answer". Git exits 128 for a missing path in a tree or the index; a WSL + * relay that never reached Git exits with anything else, or with a spawn errno. + */ +function isProvenAbsentError(error: unknown): boolean { + return (error as { code?: unknown } | null)?.code === 128 } export async function readUnstagedLeftBlob( @@ -24,7 +39,9 @@ export async function readUnstagedLeftBlob( return indexBlob } - return readGitBlobAtOidPath(worktreePath, 'HEAD', filePath, options) + const headBlob = await readGitBlobAtOidPath(worktreePath, 'HEAD', filePath, options) + // Why: if the index read never got an answer, falling back to HEAD is a guess, not a proof. + return indexBlob.failed ? { ...headBlob, failed: true } : headBlob } export async function readGitBlobAtIndexPath( @@ -36,7 +53,7 @@ export async function readGitBlobAtIndexPath( const gitPath = filePath.replace(/\\/g, '/') try { const { stdout } = await gitExecFileAsyncBuffer(['show', `:${gitPath}`], { - ...gitOptionsForWorktree(worktreePath, options), + ...gitReadOptionsForWorktree(worktreePath, options), maxBuffer: MAX_GIT_SHOW_BYTES }) @@ -45,7 +62,7 @@ export async function readGitBlobAtIndexPath( if (isMaxBufferOverflowError(error)) { return { content: '', isBinary: true, exists: true } } - return { content: '', isBinary: false, exists: false } + return { content: '', isBinary: false, exists: false, failed: !isProvenAbsentError(error) } } } @@ -61,7 +78,7 @@ export async function readGitBlobAtOidPath( const { stdout } = await gitExecFileAsyncBuffer( ['show', '--end-of-options', `${oid}:${gitPath}`], { - ...gitOptionsForWorktree(worktreePath, options), + ...gitReadOptionsForWorktree(worktreePath, options), maxBuffer: MAX_GIT_SHOW_BYTES } ) @@ -71,7 +88,7 @@ export async function readGitBlobAtOidPath( if (isMaxBufferOverflowError(error)) { return { content: '', isBinary: true, exists: true } } - return { content: '', isBinary: false, exists: false } + return { content: '', isBinary: false, exists: false, failed: !isProvenAbsentError(error) } } } @@ -81,11 +98,8 @@ export async function readWorkingTreeFile(filePath: string): Promise { + try { + await access(target) + return true + } catch { + return false + } +} + export async function abortMerge( worktreePath: string, options: GitRuntimeOptions = {} diff --git a/src/main/git/source-control/git-read-cache-invalidation.ts b/src/main/git/source-control/git-read-cache-invalidation.ts index 7d4da0cc8ed..13abe4706a8 100644 --- a/src/main/git/source-control/git-read-cache-invalidation.ts +++ b/src/main/git/source-control/git-read-cache-invalidation.ts @@ -7,15 +7,20 @@ import { GitStatusReadLeaseOwner } from '../git-status-read-lease-owner' import { invalidateGitUpstreamStatusReads } from '../upstream' import { clearSubmodulePathsCache } from './submodule-paths' import { resolvedUpstreamNameCache } from './resolved-upstream-name-cache' +import { SettledDiffCache } from './settled-diff-cache' export const gitDiffReadDedupe = new InFlightPromiseDedupe() +/** Settled diff results, valid only while their stamped git state holds. */ +export const settledDiffCache = new SettledDiffCache() + export const statusReadLeaseOwner = new GitStatusReadLeaseOwner() // Why: clear every in-flight git read cache; clearing only some would let a post-mutation // getStatus() join a pre-mutation read and publish it as current. export function invalidateGitReadCaches(): void { gitDiffReadDedupe.clear() + settledDiffCache.clear() statusReadLeaseOwner.invalidate() invalidateGitBranchLineTotalInFlight() invalidateGitUpstreamStatusReads() diff --git a/src/main/git/source-control/settled-diff-cache.ts b/src/main/git/source-control/settled-diff-cache.ts new file mode 100644 index 00000000000..5ffd7f03e76 --- /dev/null +++ b/src/main/git/source-control/settled-diff-cache.ts @@ -0,0 +1,155 @@ +import { BoundedMap } from '../../../shared/bounded-map' +import type { GitDiffResult } from '../../../shared/git-diff-compare-types' +import { + canProveUnchangedByStamp, + isDiffStampClockSkewed, + type WorktreeDiffStamp +} from './worktree-diff-stamp' + +/** + * Diff results that survive their read, guarded by a stamp of the git state they + * were built from. + * + * Why this is not a TTL: nothing here expires on a clock, so no window exists in + * which a stale diff can be served. An entry is returned only when a freshly + * taken stamp equals the one captured *before* the read that produced it, which + * means no input moved from then until now. Everything else — an unprovable + * stamp, a write too recent for its mtime to be conclusive, a read that failed + * partway, a mutation that ran while the read was in flight — declines to cache + * rather than risk it. + */ + +// A diff result carries whole file contents on both sides, so an entry count alone bounds +// nothing: the real budget is characters, and one entry can legitimately be huge. +export const MAX_SETTLED_DIFF_CACHE_ENTRIES = 32 +export const MAX_SETTLED_DIFF_CACHE_RESULT_CHARACTERS = 1_000_000 +export const MAX_SETTLED_DIFF_CACHE_TOTAL_CHARACTERS = 8_000_000 + +export type SettledDiffCacheStats = { + hits: number + /** Stamp was taken but no entry matched it. */ + misses: number + /** No stamp could be taken, so the read could never be cached. */ + unprovable: number + stores: number + /** Store declined because a write was too recent for its mtime to be conclusive. */ + racyWrites: number + /** + * Subset of `racyWrites` where a component's mtime was in this host's future, so + * the clocks disagree and the refusal will persist until they converge. A nonzero + * count here means the cache is off for a reason no amount of idling will fix. + */ + clockSkewedWrites: number + /** Store declined because a mutation invalidated the cache while the read ran. */ + invalidatedDuringRead: number + entries: number + retainedCharacters: number +} + +type CacheEntry = { stamp: string; result: GitDiffResult; characters: number } + +export class SettledDiffCache { + private readonly entries = new BoundedMap({ + maxEntries: MAX_SETTLED_DIFF_CACHE_ENTRIES, + maxBytes: MAX_SETTLED_DIFF_CACHE_TOTAL_CHARACTERS, + maxEntryBytes: MAX_SETTLED_DIFF_CACHE_RESULT_CHARACTERS, + sizeOf: (entry) => entry.characters + }) + private generation = 0 + private hits = 0 + private misses = 0 + private unprovable = 0 + private stores = 0 + private racyWrites = 0 + private clockSkewedWrites = 0 + private invalidatedDuringRead = 0 + + /** + * Take before starting a read; hand back to `set`. A mutation that lands while + * the read is in flight bumps the generation, and the store is refused — the + * result describes pre-mutation state and must not outlive it. + */ + beginRead(): number { + return this.generation + } + + get(key: string, stamp: WorktreeDiffStamp | null): GitDiffResult | undefined { + if (!stamp) { + this.unprovable += 1 + return undefined + } + // BoundedMap.get() already refreshes the LRU position. + const entry = this.entries.get(key) + if (!entry || entry.stamp !== stamp.value) { + this.misses += 1 + return undefined + } + this.hits += 1 + return entry.result + } + + set( + key: string, + stamp: WorktreeDiffStamp | null, + result: GitDiffResult, + readGeneration: number + ): void { + if (!stamp) { + return + } + if (readGeneration !== this.generation) { + this.invalidatedDuringRead += 1 + return + } + if (!canProveUnchangedByStamp(stamp)) { + this.racyWrites += 1 + if (isDiffStampClockSkewed(stamp)) { + this.clockSkewedWrites += 1 + } + return + } + const characters = resultCharacterCount(result) + if (this.entries.set(key, { stamp: stamp.value, result, characters })) { + this.stores += 1 + } + } + + clear(): void { + this.entries.clear() + // Why: bump so a read that started pre-mutation can't repopulate the invalidated cache. + this.generation += 1 + } + + /** + * Why exposed: a stamp that can never match — an inode or mtime the filesystem + * reports unstably — looks exactly like having no cache at all. These counters + * are what tells a miss storm apart from a cold start. + */ + stats(): SettledDiffCacheStats { + return { + hits: this.hits, + misses: this.misses, + unprovable: this.unprovable, + stores: this.stores, + racyWrites: this.racyWrites, + clockSkewedWrites: this.clockSkewedWrites, + invalidatedDuringRead: this.invalidatedDuringRead, + entries: this.entries.size, + retainedCharacters: this.entries.retainedBytes + } + } + + resetStatsForTests(): void { + this.hits = 0 + this.misses = 0 + this.unprovable = 0 + this.stores = 0 + this.racyWrites = 0 + this.clockSkewedWrites = 0 + this.invalidatedDuringRead = 0 + } +} + +function resultCharacterCount(result: GitDiffResult): number { + return result.originalContent.length + result.modifiedContent.length +} diff --git a/src/main/git/source-control/status-branch-line-total-input.ts b/src/main/git/source-control/status-branch-line-total-input.ts index d2ac795bb55..5603d55bd4c 100644 --- a/src/main/git/source-control/status-branch-line-total-input.ts +++ b/src/main/git/source-control/status-branch-line-total-input.ts @@ -6,7 +6,7 @@ import { type GitBranchLineTotal } from '../../../shared/git-branch-line-total' import type { GitRuntimeOptions } from '../git-runtime-options' -import { gitStatusReadOptionsForWorktree } from '../git-runtime-options' +import { gitReadOptionsForWorktree } from '../git-runtime-options' import { gitExecFileAsync, gitOptionalLocksDisabledEnv } from '../runner' import type { GetStatusOptions } from './get-status-options' @@ -36,7 +36,7 @@ export function createBranchLineTotalInput( .map((entry) => entry.path), runDiffNumstat: (args, signal) => gitExecFileAsync(args, { - ...gitStatusReadOptionsForWorktree(worktreePath, options), + ...gitReadOptionsForWorktree(worktreePath, options), // Why: after the spread, so the shared lease signal wins over this caller's own. signal, env: gitOptionalLocksDisabledEnv(), diff --git a/src/main/git/source-control/status-line-stats.ts b/src/main/git/source-control/status-line-stats.ts index 660c9ea018e..32e229cdbbd 100644 --- a/src/main/git/source-control/status-line-stats.ts +++ b/src/main/git/source-control/status-line-stats.ts @@ -6,7 +6,7 @@ import { type GitLineStats } from '../../../shared/git-uncommitted-line-stats' import type { GitRuntimeOptions } from '../git-runtime-options' -import { gitStatusReadOptionsForWorktree } from '../git-runtime-options' +import { gitReadOptionsForWorktree } from '../git-runtime-options' import { gitExecFileAsync, gitOptionalLocksDisabledEnv } from '../runner' async function runNumstat( @@ -26,7 +26,7 @@ async function runNumstat( '-M' ], { - ...gitStatusReadOptionsForWorktree(worktreePath, options), + ...gitReadOptionsForWorktree(worktreePath, options), env: gitOptionalLocksDisabledEnv() } ) diff --git a/src/main/git/source-control/status-read.ts b/src/main/git/source-control/status-read.ts index 898b46e1ba9..7893cabde15 100644 --- a/src/main/git/source-control/status-read.ts +++ b/src/main/git/source-control/status-read.ts @@ -15,7 +15,7 @@ import { import { gitOptionalLocksDisabledEnv, gitStreamStdout } from '../runner' import { findExistingWorktreeSymlinkPaths } from '../worktree-symlink-detection' import type { GetStatusOptions } from './get-status-options' -import { gitDiffReadDedupe, statusReadLeaseOwner } from './git-read-cache-invalidation' +import { statusReadLeaseOwner } from './git-read-cache-invalidation' import { detectConflictOperation } from './git-conflict-operation' import { parseUnmergedEntry } from './status-conflict-entries' import { getEffectiveUpstreamStatusCacheKey } from './effective-upstream-status-cache' @@ -37,8 +37,10 @@ export async function getStatus( worktreePath: string, options: GetStatusOptions = {} ): Promise { - gitDiffReadDedupe.clear() - // Why: dedupe only concurrent identical reads; after settle, callers must run a fresh read. + // Why nothing is cleared here: a status poll is a read. Dropping the in-flight diff entry + // mid-read only made a concurrent identical request start duplicate git work, and the + // settled diff cache is keyed on stamped git state, which a read cannot change anyway. + // Mutations invalidate both, through invalidateGitReadCaches. const cacheKey = getStatusReadKey(worktreePath, options) return statusReadLeaseOwner.lease(cacheKey, options.signal, (sharedSignal) => runGetStatus(worktreePath, { ...options, signal: sharedSignal }) diff --git a/src/main/git/source-control/worktree-diff-stamp.ts b/src/main/git/source-control/worktree-diff-stamp.ts new file mode 100644 index 00000000000..f24130d274b --- /dev/null +++ b/src/main/git/source-control/worktree-diff-stamp.ts @@ -0,0 +1,226 @@ +import { readFile, stat } from 'node:fs/promises' +import * as path from 'node:path' +import { resolveGitDir } from './resolve-git-dir' + +/** + * Subprocess-free proof of every input one working-tree file diff is built from: + * the HEAD tree, the index, `.gitmodules` (which decides submodule routing) and + * the working-tree file itself. + * + * Two equal stamps mean `git show HEAD:`, `git show :` and the + * working-tree bytes would still return what they returned when the stamp was + * taken, so a caller may reuse a settled diff instead of respawning Git — on a + * WSL/UNC worktree that trades two `wsl.exe` spawns for a handful of 9p stats. + * + * `null` means "cannot prove unchanged" — not a repo, an unreadable layout, or a + * filesystem that reports no usable mtime — and callers must not cache. + */ +export type WorktreeDiffStamp = { + /** Opaque; compare for equality only. */ + value: string + /** Newest mtime any component reported, or -Infinity when every one was absent. */ + newestMtimeMs: number + capturedAtMs: number +} + +/** + * How long after a component's mtime the stamp has to be taken before a later + * write is guaranteed to move that mtime. + * + * Why 2s: FAT/exFAT truncate mtime to a 2s bucket, the coarsest granularity a + * repo can realistically sit on. Below the margin a second write inside the same + * bucket would leave the stamp identical, so the read stays uncached instead. + */ +export const DIFF_STAMP_RACY_WRITE_MARGIN_MS = 2_000 + +const MISSING = '-' +const ABSENT: StampComponent = { text: MISSING, mtimeMs: Number.NEGATIVE_INFINITY } + +type StampComponent = { text: string; mtimeMs: number } + +/** + * A write that lands in the same timestamp bucket as the stamp is invisible, so + * only a stamp taken a full bucket after its newest component can be trusted. + * + * The margin compares two clocks: `capturedAtMs` is this host's, while the mtimes + * come from whatever wrote the files. On a `\\wsl.localhost` worktree the guest + * sets them, and a guest running ahead pushes every recently-touched file past the + * margin — silently, and for as long as the skew lasts. `isDiffStampClockSkewed` + * separates that from an honestly-too-fresh file so the caller can count it. + */ +export function canProveUnchangedByStamp(stamp: WorktreeDiffStamp): boolean { + return stamp.capturedAtMs - stamp.newestMtimeMs >= DIFF_STAMP_RACY_WRITE_MARGIN_MS +} + +/** + * True when a component's mtime is in this host's future, which no local write can + * produce — so the margin above is measuring skew, not freshness, and will keep + * refusing to store until the clocks converge. + */ +export function isDiffStampClockSkewed(stamp: WorktreeDiffStamp): boolean { + return Number.isFinite(stamp.newestMtimeMs) && stamp.newestMtimeMs > stamp.capturedAtMs +} + +/** + * Stamp the inputs of one file diff. Pass `includeWorkingTree: false` for a + * staged diff, which compares HEAD to the index and never reads the working tree. + */ +export async function readWorktreeDiffStamp( + worktreePath: string, + filePath: string, + includeWorkingTree: boolean +): Promise { + const capturedAtMs = Date.now() + try { + const gitDir = await resolveGitDir(worktreePath) + const [head, index, gitmodules, workingTree] = await Promise.all([ + readHeadComponent(gitDir), + // Over-invalidates on purpose: git run outside Orca (a terminal `git status`/`git add`) + // can refresh a stat-dirty index and rewrite this file without changing a single blob, + // which costs one re-read. That is the safe direction — do not "fix" it by dropping the + // index from the stamp, because `git add` then becomes invisible and the cache serves a + // pre-staging diff. + readFileStampComponent(path.join(gitDir, 'index')), + readFileStampComponent(path.join(worktreePath, '.gitmodules')), + includeWorkingTree + ? readWorkingTreeComponent(path.join(worktreePath, filePath)) + : Promise.resolve(ABSENT) + ]) + if (!head) { + return null + } + const components = [head, index, gitmodules, workingTree] + return { + // Why JSON: a path or a ref can contain any separator character, and an ambiguous + // join is a stamp collision — two different states that compare equal. + value: JSON.stringify([ + worktreePath, + filePath, + ...components.map((component) => component.text) + ]), + newestMtimeMs: Math.max(...components.map((component) => component.mtimeMs)), + capturedAtMs + } + } catch { + return null + } +} + +/** + * Identify the commit HEAD resolves to. Reading the tip is what makes a plain + * commit visible: it rewrites `refs/heads/` and leaves HEAD untouched. + * + * A loose tip is recorded by content, so it needs no mtime margin. Only when no + * loose ref exists at all — packed refs, the reftable backend, or an unborn + * branch — does this fall back to the coarser stamps of the files a packed tip + * moves, and the recorded "no loose ref" marker still catches the ref appearing. + */ +async function readHeadComponent(gitDir: string): Promise { + const [head, commonDirEntry] = await Promise.all([ + readTrimmedFile(path.join(gitDir, 'HEAD')), + readTrimmedFile(path.join(gitDir, 'commondir')) + ]) + if (!head) { + return null + } + const commonDir = commonDirEntry ? path.resolve(gitDir, commonDirEntry) : gitDir + const refName = head.match(/^ref:\s*(.+?)\s*$/)?.[1] + if (!refName || !isSafeRefName(refName)) { + // Detached HEAD already holds the object id; an unrecognized HEAD is covered by its own text. + return { text: head, mtimeMs: Number.NEGATIVE_INFINITY } + } + // Per-worktree refs (`refs/bisect`, `refs/worktree`) live beside the checkout; branches are shared. + const [perWorktreeTip, sharedTip] = await Promise.all([ + readTrimmedFile(path.join(gitDir, refName)), + commonDir === gitDir ? Promise.resolve(null) : readTrimmedFile(path.join(commonDir, refName)) + ]) + const looseTip = perWorktreeTip ?? sharedTip + if (looseTip) { + // A loose ref shadows any packed entry, so its bytes settle the tip on their own. + return { + text: JSON.stringify([head, 'loose', perWorktreeTip ? 'worktree' : 'common', looseTip]), + mtimeMs: Number.NEGATIVE_INFINITY + } + } + const [packedRefs, reftable] = await Promise.all([ + readFileStampComponent(path.join(commonDir, 'packed-refs')), + readFileStampComponent(path.join(commonDir, 'reftable')) + ]) + return { + text: JSON.stringify([head, 'packed', packedRefs.text, reftable.text]), + mtimeMs: Math.max(packedRefs.mtimeMs, reftable.mtimeMs) + } +} + +/** Keep a hand-edited HEAD from steering the stamp outside the repo's ref store. */ +function isSafeRefName(refName: string): boolean { + const segments = refName.split(/[\\/]/) + return ( + segments[0] === 'refs' && + segments.length > 1 && + segments.every((segment) => segment.length > 0 && segment !== '.' && segment !== '..') && + !path.isAbsolute(refName) + ) +} + +async function readTrimmedFile(filePath: string): Promise { + try { + const trimmed = (await readFile(filePath, 'utf-8')).trim() + return trimmed.length > 0 ? trimmed : null + } catch (error) { + if (isMissingEntryError(error)) { + return null + } + throw error + } +} + +async function readFileStampComponent(filePath: string): Promise { + try { + const stats = await stat(filePath) + return { text: `${requireMtimeMs(stats.mtimeMs)}:${stats.size}`, mtimeMs: stats.mtimeMs } + } catch (error) { + if (isMissingEntryError(error)) { + return ABSENT + } + throw error + } +} + +async function readWorkingTreeComponent(filePath: string): Promise { + try { + const stats = await stat(filePath) + // Why ino is optional: an atomic-rename save can keep both the size and the mtime bucket, + // so a changed inode is extra proof — but Windows reports 0 for it on the network + // redirector behind `\\wsl.localhost`, and requiring an unstable 0 to match would make + // the stamp never hit on exactly the host this cache exists for. Fold it in only when + // the filesystem gives a real one. + const inode = isUsableInode(stats.ino) ? String(stats.ino) : MISSING + return { + text: `${requireMtimeMs(stats.mtimeMs)}:${stats.size}:${inode}`, + mtimeMs: stats.mtimeMs + } + } catch (error) { + if (isMissingEntryError(error)) { + return ABSENT + } + throw error + } +} + +function isUsableInode(ino: unknown): boolean { + return typeof ino === 'number' && Number.isFinite(ino) && ino !== 0 +} + +/** A filesystem (or a stub) that reports no usable mtime cannot prove anything unchanged. */ +function requireMtimeMs(mtimeMs: unknown): number { + if (typeof mtimeMs !== 'number' || !Number.isFinite(mtimeMs)) { + throw new Error('stat reported no usable mtime') + } + return mtimeMs +} + +function isMissingEntryError(error: unknown): boolean { + const code = (error as NodeJS.ErrnoException | null)?.code + return code === 'ENOENT' || code === 'ENOTDIR' +} diff --git a/src/main/git/status-conflict-operations.test.ts b/src/main/git/status-conflict-operations.test.ts index e99c43532f4..5b61932578f 100644 --- a/src/main/git/status-conflict-operations.test.ts +++ b/src/main/git/status-conflict-operations.test.ts @@ -15,7 +15,7 @@ const { readFileMock, statMock, rmMock, - existsSyncMock + accessMock } = vi.hoisted(() => ({ gitExecFileAsyncMock: vi.fn(), gitExecFileAsyncBufferMock: vi.fn(), @@ -25,7 +25,7 @@ const { readFileMock: vi.fn(), statMock: vi.fn(), rmMock: vi.fn(), - existsSyncMock: vi.fn() + accessMock: vi.fn() })) vi.mock('./runner', () => @@ -37,13 +37,16 @@ vi.mock('./runner', () => ) vi.mock('fs/promises', () => - createFsPromisesModuleMock({ lstatMock, realpathMock, readFileMock, statMock, rmMock }) + createFsPromisesModuleMock({ + lstatMock, + realpathMock, + readFileMock, + statMock, + rmMock, + accessMock + }) ) -vi.mock('fs', () => ({ - existsSync: existsSyncMock -})) - vi.mock('../../shared/node-bounded-file-reader', async (importOriginal) => createBoundedFileReaderModuleMock(await importOriginal(), { readFileMock, @@ -83,32 +86,65 @@ describe('abortRebase', () => { describe('detectConflictOperation', () => { beforeEach(() => { readFileMock.mockReset() - existsSyncMock.mockReset() + accessMock.mockReset() }) it('ignores a stale REBASE_HEAD when no rebase directory exists', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockImplementation((target: string) => { - if (target.endsWith('MERGE_HEAD')) { - return false - } - if (target.endsWith('CHERRY_PICK_HEAD')) { - return false - } - if (target.endsWith('rebase-merge')) { - return false - } - if (target.endsWith('rebase-apply')) { - return false - } + accessMock.mockImplementation(async (target: string) => { + // Only REBASE_HEAD is present: the marker git leaves behind after a rebase finishes. if (target.endsWith('REBASE_HEAD')) { - return true + return undefined } - return false + throw Object.assign(new Error(`ENOENT: ${target}`), { code: 'ENOENT' }) }) const result = await detectConflictOperation('/repo') expect(result).toBe('unknown') }) + + it.each([ + ['MERGE_HEAD', 'merge'], + ['rebase-merge', 'rebase'], + ['rebase-apply', 'rebase'], + ['CHERRY_PICK_HEAD', 'cherry-pick'] + ])('reports %s as %s', async (marker, expected) => { + readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') + accessMock.mockImplementation(async (target: string) => { + if (target.endsWith(marker)) { + return undefined + } + throw Object.assign(new Error(`ENOENT: ${target}`), { code: 'ENOENT' }) + }) + + await expect(detectConflictOperation('/repo')).resolves.toBe(expected) + }) + + // The four markers are independent, so serializing them costs four round trips + // on a UNC git dir for something one wave answers. + it('probes every marker concurrently', async () => { + readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') + let concurrent = 0 + let peakConcurrent = 0 + accessMock.mockImplementation(async () => { + concurrent += 1 + peakConcurrent = Math.max(peakConcurrent, concurrent) + await Promise.resolve() + concurrent -= 1 + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + + await detectConflictOperation('/repo') + + expect(accessMock).toHaveBeenCalledTimes(4) + expect(peakConcurrent).toBe(4) + }) + + it('reads as unknown when the git dir cannot be reached at all', async () => { + readFileMock.mockRejectedValue(Object.assign(new Error('EIO'), { code: 'EIO' })) + accessMock.mockRejectedValue(Object.assign(new Error('EIO'), { code: 'EIO' })) + + await expect(detectConflictOperation('/repo')).resolves.toBe('unknown') + }) }) diff --git a/src/main/git/status-diff-settled-cache.test.ts b/src/main/git/status-diff-settled-cache.test.ts new file mode 100644 index 00000000000..1e5cf2231b1 --- /dev/null +++ b/src/main/git/status-diff-settled-cache.test.ts @@ -0,0 +1,331 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type * as BoundedFileReader from '../../shared/node-bounded-file-reader' +import { createBoundedFileReaderModuleMock, createGitRunnerModuleMock } from './status-test-harness' + +const { + gitExecFileAsyncMock, + gitExecFileAsyncBufferMock, + gitStreamOptionsMock, + lstatMock, + realpathMock, + rmMock, + existsSyncMock +} = vi.hoisted(() => ({ + gitExecFileAsyncMock: vi.fn(), + gitExecFileAsyncBufferMock: vi.fn(), + gitStreamOptionsMock: vi.fn(), + lstatMock: vi.fn(), + realpathMock: vi.fn(), + rmMock: vi.fn(), + existsSyncMock: vi.fn() +})) + +/** + * A tiny in-memory filesystem, because every assertion here is about what the + * cache does when one specific path's mtime or bytes move. Sequenced + * `mockResolvedValueOnce` stacks cannot express that: the stamp and the diff + * read touch overlapping paths in an order the test should not have to know. + */ +type FakeFile = { content: Buffer; mtimeMs: number; ino: number } + +const { files } = vi.hoisted(() => ({ files: new Map() })) + +const { readFileMock, statMock, accessMock } = vi.hoisted(() => { + const missing = (target: string): NodeJS.ErrnoException => + Object.assign(new Error(`ENOENT: ${target}`), { code: 'ENOENT' }) + return { + readFileMock: vi.fn(async (target: string, encoding?: BufferEncoding) => { + const file = files.get(target) + if (!file) { + throw missing(target) + } + return encoding ? file.content.toString(encoding) : file.content + }), + statMock: vi.fn(async (target: string) => { + const file = files.get(target) + if (!file) { + throw missing(target) + } + return { + isFile: () => true, + size: file.content.byteLength, + mtimeMs: file.mtimeMs, + ino: file.ino + } + }), + accessMock: vi.fn(async (target: string) => { + if (!files.has(target)) { + throw missing(target) + } + }) + } +}) + +vi.mock('./runner', () => + createGitRunnerModuleMock({ + gitExecFileAsyncMock, + gitExecFileAsyncBufferMock, + gitStreamOptionsMock + }) +) + +vi.mock('fs/promises', () => ({ + lstat: lstatMock, + realpath: realpathMock, + readFile: readFileMock, + stat: statMock, + rm: rmMock, + access: accessMock +})) + +vi.mock('fs', () => ({ existsSync: existsSyncMock })) + +vi.mock('../../shared/node-bounded-file-reader', async (importOriginal) => + createBoundedFileReaderModuleMock(await importOriginal(), { + readFileMock, + statMock + }) +) + +import { getDiff, getStatus, invalidateGitReadCaches, stageFile } from './status' +import { settledDiffCache } from './source-control/git-read-cache-invalidation' + +const REPO = '/repo' +const FILE = 'src/file.ts' +const WORKING_TREE_PATH = `${REPO}/${FILE}` +const HEAD_PATH = `${REPO}/.git/HEAD` +const REF_PATH = `${REPO}/.git/refs/heads/main` +const INDEX_PATH = `${REPO}/.git/index` +const GITMODULES_PATH = `${REPO}/.gitmodules` + +// Old enough that a further write is guaranteed to move the mtime, which is what +// lets the cache store at all. +const SETTLED_MTIME_MS = Date.now() - 60_000 +let nextInode = 100 + +function writeFile(target: string, content: string, mtimeMs = SETTLED_MTIME_MS): void { + files.set(target, { content: Buffer.from(content), mtimeMs, ino: (nextInode += 1) }) +} + +function blobReadCount(): number { + return gitExecFileAsyncBufferMock.mock.calls.length +} + +function seedRepo(): void { + files.clear() + // No `.git` file entry: `.git` is a directory, so reading it as a pointer misses. + writeFile(HEAD_PATH, 'ref: refs/heads/main\n') + writeFile(REF_PATH, `${'a'.repeat(40)}\n`) + writeFile(INDEX_PATH, 'index-bytes') + writeFile(WORKING_TREE_PATH, 'working-tree-content') +} + +describe('settled diff cache', () => { + beforeEach(() => { + gitExecFileAsyncMock.mockReset() + gitExecFileAsyncBufferMock.mockReset() + gitStreamOptionsMock.mockReset() + readFileMock.mockClear() + statMock.mockClear() + accessMock.mockClear() + existsSyncMock.mockReset() + invalidateGitReadCaches() + settledDiffCache.resetStatsForTests() + seedRepo() + // `.gitmodules` is absent, so submodule routing resolves to "no submodules". + gitExecFileAsyncMock.mockResolvedValue({ stdout: '', stderr: '' }) + gitExecFileAsyncBufferMock.mockResolvedValue({ stdout: Buffer.from('index-content\n') }) + }) + + it('serves the second read of an unchanged file without respawning git', async () => { + const first = await getDiff(REPO, FILE, false) + const spawnsAfterFirst = blobReadCount() + + const second = await getDiff(REPO, FILE, false) + + expect(spawnsAfterFirst).toBeGreaterThan(0) + expect(blobReadCount()).toBe(spawnsAfterFirst) + expect(second).toEqual(first) + expect(settledDiffCache.stats().hits).toBe(1) + }) + + // The four invalidation axes, one per diff input. Each proves the stale result is + // never served, which matters more than any of the hits above. + it.each([ + [ + 'the working tree file is edited', + () => writeFile(WORKING_TREE_PATH, 'edited-in-another-editor') + ], + ['the index is rewritten by git add', () => writeFile(INDEX_PATH, 'index-bytes-after-add')], + ['HEAD moves to a new commit', () => writeFile(REF_PATH, `${'b'.repeat(40)}\n`)], + ['HEAD is detached onto another commit', () => writeFile(HEAD_PATH, `${'c'.repeat(40)}\n`)], + ['.gitmodules appears', () => writeFile(GITMODULES_PATH, '[submodule "vendor"]\n')] + ])('re-reads after %s', async (_name, mutate) => { + await getDiff(REPO, FILE, false) + const spawnsAfterFirst = blobReadCount() + + mutate() + gitExecFileAsyncBufferMock.mockResolvedValue({ stdout: Buffer.from('fresh-index-content\n') }) + const second = await getDiff(REPO, FILE, false) + + expect(blobReadCount()).toBeGreaterThan(spawnsAfterFirst) + expect(second).toMatchObject({ originalContent: 'fresh-index-content\n' }) + }) + + it('re-reads after a mutation runs through the shared invalidation point', async () => { + await getDiff(REPO, FILE, false) + const spawnsAfterFirst = blobReadCount() + + await stageFile(REPO, FILE) + await getDiff(REPO, FILE, false) + + expect(blobReadCount()).toBeGreaterThan(spawnsAfterFirst) + }) + + // The dangerous ordering: the read observed pre-mutation state, so storing its + // result after the mutation would pin a diff that was already wrong. + it('refuses to store a result for a read that a mutation overtook', async () => { + let releaseBlob = (): void => {} + const blocked = new Promise<{ stdout: Buffer }>((resolve) => { + releaseBlob = () => resolve({ stdout: Buffer.from('pre-mutation\n') }) + }) + gitExecFileAsyncBufferMock.mockReturnValue(blocked) + + const inFlight = getDiff(REPO, FILE, false) + await vi.waitFor(() => expect(blobReadCount()).toBeGreaterThan(0)) + invalidateGitReadCaches() + releaseBlob() + await inFlight + + expect(settledDiffCache.stats().invalidatedDuringRead).toBe(1) + expect(settledDiffCache.stats().entries).toBe(0) + + const spawnsAfterFirst = blobReadCount() + gitExecFileAsyncBufferMock.mockResolvedValue({ stdout: Buffer.from('post-mutation\n') }) + const second = await getDiff(REPO, FILE, false) + + expect(blobReadCount()).toBeGreaterThan(spawnsAfterFirst) + expect(second).toMatchObject({ originalContent: 'post-mutation\n' }) + }) + + // The stamp is itself several awaited stats, so a mutation can begin and end entirely + // inside it — leaving a stamp torn across the mutation that no later stamp can match. + it('refuses to store a result for a mutation that landed inside the stamp read', async () => { + const baseStat = statMock.getMockImplementation() + if (!baseStat) { + throw new Error('the fake filesystem lost its stat implementation') + } + let invalidated = false + statMock.mockImplementation(async (target: string) => { + if (!invalidated && target === INDEX_PATH) { + invalidated = true + invalidateGitReadCaches() + } + return baseStat(target) + }) + try { + await getDiff(REPO, FILE, false) + } finally { + statMock.mockImplementation(baseStat) + } + + expect(invalidated).toBe(true) + expect(settledDiffCache.stats().invalidatedDuringRead).toBe(1) + expect(settledDiffCache.stats().entries).toBe(0) + }) + + // A write inside the mtime granularity window could be overwritten again without + // moving the timestamp, so that read is not allowed to become a cache entry. + it('refuses to store a diff of a file written moments ago', async () => { + writeFile(WORKING_TREE_PATH, 'just-saved', Date.now()) + + await getDiff(REPO, FILE, false) + const spawnsAfterFirst = blobReadCount() + await getDiff(REPO, FILE, false) + + expect(blobReadCount()).toBeGreaterThan(spawnsAfterFirst) + expect(settledDiffCache.stats().racyWrites).toBeGreaterThan(0) + expect(settledDiffCache.stats().entries).toBe(0) + }) + + // A folder workspace, or any path that is not a git checkout, cannot be stamped. + it('never caches when the repo layout cannot be stamped', async () => { + files.delete(HEAD_PATH) + + await getDiff(REPO, FILE, false) + const spawnsAfterFirst = blobReadCount() + await getDiff(REPO, FILE, false) + + expect(blobReadCount()).toBeGreaterThan(spawnsAfterFirst) + expect(settledDiffCache.stats().unprovable).toBeGreaterThan(0) + expect(settledDiffCache.stats().entries).toBe(0) + }) + + // A WSL relay that never reached git returns the same empty left side as a new + // file does, so persisting it would pin a wrong diff until something else moved. + it('refuses to store a diff whose blob read failed rather than proved absence', async () => { + gitExecFileAsyncBufferMock.mockRejectedValue( + Object.assign(new Error('wsl.exe failed'), { code: 1 }) + ) + + await getDiff(REPO, FILE, false) + const spawnsAfterFirst = blobReadCount() + await getDiff(REPO, FILE, false) + + expect(blobReadCount()).toBeGreaterThan(spawnsAfterFirst) + expect(settledDiffCache.stats().entries).toBe(0) + }) + + it('caches a new file whose absence from the index git actually reported', async () => { + gitExecFileAsyncBufferMock.mockRejectedValue( + Object.assign(new Error("fatal: path 'src/file.ts' does not exist"), { code: 128 }) + ) + + await getDiff(REPO, FILE, false) + const spawnsAfterFirst = blobReadCount() + await getDiff(REPO, FILE, false) + + expect(blobReadCount()).toBe(spawnsAfterFirst) + expect(settledDiffCache.stats().hits).toBe(1) + }) + + it('keeps staged and unstaged diffs of one file in separate entries', async () => { + await getDiff(REPO, FILE, false) + const spawnsAfterUnstaged = blobReadCount() + + await getDiff(REPO, FILE, true) + + expect(blobReadCount()).toBeGreaterThan(spawnsAfterUnstaged) + }) + + // A staged diff compares HEAD to the index, so a working-tree edit must not evict it. + it('keeps a staged diff across a working-tree edit', async () => { + await getDiff(REPO, FILE, true) + const spawnsAfterFirst = blobReadCount() + + writeFile(WORKING_TREE_PATH, 'edited-after-staging') + await getDiff(REPO, FILE, true) + + expect(blobReadCount()).toBe(spawnsAfterFirst) + }) + + it('does not let a status poll drop the in-flight diff read', async () => { + let releaseBlob = (): void => {} + const blocked = new Promise<{ stdout: Buffer }>((resolve) => { + releaseBlob = () => resolve({ stdout: Buffer.from('index-content\n') }) + }) + gitExecFileAsyncBufferMock.mockReturnValue(blocked) + + const first = getDiff(REPO, FILE, false) + await vi.waitFor(() => expect(blobReadCount()).toBeGreaterThan(0)) + const spawnsBeforePoll = blobReadCount() + + await getStatus(REPO) + const second = getDiff(REPO, FILE, false) + releaseBlob() + await Promise.all([first, second]) + + // Why exactly equal: the second read must join the first, not start its own spawns. + expect(blobReadCount()).toBe(spawnsBeforePoll) + }) +}) diff --git a/src/main/git/status-diff.test.ts b/src/main/git/status-diff.test.ts index df59b1b9a3f..d66c64e7a89 100644 --- a/src/main/git/status-diff.test.ts +++ b/src/main/git/status-diff.test.ts @@ -102,7 +102,8 @@ describe('getDiff', () => { expect(gitExecFileAsyncBufferMock).toHaveBeenCalledWith(['show', ':src/file.ts'], { cwd: '/repo', - maxBuffer: 10 * 1024 * 1024 + maxBuffer: 10 * 1024 * 1024, + preferWslDirectGit: true }) expect(readFileMock).toHaveBeenCalledWith(path.join('/repo', 'src/file.ts')) expect(result).toEqual({ @@ -122,7 +123,8 @@ describe('getDiff', () => { expect(gitExecFileAsyncBufferMock).toHaveBeenCalledWith(['show', ':src/file.ts'], { cwd: '/repo', - maxBuffer: 10 * 1024 * 1024 + maxBuffer: 10 * 1024 * 1024, + preferWslDirectGit: true }) }) @@ -139,7 +141,8 @@ describe('getDiff', () => { ['show', '--end-of-options', 'HEAD:src/file.ts'], { cwd: '/repo', - maxBuffer: 10 * 1024 * 1024 + maxBuffer: 10 * 1024 * 1024, + preferWslDirectGit: true } ) expect(result.originalContent).toBe('head-content\n') @@ -158,15 +161,19 @@ describe('getDiff', () => { }) it('does not read oversized working-tree files into memory', async () => { + const workingTreePath = path.join('/repo', 'dist/large.log') gitExecFileAsyncBufferMock.mockResolvedValueOnce({ stdout: Buffer.from('index-content\n') }) - statMock.mockResolvedValueOnce({ - isFile: () => true, - size: 10 * 1024 * 1024 + 1 - }) + // Why by path: the diff also stats git-dir entries to stamp its inputs, so a + // one-shot queue would hand the oversized size to whichever stat ran first. + statMock.mockImplementation(async (target: string) => + target === workingTreePath + ? { isFile: () => true, size: 10 * 1024 * 1024 + 1 } + : { isFile: () => true, size: 12 } + ) const result = await getDiff('/repo', 'dist/large.log', false) - expect(readFileMock).not.toHaveBeenCalled() + expect(readFileMock).not.toHaveBeenCalledWith(workingTreePath) expect(result.kind).toBe('binary') expect(result.modifiedIsBinary).toBe(true) expect(result.modifiedContent).toBe('') @@ -259,8 +266,14 @@ describe('getDiff', () => { it('flags a deleted image so previewers can fall back to the original bytes', async () => { const pngBuffer = Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x00]) + const workingTreePath = path.join('/repo', 'assets/deleted.png') gitExecFileAsyncBufferMock.mockResolvedValueOnce({ stdout: pngBuffer }) - statMock.mockRejectedValueOnce(Object.assign(new Error('missing'), { code: 'ENOENT' })) + statMock.mockImplementation(async (target: string) => { + if (target === workingTreePath) { + throw Object.assign(new Error('missing'), { code: 'ENOENT' }) + } + return { isFile: () => true, size: 12 } + }) const result = await getDiff('/repo', 'assets/deleted.png', false) @@ -275,9 +288,15 @@ describe('getDiff', () => { it('does not treat an unreadable working-tree image as a deletion', async () => { const pngBuffer = Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x00]) + const workingTreePath = path.join('/repo', 'assets/unreadable.png') gitExecFileAsyncBufferMock.mockResolvedValueOnce({ stdout: pngBuffer }) - statMock.mockResolvedValueOnce({ isFile: () => true, size: 5 }) - readFileMock.mockRejectedValueOnce(new Error('EIO')) + statMock.mockImplementation(async () => ({ isFile: () => true, size: 5 })) + readFileMock.mockImplementation(async (target: string) => { + if (target === workingTreePath) { + throw new Error('EIO') + } + return Buffer.from('') + }) const result = await getDiff('/repo', 'assets/unreadable.png', false) diff --git a/src/main/git/status-submodule.test.ts b/src/main/git/status-submodule.test.ts index 138d2874b19..b67a4eca9a0 100644 --- a/src/main/git/status-submodule.test.ts +++ b/src/main/git/status-submodule.test.ts @@ -123,11 +123,11 @@ describe('submodule diff routing', () => { expect(gitExecFileAsyncBufferMock).toHaveBeenCalledWith( ['show', '--end-of-options', `${OLD_OID}:lib/main.dart`], - { cwd: SUBMODULE, maxBuffer: 10 * 1024 * 1024 } + { cwd: SUBMODULE, maxBuffer: 10 * 1024 * 1024, preferWslDirectGit: true } ) expect(gitExecFileAsyncBufferMock).toHaveBeenCalledWith( ['show', '--end-of-options', `${NEW_OID}:lib/main.dart`], - { cwd: SUBMODULE, maxBuffer: 10 * 1024 * 1024 } + { cwd: SUBMODULE, maxBuffer: 10 * 1024 * 1024, preferWslDirectGit: true } ) expect(result.kind).toBe('text') expect(result.originalContent).toBe('v1\n') @@ -167,11 +167,11 @@ describe('submodule diff routing', () => { expect(gitExecFileAsyncBufferMock).toHaveBeenCalledWith( ['show', '--end-of-options', `${OLD_OID}:lib/main.dart`], - { cwd: SUBMODULE, maxBuffer: 10 * 1024 * 1024 } + { cwd: SUBMODULE, maxBuffer: 10 * 1024 * 1024, preferWslDirectGit: true } ) expect(gitExecFileAsyncBufferMock).toHaveBeenCalledWith( ['show', '--end-of-options', `${NEW_OID}:lib/main.dart`], - { cwd: SUBMODULE, maxBuffer: 10 * 1024 * 1024 } + { cwd: SUBMODULE, maxBuffer: 10 * 1024 * 1024, preferWslDirectGit: true } ) expect(result.kind).toBe('text') expect(result.originalContent).toBe('v1\n') @@ -202,7 +202,8 @@ describe('submodule diff routing', () => { expect(gitExecFileAsyncBufferMock).toHaveBeenCalledWith(['show', ':lib/main.dart'], { cwd: SUBMODULE, - maxBuffer: 10 * 1024 * 1024 + maxBuffer: 10 * 1024 * 1024, + preferWslDirectGit: true }) expect(readFileMock).toHaveBeenCalledWith(path.join(SUBMODULE, 'lib/main.dart')) expect(result.kind).toBe('text') diff --git a/src/main/git/status-test-harness.ts b/src/main/git/status-test-harness.ts index eda7e2f5014..99925f90363 100644 --- a/src/main/git/status-test-harness.ts +++ b/src/main/git/status-test-harness.ts @@ -15,6 +15,8 @@ export type FsPromisesMocks = { readFileMock: MockFn statMock: MockFn rmMock: MockFn + /** Optional: defaults to "nothing exists", which is what most git-read tests assume. */ + accessMock?: MockFn } export function createGitRunnerModuleMock(mocks: GitRunnerMocks): Record { @@ -49,7 +51,12 @@ export function createFsPromisesModuleMock(mocks: FsPromisesMocks): Record { + throw Object.assign(new Error(`ENOENT: ${target}`), { code: 'ENOENT' }) + }) } } diff --git a/src/main/git/status.test.ts b/src/main/git/status.test.ts index e658e773743..9663becb622 100644 --- a/src/main/git/status.test.ts +++ b/src/main/git/status.test.ts @@ -15,6 +15,7 @@ const { readFileMock, statMock, rmMock, + accessMock, existsSyncMock } = vi.hoisted(() => ({ gitExecFileAsyncMock: vi.fn(), @@ -25,6 +26,7 @@ const { readFileMock: vi.fn(), statMock: vi.fn(), rmMock: vi.fn(), + accessMock: vi.fn(), existsSyncMock: vi.fn() })) @@ -37,9 +39,17 @@ vi.mock('./runner', () => ) vi.mock('fs/promises', () => - createFsPromisesModuleMock({ lstatMock, realpathMock, readFileMock, statMock, rmMock }) + createFsPromisesModuleMock({ + lstatMock, + realpathMock, + readFileMock, + statMock, + rmMock, + accessMock + }) ) +// Why still here: unmerged-entry parsing probes the working tree through node:fs directly. vi.mock('fs', () => ({ existsSync: existsSyncMock })) @@ -62,6 +72,8 @@ describe('getStatus', () => { lstatMock.mockReset() readFileMock.mockReset() existsSyncMock.mockReset() + accessMock.mockReset() + accessMock.mockRejectedValue(Object.assign(new Error('ENOENT'), { code: 'ENOENT' })) // Why: untracked line counting stats a file before reading it; any // under-limit size routes the read to readFileMock. statMock.mockReset() @@ -75,7 +87,12 @@ describe('getStatus', () => { it('parses unmerged porcelain v2 entries into unresolved conflict rows', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockImplementation((target: string) => target.endsWith('MERGE_HEAD')) + accessMock.mockImplementation(async (target: string) => { + if (target.endsWith('MERGE_HEAD')) { + return undefined + } + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: 'u UU N... 100644 100644 100644 100644 aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb cccccccccccccccccccccccccccccccccccccccc src/app.ts\n' @@ -97,7 +114,6 @@ describe('getStatus', () => { it('maps deleted conflicts to deleted when the working tree file is absent', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: 'u UD N... 100644 100644 000000 100644 aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb cccccccccccccccccccccccccccccccccccccccc src/deleted.ts\n' @@ -132,7 +148,6 @@ describe('getStatus', () => { it('passes core.quotePath=false and round-trips UTF-8 paths', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: '1 .M N... 100644 100644 100644 ce013625030ba8dba906f756967f9e9ca394464a ce013625030ba8dba906f756967f9e9ca394464a docs/日本語/sample.md\n' @@ -159,7 +174,6 @@ describe('getStatus', () => { it('preserves porcelain v2 submodule dirtiness flags on status rows', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: '1 AM S..U 000000 160000 160000 0000000000000000000000000000000000000000 7844cb64e631f17a9ca5b548f3500ef7cecd2f17 nested-repo\n' @@ -185,7 +199,6 @@ describe('getStatus', () => { it('omits ignored files by default and parses them when requested', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: '! dist/\n! generated/file.js\n' }) @@ -206,7 +219,6 @@ describe('getStatus', () => { it('parses branch identity from porcelain v2 branch headers', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: '# branch.oid abcdef1234567890\n# branch.head feature/prompts\n1 .M N... 100644 100644 100644 ce013625030ba8dba906f756967f9e9ca394464a ce013625030ba8dba906f756967f9e9ca394464a src/app.ts\n' @@ -222,7 +234,6 @@ describe('getStatus', () => { it('folds upstream ahead/behind from porcelain v2 into the status result', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: '# branch.oid abcdef1234567890\n# branch.head feature/prompts\n# branch.upstream origin/feature/prompts\n# branch.ab +2 -3\n' @@ -241,7 +252,6 @@ describe('getStatus', () => { it('reports no upstream from porcelain v2 status when no same-name origin branch exists', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockImplementation((args: string[]) => { if (args[0] === '-c' && args.includes('status')) { return Promise.resolve({ @@ -274,7 +284,6 @@ describe('getStatus', () => { it('uses same-name origin branch status for legacy base-tracking worktrees', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock .mockResolvedValueOnce({ stdout: @@ -297,7 +306,6 @@ describe('getStatus', () => { it('omits --ignored and ignoredPaths when includeIgnored is not requested', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: '' }) const result = await getStatus('/repo') @@ -315,7 +323,6 @@ describe('getStatus', () => { it('parses ! porcelain v2 records into ignoredPaths when includeIgnored is true', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: '! dist/\n! .env\n! coverage/\n' }) @@ -337,7 +344,6 @@ describe('getStatus', () => { it('attaches per-area line counts from staged and unstaged numstat', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockImplementation((args: string[]) => { if (args.includes('status')) { return Promise.resolve({ @@ -364,7 +370,6 @@ describe('getStatus', () => { it('reuses unchanged line stats only when the safety hint is present', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockImplementation((args: string[]) => { if (args.includes('status')) { return Promise.resolve({ @@ -392,7 +397,6 @@ describe('getStatus', () => { it('recomputes after a scan whose numstat failed instead of pinning missing counts', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) let failNumstat = true gitExecFileAsyncMock.mockImplementation((args: string[]) => { if (args.includes('status')) { @@ -422,7 +426,6 @@ describe('getStatus', () => { it('invalidates safety reuse for a new head and for known mutations', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) let head = 'head-1' gitExecFileAsyncMock.mockImplementation((args: string[]) => { if (args.includes('status')) { @@ -450,7 +453,6 @@ describe('getStatus', () => { it('isolates line-stat reuse between WSL distributions', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockImplementation((args: string[]) => { if (args.includes('status')) { return Promise.resolve({ @@ -474,7 +476,6 @@ describe('getStatus', () => { it('attaches numstat counts for literal paths containing rename markers', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockImplementation((args: string[]) => { if (args.includes('status')) { return Promise.resolve({ @@ -503,7 +504,6 @@ describe('getStatus', () => { it('attaches staged rename counts to the new path', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockImplementation((args: string[]) => { if (args.includes('status')) { return Promise.resolve({ @@ -531,7 +531,6 @@ describe('getStatus', () => { }) it('counts untracked file contents as additions', async () => { - existsSyncMock.mockReturnValue(false) lstatMock.mockResolvedValue({ size: 14, mtimeMs: 1, @@ -555,7 +554,6 @@ describe('getStatus', () => { it('leaves binary working-tree changes without counts', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockImplementation((args: string[]) => { if (args.includes('status')) { return Promise.resolve({ @@ -578,7 +576,6 @@ describe('getStatus', () => { it('skips numstat entirely for a clean working tree', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: '' }) await getStatus('/repo') @@ -588,7 +585,6 @@ describe('getStatus', () => { it('truncates and flags didHitLimit when entries exceed the limit', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) const stdout = `${Array.from({ length: 25 }, (_, i) => `? file${i}.txt`).join('\n')}\n` gitExecFileAsyncMock.mockReset() gitExecFileAsyncMock.mockResolvedValue({ stdout: '' }) @@ -651,7 +647,6 @@ describe('getStatus', () => { it('does not flag didHitLimit for a normal repo under the limit', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') - existsSyncMock.mockReturnValue(false) gitExecFileAsyncMock.mockReset() gitExecFileAsyncMock.mockResolvedValue({ stdout: '' }) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: '? a.txt\n? b.txt\n' }) diff --git a/src/main/git/wsl-git-read-environment.ts b/src/main/git/wsl-git-read-environment.ts index 7c809453498..e75efee30d4 100644 --- a/src/main/git/wsl-git-read-environment.ts +++ b/src/main/git/wsl-git-read-environment.ts @@ -7,10 +7,18 @@ import { export type WslGitReadEnvironment = { gitPath: string; home: string; path: string } const PROBE_TIMEOUT_MS = 10_000 +/** + * How long a read may wait for a cold probe before taking the login shell. + * Short enough that a wedged distro cannot stall the panel, long enough that a + * healthy one resolves and every later read runs shell-free. + */ +export const WSL_GIT_READ_ENVIRONMENT_WAIT_MS = 1_500 const PROBE_MAX_BUFFER = 64 * 1024 const TRANSIENT_PROBE_RETRY_MS = 30_000 const environmentByDistro = new Map>() -const resolvedEnvironmentByDistro = new Map() +// Why the null entries matter: a settled "no direct route" answer is what lets a read skip the +// bounded probe wait entirely instead of racing an already-decided promise on every call. +const settledEnvironmentByDistro = new Map() const transientRetryAfterByDistro = new Map() type ProbeOutcome = @@ -76,6 +84,7 @@ export function getWslGitReadEnvironment(distro: string): Promise= retryAfter) { environmentByDistro.delete(distro) + settledEnvironmentByDistro.delete(distro) transientRetryAfterByDistro.delete(distro) } let environment = environmentByDistro.get(distro) @@ -85,10 +94,11 @@ export function getWslGitReadEnvironment(distro: string): Promise ({ diffCache: settledDiffCache.stats() }) }) function focusExistingWindow(): void { focusExistingMainWindow({ From 0096e47850c37dc51a643af531c1078c469e0638 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 26 Aug 2026 16:15:03 -0700 Subject: [PATCH 13/19] fix(windows): keep windows-process-tree gyp paths absolute under pnpm (#16688) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(windows): keep windows-process-tree gyp paths absolute under pnpm Hourly Windows builds have failed since #16598 at `build-windows-process-tree-relay-addon`: `require('node-addon-api').targets` is cwd-relative, so node-gyp evaluates it from the pnpm store realpath and then loads it from the `node_modules` symlink. That resolves `node_addon_api.gyp` outside the repo. Use `require.resolve` for an absolute path, matching the node-pty patch. * i18n: keep ja skill-filter labels on the catalog's Agent brand #16682 merged with a failing localization catalog: ja used エージェント in three new skill-filter strings, and repair-locale-catalog rewrites those to Agent. Match the rest of ja.json so static analysis can pass. --- .../@vscode__windows-process-tree@0.8.0.patch | 17 +++++-- ...build-windows-process-tree-relay-addon.mjs | 16 +++++-- .../windows-process-tree-gyp-path.test.mjs | 38 ++++++++++++++++ docs/reference/windows-process-enumeration.md | 17 ++++--- pnpm-lock.yaml | 6 +-- src/renderer/src/i18n/locales/ja.json | 44 ++++++++++++++++--- 6 files changed, 114 insertions(+), 24 deletions(-) create mode 100644 config/scripts/windows-process-tree-gyp-path.test.mjs diff --git a/config/patches/@vscode__windows-process-tree@0.8.0.patch b/config/patches/@vscode__windows-process-tree@0.8.0.patch index cf68eea1985..8e6e8e86e25 100644 --- a/config/patches/@vscode__windows-process-tree@0.8.0.patch +++ b/config/patches/@vscode__windows-process-tree@0.8.0.patch @@ -2,13 +2,22 @@ diff --git a/binding.gyp b/binding.gyp index 855bd4b86f0a3c18c7594212c0e42b6e35bc4001..5f15551cb1520af996f216500a83ac68b57f5104 100644 --- a/binding.gyp +++ b/binding.gyp +@@ -3,7 +3,7 @@ + { + "target_name": "windows_process_tree", + "dependencies": [ +- " { + it('keeps the gyp project path absolute so pnpm Windows source builds find it', () => { + expect(PATCH).toContain( + `+ " { + const resolved = execFileSync(process.execPath, ['-p', ABSOLUTE_GYP], { + cwd: PACKAGE_DIR, + encoding: 'utf8' + }).trim() + expect(isAbsolute(resolved)).toBe(true) + expect(existsSync(resolved)).toBe(true) + }) +}) diff --git a/docs/reference/windows-process-enumeration.md b/docs/reference/windows-process-enumeration.md index 236d6adbf47..115cc7040a9 100644 --- a/docs/reference/windows-process-enumeration.md +++ b/docs/reference/windows-process-enumeration.md @@ -116,11 +116,11 @@ it as an optional relay artifact. `config/scripts/build-windows-process-tree-relay-addon.mjs` builds it from the source pnpm has already patched, on a Windows runner, and refuses to run if -either patch hunk is missing — the Spectre hunk fails loudly, but the -1024-process hunk fails *silently*, so the source is checked rather than the -install trusted. It also reads the PE machine field of the output, because a -cross-build that quietly emitted host arch would ship a binary the target cannot -load. +any patch hunk is missing — the Spectre hunk fails loudly, the 1024-process +hunk fails *silently*, and the relative gyp path dies at configure on Windows. +The source is checked rather than the install trusted. It also reads the PE +machine field of the output, because a cross-build that quietly emitted host +arch would ship a binary the target cannot load. Windows arm64 cross-compiles from the x64 runner — verified on real hardware, producing `IMAGE_FILE_MACHINE_ARM64` (0xaa64) against x64's 0x8664. It needs the @@ -143,7 +143,7 @@ on any other OS keeps using the scan. ## Why the package is patched -`config/patches/@vscode__windows-process-tree@0.8.0.patch` carries two hunks. +`config/patches/@vscode__windows-process-tree@0.8.0.patch` carries three hunks. 1. **Spectre mitigation.** The upstream `binding.gyp` requires Spectre-mitigated libraries, which Orca's Windows build agents do not install. `node-pty` is @@ -153,6 +153,11 @@ on any other OS keeps using the scan. 1024 and the querying process was itself among the 27 missing. A truncated snapshot silently hides the descendants a teardown is trying to reap — the exact failure the native path exists to remove. +3. **Absolute `node-addon-api` gyp path.** `require('node-addon-api').targets` + is cwd-relative. node-gyp on Windows evaluates it from the pnpm store + realpath, then loads the relative path from the `node_modules` symlink, so + `node_addon_api.gyp` resolves outside the repo and hourly Windows builds + die at configure. `node-pty` is patched the same way for the same reason. The typings claim `commandLine` is truncated at 512 characters. Measured, it is not: the longest observed on a real host was 26,059. diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 6d739a42d25..0d80c024dff 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -9,7 +9,7 @@ overrides: patchedDependencies: '@vscode/windows-process-tree@0.8.0': - hash: 7c08bebce9b36829be6035218db2383f1a889c21ec907ea7e85c36fc54795a05 + hash: 73a90530e6b95c05dac50f2c48787fb51cb292587773016ebe57eb05e41075a0 path: config/patches/@vscode__windows-process-tree@0.8.0.patch '@xterm/addon-ligatures@0.11.0-beta.287': hash: 47405b9994b5acf1b4e90b49250358c1ca03649854d59560e7732b72fe336920 @@ -407,7 +407,7 @@ importers: optionalDependencies: '@vscode/windows-process-tree': specifier: 0.8.0 - version: 0.8.0(patch_hash=7c08bebce9b36829be6035218db2383f1a889c21ec907ea7e85c36fc54795a05) + version: 0.8.0(patch_hash=73a90530e6b95c05dac50f2c48787fb51cb292587773016ebe57eb05e41075a0) sherpa-onnx-darwin-arm64: specifier: 1.12.37 version: 1.12.37 @@ -9605,7 +9605,7 @@ snapshots: convert-source-map: 2.0.0 tinyrainbow: 3.1.0 - '@vscode/windows-process-tree@0.8.0(patch_hash=7c08bebce9b36829be6035218db2383f1a889c21ec907ea7e85c36fc54795a05)': + '@vscode/windows-process-tree@0.8.0(patch_hash=73a90530e6b95c05dac50f2c48787fb51cb292587773016ebe57eb05e41075a0)': dependencies: node-addon-api: 7.1.0 optional: true diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index 3b9ebdd2393..1c72acfa367 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -3888,7 +3888,7 @@ "35b9a724a0": "利用可能", "c13b82793c": "インストールを管理", "aee7b99cc6": "リンクからインストール", - "filterProvider": "エージェントで絞り込み", + "filterProvider": "Agent で絞り込み", "filterSource": "ソースで絞り込み", "allSources": "すべて", "clearFilters": "フィルターをクリア", @@ -3914,13 +3914,43 @@ "deleteSkill": "削除…" }, "SkillsList": { "listLabel": "スキル" }, - "sourceStatus": { "missing": "フォルダーが見つかりません", "remoteRepo": "リモートリポジトリ — 未スキャン", "unavailable": "未スキャン" }, + "sourceStatus": { + "missing": "フォルダーが見つかりません", + "remoteRepo": "リモートリポジトリ — 未スキャン", + "unavailable": "未スキャン" + }, "sources": { "heading": "スキルフォルダー" }, - "sourceKind": { "home": "ホーム", "workspace": "ワークスペース", "bundled": "バンドル済み", "plugin": "プラグイン" }, - "count": { "skillOne": "{{count}} 件のスキル", "skillOther": "{{count}} 件のスキル", "sourceOne": "{{count}} 件のソース", "sourceOther": "{{count}} 件のソース", "fileOne": "{{count}} 個のファイル", "fileOther": "{{count}} 個のファイル", "resultOne": "{{count}} 件の結果", "resultOther": "{{count}} 件の結果", "selected": "{{count}} 件を選択", "shareOne": "{{count}} 件のスキルを共有", "shareOther": "{{count}} 件のスキルを共有", "linkOne": "{{count}} 件のリンク", "linkOther": "{{count}} 件のリンク" }, - "filter": { "allAgents": "すべてのエージェント", "sharedAgent": "共有 (.agents)" }, - "SkillsSelectionHeader": { "exit": "選択を終了", "exitTooltip": "選択を終了 · Esc", "title": "共有するスキルを選択", "selectAll": "対象の {{count}} 件をすべて選択", "clear": "クリア", "deleteTitle": "削除するスキルを選択" }, - "SkillDetailDialog": { "agents": "エージェント", "updated": "更新日", "copy": "コピー" }, + "sourceKind": { + "home": "ホーム", + "workspace": "ワークスペース", + "bundled": "バンドル済み", + "plugin": "プラグイン" + }, + "count": { + "skillOne": "{{count}} 件のスキル", + "skillOther": "{{count}} 件のスキル", + "sourceOne": "{{count}} 件のソース", + "sourceOther": "{{count}} 件のソース", + "fileOne": "{{count}} 個のファイル", + "fileOther": "{{count}} 個のファイル", + "resultOne": "{{count}} 件の結果", + "resultOther": "{{count}} 件の結果", + "selected": "{{count}} 件を選択", + "shareOne": "{{count}} 件のスキルを共有", + "shareOther": "{{count}} 件のスキルを共有", + "linkOne": "{{count}} 件のリンク", + "linkOther": "{{count}} 件のリンク" + }, + "filter": { "allAgents": "すべての Agent", "sharedAgent": "共有 (.agents)" }, + "SkillsSelectionHeader": { + "exit": "選択を終了", + "exitTooltip": "選択を終了 · Esc", + "title": "共有するスキルを選択", + "selectAll": "対象の {{count}} 件をすべて選択", + "clear": "クリア", + "deleteTitle": "削除するスキルを選択" + }, + "SkillDetailDialog": { "agents": "Agent", "updated": "更新日", "copy": "コピー" }, "SkillFreshnessNudge": { "titleOne": "インストール済みの Orca スキルが古くなっています", "titleMany": "インストール済みの Orca スキル {{value0}} 件が古くなっています", From 9135b6f004154cc2b37769c26b102996bbec8ee0 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 26 Aug 2026 16:16:05 -0700 Subject: [PATCH 14/19] feat(orchestration): surface nested worker depth and propagate it across hosts (#16669) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(orchestration): surface nested worker depth and propagate it across hosts Builds on the depth enforcement in the previous commit, which shipped with the setting reachable only by editing settings.json and with workers never told they could nest. Adds the Settings -> Agents control (a 1/2/3 select rather than a free-form number, which bounds the value without inventing a numeric input primitive). The key stays absent from the SettingsUpdate RPC schema, matching agentSkillSharingEnabled: settings.update is reachable from the CLI, so an RPC-writable depth would let a worker raise its own cap. Adds a SUB-DISPATCH block to the dispatch preamble, emitted only when the worker actually has budget left. A worker told it "usually cannot" delegate still tries and then reports the refusal as a blocker, so the section is omitted entirely rather than softened. Propagates depth to federated worker hosts. Previously the home side computed and stored a depth the remote host never received, so a remote attachment always read as depth 1. That is correct at the default cap and wrong as soon as the cap is raised — precisely when someone starts relying on nesting. The field is optional, so an older Run home simply omits it and the attachment's NOT NULL DEFAULT 1 keeps the fail-closed behaviour. Enforcement still runs on the executing host against that host's own cap, consistent with the SSH execution boundary. * fix(orchestration): close nested depth readiness gaps * fix(settings): defer nested depth translations * fix(orchestration): drop federated depth keys that main already landed The enforcement PR's review pass added the same federated depth propagation before it merged, so replaying this branch onto main produced duplicate object keys. Keep main's versions -- its schema entry validates an integer >= 1 rather than any finite number. * fix(settings): label nested worker depth select * fix(settings): move nested depth to orchestration * fix(settings): refine nested depth placement --- skill-guides/orchestration.md | 2 +- src/cli/bundled-skill-guides.ts | 2 +- .../coordinator-task-dispatch.ts | 1 + .../runtime/orchestration/preamble.test.ts | 34 +++++++ src/main/runtime/orchestration/preamble.ts | 26 +++++- .../rpc/methods/orchestration-federation.ts | 3 + .../orchestration-tasks-dispatch.test.ts | 24 +++++ .../rpc/methods/orchestration-workers.ts | 1 + src/main/runtime/rpc/methods/orchestration.ts | 8 ++ .../settings/OrchestrationPane.test.tsx | 93 ++++++++++++++++++- .../components/settings/OrchestrationPane.tsx | 40 +++++++- .../src/components/settings/Settings.tsx | 4 +- .../settings/SettingsFormControls.tsx | 9 +- .../settings/nested-worker-depth-copy.ts | 29 ++++++ .../settings/orchestration-search.ts | 30 +++++- .../useSettingsNavigationMetadata.test.ts | 22 +++++ .../hooks/useSettingsNavigationMetadata.ts | 4 +- src/renderer/src/i18n/locales/en.json | 4 +- src/shared/nested-worker-depth.test.ts | 1 + src/shared/nested-worker-depth.ts | 6 +- 20 files changed, 323 insertions(+), 20 deletions(-) create mode 100644 src/renderer/src/components/settings/nested-worker-depth-copy.ts diff --git a/skill-guides/orchestration.md b/skill-guides/orchestration.md index ef8181dd4e3..5236a7f47fc 100644 --- a/skill-guides/orchestration.md +++ b/skill-guides/orchestration.md @@ -182,7 +182,7 @@ A dispatched worker normally cannot dispatch sub-workers. Attempting it fails wi `nested_worker_depth_exceeded` and a message telling the worker to complete the task itself. Do that — do not try to route around it. -The limit is a number, not an on/off switch. `Settings -> Agents -> Nested worker depth` +The limit is a number, not an on/off switch. `Settings -> Orchestration -> Nested worker depth` sets how many generations are allowed: - `1` (default): a coordinator dispatches workers; those workers do not dispatch. diff --git a/src/cli/bundled-skill-guides.ts b/src/cli/bundled-skill-guides.ts index 66f55c065e7..4385841d2f8 100644 --- a/src/cli/bundled-skill-guides.ts +++ b/src/cli/bundled-skill-guides.ts @@ -30,7 +30,7 @@ const ORCA_LINEAR_MARKDOWN = "---\nname: orca-linear\ndescription: >-\n Use Orc const ORCA_PER_WORKSPACE_ENV_MARKDOWN = "---\nname: orca-per-workspace-env\ndescription: >-\n Set up, review, debug, or validate Orca per-workspace environment recipes —\n on-demand, disposable runtimes (cloud sandboxes, VMs, or local) created fresh\n for each workspace. Covers first-time setup (provider prerequisites, the\n reusable base snapshot, the coding-agent auth snapshot, credentials, and\n state), not just the per-workspace lifecycle scripts. Use to stand up\n per-workspace environments, fix an `environmentRecipes` entry in `orca.yaml`, scaffold\n provider lifecycle scripts, or resolve an `orca vm recipe doctor` failure.\n---\n\n# Per-Workspace Environments\n\nHelp a user stand up and maintain a repo-owned per-workspace environment recipe end to end. Each\nworkspace gets its own on-demand, disposable runtime (a cloud sandbox, a VM, or a local one),\ncreated fresh and torn down after.\n\nOrca is a **thin wrapper**: you guide, detect, and scaffold; you never own the user's cloud account,\nbilling, images, or credentials.\n\n- **You DO:** sequence the setup, detect what's detectable (provider CLI present/logged-in? recipe\n present? `doctor` passing?), scaffold provider-templated scripts the user fills in, drive the slow\n snapshot/auth phases with the user, and always show the next action.\n- **You DO NOT:** create accounts, choose plans/regions, invent org/project/scope ids, store or print\n secrets, or run anything that spends money without an explicit user OK.\n\nFirst-time setup has **four phases before the per-workspace recipe runs** — easy to miss, so walk\nthem in order:\n\n1. **Prerequisites** — cloud account, provider CLI, scope/project, plan limits, git token (§2).\n2. **Base snapshot** — reusable image: tools + repo + headless build, snapshotted once (§3).\n3. **Agent-auth snapshot** — boot the base, run interactive device-auth, re-snapshot (§4).\n4. **State** — thread snapshot id / scope / project / port between phases via a state file (§6).\n\nThen the **per-workspace contract** (create/suspend/resume/destroy) runs fast (§8).\n\n**The one branch that shapes everything — connection mode:** **Orca-server** (`create` runs `orca serve`\nin the env and emits a `pairingCode`; §7c/§7f) vs **SSH** (`create` runs no server and emits a\n`connection.type:\"ssh\"` block Orca dials into; §7g/§7h). Settle this first — it changes the `create`\noutput shape and half the templates.\n\nKeep Orca's checkout behavior unchanged by default: omit `checkoutMode`, emit schema version 1, and\nlet Orca create a linked worktree. Only use `checkoutMode: provisioned-root` when the user explicitly\nwants one ephemeral machine to clone the finished workspace itself. This niche mode currently requires\ndirect SSH, an ordinary non-bare/non-sparse primary checkout at `projectRoot`, and schema version 2.\n\n**Quick-start (happy path):** interview the user (connection mode Orca-server vs SSH, provider, agent CLI,\ngit auth — §1.2) + read the provider's CLI docs → scaffold `scripts/orca-vm/` from §7 → run the\nbase-snapshot script, then the auth script (you invoke these by hand; not via `orca.yaml`) → wire\n`environmentRecipes` in `orca.yaml` → `orca vm recipe doctor --json` (free) → then the `--provision`\nself-test loop (§9) until it passes.\n\n---\n\n## 1. Setup workflow\n\nDrive these with the user. **[CHECKPOINT]** steps need explicit confirmation — they spend money, take\na long time, or need the user at the keyboard. Never create an Orca workspace or commit unless asked.\n\n1. **Inspect the repo** for an existing `environmentRecipes` entry, `scripts/orca-vm/`, a state file, or setup\n notes. If a working recipe exists, jump to Doctor (§9) instead of rebuilding.\n2. **Interview the user up front** — gather these choices and confirm them back before scaffolding\n anything. Don't pick for them (§11); don't guess.\n - **Connection mode:** how Orca attaches to the environment — an **Orca server** (the VM runs\n `orca serve` and Orca pairs over its pairing URL; worked example §7f) or **SSH** (Orca connects to\n the host over SSH; §7g). This decides the recipe's connection shape, so settle it first.\n - **Checkout ownership:** do not ask by default. Only when the user requires the environment to\n create the exact final checkout, confirm `provisioned-root` and direct SSH; otherwise omit it.\n - **Provider:** Vercel Sandbox, Fly, Modal, an existing SSH host, … For non-obvious providers, also\n ask scope/project/region and plan limits (§2). Then **read that provider's CLI/SDK docs** (or\n ` --help`) before scaffolding — you need its exact create/exec/snapshot/remove verbs.\n If a provider advertises `ssh`, verify whether it exposes a real dialable SSH target\n (host/port/user/key or proxy command) or only a provider-mediated interactive shell; Orca SSH mode\n needs the former.\n - **Coding-agent CLI + account:** which agent runs in the VM (`codex`, `claude`, …) and that the user\n has an account for it — it gets logged in during the Phase-3 auth snapshot (§4).\n - **Git auth:** the token source for cloning a private repo (`GH_TOKEN`/`GITHUB_TOKEN` or `gh auth\n token`; §5).\n3. **Check prerequisites (§2)** — detect the provider CLI + auth and confirm the items above are in\n place before any paid step.\n4. **Scaffold scripts + state file** from §7 (worked Vercel example: §7f; SSH host: §7g; Docker SSH:\n §7h; Windows: §7i), filling in the provider's real commands. Make them executable.\n5. **[CHECKPOINT] Build the base snapshot (§3)** — paid, slow.\n6. **[CHECKPOINT] Authenticate the agent (§4)** — interactive; the user follows a URL/code. **You cannot\n drive this step** — you run commands non-interactively, so there's no TTY for `docker exec -it` /\n `ssh -t` to prompt against. The **user** runs the Phase-3 login in their own terminal (or via the\n Claude Code harness bang-prefix — `! `, with the required space after `!`); you scaffold and drive\n the non-interactive phases around it. After kicking it off, **ask the user to report back once the login\n finishes** — you can't observe it completing, and you need that confirmation before resuming the\n non-interactive steps (base/auth commit, doctor, provision).\n7. **Wire the recipe** so `orca.yaml` points create/suspend/resume/destroy at the scripts (§8). The\n workspace composer reads `environmentRecipes` from the project's primary checkout of `orca.yaml`, **not** from\n a feature branch or worktree. So a recipe added only on a branch won't appear as a \"Run on\" option\n until that `orca.yaml` change is committed and merged to the project's primary branch. Tell the user\n this up front: `doctor`/`--provision` validate the scripts from the working copy on any branch, but\n creating a workspace from the recipe in the picker needs it on primary.\n8. **Dry-run doctor** — `orca vm recipe doctor --repo-path --json` (free, static; §9).\n Fix every failure before going live.\n9. **[CHECKPOINT] Live self-test** — get the user's OK once, then run\n `orca vm recipe doctor --provision --json` as a loop: it runs create → validates →\n destroys, and on failure returns a full transcript. Read it, fix the scripts, and re-run yourself until\n it passes (§9). Spends cloud money; the one approval covers the loop.\n10. **[CHECKPOINT] Optional workspace test** — only if asked: create a workspace via the picker, then\n verify sleep/wake/delete.\n\n---\n\n## 2. Phase 1 — Prerequisites\n\nThe user's responsibility; verify what's verifiable, ask for the rest, invent nothing. State which\nitems you verified vs. which the user asserted.\n\n- **Connection mode** (Orca server vs SSH) confirmed with the user — see §1 step 2; it shapes the recipe.\n- **Cloud account + plan** that allows sandboxes/VMs. Ask.\n- **Provider CLI installed + authenticated** — detect (`command -v `), check auth (e.g.\n `vercel whoami`). If missing, point at the provider's docs; don't log them in.\n- **Scope / project / region** the sandboxes live under. Ask; flows into every script via state.\n- **Plan / timeout / RAM caps.** Record them — e.g. Vercel Hobby caps sandbox timeout at **45m**,\n which limits both the base build and per-workspace runtime (see §10).\n- **Git token for private repos** (`GH_TOKEN`/`GITHUB_TOKEN`, or the provider's git auth; can fall back\n to `gh auth token`). See §5.\n- **Coding-agent CLI choice** (`codex`, `claude`…) and that the user has an account — it gets\n authenticated into the VM in Phase 3.\n\n---\n\n## 3. Phase 2 — Base snapshot (the reusable image)\n\nBuild **once**, snapshot, and every workspace boots from it in seconds instead of rebuilding.\nProvisioning + building takes a while (often ~20–30 min), so it runs behind a checkpoint. The script\nshape is §7a; key points:\n\n- Build the **headless Electron main only** (not the renderer) so it fits in plan RAM.\n- Use the VM image's package manager (`apt`/`dnf`/`apk`, per the base distro — not the provider brand).\n- Clone with the git token via `GIT_ASKPASS` (§5).\n- **Trap errors and remove the half-built sandbox** so a crash doesn't leave a paid resource running.\n- Snapshot the stopped sandbox, parse the snapshot id, and write it + scope/project/port/repo to state.\n\n---\n\n## 4. Phase 3 — Agent-auth snapshot (interactive)\n\nThe base snapshot has the agent CLI installed but **not logged in**, and per-workspace VMs are\nephemeral — so authenticate once and bake it into a second snapshot layer. Script shape is §7b:\n\n1. Boot a sandbox from the base `snapshotId` (from state).\n2. Run the agent's login **interactively** (`--interactive --tty`); the user completes the URL/code in\n their browser. On a **headless VM this must be the device-auth flow** (e.g. `codex login --device-auth`),\n **not** plain `codex login`: the default OAuth login starts a loopback callback server on a container\n port the host browser can't reach, so it hangs. Device-auth instead prints a URL + code the user opens\n on the **host**.\n3. Verify login; **refuse to snapshot an unauthenticated VM.** Prefer the status command's **exit code**\n (most agent CLIs exit non-zero when unauthenticated). If you grep instead, agent status often goes to\n **stderr** (e.g. `codex login status` prints \"Logged in using ChatGPT\" there), so **fold stderr first**\n (`... 2>&1 | grep …`) and match the agent's **exact success line** — never `grep -qi 'logged in'`, which\n also matches \"**not** logged in\" and would commit an unauthenticated image.\n4. Re-snapshot, parse the new id, and overwrite `snapshotId` in state to the authenticated image\n (recording `authSourceSnapshotId`). Remove the auth sandbox.\n\n**You can't drive step 2 yourself** (you run commands non-interactively — no TTY). The **user** runs it in\ntheir own terminal, or via the Claude Code harness bang-prefix (`! `, with the required space after\n`!`). You scaffold/boot the sandbox and run steps 3–4, but **you cannot observe the interactive login\nfinishing** — so **ask the user to tell you when it's done** before you verify and re-snapshot.\n\nIf the agent's credentials are short-lived, warn that the snapshot may need periodic re-auth (§10).\n\nFor disposable runtimes, do **not** treat a host agent config directory (for example `~/.codex`) as the\nauth snapshot by bind-mounting or copying it wholesale. Agent homes often contain sqlite state, hook\napproval state, caches, logs, and host-specific env/config. Instead, authenticate/configure the agent\ninside the disposable runtime and snapshot/commit that runtime layer.\n\n---\n\n## 5. Credentials\n\n- **Never** commit secrets or put them in `userData`, recipe JSON, comments, docs, or the state file.\n- **Git token:** read from env (`GH_TOKEN`/`GITHUB_TOKEN`), falling back to `gh auth token`. Pass to the\n VM only via the provider's ephemeral `--env`. Inside the VM, use a `GIT_ASKPASS` helper with\n `x-access-token` (not the token in the clone URL) and `GIT_TERMINAL_PROMPT=0` so a missing token fails\n fast instead of hanging. When you write the helper from inside `bash -lc` under `set -u`, escape the\n positional arg and the token (`\\$1`, `\\$GH_TOKEN`) so they land **literally** and resolve at git-runtime\n — an unescaped `$1` aborts with \"unbound variable\", and a literal `$GH_TOKEN` keeps the real token out of\n the written file. `rm -f` the helper after the clone/fetch.\n- **Provider auth:** rely on the provider CLI's logged-in session, not checked-in keys.\n- **Agent auth:** lives in the authenticated snapshot (Phase 3) — never a file you write or commit.\n- State holds only **non-secret** wiring (snapshot ids, scope, project, port, repo url/ref).\n\n---\n\n## 6. State file\n\nA repo-local JSON file (e.g. `scripts/orca-vm/-state.json`) threads non-secret values between\nphases. Each script resolves values as **env var → state → built-in fallback**, and merges its outputs\nback. Phase 2 writes the base `snapshotId`; Phase 3 overwrites it with the authenticated snapshot;\nper-workspace `create` boots from `snapshotId`.\n\n```json\n{\n \"baseName\": \"orca-base\",\n \"snapshotId\": \"snap_authenticated_image_id\",\n \"authSourceSnapshotId\": \"snap_base_image_id\",\n \"scope\": \"\",\n \"project\": \"\",\n \"port\": 7331,\n \"repoUrl\": \"https://host/org/repo.git\",\n \"repoRef\": \"main\",\n \"projectRoot\": \"/abs/path/on/remote/repo\"\n}\n```\n\n---\n\n## 7. Script templates (provider-agnostic shapes)\n\nScaffold under `scripts/orca-vm/`. These are **shapes** — fill in the provider's real commands. All\nreserve stdout for the final JSON and log progress to stderr. Include a shared `json_value ` /\n`env_value ` reader (env → state → fallback) in each.\n\n**Where each script runs:**\n\n- **Local-side** (`create`/`suspend`/`resume`/`destroy` + the base-snapshot/auth scripts the user\n invokes) runs **on the user's desktop**, so it must run on their OS. macOS/Linux: `#!/usr/bin/env\n bash`, `set -euo pipefail`, quoted paths. **Windows:** a bare `.sh` won't run — scaffold `.ps1`/`.cmd`\n or require WSL/Git-Bash and point `orca.yaml` at the right launcher.\n- **Remote-side** (commands you `exec` *inside* the Linux VM) always runs in the VM's Linux shell, so\n bash is fine there regardless of the user's OS.\n\n### 7a. Base-snapshot (`-base-snapshot.sh`) — Phase 2\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve base_name/repo_url/repo_ref/project_root/port/scope/project/timeout (env→state→fallback)\n# resolve gh token: GH_TOKEN | GITHUB_TOKEN | `gh auth token`\n# 1. provision a sandbox (timeout/vcpus/published port/snapshot retention); trap: remove on error\n# 2. remote exec (long timeout): install pkgs + gh + corepack/pnpm + agent CLI;\n# clone with GIT_ASKPASS(token); write headless main-only build config;\n# dev setup; pnpm install; build CLI; build headless electron main; smoke-check tools\n# 3. snapshot stopped sandbox; parse snapshot id (fail if unparseable)\n# 4. merge { baseName, snapshotId, projectRoot, repoUrl, repoRef, port, scope, project } into state\n# print only the state JSON to stdout\n```\n\nWorked Vercel commands for this phase are in §7f. You run this script by hand (not via `orca.yaml`),\nafter exporting the first-run inputs the state file doesn't have yet — e.g. provider scope/project, the\nrepo URL/ref, and a git token (`GH_TOKEN`); later runs read them back from state.\n\n### 7b. Auth (`-base-auth.sh`) — Phase 3\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# read source snapshot from state.snapshotId (fail if absent); auth_name=\"${base_name}-auth\"\n# 1. boot sandbox from source snapshot; trap: remove on error\n# 2. INTERACTIVE/TTY remote exec: agent login — user completes URL/code. Headless VM: MUST use the\n# device-auth flow (e.g. `codex login --device-auth`) — plain OAuth login binds a loopback callback\n# port the host can't reach and hangs. User runs this themselves (you have no interactive TTY); ask\n# them to report back when it's done before continuing.\n# 3. verify login, then refuse to snapshot if not logged in. Prefer the status command's EXIT CODE (most\n# agent CLIs exit non-zero when unauthenticated) over string-matching. If you must grep, fold stderr\n# first (`status 2>&1 | grep …` — many agents print the success line there) and match the agent's exact\n# success line; never `grep -qi 'logged in'`, which also matches \"not logged in\". Codex example: §7f.\n# 4. snapshot; parse new id\n# 5. merge { snapshotId:, authSourceSnapshotId: } into state; remove auth sandbox\n# print only the state JSON to stdout\n```\n\n### 7c. Create (`-create.sh`) — per workspace\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# read authenticated snapshotId/scope/project/port/repo*/project_root (env→state→fallback)\n# fail clearly if snapshotId is missing (point back to Phases 2–3)\n# name = orca-${ORCA_RECIPE_ID}-${ORCA_VM_INSTANCE_ID} (sanitized, length-capped)\n# 1. boot sandbox from snapshotId with a published port; capture the public URL → pairing address\n# (an externally reachable wss:// URL); trap: remove sandbox on error\n# 2. remote exec: ensure repo at desired commit; rebuild only if commit changed (cache marker)\n# 3. remote exec: start orca serve in the background and read the recipe JSON it writes (see below)\n# 4. print serve's JSON to stdout, optionally enriched with userData:\n# { schemaVersion:1, pairingCode, projectRoot, userData:{ provider, resourceId:name, snapshotId } }\n```\n\n**The exact `orca serve` invocation and its output (verified — do not improvise the flags).** Inside the\nVM, run:\n\n```bash\norca serve \\\n --port \"$PORT\" \\\n --project-root \"$ABS_REPO_PATH_ON_REMOTE\" \\\n --pairing-address \"$EXTERNAL_WSS_URL\" \\\n --recipe-json\n```\n\n**Binary name:** in a VM built from source (the Phase-2 flow), run it as `pnpm exec orca-dev serve …`\nfrom the repo root — `orca-dev` is the in-repo entrypoint and is what the §7f example uses. Plain\n`orca serve …` is the same command when the built CLI is installed on the VM's PATH. The flags/output\nare identical either way.\n\nThere is **no `--host` flag**. `--project-root` must be an absolute directory on the remote. With\n`--recipe-json` the server **stays running** and prints exactly this single object to **stdout**, then\nkeeps serving:\n\n```json\n{ \"schemaVersion\": 1, \"pairingCode\": \"\", \"projectRoot\": \"\" }\n```\n\n`pairingCode` is the pairing URL, already pointing at whatever you passed as `--pairing-address` — so set\n`--pairing-address` to the externally reachable address and **pass `pairingCode` through unchanged; never\nhand-rewrite it**. Because serve runs in the foreground and doesn't exit, redirect its stdout to a file\nand poll until that file parses as JSON (and bail if the process dies — dump its stderr log). Your\n`create` script then prints that JSON (optionally merging `userData`). Concrete pattern: §7f.\n\n### 7d. Suspend / resume / destroy — per workspace\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\npayload=\"$(cat)\" # Orca passes lifecycle JSON on stdin\nresource_id=\"$(node -e 'const d=JSON.parse(process.argv[1]); process.stdout.write(d.recipeResult?.userData?.resourceId ?? \"\")' \"$payload\")\"\n[ -n \"$resource_id\" ] || { echo \"No resource id in lifecycle payload\" >&2; exit 1; }\n# suspend: provider suspend \"$resource_id\"\n# resume: provider resume \"$resource_id\"; then RE-EMIT fresh recipe JSON (pairing may change)\n# destroy: provider remove \"$resource_id\" (or set destroy: none in orca.yaml)\n```\n\n### 7e. State file — scaffold with scope/project/repo filled in and snapshot ids empty (§6).\n\n### 7f. Worked example — Vercel Sandbox (all three phases)\n\nA real, working shape (the Vercel surface is a CLI: `vercel sandbox create|exec|snapshot|remove`). Adapt\nnames; verify flags against `vercel sandbox --help` for the user's CLI version before relying on them.\nThese ground §7a (base snapshot) and §7b (auth), which are otherwise generic skeletons.\n\n**Phase 2 — base snapshot (§7a):** provision → install tools + clone + headless build → snapshot.\n\n```bash\n# provision a fresh build sandbox (retain a couple of snapshots); trap-remove on error\nvercel sandbox create --name \"$base\" --runtime node24 --timeout 30m --vcpus 4 --publish-port \"$port\" \\\n --snapshot-expiration 30d --keep-last-snapshots 2 \"${vercel_args[@]}\" >&2\n# remote build (long timeout): install pkgs+gh+pnpm+agent CLI, clone with GIT_ASKPASS (write the helper\n# with LITERAL \\$1/\\$GH_TOKEN so they resolve at git-runtime, not write-time — see §5/§7f create — then\n# `rm -f /tmp/askpass.sh`), write the headless main-only build config (drop the renderer), dev setup,\n# build CLI + headless main, smoke-check\nvercel sandbox exec \"$base\" \"${vercel_args[@]}\" --timeout 25m --env \"GH_TOKEN=$gh_token\" … -- bash -lc '…build…' >&2\n# snapshot the STOPPED sandbox and parse the id from CLI output (fail if unparseable)\nout=\"$(vercel sandbox snapshot \"$base\" --stop --expiration 30d \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$out\" >&2\nsnapshot_id=\"$(printf '%s\\n' \"$out\" | sed -nE 's/.*(snap_[A-Za-z0-9]+).*/\\1/p' | tail -1)\"\n# merge { baseName, snapshotId, scope, project, port, repoUrl, repoRef, projectRoot } into state; print state JSON\n```\n\n**Phase 3 — agent-auth snapshot (§7b):** boot the base, log the agent in interactively, re-snapshot.\n(`codex` below is an example — substitute the user's chosen agent's login/status verbs, e.g. `claude`.)\n\n```bash\nvercel sandbox create --name \"$auth\" --snapshot \"$snapshot_id\" --timeout 30m --publish-port \"$port\" \"${vercel_args[@]}\" >&2\n# INTERACTIVE — the USER runs this in their own terminal (you have no interactive TTY) and completes the\n# URL/code on the HOST. --device-auth is MANDATORY on a headless VM: plain `codex login` binds a loopback\n# callback port the host browser can't reach and hangs. Ask the user to report back when login finishes.\nvercel sandbox exec --interactive --tty \"$auth\" \"${vercel_args[@]}\" -- bash -lc 'codex login --device-auth'\n# refuse to snapshot an unauthenticated VM — fold stderr, match codex's exact success line (§4)\nvercel sandbox exec \"$auth\" \"${vercel_args[@]}\" --timeout 30s -- bash -lc 'codex login status 2>&1' | grep -Eqi 'Logged in using ChatGPT|Logged in via device' \\\n || { echo \"agent not logged in; not snapshotting\" >&2; exit 1; }\nout=\"$(vercel sandbox snapshot \"$auth\" --stop --expiration 30d \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$out\" >&2\nnew_id=\"$(printf '%s\\n' \"$out\" | sed -nE 's/.*(snap_[A-Za-z0-9]+).*/\\1/p' | tail -1)\"\n# overwrite state.snapshotId = new_id, record authSourceSnapshotId = snapshot_id; remove the auth sandbox\n```\n\n**Per-workspace `create`** (the fast path):\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve from env→state→fallback: snapshot_id, scope, project, port, repo_url, repo_ref, project_root\nvercel_args=(); [ -n \"$scope\" ] && vercel_args+=(--scope \"$scope\"); [ -n \"$project\" ] && vercel_args+=(--project \"$project\")\n[ -n \"$snapshot_id\" ] || { echo \"snapshotId missing — run Phases 2–3 first\" >&2; exit 1; }\ngh_token=\"${GH_TOKEN:-${GITHUB_TOKEN:-$(command -v gh >/dev/null 2>&1 && gh auth token 2>/dev/null || true)}}\"\nrecipe_id=\"${ORCA_RECIPE_ID:-vercel-sandbox}\"\nrecipe_id=\"${recipe_id//./-}\" # Vercel names forbid dots.\ninstance_id=\"${ORCA_VM_INSTANCE_ID:-$(date +%s)}\"\nmax_recipe_id_length=$((128 - ${#instance_id} - 6)) # Preserve the unique instance suffix.\n[ \"$max_recipe_id_length\" -gt 0 ] || { echo \"ORCA_VM_INSTANCE_ID is too long for a Vercel sandbox name\" >&2; exit 1; }\nname=\"orca-${recipe_id:0:max_recipe_id_length}-${instance_id}\"\n\n# Arm cleanup BEFORE create so a failing create can't leak a half-built paid sandbox.\ncleanup_on_error() { [ \"$?\" -ne 0 ] && vercel sandbox remove \"$name\" \"${vercel_args[@]}\" >/dev/null 2>&1 || true; }\ntrap cleanup_on_error EXIT\n\n# 1. boot from the authenticated snapshot, publish the serve port\ncreate_output=\"$(vercel sandbox create --name \"$name\" --snapshot \"$snapshot_id\" \\\n --timeout 30m --publish-port \"$port\" \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$create_output\" >&2\n# Vercel prints the published https URL; derive the external wss:// pairing address from it\npublic_url=\"$(printf '%s\\n' \"$create_output\" | sed -nE 's#.*(https://[^[:space:]]+\\.vercel\\.run).*#\\1#p' | head -1)\"\n[ -n \"$public_url\" ] || { echo \"no published URL in create output\" >&2; exit 1; }\npairing_ws=\"${public_url/https:\\/\\//wss://}\"\n\n# 2. (remote) ensure the repo is at the right commit; rebuild only if the commit changed (cache marker)\nvercel sandbox exec \"$name\" \"${vercel_args[@]}\" --timeout 20m \\\n --env \"GH_TOKEN=$gh_token\" --env \"ORCA_PROJECT_ROOT=$project_root\" \\\n --env \"ORCA_REPO_URL=$repo_url\" --env \"ORCA_REPO_REF=$repo_ref\" \\\n -- bash -lc 'set -euo pipefail; cd \"$ORCA_PROJECT_ROOT\"; \\\n # Re-establish git auth for the private-repo fetch (why + full rationale: §5); else it hangs on a prompt.\n # Load-bearing escaping: \\$1 and \\$GH_TOKEN must land LITERALLY and resolve at git-runtime. Test after\n # any edit here — reformatting the nested printf/node quoting silently breaks the fetch or leaks the token.\n if [ -n \"${GH_TOKEN:-}\" ]; then \\\n printf \"%s\\n\" \"#!/usr/bin/env bash\" \"case \\\"\\$1\\\" in *Username*) echo x-access-token;; *Password*) echo \\\"\\$GH_TOKEN\\\";; esac\" > /tmp/askpass.sh; \\\n chmod 700 /tmp/askpass.sh; export GIT_ASKPASS=/tmp/askpass.sh GIT_TERMINAL_PROMPT=0; fi; \\\n git fetch origin \"$ORCA_REPO_REF\"; \\\n git checkout -B \"$ORCA_REPO_REF\" FETCH_HEAD; \\\n rm -f /tmp/askpass.sh; \\\n c=\"$(git rev-parse HEAD)\"; [ -f .orca-built ] && [ \"$(cat .orca-built)\" = \"$c\" ] || { \\\n pnpm install --prefer-offline && pnpm run build:cli && \\\n node config/scripts/run-electron-vite-build.mjs --config config/electron-vite.vm-serve.config.ts && \\\n printf \"%s\" \"$c\" > .orca-built; }' >&2\n\n# 3. (remote) start orca serve in the background, writing recipe JSON to a file; poll until it parses\nrecipe_json=\"$(vercel sandbox exec \"$name\" \"${vercel_args[@]}\" --timeout 60s \\\n --env \"ORCA_PORT=$port\" --env \"ORCA_PROJECT_ROOT=$project_root\" --env \"ORCA_PAIRING_ADDRESS=$pairing_ws\" \\\n -- bash -lc 'set -euo pipefail; cd \"$ORCA_PROJECT_ROOT\"; rm -f /tmp/orca-recipe.json /tmp/orca-serve.log; \\\n nohup pnpm exec orca-dev serve --port \"$ORCA_PORT\" --project-root \"$ORCA_PROJECT_ROOT\" \\\n --pairing-address \"$ORCA_PAIRING_ADDRESS\" --recipe-json >/tmp/orca-recipe.json 2>/tmp/orca-serve.log /dev/null 2>&1 && { cat /tmp/orca-recipe.json; exit 0; }; \\\n kill -0 \"$pid\" 2>/dev/null || { cat /tmp/orca-serve.log >&2; exit 1; }; sleep 0.25; \\\n done; cat /tmp/orca-serve.log >&2; echo \"serve recipe JSON timed out\" >&2; exit 1')\"\n\n# 4. print serve's JSON enriched with userData (single object on stdout)\nnode -e 'const p=JSON.parse(process.argv[1]); console.log(JSON.stringify({...p, schemaVersion:1,\n userData:{...p.userData, provider:\"vercel-sandbox\", resourceId:process.argv[2], snapshotId:process.argv[3]}}))' \\\n \"$recipe_json\" \"$name\" \"$snapshot_id\"\ntrap - EXIT\n```\n\n`suspend`/`resume`/`destroy` use `vercel sandbox stop|...|remove \"$resource_id\"` reading\n`userData.resourceId` from stdin (§7d). This is the **Orca-server** connection mode (the recipe emits a\npairing URL). If the user chose **SSH** in the §1 interview, use §7g instead.\n\n### 7g. Worked example — existing SSH host (SSH connection mode)\n\nSSH mode is **fundamentally different from §7c/§7f**, not a relabeling of them:\n\n- **`create` does NOT run `orca serve` and does NOT emit a `pairingCode`.** Orca itself connects to the\n host over its SSH relay, brings up the git + filesystem providers, and imports the repo. The script's\n only job is to make the host ready and **print SSH connection details** Orca will dial.\n- The result uses a `connection` block with `type: \"ssh\"` and a `target`, **not** the flat\n `pairingCode`/`projectRoot` shape. Exact shape (Orca rejects anything else):\n\n```json\n{\n \"schemaVersion\": 1,\n \"connection\": {\n \"type\": \"ssh\",\n \"projectRoot\": \"/abs/path/to/repo/on/host\",\n \"target\": {\n \"label\": \"my-box\",\n \"host\": \"192.0.2.10\",\n \"port\": 22,\n \"username\": \"ubuntu\",\n \"identityFile\": \"~/.ssh/id_ed25519\",\n \"jumpHost\": \"bastion.example.com\",\n \"proxyCommand\": \"cloudflared access ssh --hostname %h\",\n \"relayGracePeriodSeconds\": 0,\n \"portForwards\": []\n }\n }\n}\n```\n\n`label`, `host`, `port`, `username` are required; the rest are optional — omit any you don't need.\n\nFor an explicitly requested one-VM-per-workspace checkout, the create script must read\n`ORCA_RECIPE_RESULT_SCHEMA_VERSION`, `ORCA_REPO_URL`, `ORCA_REPO_REF`, `ORCA_REPO_REF_HEAD`, and\n`ORCA_REPO_BRANCH`. Use `ORCA_REPO_REF` to fetch the selected source, but create\n`ORCA_REPO_BRANCH` at the exact `ORCA_REPO_REF_HEAD` commit; resolving the symbolic ref again can race\nwith an upstream update. `ORCA_REPO_URL` and `ORCA_REPO_REF` are a matched fetch pair, including when\nthe desktop source uses multiple remotes. Return that primary checkout at `projectRoot` and emit the\nsame SSH result with:\n\n```bash\n[ -n \"${ORCA_REPO_REF_HEAD:-}\" ] || { echo \"missing pinned source commit\" >&2; exit 1; }\ngit fetch origin \"$ORCA_REPO_REF\"\ngit cat-file -e \"${ORCA_REPO_REF_HEAD}^{commit}\"\ngit checkout -B \"$ORCA_REPO_BRANCH\" \"$ORCA_REPO_REF_HEAD\"\n```\n\n```json\n{\n \"schemaVersion\": 2,\n \"checkoutMode\": \"provisioned-root\",\n \"connection\": {\n \"type\": \"ssh\",\n \"projectRoot\": \"/abs/repo\",\n \"target\": { \"label\": \"my-box\", \"host\": \"192.0.2.10\", \"port\": 22, \"username\": \"ubuntu\" }\n }\n}\n```\n\nFail if the requested schema is not `2`; do not silently fall back to the ordinary recipe shape.\n\n**Networking → which `target` fields to set** (how *your desktop* reaches the box — there is no\n`orca serve` URL in SSH mode):\n\n- Public IP / DNS, or a Tailscale/VPN address → `host`; SSH port → `port` (usually 22).\n- Key auth → `identityFile` (add `identitiesOnly: true` if the agent has many keys).\n- Through a bastion → `jumpHost` (a `user@host` ProxyJump) **or** a full `proxyCommand` (e.g. an access\n proxy). Use one, not both.\n- A service port the workspace needs → add entries to `portForwards`.\n- `relayGracePeriodSeconds` (optional): how long Orca keeps the SSH relay alive after the workspace\n detaches before tearing it down; `0` = tear down immediately. Leave it off unless the user wants a\n reconnect grace window.\n\n**Toolchain & agent auth on a persistent (no-snapshot) host — do this ONCE, by hand, before wiring the\nrecipe** (there's no base image to bake; the host *is* the base). Run the §7f Phase-2 install steps and\nthe §7f Phase-3 ` login --device-auth` **directly over SSH on the host** (interactive, e.g.\n`ssh -t user@host ' login --device-auth'`). After that the host stays ready across workspaces.\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve from env→state→fallback (default unset optionals to \"\"): ssh_username, host,\n# ssh_port (default 22), identity_file, jump_host, proxy_command, project_root, repo_url, repo_ref\n: \"${identity_file:=}\"; : \"${jump_host:=}\"; : \"${proxy_command:=}\" # avoid set -u aborts on optionals\ngh_token=\"${GH_TOKEN:-${GITHUB_TOKEN:-$(command -v gh >/dev/null 2>&1 && gh auth token 2>/dev/null || true)}}\"\nssh_target=\"${ssh_username}@${host}\"\nssh_opts=(-p \"$ssh_port\"); [ -n \"$identity_file\" ] && ssh_opts+=(-i \"$identity_file\")\n# Why: a fresh host's key isn't in known_hosts; a StrictHostKeyChecking prompt would HANG a\n# non-interactive create. Pre-add the key (or set the option) so it can't block.\nssh-keyscan -p \"$ssh_port\" \"$host\" >> \"$HOME/.ssh/known_hosts\" 2>/dev/null || true\n\n# 1. ensure the repo is present and at the right commit on the host (NO orca serve here)\nssh \"${ssh_opts[@]}\" \"$ssh_target\" \\\n \"GH_TOKEN='$gh_token' GIT_TERMINAL_PROMPT=0 bash -lc '\n set -euo pipefail\n [ -d \\\"$project_root/.git\\\" ] || git clone \\\"$repo_url\\\" \\\"$project_root\\\"\n cd \\\"$project_root\\\" && git fetch origin \\\"$repo_ref\\\" && git checkout -B \\\"$repo_ref\\\" FETCH_HEAD\n '\" >&2\n\n# 2. print the SSH connection block (NO pairingCode, NO orca serve). host/port/username tell Orca's\n# relay how to dial in; identityFile/jumpHost/proxyCommand/portForwards are emitted when set.\nnode -e 'const [host,port,user,idf,jh,pc,root]=process.argv.slice(1);\n const target={ label:\"per-workspace-host\", host, port:Number(port), username:user };\n if(idf) target.identityFile=idf; if(jh) target.jumpHost=jh; if(pc) target.proxyCommand=pc;\n // add target.portForwards=[...] here if the workspace needs forwarded service ports\n console.log(JSON.stringify({ schemaVersion:1, connection:{ type:\"ssh\", projectRoot:root, target } }))' \\\n \"$host\" \"$ssh_port\" \"$ssh_username\" \"$identity_file\" \"$jump_host\" \"$proxy_command\" \"$project_root\"\n```\n\n`suspend`/`resume`/`destroy`: on a persistent host there's usually nothing to tear down — set\n`destroy: none` and omit suspend/resume. (Orca still disconnects/reconnects its own SSH relay on\nsleep/wake/delete — that's separate from these scripts.)\n\nIf the SSH host is instead an **ephemeral/snapshot-capable VM** (your hypervisor, or a cloud VM with\nimage support), keep the §7f Phase-2/3 base-image model for provisioning, but still emit the\n`connection.type:\"ssh\"` block above instead of starting `orca serve`.\n\n### 7h. Worked example — local Docker SSH (SSH connection mode)\n\nLocal Docker can model an ephemeral SSH VM without cloud cost: build a base image with `sshd`, tools,\nrepo prerequisites, and the agent CLI; run an **interactive auth container** once; then `docker commit`\nthat container as the authenticated image used by per-workspace `create`.\n\nKey points:\n\n- Publish container SSH to a random localhost port (`-p 127.0.0.1::22`) and emit\n `connection.type:\"ssh\"` with `host:\"127.0.0.1\"`, that port, `username`, `identityFile`, and\n `identitiesOnly:true`.\n- Generate a repo-local SSH key if needed, but gitignore the private/public key files.\n- **Bake SSH host keys into the base image** (`ssh-keygen -A` at **build** time; at runtime only generate\n if absent). Ephemeral containers all present the **same** host key, so `known_hosts` on `127.0.0.1`\n doesn't churn as the published port rotates across workspaces (otherwise every container's freshly\n generated key collides on `localhost` and trips host-key-changed warnings).\n- The auth image is the Docker equivalent of Phase 3: the **user** runs the agent login **inside** the\n container (you can't drive it — you have no interactive TTY), configures proxy env/config, approves\n hooks, and you commit once they report it's done. On a headless container use the **device-auth** flow\n (§4). Verify login before committing — exit code, or fold stderr and match the exact success line (§4).\n- Do not bind-mount or copy the host's full agent home into the image. Let each container have writable\n agent state; only the committed auth image should carry reusable authenticated state.\n- If committing from an interactive shell, force the runtime entrypoint back to `sshd`:\n `docker commit --change='ENTRYPOINT [\"/usr/local/bin/orca-docker-ssh-entrypoint\"]' …`.\n- `destroy` should read `recipeResult.userData.resourceId` and run `docker rm -f \"$resource_id\"`.\n\nValidation before wiring/live use:\n\n```bash\ndocker image inspect \"$auth_image\" --format '{{json .Config.Entrypoint}}'\ndocker run -d --name \"$name\" -p 127.0.0.1::22 -e \"ORCA_SSH_PUBLIC_KEY=$pubkey\" \"$auth_image\"\ndocker ps -a --filter \"name=$name\"\ndocker logs \"$name\"\nssh -i \"$key\" -p \"$port\" -o IdentitiesOnly=yes user@127.0.0.1 'codex --version'\n```\n\nIf the container exits immediately, inspect logs before the cleanup trap removes it; a committed\ninteractive image with `ENTRYPOINT [\"bash\"]` is a common cause.\n\nAlso confirm the **host key is stable** across containers: the SSH `ssh -i … 127.0.0.1` dial should not\ntrigger a host-key-changed warning when a second container reuses the port. If it does, the host keys\nweren't baked into the base image (see the `ssh-keygen -A` point above).\n\n### 7i. Windows local-side scripts\n\nThe local-side scripts run on the user's desktop. On **Windows**, a bare `.sh` won't execute. Either\nrequire WSL/Git-Bash (and point `orca.yaml` at e.g. `bash ./scripts/orca-vm/.sh` via a `.cmd`\nlauncher), or scaffold PowerShell equivalents. Minimal PowerShell shape:\n\n```powershell\n#requires -Version 5\n$ErrorActionPreference = 'Stop'\n# resolve env→state→fallback; run the provider CLI / ssh the same way;\n# capture provider output; build the result object for the chosen mode and write ONE line of JSON to stdout.\n# Orca-server mode: @{ schemaVersion=1; pairingCode=$pairingCode; projectRoot=$projectRoot; userData=@{...} }\n# SSH mode: @{ schemaVersion=1; connection=@{ type=\"ssh\"; projectRoot=$projectRoot;\n# target=@{ label=$label; host=$host; port=$port; username=$user } } } (see §7g/§7h)\n($result | ConvertTo-Json -Compress -Depth 6)\n# progress/errors → Write-Error / the error stream, never stdout.\n```\n\nThe remote-side commands you run *inside* the Linux VM stay bash regardless of the desktop OS.\n\n---\n\n## 8. Per-workspace recipe contract (the fast path)\n\nOnce the authenticated snapshot exists, this runs on every workspace create. Define recipes in\n`orca.yaml`:\n\n```yaml\nenvironmentRecipes:\n - id: cloud-sandbox\n name: Cloud Sandbox\n create: ./scripts/orca-vm/cloud-sandbox-create.sh\n suspend: ./scripts/orca-vm/cloud-sandbox-suspend.sh\n resume: ./scripts/orca-vm/cloud-sandbox-resume.sh\n destroy: ./scripts/orca-vm/cloud-sandbox-destroy.sh\n```\n\n`create` runs **locally from the repo root** and prints **one** JSON object to stdout. Its shape depends\non the connection mode chosen in §1:\n\n**Orca-server mode** — boot the env, start `orca serve` in it, and print serve's result:\n\n```json\n{\n \"schemaVersion\": 1,\n \"pairingCode\": \"orca-pairing-code-or-url\",\n \"projectRoot\": \"/absolute/path/to/repo/on/remote\",\n \"userData\": { \"provider\": \"example\", \"resourceId\": \"provider-resource-id\" }\n}\n```\n\nHere `pairingCode` (from `orca serve --recipe-json`) and `projectRoot` are required; `schemaVersion` (`1`)\nand `userData` are optional.\n\n**SSH mode** — do **not** run `orca serve`; print the `connection.type:\"ssh\"` block instead (full shape +\nworked script in §7g). `pairingCode` is **not** used in SSH mode.\n\n**Optional provisioned root** — only for direct SSH and only when explicitly requested. Add\n`checkoutMode: provisioned-root` to the recipe, require `ORCA_RECIPE_RESULT_SCHEMA_VERSION=2`, create\nthe requested `ORCA_REPO_BRANCH` at the pinned `ORCA_REPO_REF_HEAD` commit (use `ORCA_REPO_REF` only\nto fetch that commit) at the returned `projectRoot`, and emit schema version 2 with\n`checkoutMode: \"provisioned-root\"`. All recipes without this field retain the schema-v1 behavior above.\n\nLifecycle hooks (all run locally):\n\n- `create`: required. Prints recipe result JSON.\n- `suspend`: optional. Sleep; reads lifecycle payload on stdin.\n- `resume`: optional. Wake; reads payload on stdin and **prints fresh recipe JSON** (pairing may change).\n- `destroy`: optional unless `destroy: none`. Delete/cleanup; reads payload on stdin.\n\nStart Orca remotely with `orca serve --port \"$PORT\" --project-root \"$ABS_ROOT\" --pairing-address\n\"$EXTERNAL_WSS_URL\" --recipe-json` (exact flags + output in §7c). Set `--pairing-address` to the\nexternally reachable address so the emitted `pairingCode` is reachable; tunneling/port mapping is the\nscript's job.\n\nBackward compatibility: `command`→`create`, `cleanup`→`destroy`, `cleanup: none`→`destroy: none`.\nPrefer the lifecycle names.\n\n---\n\n## 9. Doctor and validation\n\nValidate in two stages — the cheap dry run first, then the live self-test.\n\n### Dry run (free, non-destructive) — always do this first\n\n`orca vm recipe doctor --repo-path --json` validates **static wiring only** — it does\n**not** boot anything. It checks: local-host execution (v1), repo path, recipe id exists,\ncreate/destroy/suspend/resume command paths resolve, suspend/resume are paired, and each script is\nexecutable (POSIX exec bit; skipped on Windows). Fix every failure here before spending any cloud money.\n\n### Live self-test (`--provision`) — diagnose and iterate yourself\n\n`orca vm recipe doctor --repo-path --provision --json` actually runs the recipe end\nto end: it executes `create`, validates the returned recipe JSON, then runs `destroy` to **tear the\nenvironment back down** (so the test leaves nothing running, as long as `destroy` works). It spends real\ncloud money, so get the user's OK **once** before starting — that one approval covers the whole loop\nbelow; do not re-ask before each run.\n\nOn failure, the JSON result includes a `provisionTranscript` with the **complete** captured output of\neach stage so you can self-diagnose without asking the user to relay logs:\n\n```json\n{\n \"ok\": false,\n \"checks\": [ { \"id\": \"recipe.provision\", \"status\": \"fail\", \"message\": \"…\" } ],\n \"provisionTranscript\": {\n \"provision\": { \"exitCode\": 0, \"signal\": null, \"stdout\": \"…\", \"stderr\": \"…\", \"parseError\": \"…\" },\n \"destroy\": { \"exitCode\": 0, \"signal\": null, \"stdout\": \"…\", \"stderr\": \"…\" }\n }\n}\n```\n\n**Run it as a loop:** read `provisionTranscript.provision.stderr` / `.stdout` / `.parseError` (and\n`destroy.*`), fix the script, and re-run `--provision` until `ok` is `true` — iterating on your own\nrather than waiting for the user to paste errors. Common reads: a non-empty `stderr` with `exitCode 0`\nplus a `parseError` means `create` ran but printed something other than the single recipe-result JSON on\nstdout (often a stray `echo` — route it to stderr, see §10); a non-zero `exitCode` is a provider/script\nfailure described in `stderr`. Each stream is redacted and capped (head+tail) — large logs keep both the\nsetup context and the failure.\n\nThe self-test cannot see provider-side truth beyond what the scripts print, so still confirm: state has a\npopulated **authenticated** `snapshotId` (Phases 2–3 done), and `destroy` is implemented/tested (or\nexplicitly `none` — in which case the self-test won't tear down, so clean up manually).\n\nFor SSH recipes, also smoke-test the exact emitted target before declaring success: dial the host/port\nwith the identity/proxy settings, run `pwd`, verify the repo path, check the agent binary, and confirm\n`destroy` removes the provider resource/container. For Docker, inspect the auth image entrypoint and do a\nstartup-only `docker run` before the full clone/install path.\n\n---\n\n## 10. Failure modes\n\n- **Build exceeds plan timeout (e.g. Hobby 45m).** Use enough vCPUs and a timeout covering the build;\n else split work or use a higher plan. The cap also limits per-workspace runtime — surface it.\n- **Build exceeds plan RAM.** Build the **headless main only** (drop the renderer) — the biggest fitter.\n- **Private-repo clone hangs/fails.** Wrong/missing token. Use `GIT_ASKPASS` + `GIT_TERMINAL_PROMPT=0`\n so it fails fast instead of prompting.\n- **`GIT_ASKPASS` helper aborts the clone with \"`$1: unbound variable`\".** The `printf`/heredoc that writes\n the helper inside `bash -lc` under `set -u` expanded `$1`/`$GH_TOKEN` at **write** time. Escape them\n (`\\$1`, `\\$GH_TOKEN`) so they land literally and resolve at git-runtime; this also keeps the real token\n out of the file. `rm -f` the helper afterward (§5, §7f).\n- **Agent verified as \"not logged in\" despite a good login.** `codex login status` (and similar) print\n \"Logged in …\" to **stderr**; an stdout-only `grep` misses it. Prefer the status **exit code**; if you\n grep, fold stderr first (`status 2>&1 | grep …`) and match the exact success line — not `grep -qi\n 'logged in'`, which also matches \"not logged in\".\n- **Headless agent login hangs.** Plain OAuth `login` starts a loopback callback server on a VM/container\n port the host browser can't reach. Use the **device-auth** flow (`login --device-auth`) — it prints a\n URL + code the user opens on the host.\n- **`known_hosts` host-key churn on local Docker.** Each ephemeral container regenerating its SSH host key\n collides on `127.0.0.1` as the published port rotates. Bake host keys into the base image at build time\n (`ssh-keygen -A`; runtime generates only if absent) so all containers share one stable key (§7h).\n- **Snapshot expired/evicted.** If `create` hits an unknown snapshot id, rerun Phases 2–3 and update\n `snapshotId`.\n- **Agent auth didn't persist.** Confirm `snapshotId` points at the **authenticated** snapshot; re-run\n Phase 3. Warn that short-lived tokens may need periodic re-auth.\n- **Agent auth copied from the host breaks.** Do not bind-mount/copy a full host agent home; sqlite\n files can be unwritable or host-specific, hooks may need approval again, and config may reference\n local-only env vars. Authenticate inside the runtime and snapshot/commit that layer.\n- **Docker auth image exits immediately.** Inspect `docker image inspect … .Config.Entrypoint` and\n `docker logs`. If the image was committed from an interactive shell, reset the entrypoint to the SSH\n entrypoint during `docker commit`.\n- **Leaked paid resource.** Every long script must trap errors and remove the sandbox it created.\n- **`create` emits non-JSON on stdout.** A stray `echo` corrupts the result — stdout is for the final\n JSON only; everything else to stderr. The `--provision` self-test surfaces this as `exitCode 0` + a\n `parseError` with the offending stdout in `provisionTranscript` (§9).\n\n---\n\n## 11. Boundaries\n\n- Don't create accounts, choose plans/regions, or invent scope/project/org/image/billing ids.\n- Don't invent or store credentials; no secrets in `userData`, state, comments, docs, or commits.\n- Don't run paid/long phases (base snapshot, auth, live test) without an explicit OK.\n- Don't hide provider errors behind generic messages — preserve actionable stderr.\n- Don't make Orca own provider lifecycle beyond invoking the configured scripts.\n- Don't commit or create an Orca workspace unless asked.\n" // oxfmt-ignore -const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Use Orca orchestration for structured multi-agent coordination: threaded\n messages, blocking ask/reply flows, task dispatch, worker_done/escalation\n waits, task DAGs, decision gates, coordinator loops, or decomposing work\n across agents. Use `orca-cli` instead for full ownership handoffs, including\n requests phrased as \"hand off\", \"handoff\", \"handover\", \"give this to another\n agent\", or \"another worktree\" when the user did not explicitly ask to\n supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for\n ordinary terminal control, lightweight terminal prompts, shell commands, Orca\n worktree management, reading or waiting on terminals, and automation of the\n browser embedded inside Orca. Use Computer Use for browser windows, webviews,\n Orca app UI, or desktop UI outside Orca's embedded browser.\n---\n\n# Orca Inter-Agent Orchestration\n\nOrchestration is Orca's structured coordination layer for agent messages, task ownership, dispatch state, and worker completion tracking.\n\nUse this skill when coordination state matters. For lightweight terminal prompts or basic worktree/terminal/built-in-browser control, use `orca-cli`.\n\n## Tool Boundary\n\nIf a task says to use Orca orchestration, the coordinator must create or bind a Run, create the Task with `orca orchestration task-create`, then attach the worker with either the preferred `orca orchestration worker-start` composition or the low-level `orca orchestration dispatch --inject` path.\n\nDo not substitute non-Orca subagent tools, generic agent-spawn APIs, or chat-only parallel worker features. Those may create useful workers, but they do not create Orca task/dispatch provenance, injected lifecycle preambles, `worker_done` authority, or decision gates.\n\nBefore claiming a worker was orchestrated, verify the task/dispatch exists:\n\n```bash\norca orchestration task-list --json\norca orchestration dispatch-show --task --json\n```\n\nIf the work was accidentally run outside Orca orchestration, say so plainly. To repair provenance, rerun or revalidate the needed work through a fresh Orca terminal plus injected dispatch; do not retroactively describe the external worker as orchestrated.\n\n## When To Use\n\n- Send/reply/ask between agent terminals with persistent messages.\n- Dispatch structured tasks to workers and wait for `worker_done` or `escalation`.\n- Track task DAGs with dependencies.\n- Run coordinator loops or decision gates.\n\nDo not use orchestration merely because the user says \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or asks for another worktree/agent/model/effort. Those are full ownership transfers unless the user explicitly asks to supervise, monitor, wait for worker completion/results, coordinate a DAG, use decision gates, or keep a blocking ask/reply loop.\n\n## Preconditions\n\n- `orca status --json` should show a running runtime.\n- `orca` must be on PATH (`orca-ide` on Linux).\n- The orchestration experimental feature must be enabled in Settings > Experimental.\n- `orca orchestration` commands are RPC calls to the running Orca runtime.\n\n## Contract Migration\n\nOrca adopts a live pre-update orchestration assignment into an ordinary Run. Adoption preserves the existing agent process, PTY/session, terminal handle, tab/leaf/pane, worktree or folder workspace, Task, and Dispatch; it never restarts or replaces the worker. The retired scheduler is not revived, and a newly created attempt uses the current grammar.\n\nTreat the authority label on injected or formatted messages as definitive:\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported command printed with the message, using the same CLI executable and arguments that the original prompt supplied.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded, at-least-once cutover replay. Process it idempotently and acknowledge it only through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or lifecycle action.\n- An unlabeled current message uses the current guide and current grammar.\n\nAn explicitly selected current Run, attested current Run binding, current Dispatch, or federated attachment takes precedence over legacy fallback. A retained adoption record alone never turns a current command into a legacy call.\n\nDatabase provenance, an old-looking terminal, or a legacy Run ID does not prove mutation authority. If the runtime cannot prove liveness, principal ownership, capability, or the exact legacy contract, it degrades to read-only inspection and must not fall back to local execution. Exact recovery may restore the already-live PTY once in its original inactive background tab. It must not spawn, write, signal, stop, switch, focus, split, or inject a terminal. Loss of lifecycle authority does not invalidate the existing assignment, process, or filesystem work.\n\nCompatibility retries have narrow guarantees. A pending ask, a reply, a final Dispatch settlement, and a consuming check have durable recovery identities. A-era heartbeat and escalation calls remain at-least-once across a manual A-to-B retry because identical later signals may be intentional. If an A-era ask may already have been answered, run the exact non-consuming recovery check printed by the runtime first; after its answer is printed and acknowledged, a new invocation with the same question creates a new question. Never guess among multiple identical question threads.\n\nWhen a compatibility or recovery command returns structured next-step arguments, run those exact arguments with the same CLI executable. The arguments intentionally omit the executable name so the guidance works with `orca`, `orca-ide`, `orca-dev`, or another configured Orca CLI command. Do not translate the command from memory, broaden its recipient, or retry it as a current mutation unless the returned guidance explicitly says to.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The initial command durably commits the question, prints its exact `ask --resume ` command, and exits with launcher status `75`; it does not wait for the answer. Run that exact resume command after the launcher or update boundary. Resume is idempotent and read-oriented: it waits for the already-committed question and does not create another one. For a WSL process that received compatibility proof at launch, use the printed executable `orca-ide` WSL resume command so the same distro and packaged launcher authority are preserved; do not substitute a PATH-resolved local CLI. Older WSL processes that never received the hidden launch token remain lifecycle read-only after the update, even while their terminal and filesystem work continue.\n\nLegacy inspection remains available without consuming mail:\n\n```bash\norca orchestration run-list --json\n# run_legacy_local is an empty audit tombstone after adoption.\norca orchestration run-show --id run_legacy_local --json\n# In run-list, find the ordinary Run whose objective is:\n# \"Recovered orchestration work from a contract update\"\norca orchestration run-show --id --json\norca orchestration task-list --run --json\norca orchestration inbox --full --json\norca orchestration check --terminal --peek --format --json\norca terminal read --terminal --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\nIf the original coordinator is unavailable or cannot prove its retained authority, a current coordinator may explicitly take over the adopted Run from its own live agent terminal:\n\n```bash\norca orchestration run-use --id --takeover-legacy --json\norca orchestration check --run --json\n```\n\nTakeover fences only the old coordinator, binds the current one, and moves pending worker mail into current Run Delivery. It is bound to the authenticated invoking terminal; `--from` cannot name another coordinator. Live legacy workers keep their original Tasks, Dispatches, processes, filesystems, and old prompt commands; their later questions, escalations, and completion reports route to the current coordinator. Do not use takeover while the original coordinator is still actively coordinating, because its later lifecycle mutations are rejected.\n\nDo not launch a replacement editor merely because the desktop app or runtime was updated. If adoption cannot prove continuing authority, keep the original worker as the only editor until it reaches a stable handoff point, then use a new current Dispatch in a conflict-free placement for any remaining work.\n\n## Ownership\n\nNew orchestration messages and tasks belong to one explicitly bound Run. A Run is only a durable namespace and coordinator inbox; it never schedules or places workers. Lifecycle authority comes from the active Dispatch, and terminal handles remain routing metadata rather than durable identity. Send `worker_done` and `heartbeat` from the worker's own terminal; Orca routes them to that Dispatch's Run.\n\nClassify inherited context before sending lifecycle messages:\n\n- Coordinated subtask: a live coordinator owns the DAG and waits on this dispatch. Follow the preamble exactly, including `worker_done`, heartbeat/status, `ask`, and `escalation`.\n- Full handoff means ownership transfer, not supervised dispatch. The original actor is not monitoring a DAG, so do not create lifecycle obligations unless the user explicitly asks you to supervise.\n- Classify requests containing \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"another agent\", or \"another worktree\" as full handoffs by default, even when the user names a custom model or reasoning effort.\n- Use supervised orchestration only when the user explicitly asks you to \"supervise\", \"monitor\", \"wait\", \"track completion\", \"wait for worker_done\", return results, coordinate a DAG, use a decision gate, or manage ask/reply flow.\n- Do not use `orca orchestration dispatch --inject` for full handoffs. It injects a coordinator preamble that tells the worker to send `worker_done`, heartbeat, and `ask` messages, then end its turn under the original terminal's dispatch lifecycle.\n- Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. Do not peek at terminal output after prompt delivery to monitor progress.\n- A review-only `worker_done` reports findings; it does not authorize coordinator file edits. After a review-only completion, synthesize findings, ask a decision gate if ownership is unclear, and dispatch or hand off fixes unless the user explicitly asked the coordinator to own fixes.\n- If the user's plan names a next owner agent (for example, \"then use opencode to create a PR\"), post-review corrections and PR prep belong to that named owner. The coordinator routes, synthesizes, asks decision gates when needed, and supervises; the named owner edits files and creates the PR.\n\nIf unclear, inspect orchestration state before sending lifecycle messages:\n\n```bash\norca orchestration task-list --json\norca terminal list --json\n# If inherited context includes a task id:\norca orchestration dispatch-show --task --json\n```\n\n## Messaging\n\n```bash\norca orchestration send --subject [--to ] [--from ] [--body ] [--type ] [--priority ] [--thread-id ] [--payload ] [--json]\norca orchestration check [--terminal ] [--ack ] [--peek|--all] [--types ] [--format] [--wait] [--timeout-ms ] [--json]\norca orchestration reply --id --body [--from ] [--json]\norca orchestration ask (--question |--resume ) [--options ] [--timeout-ms ] [--from ] [--json]\norca orchestration inbox [--limit ] [--json]\n```\n\nRules:\n\n- Omit `--from` unless impersonating another terminal; Orca auto-resolves it from the current terminal.\n- A coordinator `check` returns the bound Run's oldest FIFO Delivery (up to 50 messages) and replays that exact batch until `--ack `. Process every message before acknowledging; `check --ack --wait` acknowledges, checks, and waits in one operation.\n- Use `--peek` and `--all` only for read-only history/debugging. Type filters decide when a waiter wakes; the returned actionable Delivery is still the oldest full batch.\n- Use `dispatch:` for coordinator guidance to one supervised worker. Orca routes that stable address locally or through the connected-server relay; do not substitute a remote terminal handle.\n- Terminal handles remain appropriate for low-level pre-Dispatch messaging. Prefer `agentTerminalHandle` from the create response, fall back to `startupTerminal.handle` for older runtimes, then re-resolve with `orca terminal list --worktree ... --json` if missing or stale. Continue with the replacement handle only; never dual-send to old and new handles.\n- `terminal list --json` omits `visualLayouts` because handle recovery does not need topology. Add `--include-visual-layouts` only for explicit tab and pane inspection.\n- `orca orchestration check --peek --format --json` returns locally formatted unread mail without consuming it; it never writes to terminal input or remotely wakes another terminal. Use `orchestration dispatch --inject` to deliver a tracked task, or `terminal send` when an existing agent needs a free-form prompt.\n- While supervising workers manually, use `check --wait --types worker_done,escalation,question --timeout-ms ` instead of sleep/poll loops. Process the whole Delivery, reply to `question` messages with `orca orchestration reply --id --body --json`, then acknowledge and keep waiting.\n- Treat a `check --wait` timeout or `{count:0}` as a checkpoint, not a worker failure. Long coding tasks routinely run 15-60 minutes; keep using rolling waits unless you receive `worker_done`/`escalation`, the terminal exits or disappears, or the user explicitly asks you to stop.\n- Heartbeats and visible terminal activity mean the worker is alive, not done. Do not stop, close, kill, or restart a worker just because it has not produced a completion message yet.\n- Use `ask` when a worker needs a blocking answer from the coordinator; it defaults to the active Dispatch's Run. Timeout or disconnect leaves the question pending, so resume by its original message ID instead of asking again.\n- `check --wait` returns one bounded Delivery, not every future completion. Process every message, acknowledge it, then keep waiting until every expected Dispatch settles.\n- Group addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, `@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`.\n- Message types include `status`, `dispatch`, `worker_done`, `merge_ready`, `escalation`, `handoff`, `question`, `decision_gate` (legacy/gates), and `heartbeat`.\n- Use group addresses only for messages that are genuinely useful to many terminals, such as `status` broadcasts or intentional fan-out questions. Do not send dispatch lifecycle messages to groups.\n- `worker_done` belongs to the active Dispatch and defaults to its Run mailbox; never target a group.\n- A valid `worker_done` for the active `taskId` + `dispatchId` marks the task and dispatch completed automatically. Do not follow it with `task-update --status completed`; reserve manual updates for explicit recovery or overrides.\n- `heartbeat` is also Dispatch-scoped. Include both IDs and omit `--to` so Orca uses the owning Run; use `status` for broad progress updates.\n\n## Tasks And Dispatch\n\nA Run is the namespace/inbox, a Task is the work item, and a Dispatch assigns one Task attempt to a terminal. Create or bind a Run once before the common loop.\n\n```bash\norca orchestration run-create --objective --json\norca orchestration task-create --spec [--deps ] [--parent ] [--json]\norca orchestration task-list [--status ] [--ready] [--brief] [--json]\norca orchestration task-update --id --status [--result ] [--json]\norca orchestration dispatch --task --to [--from ] [--inject] [--json]\norca orchestration dispatch-show --task [--json]\n```\n\nTask statuses: `pending`, `ready`, `dispatched`, `completed`, `failed`, `blocked`.\n\nDispatch rules:\n\n- `--inject` sends the task spec plus preamble into a recognized agent CLI so it can report `worker_done`.\n- If the target is a bare shell, omit `--inject`, dispatch for tracking if needed, then send the prompt manually with `orca terminal send --terminal --text --enter --json`.\n- After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed.\n- Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag.\n\n## How deep workers can nest\n\nA dispatched worker normally cannot dispatch sub-workers. Attempting it fails with\n`nested_worker_depth_exceeded` and a message telling the worker to complete the task\nitself. Do that — do not try to route around it.\n\nThe limit is a number, not an on/off switch. `Settings -> Agents -> Nested worker depth`\nsets how many generations are allowed:\n\n- `1` (default): a coordinator dispatches workers; those workers do not dispatch.\n- `2`: workers may dispatch one further generation.\n\nDepth is counted from the terminal that issues the command, not from the Run. Creating a\nnew Run does not reset it — a worker that runs `run-create` then `worker-start` is still a\nworker, and still counted. This is the part that changed: the old behaviour rejected\nsub-dispatch only because a worker's terminal was not bound to a Run, so creating a Run was\nenough to slip past it.\n\nTwo limits worth knowing:\n\n- **It is a guardrail, not a security boundary.** A caller that declares another terminal's\n handle while its own launch evidence is unverifiable (an ordinary restored terminal, for\n example) can be counted as that terminal instead. Orca does not treat workers as hostile.\n- **It applies while a Dispatch is active.** After `worker_done`, or after a coordinator\n settles the task, the terminal is no longer a worker and is counted as a root again. The\n process may still be alive; that is the documented boundary, not an accident.\n\n## Preferred Supervised Worker Loop\n\nUse `worker-start` for the normal supervised path. It composes the existing worktree, terminal, readiness, and dispatch primitives while returning exact created/reused effects. Agents still choose placement and concurrency; Orca does not schedule workers or infer conflicts.\n\nCreate the Run and every independent Task first, then start all independent workers before waiting:\n\n```bash\norca orchestration run-create --objective \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration worker-start --task --worktree current --agent codex --json\norca orchestration worker-start --task --worktree current --agent claude --json\n```\n\n`current` and exact existing worktrees create a fresh agent terminal and do not rerun setup. Reuse an existing agent only with `--terminal `.\n\nFor a per-invocation Claude, Codex, or Cursor launch, pass an opaque provider model id with `--model`; add `--effort` only when that agent/model supports the level. These options apply only to fresh agent terminals, override general agent default arguments, and are reported under `launch.requested` and `launch.effective` in the receipt:\n\n```bash\norca orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`, and neither option can combine with `--terminal`. A connected worker server must advertise launch-preference support before Orca forwards either option.\n\nFor a new worktree, setup runs by default and agent-first creation reuses the returned startup agent terminal:\n\n```bash\norca orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n# Independent/top-level:\norca orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nSetup normally starts alongside the agent. Only a repository explicitly configured with `wait-for-setup` delays agent launch until setup succeeds. Use `--setup skip` or `--setup inherit` only for a concrete reason.\n\nRead the returned receipt before continuing: `ready` plus setup `running` is normal for start-immediately, while wait-for-setup returns setup `succeeded` before accepting task input. A failed or unknown start exits nonzero; inspect its `stage`, `effects`, and `residualResources` instead of guessing or automatically retrying. A wait-for-setup timeout can honestly leave setup `running`, which is not proof of failure.\n\nTo run the worker on another connected Orca server, add `--on `. The Run and Tasks remain authoritative on the current server; later commands route by Dispatch ID, so never repeat `--on`:\n\n```bash\n# Mac Run home -> Windows worker (the reverse is identical from a Windows Run home)\norca orchestration worker-start --task --on windows --worktree new-top-level --repo --name --agent codex --setup run --json\norca orchestration worker-show --dispatch --json\norca orchestration worker-read --dispatch --limit 50 --json\norca orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nRemote `current` and `new-child` are intentionally invalid because those words are ambiguous across servers. Use an exact discovered remote worktree selector or `new-top-level` with an explicit remote repo selector.\n\nThe follow-up is structured inbox mail, not prompt injection. The worker's next\n`orchestration check` receives it even when the Dispatch is on another connected Orca server.\n\n`worker-read` defaults to `--source auto`: Orca returns the exact hook-reported Codex, Claude, OpenClaude, or Grok transcript when it can prove the worker session, otherwise it returns bounded terminal output with `source: \"terminal\"` and a typed `fallbackReason`. Continue with the returned top-level `cursor`; it stays pinned to that exact source. If Orca reports `source_changed`, start a fresh read without the old cursor. Never supply or guess a provider session ID or transcript path.\n\nWait until every expected Dispatch settles, not for a fixed number of batches:\n\n```bash\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n# Process every message. For each accepted worker_done that is not immediately reused:\norca orchestration worker-release --dispatch --json\n# Acknowledge only after every message and required release decision is handled:\norca orchestration check --ack --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\nAfter processing each accepted `worker_done`, choose the terminal's next owner before you acknowledge the Delivery or wait again. If the same exact agent has an immediate follow-up Task, read the `worker.agent_terminal_handle` field of `worker-show --dispatch --json`, then run `orca orchestration worker-start --task --terminal --json` so Orca transfers cleanup ownership to the new Dispatch. Otherwise run `orca orchestration worker-release --dispatch --json`.\n\nRun `worker-release` after both succeeded and failed `worker_done` reports unless the user explicitly asked to keep that worker live. Release is post-completion cleanup, not cancellation: Orca first preserves inspectable output, then closes only the exact agent terminal owned by that settled Dispatch. Reused or pre-existing terminals, setup terminals, coordinators, active workers, user-taken-over terminals, and identities Orca cannot prove are retained. If the user explicitly asks to keep the live terminal for debugging, record that exception with `orca orchestration worker-retain --dispatch --json` instead of silently skipping cleanup. When the user is finished, the same Dispatch can be passed to `worker-release`, which clears the requested retention and releases the terminal.\n\nDo not release a worker because of a timeout, TUI idle state, heartbeat, status, question, escalation, or rejected/stale `worker_done`. If release returns `release_pending` or `release_unknown`, do not substitute `terminal close`; follow the exact recovery action in the receipt. A replayed Delivery may repeat `worker-release` safely.\n\nWorkers report exactly once using the IDs and capability injected by Orca; they do not supply Run/server/terminal identity:\n\n```bash\norca orchestration send --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a,path/b\" --json\n# On failure, use --outcome failed; never encode failure only in prose.\n```\n\nA worker question defaults to its owning Run. Timeout leaves it pending:\n\n```bash\norca orchestration ask --question \"\" --options \"yes,no\" --timeout-ms 600000 --json\norca orchestration ask --resume --timeout-ms 600000 --json\n# Coordinator:\norca orchestration reply --id --body \"\" --json\n```\n\nRecovery is conditional, never a fixed destructive sequence:\n\n- `worker-show --dispatch ` says `ready`: keep waiting or read bounded output.\n- It proves `failed` or `stopped`: start a replacement with `worker-start --task --retry-of ` plus an explicit `--on`/`--worktree` and `--agent`/`--terminal` choice. Retry does not silently inherit placement.\n- It remains `outcome_unknown`: either `worker-stop --dispatch ` and inspect again, or explicitly `worker-abandon --dispatch ` while accepting that resources may still be live. Abandon performs no remote, process, or filesystem action.\n- `worker-stop` closes only the exact supervised agent terminal. It never deletes the worktree, setup terminal, configured tabs, or unrelated processes.\n\nLow-level `worktree create`, `terminal create`, and `dispatch --inject` remain valid recipes for custom argv or topology that `worker-start` does not express.\n\n`dispatch --inject` deliberately keeps an operator-started terminal unsupervised: it never creates a `worker_dispatches` row and `worker-stop`/`worker-abandon` never close that process. The dispatch context is still authoritative, so `worker-show`, `worker-read`, and `worker-list` report it as `unsupervised`; settled `worker-retain` and `worker-release` report `retained` with `no_owned_resource` and take no process action. Use `worker-start --terminal ` when supervision and worker lifecycle state are required.\n\n## Gates And Legacy Inspection\n\n```bash\norca orchestration gate-create --task --question [--options ] [--json]\norca orchestration gate-resolve --id --resolution [--json]\norca orchestration gate-list [--task ] [--status ] [--json]\n```\n\nUse `ask` for worker-to-coordinator questions; it creates a `question` message that the coordinator answers with `reply`. Use `gate-create` only for coordinator-managed task DAG decisions, not for answering a worker's `ask`.\n\n`coordinator-start`, `coordinator-stop`, `run`, and `run-stop` are retired scheduler commands. They perform no effects and return the current-skill recovery action. They are not aliases for lightweight Run creation or binding.\n\nRecovery only: `orca orchestration reset --tasks|--messages|--all --json` clears the selected local orchestration database state. Do not run it during active coordination unless explicitly abandoning that state.\n\n## Full Handoffs\n\nFor full ownership transfer, use non-lifecycle terminal/worktree commands and then stop monitoring unless the user asks for supervision.\n\nTreat these as full handoff requests by default: \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"send this to another agent\", \"another agent\", \"another worktree\", or \"launch another agent to own this.\" Custom model or reasoning effort words such as `gpt-5.5`, `high`, or `xhigh` do not make the handoff supervised.\n\nSupervised orchestration remains available only when the user explicitly asks for supervision or coordination: \"supervise\", \"monitor\", \"wait for worker_done\", \"wait for results\", \"track completion\", \"DAG\", \"decision gate\", \"ask/reply\", or \"coordinate workers.\"\n\nDo not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. `task-create` is also forbidden because it records coordinator-owned tracking state; if a task row is needed, the user asked for supervised orchestration. Do not create a `taskId`/`dispatchId`, inject a lifecycle preamble, wait for completion, or read the worker terminal after prompt delivery except to avoid losing the initial prompt.\n\nNew top-level worktree handoff:\n\n```bash\norca worktree create --name --no-parent --agent codex --prompt \"\" --setup run --json\n```\n\nBefore creating a new worktree from an active feature branch, decide and state whether the desired Orca lineage is child or top-level. Use child worktree lineage only when the new work is conceptually stacked under or dependent on the active worktree. For independent repo-wide fixes, standalone feature work, or unrelated follow-up tasks, create a top-level worktree with `--no-parent`.\n\nExisting terminal handoff:\n\n```bash\norca terminal send --terminal --text \"\" --enter --json\n```\n\nCustom Codex model/effort handoff:\n\n`orca worktree create --agent codex --prompt ...` launches the known Codex agent but does not accept Codex-specific `--model` or `-c model_reasoning_effort=...` arguments. When the user asks for a specific Codex model or effort, create the independent worktree first, launch Codex with the requested command in that worktree, wait only for TUI readiness if prompt delivery would otherwise race startup, send the prompt, and stop.\n\nThe two-step custom-argv path cannot enforce a repository's explicit `wait-for-setup` startup policy because the later `terminal create` is not the startup owned by `worktree create`. Use it only when the repository starts agents immediately. If the repository requires `wait-for-setup`, use an agent-first configured launcher that can preserve sequencing, or stop and ask rather than silently bypassing the policy.\n\nNote: when no repo default-terminal configuration supplies a primary terminal, bare create opens a fallback shell before `terminal create` adds the agent. Configured default tabs are materialized instead and may run real commands. Prefer `--agent` whenever custom argv is not required. With the two-step path, target only the agent handle; close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell.\n\nUse the exact full `::` worktree id returned by `orca worktree create --json`; a bare repo id cannot target the new worktree.\n\n```bash\norca worktree create --name --no-parent --setup run --json\norca terminal create --worktree id: --title --command 'codex --model gpt-5.5 -c model_reasoning_effort=\"xhigh\"' --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca terminal send --terminal --text \"\" --enter --json\n```\n\nWait only for `tui-idle` when needed to avoid losing the prompt. Do not monitor task completion.\n\n`--no-parent` only controls Orca lineage; it does not choose the Git base. If the work should start from the repo default base, omit `--base-branch` so Orca uses that default, or explicitly pass the repo default base (`origin/main`, `origin/master`, or the `orca repo show --repo --json` value); never base it on the current feature branch unless the user explicitly asks for stacked work or \"branch from current\". Put current-branch context in the prompt instead.\n\n## Worker Terminals\n\nChoose the worker location before creating a terminal. `Fresh worker` means a fresh agent session, not a new git worktree. For parallel work, create one fresh agent terminal per worker in the same required worktree, falling back to the active worktree when none is named. If the task says current worktree only, depends on uncommitted files/artifacts, or must validate/PR the current branch, keep every worker in the active worktree:\n\n```bash\norca terminal create --worktree active --title --command \"codex\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nReuse an idle agent in the required worktree only if the prompt allows reuse; otherwise create a fresh terminal there. Create a new worktree only when the user explicitly requests one or a concrete checkout or filesystem conflict makes sharing unsafe or impossible; if the user did not request it, state that conflict before running `worktree create`. Independent tasks, parallel execution, convenience, or a preference for separate checkouts are not isolation requirements.\n\nWhen a new worktree is allowed, use child lineage for isolated work that is stacked under or dependent on the active worktree, and use `--no-parent` when it is not stacked. Decide the Git base separately: `--no-parent` makes the worktree top-level in Orca, while omitted `--base-branch` uses the repo default base.\n\nFor every new worktree, pass `--setup run` so any configured repository setup hook runs. This does not mean waiting for setup before agent launch: preserve the repository's startup policy, whose default starts setup and the agent side by side. Use `--setup skip` or `--setup inherit` only when there is a concrete task-specific reason, and state that reason before creating the worktree. This rule does not rerun setup for current or existing worktrees.\n\n```bash\norca worktree create --name --agent codex --setup run --json\n# or: --agent claude | omp | pi | grok | ...\n# Read from agentTerminalHandle, falling back to startupTerminal.handle.\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nFor new-worktree workers, read the id and `agentTerminalHandle` from `worktree create`, falling back to `startupTerminal.handle` for older runtimes. Use that as the sole worker handle when present; otherwise use `terminal list` to resolve the agent handle. Omit `--repo` only inside an Orca-managed worktree; otherwise pass `--repo `.\n\n**For an allowed new worktree, use agent-first:** `--agent` reveals the new worktree and launches the selected agent **in its first terminal**, without adding a separate fallback shell for that worker. Pass `--setup run`; repo setup and default-terminal settings may add intentional tabs or splits. Do **not** run bare `worktree create` and then `terminal create --command ` for the same worker when agent-first create is available: without configured default tabs, that two-step path leaves a fallback shell + agent pair. Only use it when custom agent argv is required (for example Codex model/effort flags) or when an older CLI rejects `--agent`; if you must, message only the agent handle. Configured default tabs are intentional surfaces, so close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell. Do not run `worktree create` when the task must stay in the current worktree.\n\nUse `orca worktree create --prompt ...` or `orca terminal send ...` for full handoffs or untracked/lightweight prompts. Those paths do not attach `taskId`/`dispatchId`; the worker should not send lifecycle messages unless the prompt supplies a live orchestration preamble.\n\nSidebar lineage and orchestration lifecycle are related but not identical. A same-worktree worker may appear as a peer under that worktree in the sidebar while remaining a child dispatch in orchestration state; only an actual child worktree creates visible parent/child worktree lineage.\n\nOther terminal commands coordinators often need:\n\n```bash\norca terminal list [--worktree ] [--include-visual-layouts] [--json]\norca terminal create [--worktree ] [--title ] [--command ] [--json]\norca terminal split --terminal [--direction horizontal|vertical] [--command ] [--json]\norca terminal wait --terminal --for tui-idle --timeout-ms --json\norca terminal read --terminal --json\norca terminal send --terminal --text --enter --json\n```\n\nIf an older CLI rejects `worktree create --agent`, create the worktree normally, then run `orca terminal create --worktree --command \"codex\" --json` or `--command \"claude\"`.\n\nWait for `tui-idle` before dispatching. Always pass `--timeout-ms`; real coding tasks can take 15-60 minutes. During supervision, use rolling `check --wait` windows. If a window returns no matching message, inspect `task-list`, `terminal read`, or `terminal wait --for tui-idle` as a liveness checkpoint; if the terminal is still working or producing activity, keep waiting instead of retrying the task.\n\n## Agent Guidance\n\n- Workers with a valid live preamble must send `worker_done` exactly once from their own terminal with an explicit `--outcome succeeded` or `--outcome failed`:\n `orca orchestration send --type worker_done --subject \"\" --body \"<3-sentence summary: what you did, what you found, what's left>\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a\" --report-path \"\" --json`\n- A failed outcome is still a terminal report, but Orca records both the Dispatch and Task as failed. Never encode failure only in the subject/body.\n- After sending `worker_done`, end that dispatched turn and idle at the agent prompt. Do not autonomously start more work, poll, or attempt to close the terminal yourself. A direct user instruction takes precedence and starts ordinary user-owned work: follow it without coordinator approval or a fresh Dispatch, never refuse it because of worker/coordinator roles, and do not reuse the settled Dispatch's lifecycle IDs. A coordinator-supervised follow-up still arrives with a fresh preamble + TASK block.\n- For long tasks, send heartbeat/status only when the preamble asks for it, including both IDs:\n `orca orchestration send --type heartbeat --subject \"alive\" --payload '{\"taskId\":\"\",\"dispatchId\":\"\",\"phase\":\"implementing\"}' --json`\n- If blocked before completion, use `ask`; use `escalation` only when ownership is valid and the coordinator must intervene.\n- Treat preambles inherited through terminal history or full handoffs as stale unless the current prompt explicitly keeps that coordinator in the loop.\n- Coordinators must account for every settled worker terminal before waiting again or ending the turn: immediately reuse the exact worker for a new Dispatch, explicitly retain it at the user's request with `worker-retain`, or run `worker-release`. Do not leave a completed worker live merely to inspect output; released workers remain readable through `worker-read`.\n- Coordinators should use `task-list --ready` as external memory, dispatch parallel waves, and avoid dependency chains deeper than 3-4 steps.\n\n## Example\n\n```bash\norca terminal create --worktree active --title login-css-worker --command \"claude\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration task-create --spec \"Fix the login button CSS\" --json\norca orchestration dispatch --task --to --inject --json\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\n## Next Action\n\nCoordinator: confirm `orca status --json`, create or bind a Run, inspect `task-list`/`dispatch-show` if inheriting state, then use the explicit supervised loop (`task-create` -> `worker-start` -> `check --wait`). Use low-level terminal creation plus `dispatch --inject` only when the composed start does not express the needed topology. After every accepted `worker_done`, either transfer the exact terminal to an immediate follow-up Dispatch or run `worker-release` before the next wait.\n\nWorker: if the current prompt contains a live dispatch preamble, do the task, use `ask` for blocking questions, and send `worker_done` once with the required payload. If the preamble is stale or absent, do not send lifecycle messages; inspect state or treat the prompt as an ordinary handoff.\n" +const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Use Orca orchestration for structured multi-agent coordination: threaded\n messages, blocking ask/reply flows, task dispatch, worker_done/escalation\n waits, task DAGs, decision gates, coordinator loops, or decomposing work\n across agents. Use `orca-cli` instead for full ownership handoffs, including\n requests phrased as \"hand off\", \"handoff\", \"handover\", \"give this to another\n agent\", or \"another worktree\" when the user did not explicitly ask to\n supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for\n ordinary terminal control, lightweight terminal prompts, shell commands, Orca\n worktree management, reading or waiting on terminals, and automation of the\n browser embedded inside Orca. Use Computer Use for browser windows, webviews,\n Orca app UI, or desktop UI outside Orca's embedded browser.\n---\n\n# Orca Inter-Agent Orchestration\n\nOrchestration is Orca's structured coordination layer for agent messages, task ownership, dispatch state, and worker completion tracking.\n\nUse this skill when coordination state matters. For lightweight terminal prompts or basic worktree/terminal/built-in-browser control, use `orca-cli`.\n\n## Tool Boundary\n\nIf a task says to use Orca orchestration, the coordinator must create or bind a Run, create the Task with `orca orchestration task-create`, then attach the worker with either the preferred `orca orchestration worker-start` composition or the low-level `orca orchestration dispatch --inject` path.\n\nDo not substitute non-Orca subagent tools, generic agent-spawn APIs, or chat-only parallel worker features. Those may create useful workers, but they do not create Orca task/dispatch provenance, injected lifecycle preambles, `worker_done` authority, or decision gates.\n\nBefore claiming a worker was orchestrated, verify the task/dispatch exists:\n\n```bash\norca orchestration task-list --json\norca orchestration dispatch-show --task --json\n```\n\nIf the work was accidentally run outside Orca orchestration, say so plainly. To repair provenance, rerun or revalidate the needed work through a fresh Orca terminal plus injected dispatch; do not retroactively describe the external worker as orchestrated.\n\n## When To Use\n\n- Send/reply/ask between agent terminals with persistent messages.\n- Dispatch structured tasks to workers and wait for `worker_done` or `escalation`.\n- Track task DAGs with dependencies.\n- Run coordinator loops or decision gates.\n\nDo not use orchestration merely because the user says \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or asks for another worktree/agent/model/effort. Those are full ownership transfers unless the user explicitly asks to supervise, monitor, wait for worker completion/results, coordinate a DAG, use decision gates, or keep a blocking ask/reply loop.\n\n## Preconditions\n\n- `orca status --json` should show a running runtime.\n- `orca` must be on PATH (`orca-ide` on Linux).\n- The orchestration experimental feature must be enabled in Settings > Experimental.\n- `orca orchestration` commands are RPC calls to the running Orca runtime.\n\n## Contract Migration\n\nOrca adopts a live pre-update orchestration assignment into an ordinary Run. Adoption preserves the existing agent process, PTY/session, terminal handle, tab/leaf/pane, worktree or folder workspace, Task, and Dispatch; it never restarts or replaces the worker. The retired scheduler is not revived, and a newly created attempt uses the current grammar.\n\nTreat the authority label on injected or formatted messages as definitive:\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported command printed with the message, using the same CLI executable and arguments that the original prompt supplied.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded, at-least-once cutover replay. Process it idempotently and acknowledge it only through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or lifecycle action.\n- An unlabeled current message uses the current guide and current grammar.\n\nAn explicitly selected current Run, attested current Run binding, current Dispatch, or federated attachment takes precedence over legacy fallback. A retained adoption record alone never turns a current command into a legacy call.\n\nDatabase provenance, an old-looking terminal, or a legacy Run ID does not prove mutation authority. If the runtime cannot prove liveness, principal ownership, capability, or the exact legacy contract, it degrades to read-only inspection and must not fall back to local execution. Exact recovery may restore the already-live PTY once in its original inactive background tab. It must not spawn, write, signal, stop, switch, focus, split, or inject a terminal. Loss of lifecycle authority does not invalidate the existing assignment, process, or filesystem work.\n\nCompatibility retries have narrow guarantees. A pending ask, a reply, a final Dispatch settlement, and a consuming check have durable recovery identities. A-era heartbeat and escalation calls remain at-least-once across a manual A-to-B retry because identical later signals may be intentional. If an A-era ask may already have been answered, run the exact non-consuming recovery check printed by the runtime first; after its answer is printed and acknowledged, a new invocation with the same question creates a new question. Never guess among multiple identical question threads.\n\nWhen a compatibility or recovery command returns structured next-step arguments, run those exact arguments with the same CLI executable. The arguments intentionally omit the executable name so the guidance works with `orca`, `orca-ide`, `orca-dev`, or another configured Orca CLI command. Do not translate the command from memory, broaden its recipient, or retry it as a current mutation unless the returned guidance explicitly says to.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The initial command durably commits the question, prints its exact `ask --resume ` command, and exits with launcher status `75`; it does not wait for the answer. Run that exact resume command after the launcher or update boundary. Resume is idempotent and read-oriented: it waits for the already-committed question and does not create another one. For a WSL process that received compatibility proof at launch, use the printed executable `orca-ide` WSL resume command so the same distro and packaged launcher authority are preserved; do not substitute a PATH-resolved local CLI. Older WSL processes that never received the hidden launch token remain lifecycle read-only after the update, even while their terminal and filesystem work continue.\n\nLegacy inspection remains available without consuming mail:\n\n```bash\norca orchestration run-list --json\n# run_legacy_local is an empty audit tombstone after adoption.\norca orchestration run-show --id run_legacy_local --json\n# In run-list, find the ordinary Run whose objective is:\n# \"Recovered orchestration work from a contract update\"\norca orchestration run-show --id --json\norca orchestration task-list --run --json\norca orchestration inbox --full --json\norca orchestration check --terminal --peek --format --json\norca terminal read --terminal --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\nIf the original coordinator is unavailable or cannot prove its retained authority, a current coordinator may explicitly take over the adopted Run from its own live agent terminal:\n\n```bash\norca orchestration run-use --id --takeover-legacy --json\norca orchestration check --run --json\n```\n\nTakeover fences only the old coordinator, binds the current one, and moves pending worker mail into current Run Delivery. It is bound to the authenticated invoking terminal; `--from` cannot name another coordinator. Live legacy workers keep their original Tasks, Dispatches, processes, filesystems, and old prompt commands; their later questions, escalations, and completion reports route to the current coordinator. Do not use takeover while the original coordinator is still actively coordinating, because its later lifecycle mutations are rejected.\n\nDo not launch a replacement editor merely because the desktop app or runtime was updated. If adoption cannot prove continuing authority, keep the original worker as the only editor until it reaches a stable handoff point, then use a new current Dispatch in a conflict-free placement for any remaining work.\n\n## Ownership\n\nNew orchestration messages and tasks belong to one explicitly bound Run. A Run is only a durable namespace and coordinator inbox; it never schedules or places workers. Lifecycle authority comes from the active Dispatch, and terminal handles remain routing metadata rather than durable identity. Send `worker_done` and `heartbeat` from the worker's own terminal; Orca routes them to that Dispatch's Run.\n\nClassify inherited context before sending lifecycle messages:\n\n- Coordinated subtask: a live coordinator owns the DAG and waits on this dispatch. Follow the preamble exactly, including `worker_done`, heartbeat/status, `ask`, and `escalation`.\n- Full handoff means ownership transfer, not supervised dispatch. The original actor is not monitoring a DAG, so do not create lifecycle obligations unless the user explicitly asks you to supervise.\n- Classify requests containing \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"another agent\", or \"another worktree\" as full handoffs by default, even when the user names a custom model or reasoning effort.\n- Use supervised orchestration only when the user explicitly asks you to \"supervise\", \"monitor\", \"wait\", \"track completion\", \"wait for worker_done\", return results, coordinate a DAG, use a decision gate, or manage ask/reply flow.\n- Do not use `orca orchestration dispatch --inject` for full handoffs. It injects a coordinator preamble that tells the worker to send `worker_done`, heartbeat, and `ask` messages, then end its turn under the original terminal's dispatch lifecycle.\n- Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. Do not peek at terminal output after prompt delivery to monitor progress.\n- A review-only `worker_done` reports findings; it does not authorize coordinator file edits. After a review-only completion, synthesize findings, ask a decision gate if ownership is unclear, and dispatch or hand off fixes unless the user explicitly asked the coordinator to own fixes.\n- If the user's plan names a next owner agent (for example, \"then use opencode to create a PR\"), post-review corrections and PR prep belong to that named owner. The coordinator routes, synthesizes, asks decision gates when needed, and supervises; the named owner edits files and creates the PR.\n\nIf unclear, inspect orchestration state before sending lifecycle messages:\n\n```bash\norca orchestration task-list --json\norca terminal list --json\n# If inherited context includes a task id:\norca orchestration dispatch-show --task --json\n```\n\n## Messaging\n\n```bash\norca orchestration send --subject [--to ] [--from ] [--body ] [--type ] [--priority ] [--thread-id ] [--payload ] [--json]\norca orchestration check [--terminal ] [--ack ] [--peek|--all] [--types ] [--format] [--wait] [--timeout-ms ] [--json]\norca orchestration reply --id --body [--from ] [--json]\norca orchestration ask (--question |--resume ) [--options ] [--timeout-ms ] [--from ] [--json]\norca orchestration inbox [--limit ] [--json]\n```\n\nRules:\n\n- Omit `--from` unless impersonating another terminal; Orca auto-resolves it from the current terminal.\n- A coordinator `check` returns the bound Run's oldest FIFO Delivery (up to 50 messages) and replays that exact batch until `--ack `. Process every message before acknowledging; `check --ack --wait` acknowledges, checks, and waits in one operation.\n- Use `--peek` and `--all` only for read-only history/debugging. Type filters decide when a waiter wakes; the returned actionable Delivery is still the oldest full batch.\n- Use `dispatch:` for coordinator guidance to one supervised worker. Orca routes that stable address locally or through the connected-server relay; do not substitute a remote terminal handle.\n- Terminal handles remain appropriate for low-level pre-Dispatch messaging. Prefer `agentTerminalHandle` from the create response, fall back to `startupTerminal.handle` for older runtimes, then re-resolve with `orca terminal list --worktree ... --json` if missing or stale. Continue with the replacement handle only; never dual-send to old and new handles.\n- `terminal list --json` omits `visualLayouts` because handle recovery does not need topology. Add `--include-visual-layouts` only for explicit tab and pane inspection.\n- `orca orchestration check --peek --format --json` returns locally formatted unread mail without consuming it; it never writes to terminal input or remotely wakes another terminal. Use `orchestration dispatch --inject` to deliver a tracked task, or `terminal send` when an existing agent needs a free-form prompt.\n- While supervising workers manually, use `check --wait --types worker_done,escalation,question --timeout-ms ` instead of sleep/poll loops. Process the whole Delivery, reply to `question` messages with `orca orchestration reply --id --body --json`, then acknowledge and keep waiting.\n- Treat a `check --wait` timeout or `{count:0}` as a checkpoint, not a worker failure. Long coding tasks routinely run 15-60 minutes; keep using rolling waits unless you receive `worker_done`/`escalation`, the terminal exits or disappears, or the user explicitly asks you to stop.\n- Heartbeats and visible terminal activity mean the worker is alive, not done. Do not stop, close, kill, or restart a worker just because it has not produced a completion message yet.\n- Use `ask` when a worker needs a blocking answer from the coordinator; it defaults to the active Dispatch's Run. Timeout or disconnect leaves the question pending, so resume by its original message ID instead of asking again.\n- `check --wait` returns one bounded Delivery, not every future completion. Process every message, acknowledge it, then keep waiting until every expected Dispatch settles.\n- Group addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, `@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`.\n- Message types include `status`, `dispatch`, `worker_done`, `merge_ready`, `escalation`, `handoff`, `question`, `decision_gate` (legacy/gates), and `heartbeat`.\n- Use group addresses only for messages that are genuinely useful to many terminals, such as `status` broadcasts or intentional fan-out questions. Do not send dispatch lifecycle messages to groups.\n- `worker_done` belongs to the active Dispatch and defaults to its Run mailbox; never target a group.\n- A valid `worker_done` for the active `taskId` + `dispatchId` marks the task and dispatch completed automatically. Do not follow it with `task-update --status completed`; reserve manual updates for explicit recovery or overrides.\n- `heartbeat` is also Dispatch-scoped. Include both IDs and omit `--to` so Orca uses the owning Run; use `status` for broad progress updates.\n\n## Tasks And Dispatch\n\nA Run is the namespace/inbox, a Task is the work item, and a Dispatch assigns one Task attempt to a terminal. Create or bind a Run once before the common loop.\n\n```bash\norca orchestration run-create --objective --json\norca orchestration task-create --spec [--deps ] [--parent ] [--json]\norca orchestration task-list [--status ] [--ready] [--brief] [--json]\norca orchestration task-update --id --status [--result ] [--json]\norca orchestration dispatch --task --to [--from ] [--inject] [--json]\norca orchestration dispatch-show --task [--json]\n```\n\nTask statuses: `pending`, `ready`, `dispatched`, `completed`, `failed`, `blocked`.\n\nDispatch rules:\n\n- `--inject` sends the task spec plus preamble into a recognized agent CLI so it can report `worker_done`.\n- If the target is a bare shell, omit `--inject`, dispatch for tracking if needed, then send the prompt manually with `orca terminal send --terminal --text --enter --json`.\n- After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed.\n- Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag.\n\n## How deep workers can nest\n\nA dispatched worker normally cannot dispatch sub-workers. Attempting it fails with\n`nested_worker_depth_exceeded` and a message telling the worker to complete the task\nitself. Do that — do not try to route around it.\n\nThe limit is a number, not an on/off switch. `Settings -> Orchestration -> Nested worker depth`\nsets how many generations are allowed:\n\n- `1` (default): a coordinator dispatches workers; those workers do not dispatch.\n- `2`: workers may dispatch one further generation.\n\nDepth is counted from the terminal that issues the command, not from the Run. Creating a\nnew Run does not reset it — a worker that runs `run-create` then `worker-start` is still a\nworker, and still counted. This is the part that changed: the old behaviour rejected\nsub-dispatch only because a worker's terminal was not bound to a Run, so creating a Run was\nenough to slip past it.\n\nTwo limits worth knowing:\n\n- **It is a guardrail, not a security boundary.** A caller that declares another terminal's\n handle while its own launch evidence is unverifiable (an ordinary restored terminal, for\n example) can be counted as that terminal instead. Orca does not treat workers as hostile.\n- **It applies while a Dispatch is active.** After `worker_done`, or after a coordinator\n settles the task, the terminal is no longer a worker and is counted as a root again. The\n process may still be alive; that is the documented boundary, not an accident.\n\n## Preferred Supervised Worker Loop\n\nUse `worker-start` for the normal supervised path. It composes the existing worktree, terminal, readiness, and dispatch primitives while returning exact created/reused effects. Agents still choose placement and concurrency; Orca does not schedule workers or infer conflicts.\n\nCreate the Run and every independent Task first, then start all independent workers before waiting:\n\n```bash\norca orchestration run-create --objective \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration worker-start --task --worktree current --agent codex --json\norca orchestration worker-start --task --worktree current --agent claude --json\n```\n\n`current` and exact existing worktrees create a fresh agent terminal and do not rerun setup. Reuse an existing agent only with `--terminal `.\n\nFor a per-invocation Claude, Codex, or Cursor launch, pass an opaque provider model id with `--model`; add `--effort` only when that agent/model supports the level. These options apply only to fresh agent terminals, override general agent default arguments, and are reported under `launch.requested` and `launch.effective` in the receipt:\n\n```bash\norca orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`, and neither option can combine with `--terminal`. A connected worker server must advertise launch-preference support before Orca forwards either option.\n\nFor a new worktree, setup runs by default and agent-first creation reuses the returned startup agent terminal:\n\n```bash\norca orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n# Independent/top-level:\norca orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nSetup normally starts alongside the agent. Only a repository explicitly configured with `wait-for-setup` delays agent launch until setup succeeds. Use `--setup skip` or `--setup inherit` only for a concrete reason.\n\nRead the returned receipt before continuing: `ready` plus setup `running` is normal for start-immediately, while wait-for-setup returns setup `succeeded` before accepting task input. A failed or unknown start exits nonzero; inspect its `stage`, `effects`, and `residualResources` instead of guessing or automatically retrying. A wait-for-setup timeout can honestly leave setup `running`, which is not proof of failure.\n\nTo run the worker on another connected Orca server, add `--on `. The Run and Tasks remain authoritative on the current server; later commands route by Dispatch ID, so never repeat `--on`:\n\n```bash\n# Mac Run home -> Windows worker (the reverse is identical from a Windows Run home)\norca orchestration worker-start --task --on windows --worktree new-top-level --repo --name --agent codex --setup run --json\norca orchestration worker-show --dispatch --json\norca orchestration worker-read --dispatch --limit 50 --json\norca orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nRemote `current` and `new-child` are intentionally invalid because those words are ambiguous across servers. Use an exact discovered remote worktree selector or `new-top-level` with an explicit remote repo selector.\n\nThe follow-up is structured inbox mail, not prompt injection. The worker's next\n`orchestration check` receives it even when the Dispatch is on another connected Orca server.\n\n`worker-read` defaults to `--source auto`: Orca returns the exact hook-reported Codex, Claude, OpenClaude, or Grok transcript when it can prove the worker session, otherwise it returns bounded terminal output with `source: \"terminal\"` and a typed `fallbackReason`. Continue with the returned top-level `cursor`; it stays pinned to that exact source. If Orca reports `source_changed`, start a fresh read without the old cursor. Never supply or guess a provider session ID or transcript path.\n\nWait until every expected Dispatch settles, not for a fixed number of batches:\n\n```bash\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n# Process every message. For each accepted worker_done that is not immediately reused:\norca orchestration worker-release --dispatch --json\n# Acknowledge only after every message and required release decision is handled:\norca orchestration check --ack --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\nAfter processing each accepted `worker_done`, choose the terminal's next owner before you acknowledge the Delivery or wait again. If the same exact agent has an immediate follow-up Task, read the `worker.agent_terminal_handle` field of `worker-show --dispatch --json`, then run `orca orchestration worker-start --task --terminal --json` so Orca transfers cleanup ownership to the new Dispatch. Otherwise run `orca orchestration worker-release --dispatch --json`.\n\nRun `worker-release` after both succeeded and failed `worker_done` reports unless the user explicitly asked to keep that worker live. Release is post-completion cleanup, not cancellation: Orca first preserves inspectable output, then closes only the exact agent terminal owned by that settled Dispatch. Reused or pre-existing terminals, setup terminals, coordinators, active workers, user-taken-over terminals, and identities Orca cannot prove are retained. If the user explicitly asks to keep the live terminal for debugging, record that exception with `orca orchestration worker-retain --dispatch --json` instead of silently skipping cleanup. When the user is finished, the same Dispatch can be passed to `worker-release`, which clears the requested retention and releases the terminal.\n\nDo not release a worker because of a timeout, TUI idle state, heartbeat, status, question, escalation, or rejected/stale `worker_done`. If release returns `release_pending` or `release_unknown`, do not substitute `terminal close`; follow the exact recovery action in the receipt. A replayed Delivery may repeat `worker-release` safely.\n\nWorkers report exactly once using the IDs and capability injected by Orca; they do not supply Run/server/terminal identity:\n\n```bash\norca orchestration send --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a,path/b\" --json\n# On failure, use --outcome failed; never encode failure only in prose.\n```\n\nA worker question defaults to its owning Run. Timeout leaves it pending:\n\n```bash\norca orchestration ask --question \"\" --options \"yes,no\" --timeout-ms 600000 --json\norca orchestration ask --resume --timeout-ms 600000 --json\n# Coordinator:\norca orchestration reply --id --body \"\" --json\n```\n\nRecovery is conditional, never a fixed destructive sequence:\n\n- `worker-show --dispatch ` says `ready`: keep waiting or read bounded output.\n- It proves `failed` or `stopped`: start a replacement with `worker-start --task --retry-of ` plus an explicit `--on`/`--worktree` and `--agent`/`--terminal` choice. Retry does not silently inherit placement.\n- It remains `outcome_unknown`: either `worker-stop --dispatch ` and inspect again, or explicitly `worker-abandon --dispatch ` while accepting that resources may still be live. Abandon performs no remote, process, or filesystem action.\n- `worker-stop` closes only the exact supervised agent terminal. It never deletes the worktree, setup terminal, configured tabs, or unrelated processes.\n\nLow-level `worktree create`, `terminal create`, and `dispatch --inject` remain valid recipes for custom argv or topology that `worker-start` does not express.\n\n`dispatch --inject` deliberately keeps an operator-started terminal unsupervised: it never creates a `worker_dispatches` row and `worker-stop`/`worker-abandon` never close that process. The dispatch context is still authoritative, so `worker-show`, `worker-read`, and `worker-list` report it as `unsupervised`; settled `worker-retain` and `worker-release` report `retained` with `no_owned_resource` and take no process action. Use `worker-start --terminal ` when supervision and worker lifecycle state are required.\n\n## Gates And Legacy Inspection\n\n```bash\norca orchestration gate-create --task --question [--options ] [--json]\norca orchestration gate-resolve --id --resolution [--json]\norca orchestration gate-list [--task ] [--status ] [--json]\n```\n\nUse `ask` for worker-to-coordinator questions; it creates a `question` message that the coordinator answers with `reply`. Use `gate-create` only for coordinator-managed task DAG decisions, not for answering a worker's `ask`.\n\n`coordinator-start`, `coordinator-stop`, `run`, and `run-stop` are retired scheduler commands. They perform no effects and return the current-skill recovery action. They are not aliases for lightweight Run creation or binding.\n\nRecovery only: `orca orchestration reset --tasks|--messages|--all --json` clears the selected local orchestration database state. Do not run it during active coordination unless explicitly abandoning that state.\n\n## Full Handoffs\n\nFor full ownership transfer, use non-lifecycle terminal/worktree commands and then stop monitoring unless the user asks for supervision.\n\nTreat these as full handoff requests by default: \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"send this to another agent\", \"another agent\", \"another worktree\", or \"launch another agent to own this.\" Custom model or reasoning effort words such as `gpt-5.5`, `high`, or `xhigh` do not make the handoff supervised.\n\nSupervised orchestration remains available only when the user explicitly asks for supervision or coordination: \"supervise\", \"monitor\", \"wait for worker_done\", \"wait for results\", \"track completion\", \"DAG\", \"decision gate\", \"ask/reply\", or \"coordinate workers.\"\n\nDo not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. `task-create` is also forbidden because it records coordinator-owned tracking state; if a task row is needed, the user asked for supervised orchestration. Do not create a `taskId`/`dispatchId`, inject a lifecycle preamble, wait for completion, or read the worker terminal after prompt delivery except to avoid losing the initial prompt.\n\nNew top-level worktree handoff:\n\n```bash\norca worktree create --name --no-parent --agent codex --prompt \"\" --setup run --json\n```\n\nBefore creating a new worktree from an active feature branch, decide and state whether the desired Orca lineage is child or top-level. Use child worktree lineage only when the new work is conceptually stacked under or dependent on the active worktree. For independent repo-wide fixes, standalone feature work, or unrelated follow-up tasks, create a top-level worktree with `--no-parent`.\n\nExisting terminal handoff:\n\n```bash\norca terminal send --terminal --text \"\" --enter --json\n```\n\nCustom Codex model/effort handoff:\n\n`orca worktree create --agent codex --prompt ...` launches the known Codex agent but does not accept Codex-specific `--model` or `-c model_reasoning_effort=...` arguments. When the user asks for a specific Codex model or effort, create the independent worktree first, launch Codex with the requested command in that worktree, wait only for TUI readiness if prompt delivery would otherwise race startup, send the prompt, and stop.\n\nThe two-step custom-argv path cannot enforce a repository's explicit `wait-for-setup` startup policy because the later `terminal create` is not the startup owned by `worktree create`. Use it only when the repository starts agents immediately. If the repository requires `wait-for-setup`, use an agent-first configured launcher that can preserve sequencing, or stop and ask rather than silently bypassing the policy.\n\nNote: when no repo default-terminal configuration supplies a primary terminal, bare create opens a fallback shell before `terminal create` adds the agent. Configured default tabs are materialized instead and may run real commands. Prefer `--agent` whenever custom argv is not required. With the two-step path, target only the agent handle; close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell.\n\nUse the exact full `::` worktree id returned by `orca worktree create --json`; a bare repo id cannot target the new worktree.\n\n```bash\norca worktree create --name --no-parent --setup run --json\norca terminal create --worktree id: --title --command 'codex --model gpt-5.5 -c model_reasoning_effort=\"xhigh\"' --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca terminal send --terminal --text \"\" --enter --json\n```\n\nWait only for `tui-idle` when needed to avoid losing the prompt. Do not monitor task completion.\n\n`--no-parent` only controls Orca lineage; it does not choose the Git base. If the work should start from the repo default base, omit `--base-branch` so Orca uses that default, or explicitly pass the repo default base (`origin/main`, `origin/master`, or the `orca repo show --repo --json` value); never base it on the current feature branch unless the user explicitly asks for stacked work or \"branch from current\". Put current-branch context in the prompt instead.\n\n## Worker Terminals\n\nChoose the worker location before creating a terminal. `Fresh worker` means a fresh agent session, not a new git worktree. For parallel work, create one fresh agent terminal per worker in the same required worktree, falling back to the active worktree when none is named. If the task says current worktree only, depends on uncommitted files/artifacts, or must validate/PR the current branch, keep every worker in the active worktree:\n\n```bash\norca terminal create --worktree active --title --command \"codex\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nReuse an idle agent in the required worktree only if the prompt allows reuse; otherwise create a fresh terminal there. Create a new worktree only when the user explicitly requests one or a concrete checkout or filesystem conflict makes sharing unsafe or impossible; if the user did not request it, state that conflict before running `worktree create`. Independent tasks, parallel execution, convenience, or a preference for separate checkouts are not isolation requirements.\n\nWhen a new worktree is allowed, use child lineage for isolated work that is stacked under or dependent on the active worktree, and use `--no-parent` when it is not stacked. Decide the Git base separately: `--no-parent` makes the worktree top-level in Orca, while omitted `--base-branch` uses the repo default base.\n\nFor every new worktree, pass `--setup run` so any configured repository setup hook runs. This does not mean waiting for setup before agent launch: preserve the repository's startup policy, whose default starts setup and the agent side by side. Use `--setup skip` or `--setup inherit` only when there is a concrete task-specific reason, and state that reason before creating the worktree. This rule does not rerun setup for current or existing worktrees.\n\n```bash\norca worktree create --name --agent codex --setup run --json\n# or: --agent claude | omp | pi | grok | ...\n# Read from agentTerminalHandle, falling back to startupTerminal.handle.\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nFor new-worktree workers, read the id and `agentTerminalHandle` from `worktree create`, falling back to `startupTerminal.handle` for older runtimes. Use that as the sole worker handle when present; otherwise use `terminal list` to resolve the agent handle. Omit `--repo` only inside an Orca-managed worktree; otherwise pass `--repo `.\n\n**For an allowed new worktree, use agent-first:** `--agent` reveals the new worktree and launches the selected agent **in its first terminal**, without adding a separate fallback shell for that worker. Pass `--setup run`; repo setup and default-terminal settings may add intentional tabs or splits. Do **not** run bare `worktree create` and then `terminal create --command ` for the same worker when agent-first create is available: without configured default tabs, that two-step path leaves a fallback shell + agent pair. Only use it when custom agent argv is required (for example Codex model/effort flags) or when an older CLI rejects `--agent`; if you must, message only the agent handle. Configured default tabs are intentional surfaces, so close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell. Do not run `worktree create` when the task must stay in the current worktree.\n\nUse `orca worktree create --prompt ...` or `orca terminal send ...` for full handoffs or untracked/lightweight prompts. Those paths do not attach `taskId`/`dispatchId`; the worker should not send lifecycle messages unless the prompt supplies a live orchestration preamble.\n\nSidebar lineage and orchestration lifecycle are related but not identical. A same-worktree worker may appear as a peer under that worktree in the sidebar while remaining a child dispatch in orchestration state; only an actual child worktree creates visible parent/child worktree lineage.\n\nOther terminal commands coordinators often need:\n\n```bash\norca terminal list [--worktree ] [--include-visual-layouts] [--json]\norca terminal create [--worktree ] [--title ] [--command ] [--json]\norca terminal split --terminal [--direction horizontal|vertical] [--command ] [--json]\norca terminal wait --terminal --for tui-idle --timeout-ms --json\norca terminal read --terminal --json\norca terminal send --terminal --text --enter --json\n```\n\nIf an older CLI rejects `worktree create --agent`, create the worktree normally, then run `orca terminal create --worktree --command \"codex\" --json` or `--command \"claude\"`.\n\nWait for `tui-idle` before dispatching. Always pass `--timeout-ms`; real coding tasks can take 15-60 minutes. During supervision, use rolling `check --wait` windows. If a window returns no matching message, inspect `task-list`, `terminal read`, or `terminal wait --for tui-idle` as a liveness checkpoint; if the terminal is still working or producing activity, keep waiting instead of retrying the task.\n\n## Agent Guidance\n\n- Workers with a valid live preamble must send `worker_done` exactly once from their own terminal with an explicit `--outcome succeeded` or `--outcome failed`:\n `orca orchestration send --type worker_done --subject \"\" --body \"<3-sentence summary: what you did, what you found, what's left>\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a\" --report-path \"\" --json`\n- A failed outcome is still a terminal report, but Orca records both the Dispatch and Task as failed. Never encode failure only in the subject/body.\n- After sending `worker_done`, end that dispatched turn and idle at the agent prompt. Do not autonomously start more work, poll, or attempt to close the terminal yourself. A direct user instruction takes precedence and starts ordinary user-owned work: follow it without coordinator approval or a fresh Dispatch, never refuse it because of worker/coordinator roles, and do not reuse the settled Dispatch's lifecycle IDs. A coordinator-supervised follow-up still arrives with a fresh preamble + TASK block.\n- For long tasks, send heartbeat/status only when the preamble asks for it, including both IDs:\n `orca orchestration send --type heartbeat --subject \"alive\" --payload '{\"taskId\":\"\",\"dispatchId\":\"\",\"phase\":\"implementing\"}' --json`\n- If blocked before completion, use `ask`; use `escalation` only when ownership is valid and the coordinator must intervene.\n- Treat preambles inherited through terminal history or full handoffs as stale unless the current prompt explicitly keeps that coordinator in the loop.\n- Coordinators must account for every settled worker terminal before waiting again or ending the turn: immediately reuse the exact worker for a new Dispatch, explicitly retain it at the user's request with `worker-retain`, or run `worker-release`. Do not leave a completed worker live merely to inspect output; released workers remain readable through `worker-read`.\n- Coordinators should use `task-list --ready` as external memory, dispatch parallel waves, and avoid dependency chains deeper than 3-4 steps.\n\n## Example\n\n```bash\norca terminal create --worktree active --title login-css-worker --command \"claude\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration task-create --spec \"Fix the login button CSS\" --json\norca orchestration dispatch --task --to --inject --json\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\n## Next Action\n\nCoordinator: confirm `orca status --json`, create or bind a Run, inspect `task-list`/`dispatch-show` if inheriting state, then use the explicit supervised loop (`task-create` -> `worker-start` -> `check --wait`). Use low-level terminal creation plus `dispatch --inject` only when the composed start does not express the needed topology. After every accepted `worker_done`, either transfer the exact terminal to an immediate follow-up Dispatch or run `worker-release` before the next wait.\n\nWorker: if the current prompt contains a live dispatch preamble, do the task, use `ask` for blocking questions, and send `worker_done` once with the required payload. If the preamble is stale or absent, do not send lifecycle messages; inspect state or treat the prompt as an ordinary handoff.\n" // Why: no current guide has bundled reference documents, so --full is byte-identical for now. // oxfmt-ignore diff --git a/src/main/runtime/orchestration/coordinator-task-dispatch.ts b/src/main/runtime/orchestration/coordinator-task-dispatch.ts index 1a882ca3b73..8dae105726a 100644 --- a/src/main/runtime/orchestration/coordinator-task-dispatch.ts +++ b/src/main/runtime/orchestration/coordinator-task-dispatch.ts @@ -112,6 +112,7 @@ export async function dispatchTaskToWorker(params: { const preamble = buildDispatchPreamble({ taskId: task.id, dispatchId: dispatch.id, + canDispatchSubWorkers: dispatch.depth < params.nestedWorkerMaxDepth, // Why (§3.4): strippedSpec drops the allow-stale-base line so the worker doesn't read the infra flag as an instruction. taskSpec: strippedSpec, coordinatorHandle: params.coordinatorHandle, diff --git a/src/main/runtime/orchestration/preamble.test.ts b/src/main/runtime/orchestration/preamble.test.ts index 3cedb9fa242..79e06f5a55a 100644 --- a/src/main/runtime/orchestration/preamble.test.ts +++ b/src/main/runtime/orchestration/preamble.test.ts @@ -290,3 +290,37 @@ describe('buildDispatchPreamble', () => { expect(result).toMatchSnapshot() }) }) + +describe('sub-dispatch section', () => { + const base = { + taskId: 'task_1', + dispatchId: 'ctx_1', + taskSpec: 'do the thing', + coordinatorHandle: 'term_coord', + workerHandle: 'term_worker' + } + + it('is omitted when the worker has no nesting budget', () => { + const preamble = buildDispatchPreamble(base) + expect(preamble).not.toContain('=== SUB-DISPATCH ===') + expect(preamble).not.toContain('worker-start') + }) + + it('is omitted explicitly when nesting is disallowed', () => { + expect(buildDispatchPreamble({ ...base, canDispatchSubWorkers: false })).not.toContain( + '=== SUB-DISPATCH ===' + ) + }) + + it('appears with the run-create sequence when budget remains', () => { + const preamble = buildDispatchPreamble({ ...base, canDispatchSubWorkers: true }) + expect(preamble).toContain('=== SUB-DISPATCH ===') + expect(preamble).toContain('orchestration run-create') + expect(preamble).toContain('orchestration worker-start') + }) + + it('keeps the task block last so the spec is not buried', () => { + const preamble = buildDispatchPreamble({ ...base, canDispatchSubWorkers: true }) + expect(preamble.indexOf('=== SUB-DISPATCH ===')).toBeLessThan(preamble.indexOf('=== TASK ===')) + }) +}) diff --git a/src/main/runtime/orchestration/preamble.ts b/src/main/runtime/orchestration/preamble.ts index d98f33763d2..7d426184954 100644 --- a/src/main/runtime/orchestration/preamble.ts +++ b/src/main/runtime/orchestration/preamble.ts @@ -31,6 +31,8 @@ export type PreambleParams = { // Why: prompt-returning agents should idle after worker_done, while bare // shells have no agent prompt for Orca to reuse. workerKind?: 'prompt-returning-agent' | 'bare-shell' + // Why gated: advertising a verb the depth cap will reject just burns a turn. + canDispatchSubWorkers?: boolean } // Why: 5 minutes is frequent enough that the coordinator's stale-heartbeat @@ -138,7 +140,9 @@ ${postDoneInstructions}` const drift = params.baseDrift && params.baseDrift.behind > 0 ? buildDriftSection(params.baseDrift) : '' - return `${header}${drift} + const subDispatch = params.canDispatchSubWorkers ? buildSubDispatchSection(cli) : '' + + return `${header}${drift}${subDispatch} === TASK === ${params.taskSpec}` @@ -186,6 +190,26 @@ preamble + TASK block, which arrives as new input. Treat that as supervised work under the new Dispatch; ignore stale follow-ups from the settled task.` } +// Why the whole section is omitted rather than softened when nesting is off: a +// worker told it "usually cannot" delegate still tries, then reports the refusal +// as a blocker. +function buildSubDispatchSection(cli: string): string { + return ` + +=== SUB-DISPATCH === +You may dispatch sub-workers for this task. Bind your own Run first, then create +and start each one: + + ${cli} orchestration run-create --objective "" --json + ${cli} orchestration task-create --spec "" --json + ${cli} orchestration worker-start --task --worktree current --agent --json + +You own those sub-workers: wait for their worker_done, and do not report your own +until they have settled. Nesting is capped, so a sub-worker of yours may not be +able to dispatch further. +---` +} + function buildDriftSection(drift: NonNullable): string { const subjects = drift.recentSubjects.map((s) => ` - ${s}`).join('\n') return ` diff --git a/src/main/runtime/rpc/methods/orchestration-federation.ts b/src/main/runtime/rpc/methods/orchestration-federation.ts index da4e29d4fc0..4b48632c39a 100644 --- a/src/main/runtime/rpc/methods/orchestration-federation.ts +++ b/src/main/runtime/rpc/methods/orchestration-federation.ts @@ -236,6 +236,9 @@ export const ORCHESTRATION_FEDERATION_ATTACH_METHODS: RpcMethod[] = [ workerHandle: terminalHandle, dispatchCapability: capability, devMode: params.devMode, + // Why the worker host's own setting: enforcement runs here, with this + // host's code, against this host's cap. + canDispatchSubWorkers: (params.depth ?? 1) < runtime.getNestedWorkerMaxDepth(), cliCommand: runtime.getTerminalOrchestrationCliCommand(terminalHandle) }) ) diff --git a/src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts b/src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts index e77f1fd78fe..4a87c53059d 100644 --- a/src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts +++ b/src/main/runtime/rpc/methods/orchestration-tasks-dispatch.test.ts @@ -433,6 +433,30 @@ describe('orchestration RPC methods', () => { expect(db.getDispatchContext(task.id)).toBeUndefined() }) + it('dry-run previews the same sub-dispatch section as a real dispatch', async () => { + setup() + vi.spyOn(runtime, 'getNestedWorkerMaxDepth').mockReturnValue(2) + const task = db.createTask({ spec: 'work' }) + + const preview = (await call('orchestration.dispatch', { + task: task.id, + to: 'term_a', + dryRun: true, + from: 'term_coord' + })) as { preamble: string } + const dispatched = (await call('orchestration.dispatch', { + task: task.id, + to: 'term_a', + returnPreamble: true, + from: 'term_coord' + })) as { preamble: string } + + const section = (preamble: string) => + preamble.match(/=== SUB-DISPATCH ===[\s\S]*?(?=\n=== TASK ===)/)?.[0] + expect(section(preview.preamble)).toBeDefined() + expect(section(preview.preamble)).toBe(section(dispatched.preamble)) + }) + it('returnPreamble includes preamble in the response', async () => { setup() const task = db.createTask({ spec: 'work' }) diff --git a/src/main/runtime/rpc/methods/orchestration-workers.ts b/src/main/runtime/rpc/methods/orchestration-workers.ts index 6bc37f0c0b3..0b8891481cd 100644 --- a/src/main/runtime/rpc/methods/orchestration-workers.ts +++ b/src/main/runtime/rpc/methods/orchestration-workers.ts @@ -234,6 +234,7 @@ export const ORCHESTRATION_WORKER_START_METHODS: RpcMethod[] = [ failedStage = 'dispatch_input' const preamble = buildDispatchPreamble({ + canDispatchSubWorkers: started.dispatch.depth < runtime.getNestedWorkerMaxDepth(), taskId: task.id, dispatchId: started.dispatch.id, taskSpec: task.spec, diff --git a/src/main/runtime/rpc/methods/orchestration.ts b/src/main/runtime/rpc/methods/orchestration.ts index 2cf7c8b1a10..4045ce11409 100644 --- a/src/main/runtime/rpc/methods/orchestration.ts +++ b/src/main/runtime/rpc/methods/orchestration.ts @@ -1618,9 +1618,15 @@ export const ORCHESTRATION_METHODS: RpcMethod[] = [ // Why: dry-run previews the preamble without mutating state, so it skips the ready-status check and uses a placeholder dispatchId. if (params.dryRun) { + const maxDepth = runtime.getNestedWorkerMaxDepth() + const previewDepth = db.resolveChildDispatchDepth( + resolveDispatchCreator(runtime, params.from), + maxDepth + ) const preamble = buildDispatchPreamble({ taskId: task.id, dispatchId: 'ctx_dryrun', + canDispatchSubWorkers: previewDepth < maxDepth, taskSpec: task.spec, coordinatorHandle: params.from ?? 'coordinator', workerHandle: params.to ?? 'worker', @@ -1685,6 +1691,7 @@ export const ORCHESTRATION_METHODS: RpcMethod[] = [ const preamble = buildDispatchPreamble({ taskId: task.id, dispatchId: ctx.id, + canDispatchSubWorkers: ctx.depth < runtime.getNestedWorkerMaxDepth(), taskSpec: task.spec, coordinatorHandle: params.from ?? 'coordinator', workerHandle: to, @@ -1733,6 +1740,7 @@ export const ORCHESTRATION_METHODS: RpcMethod[] = [ taskId: task.id, // Why: use the real ctx.id when present so the preview matches what was injected; placeholder when no dispatch has occurred yet. dispatchId: ctx?.id ?? 'ctx_preview', + canDispatchSubWorkers: (ctx?.depth ?? 1) < runtime.getNestedWorkerMaxDepth(), taskSpec: task.spec, coordinatorHandle: params.from ?? 'coordinator', workerHandle, diff --git a/src/renderer/src/components/settings/OrchestrationPane.test.tsx b/src/renderer/src/components/settings/OrchestrationPane.test.tsx index 231991205d4..e8652197240 100644 --- a/src/renderer/src/components/settings/OrchestrationPane.test.tsx +++ b/src/renderer/src/components/settings/OrchestrationPane.test.tsx @@ -5,7 +5,12 @@ import { createRoot, type Root } from 'react-dom/client' import { renderToStaticMarkup } from 'react-dom/server' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { getOrchestrationUsageExamples } from '@/lib/orchestration-usage-examples' +import { getDefaultSettings } from '../../../../shared/constants' +import type { GlobalSettings } from '../../../../shared/global-settings-types' +import { useAppStore } from '../../store' import { OrchestrationPane } from './OrchestrationPane' +import { getOrchestrationPaneSearchEntries } from './orchestration-search' +import { matchesSettingsSearch } from './settings-search' const INSTALL_COMMAND = 'npx skills add https://github.com/stablyai/orca --skill orchestration --global' @@ -16,7 +21,8 @@ const WINDOWS_INSTALL_COMMAND = const mocks = vi.hoisted(() => ({ dialogProps: [] as Record[], panelProps: [] as Record[], - skillInstalled: true + skillInstalled: true, + updateSettings: vi.fn() })) vi.mock('./AgentSkillSetupPanel', () => ({ @@ -98,18 +104,32 @@ vi.mock('@/hooks/useDetectedAgents', () => ({ let root: Root | null = null let container: HTMLDivElement | null = null -async function renderPane(): Promise { +globalThis.IS_REACT_ACT_ENVIRONMENT = true + +function setNativeValue(input: HTMLInputElement, text: string): void { + const setValue = Object.getOwnPropertyDescriptor(HTMLInputElement.prototype, 'value')?.set + setValue?.call(input, text) +} + +function getPaneProps(settings: GlobalSettings = getDefaultSettings('/tmp')) { + return { settings, updateSettings: mocks.updateSettings } +} + +async function renderPane( + settings: GlobalSettings = getDefaultSettings('/tmp') +): Promise { container = document.createElement('div') document.body.appendChild(container) root = createRoot(container) await act(async () => { - root?.render() + root?.render() }) return container } describe('OrchestrationPane', () => { beforeEach(() => { + useAppStore.setState({ settingsSearchQuery: '' }) Object.defineProperty(window, 'api', { configurable: true, value: { @@ -146,10 +166,12 @@ describe('OrchestrationPane', () => { mocks.dialogProps.length = 0 mocks.panelProps.length = 0 mocks.skillInstalled = true + mocks.updateSettings.mockReset() + delete (globalThis as { __ORCA_WEB_CLIENT__?: boolean }).__ORCA_WEB_CLIENT__ }) it('keeps skill setup visible after install and shows agent coverage plus examples', () => { - const markup = renderToStaticMarkup() + const markup = renderToStaticMarkup() expect(markup).toContain('Orchestration skill') expect(markup).toContain('Installed') @@ -170,6 +192,69 @@ describe('OrchestrationPane', () => { expect(markup).toContain('Re-check') }) + it('renders nested worker depth as an unbounded positive whole-number input', () => { + const markup = renderToStaticMarkup() + + expect(markup).toContain('Nested worker depth') + expect(markup).toContain('type="number"') + expect(markup).toContain('aria-label="Nested worker depth"') + expect(markup).toContain('min="1"') + expect(markup).not.toContain('max=') + expect(markup).not.toContain('Default:') + expect(markup.indexOf('Nested worker depth')).toBeGreaterThan( + markup.indexOf('Orchestration skill') + ) + expect(matchesSettingsSearch('nested worker', getOrchestrationPaneSearchEntries())).toBe(true) + }) + + it('keeps the nested depth row visible when settings search routes to Orchestration', () => { + useAppStore.setState({ settingsSearchQuery: 'Nested worker' }) + + const markup = renderToStaticMarkup() + + expect(markup).toContain('Nested worker depth') + expect(markup).toContain('aria-label="Nested worker depth"') + }) + + it('commits a whole-number depth and rejects fractional values', async () => { + const rendered = await renderPane() + const input = rendered.querySelector( + 'input[aria-label="Nested worker depth"]' + ) + if (!input) { + throw new Error('Nested worker depth input was not rendered') + } + + await act(async () => { + setNativeValue(input, '5') + input.dispatchEvent(new Event('input', { bubbles: true })) + input.dispatchEvent(new FocusEvent('focusout', { bubbles: true })) + }) + expect(mocks.updateSettings).toHaveBeenCalledWith({ nestedWorkerMaxDepth: 5 }) + + mocks.updateSettings.mockClear() + await act(async () => { + setNativeValue(input, '2.5') + input.dispatchEvent(new Event('input', { bubbles: true })) + input.dispatchEvent(new FocusEvent('focusout', { bubbles: true })) + }) + expect(mocks.updateSettings).not.toHaveBeenCalled() + expect(input.value).toBe('1') + }) + + it('keeps host-only nested depth out of paired web clients', () => { + ;(globalThis as { __ORCA_WEB_CLIENT__?: boolean }).__ORCA_WEB_CLIENT__ = true + const markup = renderToStaticMarkup() + + expect(markup).not.toContain('Nested worker depth') + expect( + matchesSettingsSearch( + 'nested worker', + getOrchestrationPaneSearchEntries({ includeNestedWorkerDepth: false }) + ) + ).toBe(false) + }) + it('passes update commands to the main panel without an installed manual-copy path', async () => { const rendered = await renderPane() diff --git a/src/renderer/src/components/settings/OrchestrationPane.tsx b/src/renderer/src/components/settings/OrchestrationPane.tsx index f60290968f2..e3230f4cf86 100644 --- a/src/renderer/src/components/settings/OrchestrationPane.tsx +++ b/src/renderer/src/components/settings/OrchestrationPane.tsx @@ -30,6 +30,14 @@ import { OrchestrationSkillAgentCoverage } from './OrchestrationSkillAgentCovera import { SkillUsageExamplesSection } from './SkillUsageExamplesSection' import { OrchestrationSkillPromptDialog } from './OrchestrationSkillPromptDialog' import { translate } from '@/i18n/i18n' +import type { GlobalSettings } from '../../../../shared/global-settings-types' +import { resolveNestedWorkerMaxDepth } from '../../../../shared/nested-worker-depth' +import { isPairedWebClientWindow } from '@/lib/desktop-window-chrome' +import { NumberField } from './SettingsFormControls' +import { + getNestedWorkerDepthDescription, + getNestedWorkerDepthTitle +} from './nested-worker-depth-copy' const EXAMPLE_ICONS = { handoff: ArrowRightLeft, @@ -43,9 +51,21 @@ function resolveOrchestrationExampleIcon(example: SkillUsageExample): LucideIcon return EXAMPLE_ICONS[example.id as keyof typeof EXAMPLE_ICONS] ?? Workflow } -export function OrchestrationPane(): React.JSX.Element { +type OrchestrationPaneProps = { + settings: GlobalSettings + updateSettings: (updates: Partial) => void | Promise +} + +export function OrchestrationPane({ + settings, + updateSettings +}: OrchestrationPaneProps): React.JSX.Element { const searchQuery = useAppStore((s) => s.settingsSearchQuery) - const showOrchestration = matchesSettingsSearch(searchQuery, getOrchestrationPaneSearchEntries()) + const showNestedWorkerDepth = !isPairedWebClientWindow() + const searchEntries = getOrchestrationPaneSearchEntries({ + includeNestedWorkerDepth: showNestedWorkerDepth + }) + const showOrchestration = matchesSettingsSearch(searchQuery, searchEntries) const [skillPromptOpen, setSkillPromptOpen] = useState(false) const activeSkillRuntime = useActiveProjectSkillRuntime() const orchestrationInstallCommand = !activeSkillRuntime.installDisabledReason @@ -87,7 +107,8 @@ export function OrchestrationPane(): React.JSX.Element { 'auto.components.settings.OrchestrationPane.2aacdb0517', 'Coordinate coding agents across handoffs, worktree handovers, and child-agent work.' )} - keywords={getOrchestrationPaneSearchEntries()[0].keywords} + keywords={searchEntries[0].keywords} + forceVisible className="space-y-5 py-2" > + {showNestedWorkerDepth ? ( + { + void updateSettings({ nestedWorkerMaxDepth }) + }} + /> + ) : null} + - {isSectionMounted('orchestration') ? : null} + {isSectionMounted('orchestration') ? ( + + ) : null} {linearConnected ? ( diff --git a/src/renderer/src/components/settings/SettingsFormControls.tsx b/src/renderer/src/components/settings/SettingsFormControls.tsx index b82e90a0984..bb38bcc8d4e 100644 --- a/src/renderer/src/components/settings/SettingsFormControls.tsx +++ b/src/renderer/src/components/settings/SettingsFormControls.tsx @@ -265,8 +265,9 @@ type NumberFieldProps = { value: number defaultValue?: number min: number - max: number + max?: number step?: number + integer?: boolean onChange: (value: number) => void suffix?: string } @@ -312,6 +313,7 @@ export function NumberField({ min, max, step = 1, + integer = false, onChange, suffix }: NumberFieldProps): React.JSX.Element { @@ -332,8 +334,8 @@ export function NumberField({ return } const next = Number(trimmed) - if (Number.isFinite(next)) { - const clamped = Math.min(max, Math.max(min, next)) + if (Number.isFinite(next) && (!integer || Number.isSafeInteger(next))) { + const clamped = max === undefined ? Math.max(min, next) : Math.min(max, Math.max(min, next)) onChange(clamped) setDraft(String(clamped)) } else { @@ -363,6 +365,7 @@ export function NumberField({ min={min} max={max} step={step} + aria-label={label} value={draft} onChange={(e) => setDraft(e.target.value)} onBlur={commit} diff --git a/src/renderer/src/components/settings/nested-worker-depth-copy.ts b/src/renderer/src/components/settings/nested-worker-depth-copy.ts new file mode 100644 index 00000000000..6f2754b637d --- /dev/null +++ b/src/renderer/src/components/settings/nested-worker-depth-copy.ts @@ -0,0 +1,29 @@ +import { translate } from '@/i18n/i18n' +import { searchKeywords } from './settings-search-keywords' + +export function getNestedWorkerDepthTitle(): string { + return translate( + 'auto.components.settings.OrchestrationPane.nestedWorkerDepthTitle', + 'Nested worker depth' + ) +} + +export function getNestedWorkerDepthDescription(): string { + return translate( + 'auto.components.settings.OrchestrationPane.nestedWorkerDepthDescription', + 'How many generations of dispatched workers may spawn their own workers. 1 keeps the agent tree flat: a coordinator dispatches workers, and those workers do not dispatch.' + ) +} + +export function getNestedWorkerDepthSearchKeywords(): string[] { + return searchKeywords([ + { key: 'auto.components.settings.agents.search.96ba2373b6', fallback: 'agent' }, + { key: 'auto.components.settings.general.search.ec5049e510', fallback: 'nested' }, + { key: 'auto.components.settings.orchestration.search.741dfc03fa', fallback: 'worker' }, + { key: 'auto.components.settings.orchestration.search.eee028ae14', fallback: 'dispatch' }, + { + key: 'auto.components.settings.orchestration.search.f5d39af41e', + fallback: 'child agents' + } + ]) +} diff --git a/src/renderer/src/components/settings/orchestration-search.ts b/src/renderer/src/components/settings/orchestration-search.ts index a9d22636030..5d15a6dd88c 100644 --- a/src/renderer/src/components/settings/orchestration-search.ts +++ b/src/renderer/src/components/settings/orchestration-search.ts @@ -1,8 +1,19 @@ import { translate } from '@/i18n/i18n' import { translateSearchKeyword } from './settings-search-keywords' import { createLocalizedCatalog } from '@/i18n/localized-catalog' +import { + getNestedWorkerDepthDescription, + getNestedWorkerDepthSearchKeywords, + getNestedWorkerDepthTitle +} from './nested-worker-depth-copy' -export const getOrchestrationPaneSearchEntries = createLocalizedCatalog(() => [ +type OrchestrationPaneSearchOptions = { + includeNestedWorkerDepth?: boolean +} + +const NESTED_WORKER_DEPTH_SEARCH_ENTRY_ID = 'nested-worker-depth' + +const getAllOrchestrationPaneSearchEntries = createLocalizedCatalog(() => [ { title: translate( 'auto.components.settings.orchestration.search.c34045764e', @@ -68,5 +79,22 @@ export const getOrchestrationPaneSearchEntries = createLocalizedCatalog(() => [ 'child agents' ) ] + }, + { + id: NESTED_WORKER_DEPTH_SEARCH_ENTRY_ID, + title: getNestedWorkerDepthTitle(), + description: getNestedWorkerDepthDescription(), + keywords: getNestedWorkerDepthSearchKeywords() } ]) + +export function getOrchestrationPaneSearchEntries({ + includeNestedWorkerDepth = true +}: OrchestrationPaneSearchOptions = {}) { + const entries = getAllOrchestrationPaneSearchEntries() + return includeNestedWorkerDepth + ? entries + : entries.filter( + (entry) => !('id' in entry) || entry.id !== NESTED_WORKER_DEPTH_SEARCH_ENTRY_ID + ) +} diff --git a/src/renderer/src/hooks/useSettingsNavigationMetadata.test.ts b/src/renderer/src/hooks/useSettingsNavigationMetadata.test.ts index b0b694afcf1..7cc5c009f9d 100644 --- a/src/renderer/src/hooks/useSettingsNavigationMetadata.test.ts +++ b/src/renderer/src/hooks/useSettingsNavigationMetadata.test.ts @@ -47,6 +47,22 @@ describe('settings navigation metadata', () => { ]) }) + it('owns nested worker depth under Orchestration on desktop', () => { + const sections = buildSettingsNavigationMetadata({ + isMac: false, + isWindows: false, + isWebClient: false, + repos: [repo] + }) + const agents = sections.find((section) => section.id === 'agents') + const orchestration = sections.find((section) => section.id === 'orchestration') + + expect(agents?.searchEntries.map((entry) => entry.title)).not.toContain('Nested worker depth') + expect(orchestration?.searchEntries.map((entry) => entry.title)).toContain( + 'Nested worker depth' + ) + }) + it('adds the Linear capability section right after Orchestration only when connected', () => { expect(ids()).not.toContain('linear') @@ -157,6 +173,12 @@ describe('settings navigation metadata', () => { expect(shortcuts?.searchEntries.map((entry) => entry.title)).not.toContain( 'New mobile emulator tab' ) + const agents = webSections.find((section) => section.id === 'agents') + expect(agents?.searchEntries.map((entry) => entry.title)).not.toContain('Nested worker depth') + const orchestration = webSections.find((section) => section.id === 'orchestration') + expect(orchestration?.searchEntries.map((entry) => entry.title)).not.toContain( + 'Nested worker depth' + ) }) it('keeps the Browser shortcut searchable for a capable web runtime', () => { diff --git a/src/renderer/src/hooks/useSettingsNavigationMetadata.ts b/src/renderer/src/hooks/useSettingsNavigationMetadata.ts index 07f69bbbcb1..9983ba38f0a 100644 --- a/src/renderer/src/hooks/useSettingsNavigationMetadata.ts +++ b/src/renderer/src/hooks/useSettingsNavigationMetadata.ts @@ -202,7 +202,9 @@ export function buildSettingsNavigationMetadata({ 'Coordinate multiple coding agents through Orca.' ), icon: Network, - searchEntries: getOrchestrationPaneSearchEntries(), + searchEntries: getOrchestrationPaneSearchEntries({ + includeNestedWorkerDepth: !isWebClient + }), group: 'capabilities' }, // Why: only surfaced once Linear is connected — a capability that needs a diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 2432d4c5952..bbffbf8a974 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -7459,7 +7459,9 @@ "9bedd2a6e5": "Enables agents to hand off context and coordinate work through Orca.", "07641b9768": "Orchestration skill", "2aacdb0517": "Coordinate coding agents across handoffs, worktree handovers, and child-agent work.", - "191ac34567": "Agent Orchestration" + "191ac34567": "Agent Orchestration", + "nestedWorkerDepthTitle": "Nested worker depth", + "nestedWorkerDepthDescription": "How many generations of dispatched workers may spawn their own workers. 1 keeps the agent tree flat: a coordinator dispatches workers, and those workers do not dispatch." }, "OrchestrationSetupCard": { "e7d2a5146c": "Enables agents to hand off context and coordinate work through Orca.", diff --git a/src/shared/nested-worker-depth.test.ts b/src/shared/nested-worker-depth.test.ts index a91dd87ad00..f3c173988a6 100644 --- a/src/shared/nested-worker-depth.test.ts +++ b/src/shared/nested-worker-depth.test.ts @@ -27,6 +27,7 @@ describe('resolveNestedWorkerMaxDepth', () => { ['fractional', 1.5], ['NaN', Number.NaN], ['Infinity', Number.POSITIVE_INFINITY], + ['unsafe integer', Number.MAX_SAFE_INTEGER + 1], ['null', null] ])('falls back to the default for %s', (_label, value) => { expect(resolveNestedWorkerMaxDepth({ nestedWorkerMaxDepth: value as unknown as number })).toBe( diff --git a/src/shared/nested-worker-depth.ts b/src/shared/nested-worker-depth.ts index 927f9801c12..d927bfea77e 100644 --- a/src/shared/nested-worker-depth.ts +++ b/src/shared/nested-worker-depth.ts @@ -23,11 +23,11 @@ export function nestedWorkerDepthExceededMessage(childDepth: number, maxDepth: n export const NESTED_WORKER_DEPTH_EXCEEDED_NEXT_STEPS: readonly string[] = [ 'Do the work in this terminal instead of dispatching a sub-worker.', - 'To allow deeper nesting, open Settings → Agents in the Orca desktop app and raise "Nested worker depth".' + 'To allow deeper nesting, open Settings → Orchestration in the Orca desktop app and raise "Nested worker depth".' ] /** - * Clamp to a usable integer. Anything that is not a whole number >= 1 falls back + * Clamp to a usable integer. Anything that is not a safe whole number >= 1 falls back * to the default rather than disabling the fence: a malformed setting must not * be a way to get unlimited nesting. */ @@ -35,7 +35,7 @@ export function resolveNestedWorkerMaxDepth( settings: Pick | null | undefined ): number { const raw = settings?.nestedWorkerMaxDepth - if (typeof raw !== 'number' || !Number.isInteger(raw) || raw < 1) { + if (typeof raw !== 'number' || !Number.isSafeInteger(raw) || raw < 1) { return NESTED_WORKER_MAX_DEPTH_DEFAULT } return raw From 08a447dfb2413381c99be53a661125dc7a01582d Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 26 Aug 2026 16:30:26 -0700 Subject: [PATCH 15/19] fix(terminal): size the pre-Enter wait to what the host actually ingests (#15925) (#16586) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(terminal): size the pre-Enter wait to what the host actually ingests The Windows agent-prompt submit delay was a flat 1_500 ms frozen from the client's process.platform at import. Measured on two real Win11 hosts, ConPTY ingests a bracketed paste linearly at ~0.009-0.010 ms/byte, so the constant was both far too long for a 2-8 KB prompt (14-89 ms of real cost) and too short past ~145 KB — at 160 KB one host took 1_499 ms, meaning Enter landed mid-paste, exactly the corruption the delay exists to prevent, up to the 16 MB input ceiling. Replace it with getTerminalPasteIngestMs(platform, byteLength) and derive every pre-Enter wait from it: - open-loop fallback = 500 ms settle + ingest bound, uncapped - claude/codex render gate cannot start its quiet window before the ingest bound elapses (an agent that repaints mid-ingest could otherwise satisfy marker-then-quiet while ConPTY was still feeding the paste), and its 8 s hard cap now sits on top of the ingest bound instead of standing in for it - the plain terminal.send suffix path, which had an undocumented flat 500 ms The rate follows the host that owns the pty transport, not the client: a WSL pane is spawned as wsl.exe behind the Windows pseudoconsole so it still pays ConPTY, while an SSH pane follows the relay's reported remotePlatform. Also swap the inter-chunk setTimeout(0) for setImmediate. It cost a full ~15 ms Windows timer tick per 16 KiB chunk (~0.95 s/MB) while pacing ~1.07 MB/s — 11x above ConPTY's drain rate — so it never provided backpressure; the event-loop yield it did provide is preserved. * fix(terminal): stop double-charging paste ingest in the render gate The render gate's hard cap is armed twice -- once at arm() and again when the show-cursor marker arrives -- but it re-added the whole ingest window each time while the ingest clock itself runs once from gate construction. A marker seen mid-ingest pushed the cap out by a second full ingest term (~34 s instead of ~24 s for a 1 MB prompt on ConPTY). Capture the ingest deadline absolutely and arm with what is left of it. Also thread the request AbortSignal through terminal.send so the now payload-scaled suffix wait can be cancelled: at 16 MB it runs ~262 s, well past the CLI's 60 s request budget, and previously nothing stopped the eventual Enter. Cleanups: a pty record's connectionId is only ever an SSH target id, so the wsl: relay-id guard in getPtyWriteHostPlatform was dead; and hoisting action.text removes both non-null assertions in writeTerminalAction. --- .../agent-prompt-submission-runtime.test.ts | 13 +- ...pt-submission-windows-submit-delay.test.ts | 377 +++++++++++++++++- src/main/runtime/orca-runtime.test.ts | 30 +- src/main/runtime/orca-runtime.ts | 143 +++++-- .../methods/terminal/terminal-send-method.ts | 1 + .../rpc/terminal-agent-prompt-send.test.ts | 31 +- src/shared/agent-prompt-injection.test.ts | 59 ++- src/shared/agent-prompt-injection.ts | 48 ++- 8 files changed, 636 insertions(+), 66 deletions(-) diff --git a/src/main/runtime/agent-prompt-submission-runtime.test.ts b/src/main/runtime/agent-prompt-submission-runtime.test.ts index f26ff8030b8..69eb0a01079 100644 --- a/src/main/runtime/agent-prompt-submission-runtime.test.ts +++ b/src/main/runtime/agent-prompt-submission-runtime.test.ts @@ -1,7 +1,8 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { AGENT_PROMPT_BRACKETED_PASTE_END, - AGENT_PROMPT_SUBMIT_DELAY_MS + buildAgentPromptPasteBytes, + getAgentPromptSubmitDelayMs } from '../../shared/agent-prompt-injection' import { AGENT_PROMPT_TEST_WORKTREE_PATH, @@ -664,8 +665,14 @@ describe('agent prompt submission runtime', () => { }) const rejected = expect(submission).rejects.toThrow('request_aborted') - // Why: the submit delay is 1_500 on Windows (ConPTY); a hardcoded 500 aborts before the Enter there. - await vi.advanceTimersByTimeAsync(AGENT_PROMPT_SUBMIT_DELAY_MS) + // Why compute it: the submit delay now follows the payload size and the executing host, + // so a hardcoded number aborts before the Enter on some lanes. + await vi.advanceTimersByTimeAsync( + getAgentPromptSubmitDelayMs( + process.platform, + Buffer.byteLength(buildAgentPromptPasteBytes('review this'), 'utf8') + ) + ) // Why: pin the phase boundary so drift fails here instead of as an empty post-abort array. expect(writes.filter((data) => data === '\r')).toHaveLength(1) controller.abort() diff --git a/src/main/runtime/agent-prompt-submission-windows-submit-delay.test.ts b/src/main/runtime/agent-prompt-submission-windows-submit-delay.test.ts index fdb90700de1..f6298df8b17 100644 --- a/src/main/runtime/agent-prompt-submission-windows-submit-delay.test.ts +++ b/src/main/runtime/agent-prompt-submission-windows-submit-delay.test.ts @@ -1,18 +1,22 @@ import { afterEach, describe, expect, it, vi } from 'vitest' -import type * as AgentPromptInjection from '../../shared/agent-prompt-injection' +import { + buildAgentPromptPasteBytes, + getAgentPromptSubmitDelayMs, + getTerminalPasteIngestMs +} from '../../shared/agent-prompt-injection' +import { setSshTargetRegistryHandlers } from '../ssh/ssh-target-registry' import { OrcaRuntimeService } from './orca-runtime' import { makeStore } from './runtime-rpc-worktree-store-fixtures' -// Why: vi.mock factories are hoisted above module-scope consts, so the delay must be hoisted too. -const { WINDOWS_SUBMIT_DELAY_MS } = vi.hoisted(() => ({ WINDOWS_SUBMIT_DELAY_MS: 1_500 })) const WORKTREE_PATH = '/tmp/worktree-a' +const PTY_ID = 'pty-prompt' +// Why: the submit delay is resolved per send from the *executing* host and the payload size, +// so these suites drive it by stubbing that host rather than by mocking the shared module. +const originalPlatform = Object.getOwnPropertyDescriptor(process, 'platform')! -// Why: AGENT_PROMPT_SUBMIT_DELAY_MS is frozen from process.platform at import, so the 1_500 ConPTY -// path is otherwise unreachable on the Linux/macOS lanes that run this suite. -vi.mock('../../shared/agent-prompt-injection', async (importOriginal) => ({ - ...(await importOriginal()), - AGENT_PROMPT_SUBMIT_DELAY_MS: WINDOWS_SUBMIT_DELAY_MS -})) +function useHostPlatform(platform: NodeJS.Platform): void { + Object.defineProperty(process, 'platform', { configurable: true, value: platform }) +} vi.mock('../git/worktree', () => ({ listWorktrees: vi.fn().mockResolvedValue([ @@ -35,18 +39,24 @@ vi.mock('../git/worktree', () => ({ ]) })) -// Why: 'aider' is not a settlement agent, so submission takes the fixed-delay branch under test. +// Why: 'aider' is not a settlement agent, so submission takes the open-loop delay under test. async function createPromptRuntime(): Promise<{ runtime: OrcaRuntimeService handle: string writes: string[] + submitTimes: number[] }> { const runtime = new OrcaRuntimeService(makeStore() as never) const writes: string[] = [] + const submitTimes: number[] = [] + const startedAt = Date.now() runtime.setPtyController({ - spawn: vi.fn().mockResolvedValue({ id: 'pty-prompt' }), + spawn: vi.fn().mockResolvedValue({ id: PTY_ID }), write: (_ptyId, data) => { writes.push(data) + if (data === '\r') { + submitTimes.push(Date.now() - startedAt) + } return true }, kill: () => true, @@ -55,23 +65,56 @@ async function createPromptRuntime(): Promise<{ const terminal = await runtime.createTerminal(`path:${WORKTREE_PATH}`, { launchAgent: 'aider' }) - return { runtime, handle: terminal.handle, writes } + return { runtime, handle: terminal.handle, writes, submitTimes } +} + +// Why 12 KB: one chunk (so the write loop costs no clock), yet large enough that the ConPTY +// term and the fast-platform term are far apart. +const HOST_PROBE_PROMPT = 'p'.repeat(12_000) + +/** The pane's execution host is spawn-time state the fixture cannot express; patch the + * record the runtime actually consults so remote and WSL panes are reachable here. */ +function patchPtyRecord(runtime: OrcaRuntimeService, patch: Record): void { + const ptys = (runtime as unknown as { ptysById: Map> }).ptysById + const record = ptys.get(PTY_ID) + expect(record).toBeDefined() + Object.assign(record!, patch) +} + +function registerSshRemotePlatform(platform: NodeJS.Platform | undefined): void { + setSshTargetRegistryHandlers({ + connect: null, + getState: () => ({ remotePlatform: platform }) as never + }) } function countSubmits(writes: readonly string[]): number { return writes.filter((data) => data === '\r').length } -describe('agent prompt submission at the Windows submit delay', () => { - afterEach(() => vi.useRealTimers()) +function submitDelayFor(prompt: string, platform: NodeJS.Platform): number { + return getAgentPromptSubmitDelayMs( + platform, + Buffer.byteLength(buildAgentPromptPasteBytes(prompt), 'utf8') + ) +} - it('holds Enter until the full platform delay elapses', async () => { +describe('agent prompt submit delay on a ConPTY host', () => { + afterEach(() => { + vi.useRealTimers() + Object.defineProperty(process, 'platform', originalPlatform) + setSshTargetRegistryHandlers({ connect: null, getState: null }) + }) + + it('holds Enter for the payload ingest plus the settle window', async () => { + useHostPlatform('win32') vi.useFakeTimers() const { runtime, handle, writes } = await createPromptRuntime() + const delayMs = submitDelayFor('review this', 'win32') const submission = runtime.sendTerminalAgentPrompt(handle, 'review this') const stalled = expect(submission).rejects.toThrow('agent_prompt_stalled') - await vi.advanceTimersByTimeAsync(WINDOWS_SUBMIT_DELAY_MS - 1) + await vi.advanceTimersByTimeAsync(delayMs - 1) expect(countSubmits(writes)).toBe(0) await vi.advanceTimersByTimeAsync(1) expect(countSubmits(writes)).toBe(1) @@ -80,7 +123,46 @@ describe('agent prompt submission at the Windows submit delay', () => { await stalled }) + it('no longer burns the flat 1_500 ms on a common-sized prompt', async () => { + useHostPlatform('win32') + vi.useFakeTimers() + const { runtime, handle, writes } = await createPromptRuntime() + const prompt = 'x'.repeat(8_000) + // Measured ConPTY ingest for 8 KB is 60-89 ms; the old constant charged 1_500 ms. + const delayMs = submitDelayFor(prompt, 'win32') + expect(delayMs).toBeLessThan(700) + const submission = runtime.sendTerminalAgentPrompt(handle, prompt) + const stalled = expect(submission).rejects.toThrow('agent_prompt_stalled') + + await vi.advanceTimersByTimeAsync(delayMs) + expect(countSubmits(writes)).toBe(1) + + await vi.runAllTimersAsync() + await stalled + }) + + it('does not write Enter while ConPTY is still ingesting a large paste', async () => { + useHostPlatform('win32') + vi.useFakeTimers() + const { runtime, handle, writes, submitTimes } = await createPromptRuntime() + const prompt = 'y'.repeat(320_000) + const submission = runtime.sendTerminalAgentPrompt(handle, prompt) + const stalled = expect(submission).rejects.toThrow('agent_prompt_stalled') + + // Every byte is already handed to node-pty here -- the hazard is that the *host* is + // still feeding them to the child. 3_342 ms is the measured 320 KB ConPTY ingest. + await vi.advanceTimersByTimeAsync(3_342) + expect(writes.filter((data) => data.includes('\x1b[201~'))).toHaveLength(1) + expect(countSubmits(writes)).toBe(0) + + await vi.runAllTimersAsync() + expect(submitTimes).toHaveLength(1) + expect(submitTimes[0]).toBeGreaterThan(3_342) + await stalled + }) + it('does not send another Enter after cancellation during verification', async () => { + useHostPlatform('win32') vi.useFakeTimers() const controller = new AbortController() const { runtime, handle, writes } = await createPromptRuntime() @@ -89,7 +171,7 @@ describe('agent prompt submission at the Windows submit delay', () => { }) const rejected = expect(submission).rejects.toThrow('request_aborted') - await vi.advanceTimersByTimeAsync(WINDOWS_SUBMIT_DELAY_MS) + await vi.advanceTimersByTimeAsync(submitDelayFor('review this', 'win32')) expect(countSubmits(writes)).toBe(1) controller.abort() await vi.runAllTimersAsync() @@ -97,4 +179,265 @@ describe('agent prompt submission at the Windows submit delay', () => { await rejected expect(countSubmits(writes)).toBe(1) }) + + it('charges a non-Windows host only the settle window', async () => { + useHostPlatform('darwin') + vi.useFakeTimers() + const { runtime, handle, writes } = await createPromptRuntime() + const delayMs = submitDelayFor(HOST_PROBE_PROMPT, 'darwin') + expect(delayMs).toBeLessThan(submitDelayFor(HOST_PROBE_PROMPT, 'win32')) + const submission = runtime.sendTerminalAgentPrompt(handle, HOST_PROBE_PROMPT) + const stalled = expect(submission).rejects.toThrow('agent_prompt_stalled') + + await vi.advanceTimersByTimeAsync(delayMs - 1) + expect(countSubmits(writes)).toBe(0) + await vi.advanceTimersByTimeAsync(1) + expect(countSubmits(writes)).toBe(1) + + await vi.runAllTimersAsync() + await stalled + }) +}) + +describe('agent prompt submit delay follows the execution host', () => { + afterEach(() => { + vi.useRealTimers() + Object.defineProperty(process, 'platform', originalPlatform) + setSshTargetRegistryHandlers({ connect: null, getState: null }) + }) + + it('still charges ConPTY for a WSL pane, which is spawned through it', async () => { + useHostPlatform('win32') + vi.useFakeTimers() + const { runtime, handle, writes } = await createPromptRuntime() + // The shell is Linux, but node-pty spawned wsl.exe behind the Windows pseudoconsole, + // so the paste still pays ConPTY ingest -- unlike the reported `hostPlatform`. + patchPtyRecord(runtime, { isWsl: true, wslDistro: 'Ubuntu' }) + const delayMs = submitDelayFor(HOST_PROBE_PROMPT, 'win32') + expect(delayMs).toBeGreaterThan(submitDelayFor(HOST_PROBE_PROMPT, 'linux')) + const submission = runtime.sendTerminalAgentPrompt(handle, HOST_PROBE_PROMPT) + const stalled = expect(submission).rejects.toThrow('agent_prompt_stalled') + + await vi.advanceTimersByTimeAsync(delayMs - 1) + expect(countSubmits(writes)).toBe(0) + await vi.advanceTimersByTimeAsync(1) + expect(countSubmits(writes)).toBe(1) + + await vi.runAllTimersAsync() + await stalled + }) + + it('waits the ConPTY delay for a Windows SSH host driven from macOS', async () => { + useHostPlatform('darwin') + vi.useFakeTimers() + const { runtime, handle, writes } = await createPromptRuntime() + patchPtyRecord(runtime, { connectionId: 'ssh-target-1' }) + registerSshRemotePlatform('win32') + const clientDelayMs = submitDelayFor(HOST_PROBE_PROMPT, 'darwin') + const hostDelayMs = submitDelayFor(HOST_PROBE_PROMPT, 'win32') + const submission = runtime.sendTerminalAgentPrompt(handle, HOST_PROBE_PROMPT) + const stalled = expect(submission).rejects.toThrow('agent_prompt_stalled') + + await vi.advanceTimersByTimeAsync(clientDelayMs) + expect(countSubmits(writes)).toBe(0) + await vi.advanceTimersByTimeAsync(hostDelayMs - clientDelayMs) + expect(countSubmits(writes)).toBe(1) + + await vi.runAllTimersAsync() + await stalled + }) + + it('skips the ConPTY delay for a Linux SSH host driven from Windows', async () => { + useHostPlatform('win32') + vi.useFakeTimers() + const { runtime, handle, writes } = await createPromptRuntime() + patchPtyRecord(runtime, { connectionId: 'ssh-target-1' }) + registerSshRemotePlatform('linux') + const delayMs = submitDelayFor(HOST_PROBE_PROMPT, 'linux') + expect(delayMs).toBeLessThan(submitDelayFor(HOST_PROBE_PROMPT, 'win32')) + const submission = runtime.sendTerminalAgentPrompt(handle, HOST_PROBE_PROMPT) + const stalled = expect(submission).rejects.toThrow('agent_prompt_stalled') + + await vi.advanceTimersByTimeAsync(delayMs - 1) + expect(countSubmits(writes)).toBe(0) + await vi.advanceTimersByTimeAsync(1) + expect(countSubmits(writes)).toBe(1) + + await vi.runAllTimersAsync() + await stalled + }) + + it('falls back to the remote worktree path flavor before the relay reports a platform', async () => { + useHostPlatform('darwin') + vi.useFakeTimers() + const { runtime, handle, writes } = await createPromptRuntime() + patchPtyRecord(runtime, { + connectionId: 'ssh-target-1', + worktreeId: 'repo-1::C:\\worktrees\\worktree-a' + }) + registerSshRemotePlatform(undefined) + const delayMs = submitDelayFor(HOST_PROBE_PROMPT, 'win32') + const submission = runtime.sendTerminalAgentPrompt(handle, HOST_PROBE_PROMPT) + const stalled = expect(submission).rejects.toThrow('agent_prompt_stalled') + + await vi.advanceTimersByTimeAsync(delayMs - 1) + expect(countSubmits(writes)).toBe(0) + await vi.advanceTimersByTimeAsync(1) + expect(countSubmits(writes)).toBe(1) + + await vi.runAllTimersAsync() + await stalled + }) +}) + +describe('agent prompt render gate on a ConPTY host', () => { + afterEach(() => { + vi.useRealTimers() + Object.defineProperty(process, 'platform', originalPlatform) + }) + + /** Claude/Codex take the closed-loop gate; the paste-end write emits the show-cursor + * marker and then goes silent, which is what an agent that repaints mid-ingest looks like. */ + async function createSettlementRuntime( + // `noiseUntilMs` keeps the pane emitting inside every quiet window, so the gate can only + // end on its hard cap -- which is what the cap's arithmetic has to be measured against. + agentOutput: { markerDelayMs?: number; noiseUntilMs?: number } = {} + ): Promise<{ + runtime: OrcaRuntimeService + handle: string + writes: string[] + submitTimes: number[] + }> { + const markerDelayMs = agentOutput.markerDelayMs ?? 100 + const runtime = new OrcaRuntimeService(makeStore() as never) + const writes: string[] = [] + const submitTimes: number[] = [] + const startedAt = Date.now() + runtime.setPtyController({ + spawn: vi.fn().mockResolvedValue({ id: PTY_ID }), + write: (_ptyId, data) => { + writes.push(data) + if (data === '\r') { + submitTimes.push(Date.now() - startedAt) + } + if (data.includes('\x1b[201~')) { + setTimeout(() => runtime.onPtyData(PTY_ID, '\x1b[?25h', Date.now()), markerDelayMs) + for (let at = markerDelayMs + 500; at <= (agentOutput.noiseUntilMs ?? 0); at += 500) { + setTimeout(() => runtime.onPtyData(PTY_ID, '.', Date.now()), at) + } + } + return true + }, + kill: () => true, + getForegroundProcess: async () => null + }) + const terminal = await runtime.createTerminal(`path:${WORKTREE_PATH}`, { + launchAgent: 'claude' + }) + return { runtime, handle: terminal.handle, writes, submitTimes } + } + + it('does not let a mid-ingest marker plus quiet settle a large paste early', async () => { + useHostPlatform('win32') + vi.useFakeTimers() + const { runtime, handle, writes, submitTimes } = await createSettlementRuntime() + const submission = runtime.sendTerminalAgentPrompt(handle, 'y'.repeat(320_000)) + const stalled = expect(submission).rejects.toThrow('agent_prompt_stalled') + + // Marker at 100 ms + a 1_500 ms quiet window would have submitted at ~1_600 ms, while + // ConPTY needs 2_969-3_342 ms just to hand the paste to the child. + await vi.advanceTimersByTimeAsync(3_342) + expect(writes.filter((data) => data.includes('\x1b[201~'))).toHaveLength(1) + expect(countSubmits(writes)).toBe(0) + + await vi.runAllTimersAsync() + expect(submitTimes).toHaveLength(1) + expect(submitTimes[0]).toBeGreaterThan(3_342) + await stalled + }) + + it('caps a never-quiet pane at one ingest window past the render timeout', async () => { + useHostPlatform('win32') + vi.useFakeTimers() + const prompt = 'y'.repeat(320_000) + const ingestMs = getTerminalPasteIngestMs( + 'win32', + Buffer.byteLength(buildAgentPromptPasteBytes(prompt), 'utf8') + ) + // The marker lands mid-ingest, which re-arms the cap; the ingest term must not be charged + // a second time from that later moment. + const { runtime, handle, writes, submitTimes } = await createSettlementRuntime({ + markerDelayMs: ingestMs - 1_000, + noiseUntilMs: ingestMs + 20_000 + }) + const submission = runtime.sendTerminalAgentPrompt(handle, prompt) + const stalled = expect(submission).rejects.toThrow('agent_prompt_stalled') + + await vi.advanceTimersByTimeAsync(ingestMs + 8_000 - 1) + expect(countSubmits(writes)).toBe(0) + + await vi.runAllTimersAsync() + expect(submitTimes).toHaveLength(1) + expect(submitTimes[0]).toBeGreaterThanOrEqual(ingestMs + 8_000) + expect(submitTimes[0]).toBeLessThan(ingestMs + 8_500) + await stalled + }) + + it('still settles a normal prompt on the marker plus one quiet window', async () => { + useHostPlatform('win32') + vi.useFakeTimers() + const { runtime, handle, submitTimes } = await createSettlementRuntime() + const submission = runtime.sendTerminalAgentPrompt(handle, 'review this') + const stalled = expect(submission).rejects.toThrow('agent_prompt_stalled') + + await vi.runAllTimersAsync() + expect(submitTimes).toHaveLength(1) + // 100 ms marker + 1_500 ms quiet: a sub-chunk paste adds no measurable ingest. + expect(submitTimes[0]).toBeGreaterThanOrEqual(1_600) + expect(submitTimes[0]).toBeLessThan(1_700) + await stalled + }) +}) + +describe('plain terminal send suffix delay', () => { + afterEach(() => { + vi.useRealTimers() + Object.defineProperty(process, 'platform', originalPlatform) + }) + + it('scales the Enter that follows plain text with the payload the host must ingest', async () => { + useHostPlatform('win32') + vi.useFakeTimers() + const { runtime, handle, writes, submitTimes } = await createPromptRuntime() + const send = runtime.sendTerminal(handle, { text: 'z'.repeat(320_000), enter: true }) + + // Same hazard as the agent-prompt path: a flat 500 ms wrote Enter mid-paste here. + await vi.advanceTimersByTimeAsync(3_342) + expect(countSubmits(writes)).toBe(0) + + await vi.runAllTimersAsync() + await send + expect(submitTimes).toHaveLength(1) + expect(submitTimes[0]).toBeGreaterThan(3_342) + }) + + it('abandons the scaled suffix wait when the request is aborted', async () => { + useHostPlatform('win32') + vi.useFakeTimers() + const controller = new AbortController() + const { runtime, handle, writes } = await createPromptRuntime() + const send = runtime.sendTerminal( + handle, + { text: 'z'.repeat(320_000), enter: true }, + { signal: controller.signal } + ) + const rejected = expect(send).rejects.toThrow('request_aborted') + + await vi.advanceTimersByTimeAsync(100) + controller.abort() + await vi.runAllTimersAsync() + + await rejected + expect(countSubmits(writes)).toBe(0) + }) }) diff --git a/src/main/runtime/orca-runtime.test.ts b/src/main/runtime/orca-runtime.test.ts index 1f9d810e977..ea21fda6fff 100644 --- a/src/main/runtime/orca-runtime.test.ts +++ b/src/main/runtime/orca-runtime.test.ts @@ -97,8 +97,9 @@ import { MAX_QUICK_COMMANDS } from '../../shared/terminal-quick-commands' import { AGENT_PROMPT_BRACKETED_PASTE_END, AGENT_PROMPT_BRACKETED_PASTE_START, - AGENT_PROMPT_SUBMIT_DELAY_MS, - buildAgentPromptPasteBytes + buildAgentPromptPasteBytes, + getAgentPromptSubmitDelayMs, + getTerminalPasteIngestMs } from '../../shared/agent-prompt-injection' import { CLIPBOARD_TEXT_MEASURE_YIELD_CODE_UNITS } from '../../shared/clipboard-text' import { projectHostSetupProjectionFromRepos } from '../../shared/project-host-setup-projection' @@ -1049,6 +1050,17 @@ const TEST_REPO_ID = 'repo-1' const TEST_REPO_PATH = '/tmp/repo' const TEST_WORKTREE_PATH = '/tmp/worktree-a' const TEST_WORKTREE_ID = `${TEST_REPO_ID}::${TEST_WORKTREE_PATH}` +/** The render gate's hard cap bounds the wait *after* the paste lands, so it carries the + * payload's ingest bound on top of the flat 8 s settlement budget. */ +function renderGateCapMs(prompt: string): number { + return ( + 8_000 + + getTerminalPasteIngestMs( + process.platform, + Buffer.byteLength(buildAgentPromptPasteBytes(prompt), 'utf8') + ) + ) +} const TEST_FOLDER_PROJECT_GROUP_ID = 'folder-project-group-1' const TEST_FOLDER_WORKSPACE_ID = 'folder-workspace-1' const TEST_FOLDER_WORKSPACE_KEY = `folder:${TEST_FOLDER_WORKSPACE_ID}` @@ -17490,7 +17502,7 @@ describe('OrcaRuntimeService', () => { (Object.keys(TUI_AGENT_CONFIG) as TuiAgent[]).filter( (agent) => agent !== 'claude' && agent !== 'codex' ) - )('preserves the legacy fixed submit delay for %s', async (agent) => { + )('holds Enter for the full open-loop submit delay for %s', async (agent) => { vi.useFakeTimers() try { const writes: string[] = [] @@ -17509,8 +17521,12 @@ describe('OrcaRuntimeService', () => { launchAgent: agent }) + const submitDelayMs = getAgentPromptSubmitDelayMs( + process.platform, + Buffer.byteLength(buildAgentPromptPasteBytes('review this change'), 'utf8') + ) const sendPromise = runtime.sendTerminalAgentPrompt(handle, 'review this change') - await vi.advanceTimersByTimeAsync(AGENT_PROMPT_SUBMIT_DELAY_MS - 1) + await vi.advanceTimersByTimeAsync(submitDelayMs - 1) expect(writes).not.toContain('\r') await vi.advanceTimersByTimeAsync(1) @@ -17581,7 +17597,7 @@ describe('OrcaRuntimeService', () => { }) const sendPromise = runtime.sendTerminalAgentPrompt(handle, 'review this change') - await vi.advanceTimersByTimeAsync(7_999) + await vi.advanceTimersByTimeAsync(renderGateCapMs('review this change') - 1) expect(writes).not.toContain('\r') await vi.advanceTimersByTimeAsync(1) @@ -17662,7 +17678,9 @@ describe('OrcaRuntimeService', () => { }) const sendPromise = runtime.sendTerminalAgentPrompt(handle, 'review this change') - await vi.advanceTimersByTimeAsync(8_099) + // The marker at 100 ms re-arms the cap, but the ingest term is absolute: a prompt this + // small is already ingested by then, so the fallback is one flat render timeout later. + await vi.advanceTimersByTimeAsync(100 + 8_000 - 1) expect(writes).not.toContain('\r') await vi.advanceTimersByTimeAsync(1) diff --git a/src/main/runtime/orca-runtime.ts b/src/main/runtime/orca-runtime.ts index db9be9e7aa0..71c71b7291d 100644 --- a/src/main/runtime/orca-runtime.ts +++ b/src/main/runtime/orca-runtime.ts @@ -100,8 +100,9 @@ import { import { AGENT_PROMPT_BRACKETED_PASTE_END, AGENT_PROMPT_SUBMIT, - AGENT_PROMPT_SUBMIT_DELAY_MS, - buildAgentPromptPasteBytes + buildAgentPromptPasteBytes, + getAgentPromptSubmitDelayMs, + getTerminalPasteIngestMs } from '../../shared/agent-prompt-injection' import { type AgentPromptActivity, @@ -2120,6 +2121,10 @@ const FOREGROUND_AGENT_WRAPPER_RETRY_TIMEOUT_MS = 6_500 const BRACKETED_PASTE_BEGIN = '\x1b[200~' const BRACKETED_PASTE_END = '\x1b[201~' const BRACKETED_PASTE_QUIET_MS = 1500 +// Why: both are windows *after* the paste is ingested, so each is added to the +// payload's ingest bound rather than standing in for it (see getTerminalPasteIngestMs). +// The quiet window stays at 1500: nothing measured describes an agent's post-paste +// redraw cadence, and a shorter window submits mid-redraw. const AGENT_PROMPT_RENDER_TIMEOUT_MS = 8000 const AGENT_PROMPT_RENDER_QUIET_MS = 1500 // Why: Claude and Codex emit show-cursor after accepting bracketed paste. @@ -2163,6 +2168,19 @@ async function waitForAgentPromptPromise(promise: Promise, signal?: AbortS }) } +// Why not setTimeout(0): it costs a full ~15.19 ms Windows timer tick per chunk (~0.95 s/MB) +// and never bought backpressure -- 16 KiB per tick paces ~1.07 MB/s, 11x above ConPTY's +// ~96 KB/s drain, so the in-flight buffer grew regardless. setImmediate keeps the only thing +// the yield actually did (let abort/permission/data callbacks run between chunks) at ~0.01 ms, +// and TERMINAL_INPUT_MAX_BYTES still bounds what can be in flight either way. +// Why the global and not node:timers/promises: only the global is intercepted by fake timers, +// so a chunked paste stays observable on the test clock. +function yieldBetweenTerminalInputChunks(): Promise { + return new Promise((resolve) => { + setImmediate(resolve) + }) +} + async function waitForAgentPromptDelay(delayMs: number, signal?: AbortSignal): Promise { if (!signal) { await new Promise((resolve) => setTimeout(resolve, delayMs)) @@ -19113,6 +19131,9 @@ export class OrcaRuntimeService { reserveWrite?: (ptyId: string) => void afterWrite?: (ptyId: string) => void | Promise suffixFailureError?: string + // Why: the pre-Enter wait now scales with the payload, so an abandoned request must be + // able to stop it instead of writing Enter minutes after the caller gave up. + signal?: AbortSignal } = {} ): Promise { const pty = this.getLivePtyForHandle(handle) @@ -19809,19 +19830,28 @@ export class OrcaRuntimeService { reserveWrite?: (ptyId: string) => void afterWrite?: (ptyId: string) => void | Promise suffixFailureError?: string + signal?: AbortSignal } = {} ): Promise { // Why: direct terminal.send can carry paste-sized text from RPC/mobile // clients; chunk text before PTY/ConPTY while preserving suffix separation. - const hasText = typeof action.text === 'string' && action.text.length > 0 + const text = typeof action.text === 'string' ? action.text : '' const hasSuffix = action.enter || action.interrupt - if (hasText) { - await this.writeTerminalInputChunks(ptyId, action.text!, options) + if (text) { + await this.writeTerminalInputChunks(ptyId, text, options) } if (hasSuffix) { const suffix = (action.enter ? '\r' : '') + (action.interrupt ? '\x03' : '') - if (hasText) { - await new Promise((resolve) => setTimeout(resolve, 500)) + if (text) { + // Why: same hazard as the agent-prompt path -- Enter must not overtake text the + // execution host is still ingesting, and a flat 500 ms cannot cover 16 MB. + await waitForAgentPromptDelay( + getAgentPromptSubmitDelayMs( + this.getPtyWriteHostPlatform(ptyId), + Buffer.byteLength(text, 'utf8') + ), + options.signal + ) } try { await options.beforeWrite?.(ptyId) @@ -19839,7 +19869,7 @@ export class OrcaRuntimeService { await options.afterWrite?.(ptyId) return } - if (hasText) { + if (text) { return } @@ -19873,11 +19903,32 @@ export class OrcaRuntimeService { await options.afterWrite?.(ptyId) chunk = chunks.next() if (!chunk.done) { - await new Promise((resolve) => setTimeout(resolve, 0)) + await yieldBetweenTerminalInputChunks() } } } + /** Platform of the host whose pty transport ingests our writes -- deliberately NOT the OS + * the command runs under. A WSL pane is spawned as `wsl.exe` through the Windows ConPTY + * (see local-pty-provider), so it pays the ConPTY ingest cost even though its shell is + * Linux; an SSH pane is spawned by node-pty on the remote host, so the client's + * process.platform says nothing about it. */ + private getPtyWriteHostPlatform(ptyId: string): NodeJS.Platform { + const pty = this.ptysById.get(ptyId) + const connectionId = pty?.connectionId + if (!connectionId) { + return process.platform + } + const remotePlatform = getRegisteredSshState(connectionId)?.remotePlatform + if (remotePlatform) { + return remotePlatform + } + // Why: remotePlatform only arrives with the relay handshake; until then the worktree path + // flavor is the same signal getAgentLaunchPlatformForRepo already trusts for a remote repo. + const worktreePath = pty ? splitWorktreeIdForFilesystem(pty.worktreeId)?.worktreePath : null + return worktreePath && isWindowsAbsolutePathLike(worktreePath) ? 'win32' : 'linux' + } + private async writeTerminalAgentPrompt( handle: string, ptyId: string, @@ -19893,7 +19944,12 @@ export class OrcaRuntimeService { this.assertAgentPromptGeneration(ptyId, generation) const permissionBaseline = this.getAgentPromptActivity(handle, ptyId) this.assertAgentPromptPermissionSafe(permissionBaseline, permissionBaseline) - const renderGate = this.createAgentPromptRenderGate(ptyId) + // Why: the floor for every wait below. Enter must never overtake bytes the execution + // host is still feeding the child, and that cost is proportional to the payload. + const writeHostPlatform = this.getPtyWriteHostPlatform(ptyId) + const pasteByteLength = Buffer.byteLength(pastePayload, 'utf8') + const pasteIngestMs = getTerminalPasteIngestMs(writeHostPlatform, pasteByteLength) + const renderGate = this.createAgentPromptRenderGate(ptyId, pasteIngestMs) let wrotePasteBytes = false let completedPaste = false try { @@ -19920,7 +19976,7 @@ export class OrcaRuntimeService { wrotePasteBytes = true chunk = nextChunk if (!chunk.done) { - await new Promise((resolve) => setTimeout(resolve, 0)) + await yieldBetweenTerminalInputChunks() } } completedPaste = true @@ -19943,7 +19999,10 @@ export class OrcaRuntimeService { renderGate.dispose() } } else { - await waitForAgentPromptDelay(AGENT_PROMPT_SUBMIT_DELAY_MS, options.signal) + await waitForAgentPromptDelay( + getAgentPromptSubmitDelayMs(writeHostPlatform, pasteByteLength), + options.signal + ) } assertAgentPromptRequestActive(options.signal) this.assertAgentPromptGeneration(ptyId, generation) @@ -20044,7 +20103,13 @@ export class OrcaRuntimeService { } } - private createAgentPromptRenderGate(ptyId: string): { + /** `pasteIngestMs` is the payload's ingest bound on the executing host. Nothing here may + * settle before it elapses: the agent can repaint mid-ingest, so marker-then-quiet alone + * would fire Enter into a paste ConPTY is still feeding. */ + private createAgentPromptRenderGate( + ptyId: string, + pasteIngestMs: number + ): { arm: () => void wait: () => Promise dispose: () => void @@ -20057,19 +20122,20 @@ export class OrcaRuntimeService { let armed = false let canSettle = false let settled = false + let ingested = pasteIngestMs <= 0 + // Why absolute: the ingest clock starts once, here, but the cap is armed twice (at arm() + // and again on the marker). Re-adding the whole window would charge ingest twice. + const ingestDeadlineAt = Date.now() + pasteIngestMs let markerCarry = '' let quietTimer: NodeJS.Timeout | null = null let hardTimer: NodeJS.Timeout | null = null + let ingestTimer: NodeJS.Timeout | null = null let resolveRender!: () => void const rendered = new Promise((resolve) => { resolveRender = resolve }) - const finish = (): void => { - if (settled) { - return - } - settled = true + const clearGateTimers = (): void => { if (quietTimer) { clearTimeout(quietTimer) quietTimer = null @@ -20078,9 +20144,25 @@ export class OrcaRuntimeService { clearTimeout(hardTimer) hardTimer = null } + if (ingestTimer) { + clearTimeout(ingestTimer) + ingestTimer = null + } + } + const finish = (): void => { + if (settled) { + return + } + settled = true + clearGateTimers() resolveRender() } const armQuietTimer = (): void => { + // Why: the quiet window measures the agent going still after a *complete* paste. + // Silence during ingest is not settlement, so it cannot start the clock. + if (!ingested) { + return + } if (quietTimer) { clearTimeout(quietTimer) } @@ -20090,7 +20172,21 @@ export class OrcaRuntimeService { if (hardTimer) { clearTimeout(hardTimer) } - hardTimer = setTimeout(finish, AGENT_PROMPT_RENDER_TIMEOUT_MS) + // Why: the cap bounds the wait *after* the bytes land; a flat 8000 ms would expire + // mid-paste past ~770 KB on Windows and write Enter into it. + hardTimer = setTimeout( + finish, + AGENT_PROMPT_RENDER_TIMEOUT_MS + Math.max(0, ingestDeadlineAt - Date.now()) + ) + } + if (!ingested) { + ingestTimer = setTimeout(() => { + ingestTimer = null + ingested = true + if (canSettle) { + armQuietTimer() + } + }, pasteIngestMs) } const unsubscribe = this.subscribeToTerminalData(ptyId, (data) => { if (!armed || settled) { @@ -20122,14 +20218,7 @@ export class OrcaRuntimeService { }, dispose: () => { unsubscribe() - if (quietTimer) { - clearTimeout(quietTimer) - quietTimer = null - } - if (hardTimer) { - clearTimeout(hardTimer) - hardTimer = null - } + clearGateTimers() } } } diff --git a/src/main/runtime/rpc/methods/terminal/terminal-send-method.ts b/src/main/runtime/rpc/methods/terminal/terminal-send-method.ts index a4c58fbab68..9baa4827a55 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-send-method.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-send-method.ts @@ -179,6 +179,7 @@ export const TERMINAL_SEND_METHODS: RpcAnyMethod[] = [ }, { beforeWrite, + signal, ...(reserveWrite ? { reserveWrite } : {}), ...(params.inputKind !== 'query-reply' && mobileFloorClientId ? { afterWrite: () => commitMobileInputFloorClaim(mobileFloorClaim) } diff --git a/src/main/runtime/rpc/terminal-agent-prompt-send.test.ts b/src/main/runtime/rpc/terminal-agent-prompt-send.test.ts index eb7624c95e4..29254fbcf3a 100644 --- a/src/main/runtime/rpc/terminal-agent-prompt-send.test.ts +++ b/src/main/runtime/rpc/terminal-agent-prompt-send.test.ts @@ -81,8 +81,37 @@ describe('terminal agent prompt send RPC', () => { expect(sendTerminal).toHaveBeenCalledWith( 'terminal-1', { text: 'echo x', enter: true, interrupt: false }, - { beforeWrite: undefined } + { beforeWrite: undefined, signal: undefined } ) expect(sendTerminalAgentPrompt).not.toHaveBeenCalled() }) + + it('forwards the request signal to a plain send so an abandoned call stops before Enter', async () => { + const sendTerminal = vi.fn().mockResolvedValue({ + handle: 'terminal-1', + accepted: true, + bytesWritten: 7 + }) + const runtime = makeRuntime({ + resolveLiveLeafForHandle: vi.fn().mockReturnValue({ ptyId: 'pty-1' }), + getDriver: vi.fn().mockReturnValue({ kind: 'idle' }), + isTerminalRunningSettledPromptAgent: vi.fn().mockResolvedValue(false), + sendTerminal + }) + const dispatcher = new RpcDispatcher({ runtime, methods: TERMINAL_METHODS }) + const controller = new AbortController() + + const response = await dispatcher.dispatch( + makeRequest({ + terminal: 'terminal-1', + text: 'echo x', + enter: true, + client: { id: 'orca-cli', type: 'desktop' } + }), + { signal: controller.signal } + ) + + expect(response.ok).toBe(true) + expect(sendTerminal.mock.calls[0][2].signal).toBe(controller.signal) + }) }) diff --git a/src/shared/agent-prompt-injection.test.ts b/src/shared/agent-prompt-injection.test.ts index 54b4d6cb544..159f5d19270 100644 --- a/src/shared/agent-prompt-injection.test.ts +++ b/src/shared/agent-prompt-injection.test.ts @@ -5,6 +5,7 @@ import { buildAgentPromptPasteBytes, buildAgentPromptSubmitBytes, getAgentPromptSubmitDelayMs, + getTerminalPasteIngestMs, iterateAgentPromptPasteChunks, sanitizeAgentPromptText } from './agent-prompt-injection' @@ -24,10 +25,60 @@ describe('agent prompt injection bytes', () => { expect(buildAgentPromptSubmitBytes()).toBe('\r') }) - it('gives Windows ConPTY more time to render before submit', () => { - expect(getAgentPromptSubmitDelayMs('win32')).toBe(1_500) - expect(getAgentPromptSubmitDelayMs('darwin')).toBe(500) - expect(getAgentPromptSubmitDelayMs('linux')).toBe(500) + it('costs a common-sized prompt far less than the old flat Windows delay', () => { + // Measured ingest: 2 KB 14-25 ms, 8 KB 60-89 ms. The old constant charged 1_500 ms for both. + expect(getAgentPromptSubmitDelayMs('win32', 2_000)).toBeLessThan(700) + expect(getAgentPromptSubmitDelayMs('win32', 8_000)).toBeLessThan(700) + expect(getAgentPromptSubmitDelayMs('darwin', 8_000)).toBeLessThan(700) + }) + + it.each([ + // The slower of the two measured Win11 hosts at each size. Dipping under any of these + // reopens the mid-paste Enter bug, so pin all of them, not just the top end. + [2_000, 25], + [8_000, 89], + [40_000, 440], + [80_000, 858], + [160_000, 1_662], + [320_000, 3_342] + ])('outlasts the slowest measured ConPTY ingest of %i bytes', (bytes, measuredMs) => { + expect(getTerminalPasteIngestMs('win32', bytes)).toBeGreaterThan(measuredMs) + expect(getAgentPromptSubmitDelayMs('win32', bytes)).toBeGreaterThan(measuredMs) + }) + + it('keeps a real margin over the slowest measured slope', () => { + // 1.5x of 0.0104 ms/byte, so a host ~50% slower than either measured one is still covered. + expect(getTerminalPasteIngestMs('win32', 320_000)).toBeGreaterThan(3_342 * 1.4) + }) + + it('outgrows the old 1_500 ms constant before ConPTY ingest does', () => { + // Ingest crossed 1_500 ms at ~145 KB; the delay must already exceed it there. + expect(getAgentPromptSubmitDelayMs('win32', 145_000)).toBeGreaterThan(1_500) + }) + + it('scales without a cap all the way to the terminal input ceiling', () => { + const ceilingBytes = 16 * 1024 * 1024 + expect(getAgentPromptSubmitDelayMs('win32', ceilingBytes)).toBeGreaterThan(250_000) + // Even the fast platforms outrun a flat 500 ms at the ceiling. + expect(getAgentPromptSubmitDelayMs('linux', ceilingBytes)).toBeGreaterThan(4_000) + }) + + it('charges non-Windows hosts nothing measurable for a real prompt', () => { + expect(getTerminalPasteIngestMs('darwin', 8_000)).toBeLessThanOrEqual(2) + expect(getTerminalPasteIngestMs('linux', 0)).toBe(0) + expect(getTerminalPasteIngestMs('linux', Number.NaN)).toBe(0) + expect(getTerminalPasteIngestMs('win32', -5)).toBe(0) + }) + + it('grows monotonically with payload size on every platform', () => { + for (const platform of ['win32', 'darwin', 'linux'] as const) { + expect(getAgentPromptSubmitDelayMs(platform, 400_000)).toBeGreaterThan( + getAgentPromptSubmitDelayMs(platform, 40_000) + ) + } + expect(getTerminalPasteIngestMs('win32', 320_000)).toBeGreaterThan( + getTerminalPasteIngestMs('darwin', 320_000) + ) }) it('sanitizes embedded escape bytes before framing', () => { diff --git a/src/shared/agent-prompt-injection.ts b/src/shared/agent-prompt-injection.ts index 3db9d1cd39d..a52738a6f90 100644 --- a/src/shared/agent-prompt-injection.ts +++ b/src/shared/agent-prompt-injection.ts @@ -4,17 +4,49 @@ export const AGENT_PROMPT_BRACKETED_PASTE_START = '\x1b[200~' export const AGENT_PROMPT_BRACKETED_PASTE_END = '\x1b[201~' export const AGENT_PROMPT_SUBMIT = '\r' -const DEFAULT_AGENT_PROMPT_SUBMIT_DELAY_MS = 500 -const WINDOWS_AGENT_PROMPT_SUBMIT_DELAY_MS = 1_500 +// Why: Windows ConPTY ingests pasted input linearly (first byte written -> child observes +// ESC[201~), and the cost is input ingest, not rendering -- the child repaints in ~0 ms on +// both platforms. Two real Win11 hosts, bundled ConPTY DLL, 16 KiB chunks: +// bytes host A host B +// 2,000 14 ms 25 ms +// 8,000 60 ms 89 ms +// 40,000 347 ms 440 ms +// 160,000 1662 ms 1499 ms +// 320,000 3342 ms 2969 ms +// Slopes: 0.0104 ms/byte (A) and 0.0092 ms/byte (B), i.e. ~40% host-to-host spread in both +// directions. 64 B/ms is 1.5x the slower of the two slopes, so neither host -- nor a +// meaningfully slower one -- can still be ingesting when the wait ends. +const WINDOWS_CONPTY_INGEST_BYTES_PER_MS = 64 +// Why: the same walk on macOS drains 320 KB in 26 ms (~12.3 KB/ms), but at those magnitudes +// the samples are noise-dominated (80 KB measured faster than 40 KB), so hold a 3x margin. +// It costs 0 ms at real prompt sizes and 4.1 s at the 16 MB terminal-input ceiling. +const DEFAULT_PASTE_INGEST_BYTES_PER_MS = 4_096 +// Why: ingest only buys the child the *bytes*; it still has to attach the completed paste +// before Enter counts. Unchanged from the previous cross-platform constant -- nothing +// measured here justifies moving it, and it also absorbs the 15-25 ms fixed intercept +// both ConPTY hosts show below the linear term. +const AGENT_PROMPT_SUBMIT_SETTLE_MS = 500 -// Why: ConPTY renders long bracketed pastes more slowly; an early Enter leaves the task in the agent input buffer. -export function getAgentPromptSubmitDelayMs(platform: NodeJS.Platform): number { - return platform === 'win32' - ? WINDOWS_AGENT_PROMPT_SUBMIT_DELAY_MS - : DEFAULT_AGENT_PROMPT_SUBMIT_DELAY_MS +/** Lower bound on when a paste of `byteLength` can have reached the child, given the + * ingest rate of the host that owns the pty transport (not the OS the command runs under). */ +export function getTerminalPasteIngestMs(platform: NodeJS.Platform, byteLength: number): number { + if (!Number.isFinite(byteLength) || byteLength <= 0) { + return 0 + } + return Math.ceil( + byteLength / + (platform === 'win32' + ? WINDOWS_CONPTY_INGEST_BYTES_PER_MS + : DEFAULT_PASTE_INGEST_BYTES_PER_MS) + ) } -export const AGENT_PROMPT_SUBMIT_DELAY_MS = getAgentPromptSubmitDelayMs(process.platform) +/** Open-loop wait before Enter for agents with no settlement signal: the paste cannot have + * landed before it is ingested, and the child needs a settle window after that. Never + * capped -- a cap silently reintroduces the mid-paste Enter it exists to prevent. */ +export function getAgentPromptSubmitDelayMs(platform: NodeJS.Platform, byteLength: number): number { + return AGENT_PROMPT_SUBMIT_SETTLE_MS + getTerminalPasteIngestMs(platform, byteLength) +} const ESCAPE = '\x1b' const INERT_ESCAPE = '' From 68d5b9206e9bc79a4522d6466debe2b9fa356b8f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 26 Aug 2026 16:44:40 -0700 Subject: [PATCH 16/19] fix(agent-prompt): stop reporting delivered prompts as stalled (#16095) (#16590) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(agent-prompt): stop reporting delivered prompts as stalled (#16095) Enter is written before verification runs, so `agent_prompt_stalled` can only ever mean "turn start not observed" — never "prompt not delivered". Three of the verifier's blind spots made that misreading routine, and the coordinator then treated it as non-delivery and pasted the whole preamble a second time into a worker already running it. - Accept a hook-reported `working` recorded after the baseline. Hook rows reach the runtime through getAgentStatusSnapshot with no window involved, unlike the synthetic-title route that feeds workingSequence (suppressed for codex, absent for kimi, and gated on window visibility for everyone else). - Accept pane output after Enter when the agent was already working: a `->working` edge is unreachable there, so the old predicate could never be satisfied by a follow-up prompt. An idle agent still owes a real turn start, so a swallowed Enter stays detectable. - Give codex/kimi panes a longer effect window; their only turn-start proof is an out-of-process hook round-trip, not a TUI repaint. - Coordinator dispatch no longer fails (and re-dispatches) a task whose prompt stalled; the dispatch stays active with its capability intact so the worker's own report settles it. * fix(orchestration): let a worker's own report correct an unobserved prompt (#16095) Follow-up to f9f973c on two review findings. Anchor the hook signal on a turn, not a refresh: `receivedAt`/`updatedAt` bump on every same-state hook ping, so an in-progress turn could have passed for a new one and silently accepted every prompt to a working agent. `stateStartedAt` is the documented per-turn identity (pinned across same-state pings), so the verifier now reads that. Close the worker-start path: a `dispatch_input` stall settled the dispatch as failed *and* revoked the capability, so a worker that ran the preamble to completion had its result rejected. Revocation is now skipped for that cause, and a worker report can re-settle a dispatch whose `last_failure` is `agent_prompt_stalled` — retaining the capability alone was not enough, because settlement also gates on dispatch/task status. * fix(orchestration): let a failed worker report correct a stalled-prompt record (#16095) The duplicate short-circuit ran before the unobserved-prompt branch, and for outcome 'failed' both expected statuses are exactly the state failWorkerStart leaves behind. A worker that reported a real failure was answered duplicate:true, so its cause and result body were dropped and the record kept 'agent_prompt_stalled'. Evaluate settledByUnobservedPrompt first so one failure report can re-settle that dispatch; a repeat report is still a duplicate. Also derive the previous dispatch/worker states once instead of two parallel ternaries, reuse getPtyAgent in createAgentPromptRenderGate, cite the real 30s relay request budget the hook window is sized against, and make the coordinator test settle an actual worker report rather than calling completeDispatch under a name that promised otherwise. * test(orchestration): carry the new dispatch-depth fields into this PR's fixtures main added required creator/maxDepth on createStartingWorkerDispatch and nestedWorkerMaxDepth on dispatchTaskToWorker while this branch was open. The two fixtures added here predate them, so the merge typechecked clean on each side and failed once combined. Mechanical; no behaviour asserted here changes. --- .../agent-prompt-submission-runtime.test.ts | 118 +++++++++++++-- ...ent-prompt-submission-verification.test.ts | 110 ++++++++++++++ .../agent-prompt-submission-verification.ts | 76 +++++++++- src/main/runtime/orca-runtime.ts | 32 ++++- ...dinator-dispatch-unobserved-prompt.test.ts | 134 ++++++++++++++++++ .../coordinator-task-dispatch.ts | 15 +- .../worker-report-settlement.ts | 43 ++++-- .../worker-dispatch-outcome.ts | 11 +- ...start-unobserved-prompt-settlement.test.ts | 102 +++++++++++++ ...orker-start-outcome-classification.test.ts | 32 +++++ ...ation-worker-start-prompt-contract.test.ts | 4 +- .../orchestration-worker-start-receipt.ts | 7 +- 12 files changed, 648 insertions(+), 36 deletions(-) create mode 100644 src/main/runtime/orchestration/coordinator-dispatch-unobserved-prompt.test.ts create mode 100644 src/main/runtime/orchestration/worker-start-unobserved-prompt-settlement.test.ts create mode 100644 src/main/runtime/rpc/methods/orchestration-worker-start-outcome-classification.test.ts diff --git a/src/main/runtime/agent-prompt-submission-runtime.test.ts b/src/main/runtime/agent-prompt-submission-runtime.test.ts index 69eb0a01079..3b8fe67bea7 100644 --- a/src/main/runtime/agent-prompt-submission-runtime.test.ts +++ b/src/main/runtime/agent-prompt-submission-runtime.test.ts @@ -403,15 +403,7 @@ describe('agent prompt submission runtime', () => { it('does not treat an unchanged newer working status as submission evidence', async () => { vi.useFakeTimers() vi.setSystemTime(1_000) - const { runtime, handle, writes } = await createPromptRuntime((runtime, data) => { - if (data === '\r') { - runtime.onPtyData( - 'pty-prompt', - '\x1b]9999;{"state":"working","agentType":"aider"}\x07', - Date.now() - ) - } - }) + const { runtime, handle, writes } = await createPromptRuntime(() => undefined) runtime.onPtyData('pty-prompt', '\x1b]0;Codex waiting for permission\x07', Date.now()) vi.setSystemTime(2_000) runtime.onPtyData( @@ -428,6 +420,114 @@ describe('agent prompt submission runtime', () => { expect(writes.filter((data) => data === '\r')).toHaveLength(1) }) + // Why (#16095): a still-working agent can never produce a `→working` edge, so the old predicate + // was unsatisfiable for every follow-up prompt; pane output after Enter is the evidence left. + it('accepts pane output after Enter while the agent is already working', async () => { + vi.useFakeTimers() + vi.setSystemTime(1_000) + const { runtime, handle, writes } = await createPromptRuntime((runtime, data) => { + if (data === '\r') { + runtime.onPtyData('pty-prompt', 'queued for the current turn', Date.now()) + } + }) + runtime.onPtyData( + 'pty-prompt', + '\x1b]9999;{"state":"working","agentType":"aider"}\x07', + Date.now() + ) + + const submission = runtime.sendTerminalAgentPrompt(handle, 'review this') + await vi.runAllTimersAsync() + + await expect(submission).resolves.toMatchObject({ accepted: true }) + expect(writes.filter((data) => data === '\r')).toHaveLength(1) + }) + + // Why: hook rows reach the runtime through this provider, which has no window and no OSC title — + // the same path a headless `orca serve` host and a minimized desktop window take. + async function createHookOnlyPromptRuntime(hook: { + state: 'done' | 'working' + stateStartedAt: number + }): Promise<{ runtime: OrcaRuntimeService; handle: string; writes: string[] }> { + let handle = '' + const writes: string[] = [] + const runtime = new OrcaRuntimeService(makeStore() as never, undefined, { + getAgentStatusSnapshot: () => [ + { + paneKey: 'prompt-pane', + terminalHandle: handle, + state: hook.state, + prompt: '', + agentType: 'kimi', + connectionId: null, + // Why: every hook ping refreshes receivedAt, including same-state tool pings. + receivedAt: Date.now(), + stateStartedAt: hook.stateStartedAt + } + ] + }) + runtime.setPtyController({ + spawn: vi.fn().mockResolvedValue({ id: 'pty-prompt' }), + write: (_ptyId, data) => { + writes.push(data) + return true + }, + kill: () => true, + getForegroundProcess: async () => null + }) + handle = ( + await runtime.createTerminal(`path:${AGENT_PROMPT_TEST_WORKTREE_PATH}`, { + launchAgent: 'kimi' + }) + ).handle + return { runtime, handle, writes } + } + + it('accepts a hook working status with no window and no title coverage', async () => { + vi.useFakeTimers() + vi.setSystemTime(1_000) + const hook = { state: 'done' as 'done' | 'working', stateStartedAt: 1_000 } + const { runtime, handle, writes } = await createHookOnlyPromptRuntime(hook) + runtime.setPtyController({ + spawn: vi.fn().mockResolvedValue({ id: 'pty-prompt' }), + write: (_ptyId, data) => { + writes.push(data) + if (data === '\r') { + vi.setSystemTime(3_000) + hook.state = 'working' + hook.stateStartedAt = 3_000 + } + return true + }, + kill: () => true, + getForegroundProcess: async () => null + }) + + const submission = runtime.sendTerminalAgentPrompt(handle, 'review this') + await vi.runAllTimersAsync() + + await expect(submission).resolves.toMatchObject({ accepted: true }) + expect(writes.filter((data) => data === '\r')).toHaveLength(1) + }) + + // Why: same-state pings keep refreshing receivedAt on a turn that started before the prompt; + // only the pinned stateStartedAt separates that from a turn this prompt started. + it('does not accept a hook row refreshed without a new working turn', async () => { + vi.useFakeTimers() + vi.setSystemTime(1_000) + const { runtime, handle, writes } = await createHookOnlyPromptRuntime({ + state: 'working', + stateStartedAt: 1_000 + }) + + const submission = runtime.sendTerminalAgentPrompt(handle, 'review this') + const rejected = expect(submission).rejects.toThrow('agent_prompt_stalled') + await vi.runAllTimersAsync() + + await rejected + expect(writes.filter((data) => data === '\r')).toHaveLength(1) + }) + it('does not write Enter after the PTY generation changes during settlement', async () => { vi.useFakeTimers() const { runtime, handle, writes } = await createPromptRuntime(() => undefined) diff --git a/src/main/runtime/agent-prompt-submission-verification.test.ts b/src/main/runtime/agent-prompt-submission-verification.test.ts index a0f3e1a5213..4a802df8d33 100644 --- a/src/main/runtime/agent-prompt-submission-verification.test.ts +++ b/src/main/runtime/agent-prompt-submission-verification.test.ts @@ -1,7 +1,10 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { AGENT_PROMPT_EFFECT_TIMEOUT_MS, + AGENT_PROMPT_HOOK_EFFECT_TIMEOUT_MS, type AgentPromptActivity, + isAgentPromptStalledError, + resolveAgentPromptEffectTimeoutMs, verifyAgentPromptSubmission } from './agent-prompt-submission-verification' @@ -10,6 +13,8 @@ function activity(overrides: Partial = {}): AgentPromptActi generation: 1, permissionSequence: 2, workingSequence: 4, + explicitWorkingStartedAt: null, + outputSequence: 7, status: 'idle', ...overrides } @@ -127,6 +132,111 @@ describe('agent prompt submission verification', () => { await rejected }) + it('accepts a hook working status recorded after the baseline', async () => { + vi.useFakeTimers() + let current = activity() + const verification = verifyAgentPromptSubmission({ + baseline: current, + readActivity: () => current + }) + + // No workingSequence edge: the window-gated synthetic title never ran (hidden window/headless). + current = activity({ explicitWorkingStartedAt: 2_000, status: 'working' }) + await vi.advanceTimersByTimeAsync(50) + + await expect(verification).resolves.toBeUndefined() + }) + + it('does not accept a hook working status that predates the baseline', async () => { + vi.useFakeTimers() + const current = activity({ explicitWorkingStartedAt: 2_000, status: 'working' }) + const verification = verifyAgentPromptSubmission({ + baseline: current, + readActivity: () => current + }) + const rejected = expect(verification).rejects.toThrow('agent_prompt_stalled') + + await vi.advanceTimersByTimeAsync(AGENT_PROMPT_EFFECT_TIMEOUT_MS) + + await rejected + }) + + // Why: same-state hook pings refresh the row without starting a turn, so only the pinned + // stateStartedAt may satisfy the check — a refreshed row must stay unproven. + it('does not accept a refreshed hook row whose working turn did not restart', async () => { + vi.useFakeTimers() + let current = activity({ explicitWorkingStartedAt: 2_000 }) + const verification = verifyAgentPromptSubmission({ + baseline: current, + readActivity: () => current + }) + const rejected = expect(verification).rejects.toThrow('agent_prompt_stalled') + + current = activity({ explicitWorkingStartedAt: 2_000, outputSequence: 40 }) + await vi.advanceTimersByTimeAsync(AGENT_PROMPT_EFFECT_TIMEOUT_MS) + + await rejected + }) + + it('accepts pane output after Enter when the agent was already working', async () => { + vi.useFakeTimers() + let current = activity({ status: 'working' }) + const verification = verifyAgentPromptSubmission({ + baseline: current, + readActivity: () => current + }) + + current = activity({ status: 'working', outputSequence: 8 }) + await vi.advanceTimersByTimeAsync(50) + + await expect(verification).resolves.toBeUndefined() + }) + + it('does not accept pane output when the agent was idle at submit', async () => { + vi.useFakeTimers() + let current = activity() + const verification = verifyAgentPromptSubmission({ + baseline: current, + readActivity: () => current + }) + const rejected = expect(verification).rejects.toThrow('agent_prompt_stalled') + + current = activity({ outputSequence: 9 }) + await vi.advanceTimersByTimeAsync(AGENT_PROMPT_EFFECT_TIMEOUT_MS) + + await rejected + }) + + it('holds the longer hook window open past the default timeout', async () => { + vi.useFakeTimers() + let current = activity() + const verification = verifyAgentPromptSubmission({ + baseline: current, + readActivity: () => current, + timeoutMs: AGENT_PROMPT_HOOK_EFFECT_TIMEOUT_MS + }) + + await vi.advanceTimersByTimeAsync(AGENT_PROMPT_EFFECT_TIMEOUT_MS + 1_000) + current = activity({ explicitWorkingStartedAt: 9_000, status: 'working' }) + await vi.advanceTimersByTimeAsync(50) + + await expect(verification).resolves.toBeUndefined() + }) + + it('gives hook-observed agents the longer effect window', () => { + expect(resolveAgentPromptEffectTimeoutMs('codex')).toBe(AGENT_PROMPT_HOOK_EFFECT_TIMEOUT_MS) + expect(resolveAgentPromptEffectTimeoutMs('kimi')).toBe(AGENT_PROMPT_HOOK_EFFECT_TIMEOUT_MS) + expect(resolveAgentPromptEffectTimeoutMs('claude')).toBe(AGENT_PROMPT_EFFECT_TIMEOUT_MS) + expect(resolveAgentPromptEffectTimeoutMs(null)).toBe(AGENT_PROMPT_EFFECT_TIMEOUT_MS) + }) + + it('recognizes a stalled verdict from a message or a relayed error code', () => { + expect(isAgentPromptStalledError(new Error('agent_prompt_stalled'))).toBe(true) + expect(isAgentPromptStalledError({ code: 'agent_prompt_stalled' })).toBe(true) + expect(isAgentPromptStalledError(new Error('terminal_not_writable'))).toBe(false) + expect(isAgentPromptStalledError(null)).toBe(false) + }) + it('rejects a replaced terminal generation', async () => { const baseline = activity() diff --git a/src/main/runtime/agent-prompt-submission-verification.ts b/src/main/runtime/agent-prompt-submission-verification.ts index 41b5f5b1732..e8373bbaf1f 100644 --- a/src/main/runtime/agent-prompt-submission-verification.ts +++ b/src/main/runtime/agent-prompt-submission-verification.ts @@ -1,31 +1,68 @@ +import type { TuiAgent } from '../../shared/tui-agent' + export const AGENT_PROMPT_EFFECT_TIMEOUT_MS = 5_000 +// Why: these panes prove a turn start only through the out-of-process hook — kimi has no synthetic +// title profile and codex suppresses the hook-driven working frame (synthesizeWorkingTitle: false), +// so the first proof lags Enter by agent startup, not by one TUI repaint. Capped so the worst case +// (8s render gate + this wait + chunked paste) still fits RELAY_TO_CLIENT_REQUEST_TIMEOUT_MS +// (30s, src/relay/dispatcher.ts), the budget a paired client's submission runs under. +export const AGENT_PROMPT_HOOK_EFFECT_TIMEOUT_MS = 15_000 const AGENT_PROMPT_EFFECT_POLL_MS = 50 +const HOOK_OBSERVED_TURN_START_AGENTS = new Set(['codex', 'kimi']) + +/** The prompt bytes are written before verification, so this only ever means "not observed". */ +export const AGENT_PROMPT_STALLED_ERROR = 'agent_prompt_stalled' + export type AgentPromptActivity = Readonly<{ generation: number permissionSequence: number workingSequence: number + /** When the hook's current `working` turn began; reaches the runtime with no window and no + * title coverage. Pinned across same-state pings, so a refresh alone cannot move it. */ + explicitWorkingStartedAt: number | null + /** PTY bytes seen on this pane; delivery evidence when a turn-start edge cannot be observed. */ + outputSequence: number status: 'working' | 'permission' | 'idle' | null }> type AgentPromptVerificationOptions = { baseline: AgentPromptActivity readActivity: () => AgentPromptActivity + timeoutMs?: number signal?: AbortSignal } +export function resolveAgentPromptEffectTimeoutMs(agent: TuiAgent | null | undefined): number { + return agent && HOOK_OBSERVED_TURN_START_AGENTS.has(agent) + ? AGENT_PROMPT_HOOK_EFFECT_TIMEOUT_MS + : AGENT_PROMPT_EFFECT_TIMEOUT_MS +} + +export function isAgentPromptStalledError(error: unknown): boolean { + if (error instanceof Error && error.message === AGENT_PROMPT_STALLED_ERROR) { + return true + } + // Why: a relayed submission surfaces the same verdict as an RPC error code, not a message. + return ( + typeof error === 'object' && + error !== null && + (error as { code?: unknown }).code === AGENT_PROMPT_STALLED_ERROR + ) +} + export async function verifyAgentPromptSubmission( options: AgentPromptVerificationOptions ): Promise { throwIfAgentPromptAborted(options.signal) assertPromptNotBlocked(options.baseline, options.baseline) - const deadline = Date.now() + AGENT_PROMPT_EFFECT_TIMEOUT_MS + const deadline = Date.now() + (options.timeoutMs ?? AGENT_PROMPT_EFFECT_TIMEOUT_MS) while (Date.now() < deadline) { const current = options.readActivity() assertSamePromptGeneration(options.baseline, current) assertPromptNotBlocked(options.baseline, current) - if (agentPromptLifecycleChanged(options.baseline, current)) { + if (agentPromptEffectObserved(options.baseline, current)) { return } await waitForAgentPromptPoll(options.signal) @@ -34,17 +71,44 @@ export async function verifyAgentPromptSubmission( const current = options.readActivity() assertSamePromptGeneration(options.baseline, current) assertPromptNotBlocked(options.baseline, current) - if (agentPromptLifecycleChanged(options.baseline, current)) { + if (agentPromptEffectObserved(options.baseline, current)) { return } - throw new Error('agent_prompt_stalled') + throw new Error(AGENT_PROMPT_STALLED_ERROR) } -function agentPromptLifecycleChanged( +function agentPromptEffectObserved( baseline: AgentPromptActivity, current: AgentPromptActivity ): boolean { - return current.workingSequence > baseline.workingSequence + return ( + current.workingSequence > baseline.workingSequence || + observedHookWorkingAfterBaseline(baseline, current) || + observedDeliveryEvidence(baseline, current) + ) +} + +// Why: hook status reaches the runtime directly, so it survives a hidden window and headless serve — +// the synthetic-title route that feeds workingSequence does not (#16095). Only a turn that started +// after the baseline counts, so a same-state ping on the turn already running is not evidence. +function observedHookWorkingAfterBaseline( + baseline: AgentPromptActivity, + current: AgentPromptActivity +): boolean { + return ( + current.explicitWorkingStartedAt !== null && + current.explicitWorkingStartedAt > (baseline.explicitWorkingStartedAt ?? 0) + ) +} + +// Why: a `→working` edge is unreachable for an agent that is already working, so the honest proof +// that the prompt landed is the pane emitting bytes after Enter. An idle agent still owes a real +// turn start, which keeps a swallowed Enter detectable. +function observedDeliveryEvidence( + baseline: AgentPromptActivity, + current: AgentPromptActivity +): boolean { + return baseline.status === 'working' && current.outputSequence > baseline.outputSequence } function assertSamePromptGeneration( diff --git a/src/main/runtime/orca-runtime.ts b/src/main/runtime/orca-runtime.ts index 71c71b7291d..54c6b711173 100644 --- a/src/main/runtime/orca-runtime.ts +++ b/src/main/runtime/orca-runtime.ts @@ -106,6 +106,7 @@ import { } from '../../shared/agent-prompt-injection' import { type AgentPromptActivity, + resolveAgentPromptEffectTimeoutMs, verifyAgentPromptSubmission } from './agent-prompt-submission-verification' import { @@ -19780,16 +19781,20 @@ export class OrcaRuntimeService { private getFreshExplicitAgentStatusForHandle(handle: string): { status: NonNullable updatedAt: number + /** When this state was entered. Pinned across same-state pings, so it identifies the turn. */ + stateStartedAt: number } | null { const paneKey = this.getPaneKeyForTerminalHandle(handle) const now = Date.now() let bestStatus: NonNullable | null = null let bestUpdatedAt = -1 + let bestStateStartedAt = -1 const consider = ( state: AgentStatusEntry['state'] | undefined, updatedAt: number | null | undefined, - restoredUnconfirmed = false + restoredUnconfirmed = false, + stateStartedAt?: number | null ): void => { if (!state || restoredUnconfirmed) { return @@ -19803,22 +19808,25 @@ export class OrcaRuntimeService { if (updatedAt > bestUpdatedAt || (updatedAt === bestUpdatedAt && status === 'permission')) { bestStatus = status bestUpdatedAt = updatedAt + bestStateStartedAt = typeof stateStartedAt === 'number' ? stateStartedAt : updatedAt } } if (paneKey) { const retained = this.latestAgentStatusByPaneKey.get(paneKey) - consider(retained?.payload.state, retained?.updatedAt) + consider(retained?.payload.state, retained?.updatedAt, false, retained?.stateStartedAt) } for (const entry of this.getAgentStatusSnapshotFn?.() ?? []) { if (entry.terminalHandle !== handle && (!paneKey || entry.paneKey !== paneKey)) { continue } - consider(entry.state, entry.receivedAt, entry.restoredUnconfirmed) + consider(entry.state, entry.receivedAt, entry.restoredUnconfirmed, entry.stateStartedAt) } - return bestStatus ? { status: bestStatus, updatedAt: bestUpdatedAt } : null + return bestStatus + ? { status: bestStatus, updatedAt: bestUpdatedAt, stateStartedAt: bestStateStartedAt } + : null } private async writeTerminalAction( @@ -20025,6 +20033,7 @@ export class OrcaRuntimeService { await verifyAgentPromptSubmission({ baseline, readActivity: () => this.getAgentPromptActivity(handle, ptyId), + timeoutMs: resolveAgentPromptEffectTimeoutMs(this.getPtyAgent(ptyId)), signal: options.signal }) return 1 @@ -20081,10 +20090,21 @@ export class OrcaRuntimeService { generation: this.getPtyLifecycleGeneration(ptyId), permissionSequence: this.agentPromptPermissionSequenceByPtyId.get(ptyId) ?? 0, workingSequence: lifecycle?.workingSequence ?? 0, + // Why: hook status is the only turn-start signal agents without title coverage have, and it + // reaches here without the window-gated synthetic title frame (#16095). Anchored on + // stateStartedAt, not updatedAt — same-state tool/prompt pings refresh updatedAt and would + // otherwise pass off an in-progress turn as a new one. + explicitWorkingStartedAt: explicit?.status === 'working' ? explicit.stateStartedAt : null, + outputSequence: this.getPtyOutputSequence(ptyId), status } } + private getPtyAgent(ptyId: string): TuiAgent | null { + const pty = this.ptysById.get(ptyId) + return pty?.launchAgent ?? pty?.foregroundAgent ?? null + } + private assertAgentPromptPermissionSafe( baseline: AgentPromptActivity, current: AgentPromptActivity @@ -20114,9 +20134,7 @@ export class OrcaRuntimeService { wait: () => Promise dispose: () => void } | null { - const pty = this.ptysById.get(ptyId) - const agent = pty?.launchAgent ?? pty?.foregroundAgent - if (!isTerminalSendSettlementAgent(agent)) { + if (!isTerminalSendSettlementAgent(this.getPtyAgent(ptyId))) { return null } let armed = false diff --git a/src/main/runtime/orchestration/coordinator-dispatch-unobserved-prompt.test.ts b/src/main/runtime/orchestration/coordinator-dispatch-unobserved-prompt.test.ts new file mode 100644 index 00000000000..5f99e283de3 --- /dev/null +++ b/src/main/runtime/orchestration/coordinator-dispatch-unobserved-prompt.test.ts @@ -0,0 +1,134 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { OrchestrationDb } from './db' +import type { CoordinatorRuntime } from './coordinator-runtime-contract' +import { dispatchTaskToWorker } from './coordinator-task-dispatch' + +const WORKER_PANE_KEY = 'tab_worker:bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb' +let db: OrchestrationDb + +function createRuntime(promptError: Error | null): CoordinatorRuntime & { prompts: string[] } { + const prompts: string[] = [] + return { + prompts, + async sendTerminalAgentPrompt(_handle: string, prompt: string) { + prompts.push(prompt) + if (promptError) { + throw promptError + } + return { accepted: true } + }, + async listTerminals() { + return { terminals: [] } + }, + async createTerminal() { + return { handle: 'term_a', worktreeId: 'wt1' } + }, + async waitForTerminal(handle: string) { + return { handle, condition: 'exit' } + }, + async probeWorktreeDrift() { + return null + }, + getTerminalPaneKey() { + return WORKER_PANE_KEY + }, + getOrchestrationDispatchAuthority() { + return { + paneKey: WORKER_PANE_KEY, + processIncarnation: 'incarnation-1', + launchTokenHash: null + } + } + } +} + +async function dispatch( + runtime: CoordinatorRuntime, + taskId: string, + logs: string[] +): Promise { + return dispatchTaskToWorker({ + db, + runtime, + task: db.getTask(taskId)!, + targetHandle: 'term_a', + nestedWorkerMaxDepth: Number.MAX_SAFE_INTEGER, + baseDrift: null, + coordinatorHandle: 'coord', + worktree: undefined, + onLog: (message) => logs.push(message), + onCircuitBroken: () => undefined + }) +} + +describe('coordinator dispatch with an unobserved prompt', () => { + afterEach(() => db?.close()) + + it('never re-pastes a preamble whose turn start was not observed', async () => { + db = new OrchestrationDb(':memory:') + const task = db.createTask({ spec: 'do the work' }) + const runtime = createRuntime(new Error('agent_prompt_stalled')) + const logs: string[] = [] + + const result = await dispatch(runtime, task.id, logs) + + expect(result).toBe('dispatched-unobserved') + expect(runtime.prompts).toHaveLength(1) + // The task stays dispatched, so the next coordinator tick cannot pick it up again. + expect(db.getTask(task.id)?.status).toBe('dispatched') + const ctx = db.getDispatchContext(task.id) + expect(ctx).toMatchObject({ + status: 'dispatched', + failure_count: 0, + capability_revoked_at: null + }) + expect(db.listTasks({ status: 'ready' })).toEqual([]) + expect(logs.join('\n')).toContain('turn start was not observed') + }) + + it('lets a late worker report settle a dispatch whose prompt was unobserved', async () => { + db = new OrchestrationDb(':memory:') + const task = db.createTask({ spec: 'do the work' }) + await dispatch(createRuntime(new Error('agent_prompt_stalled')), task.id, []) + const dispatchId = db.getDispatchContext(task.id)!.id + const minted = db.mintDispatchCapability({ + dispatchId, + paneKey: WORKER_PANE_KEY, + processIncarnation: 'incarnation-1' + }) + + expect( + db.verifyDispatchCapability({ + dispatchId, + capability: minted, + paneKey: WORKER_PANE_KEY, + processIncarnation: 'incarnation-1' + }) + ).toEqual({ valid: true }) + expect( + db.settleWorkerReport({ + taskId: task.id, + dispatchId, + outcome: 'succeeded', + result: 'done the work' + }) + ).toEqual({ action: 'settled', outcome: 'succeeded', duplicate: false }) + expect(db.getTask(task.id)).toMatchObject({ status: 'completed', result: 'done the work' }) + expect(db.getDispatchContextById(dispatchId)?.status).toBe('completed') + }) + + it('still fails the dispatch when the prompt was never delivered', async () => { + db = new OrchestrationDb(':memory:') + const task = db.createTask({ spec: 'do the work' }) + const runtime = createRuntime(new Error('terminal_not_writable')) + + await expect(dispatch(runtime, task.id, [])).rejects.toThrow('terminal_not_writable') + + expect(db.getTask(task.id)?.status).toBe('ready') + expect(db.getDispatchContext(task.id)).toMatchObject({ + status: 'failed', + failure_count: 1, + last_failure: 'terminal_not_writable' + }) + }) +}) diff --git a/src/main/runtime/orchestration/coordinator-task-dispatch.ts b/src/main/runtime/orchestration/coordinator-task-dispatch.ts index 8dae105726a..685552cc2e9 100644 --- a/src/main/runtime/orchestration/coordinator-task-dispatch.ts +++ b/src/main/runtime/orchestration/coordinator-task-dispatch.ts @@ -7,8 +7,10 @@ import { DISPATCH_STALE_THRESHOLD, parseAllowStaleBaseFromSpec } from './coordinator-stale-base-flag' +import { isAgentPromptStalledError } from '../agent-prompt-submission-verification' -export type TaskDispatchResult = 'dispatched' | 'stale-base-refused' +/** `dispatched-unobserved`: the preamble landed but the worker's turn start was never observed. */ +export type TaskDispatchResult = 'dispatched' | 'dispatched-unobserved' | 'stale-base-refused' // Why: 10 min = documented heartbeat cadence (5 min) × 2, so one missed heartbeat is the earliest a dispatch can look stale. const HUNG_THRESHOLD_MS = 10 * 60 * 1000 @@ -136,6 +138,17 @@ export async function dispatchTaskToWorker(params: { try { await runtime.sendTerminalAgentPrompt(targetHandle, preamble + gateContext) } catch (err) { + // Why (#16095): Enter is written before submission is verified, so a stall is only ever an + // unobserved turn start — never proof the preamble is missing. Failing here would reset the + // task to 'ready' and paste the whole preamble a second time into a worker already running it, + // and would revoke the capability its worker_done needs. + if (isAgentPromptStalledError(err)) { + onLog( + `Dispatched task ${task.id} to ${targetHandle}; turn start was not observed. ` + + `The preamble is already in the pane, so the dispatch stays active instead of being resent.` + ) + return 'dispatched-unobserved' + } const updated = db.failDispatch(dispatch.id, err instanceof Error ? err.message : String(err)) if (updated?.status === 'circuit_broken') { params.onCircuitBroken(task.id) diff --git a/src/main/runtime/orchestration/db/dispatch-context/worker-report-settlement.ts b/src/main/runtime/orchestration/db/dispatch-context/worker-report-settlement.ts index 734c25b1264..9d6b643c070 100644 --- a/src/main/runtime/orchestration/db/dispatch-context/worker-report-settlement.ts +++ b/src/main/runtime/orchestration/db/dispatch-context/worker-report-settlement.ts @@ -1,5 +1,6 @@ import type { WorkerReportOutcome, WorkerReportSettlement } from '../../types' import type { OrchestrationDb } from '../orchestration-db' +import { AGENT_PROMPT_STALLED_ERROR } from '../../../agent-prompt-submission-verification' import { settleActiveDispatchesForTask } from './dispatch-completion' import { getActiveDispatchForTask } from './task-dispatch-reconciliation' @@ -54,10 +55,26 @@ export function settleWorkerReportInTransaction( const expectedDispatchStatus = params.outcome === 'succeeded' ? 'completed' : 'failed' const expectedTaskStatus = params.outcome === 'succeeded' ? 'completed' : 'failed' - if (dispatch.status === expectedDispatchStatus && task.status === expectedTaskStatus) { + // Why (#16095): worker-start records a stalled prompt as failed, but the preamble was written + // before verification ran — the worker may have been executing it the whole time. Its own report + // is first-hand evidence and must be able to correct that record instead of being thrown away. + // Checked before the duplicate short-circuit: a `failed` report lands on the very statuses that + // short-circuit reads as already settled, dropping the worker's real cause and result body. + const settledByUnobservedPrompt = + dispatch.status === 'failed' && + dispatch.last_failure === AGENT_PROMPT_STALLED_ERROR && + task.status === 'failed' + if ( + !settledByUnobservedPrompt && + dispatch.status === expectedDispatchStatus && + task.status === expectedTaskStatus + ) { return { action: 'settled', outcome: params.outcome, duplicate: true } } - if (dispatch.status !== 'dispatched' || task.status !== 'dispatched') { + const previous = settledByUnobservedPrompt + ? { status: 'failed', workerState: 'failed' } + : { status: 'dispatched', workerState: 'ready' } + if (dispatch.status !== previous.status || task.status !== previous.status) { return { action: 'rejected', code: 'inactive_dispatch', @@ -105,16 +122,22 @@ export function settleWorkerReportInTransaction( SET status = ?, completed_at = datetime('now'), last_failure = CASE WHEN ? = 'failed' THEN ? ELSE last_failure END, capability_revoked_at = COALESCE(capability_revoked_at, datetime('now')) - WHERE id = ? AND status = 'dispatched'` + WHERE id = ? AND status = ?` + ) + .run( + expectedDispatchStatus, + expectedDispatchStatus, + params.result, + params.dispatchId, + previous.status ) - .run(expectedDispatchStatus, expectedDispatchStatus, params.result, params.dispatchId) const taskUpdate = this.db .prepare( `UPDATE tasks SET status = ?, result = ?, completed_at = datetime('now') - WHERE id = ? AND status = 'dispatched'` + WHERE id = ? AND status = ?` ) - .run(expectedTaskStatus, params.result, params.taskId) + .run(expectedTaskStatus, params.result, params.taskId, previous.status) if (dispatchUpdate.changes !== 1 || taskUpdate.changes !== 1) { this.db.exec('ROLLBACK TO settle_worker_report') this.db.exec('RELEASE settle_worker_report') @@ -128,9 +151,13 @@ export function settleWorkerReportInTransaction( .prepare( `UPDATE worker_dispatches SET state = ?, stage = 'settled', updated_at = datetime('now') - WHERE dispatch_id = ? AND state = 'ready'` + WHERE dispatch_id = ? AND state = ?` + ) + .run( + params.outcome === 'succeeded' ? 'succeeded' : 'failed', + params.dispatchId, + previous.workerState ) - .run(params.outcome === 'succeeded' ? 'succeeded' : 'failed', params.dispatchId) settleActiveDispatchesForTask( this, params.taskId, diff --git a/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-outcome.ts b/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-outcome.ts index d9a8b23a39b..3ff796c097d 100644 --- a/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-outcome.ts +++ b/src/main/runtime/orchestration/db/worker-dispatch/worker-dispatch-outcome.ts @@ -37,7 +37,11 @@ export function failWorkerStart( this: OrchestrationDb, dispatchId: string, stage: string, - reason: string + reason: string, + // Why (#16095): revocation exists to stop a worker acting on a dispatch that never landed. A + // prompt whose turn start went unobserved provably landed, so its worker keeps the authority its + // own report needs. + options: { retainCapability?: boolean } = {} ): WorkerDispatchRow { this.db.exec('BEGIN IMMEDIATE') try { @@ -50,10 +54,11 @@ export function failWorkerStart( .prepare( `UPDATE dispatch_contexts SET status = 'failed', last_failure = ?, completed_at = datetime('now'), - capability_revoked_at = COALESCE(capability_revoked_at, datetime('now')) + capability_revoked_at = CASE WHEN ? = 1 THEN capability_revoked_at + ELSE COALESCE(capability_revoked_at, datetime('now')) END WHERE id = ?` ) - .run(reason, dispatchId) + .run(reason, options.retainCapability ? 1 : 0, dispatchId) this.db .prepare( `UPDATE worker_dispatches diff --git a/src/main/runtime/orchestration/worker-start-unobserved-prompt-settlement.test.ts b/src/main/runtime/orchestration/worker-start-unobserved-prompt-settlement.test.ts new file mode 100644 index 00000000000..35cfb94b74c --- /dev/null +++ b/src/main/runtime/orchestration/worker-start-unobserved-prompt-settlement.test.ts @@ -0,0 +1,102 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { OrchestrationDb } from './db' + +const WORKER_PANE_KEY = 'tab_worker:bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb' +const INCARNATION = 'runtime_test:term_worker:1' +let db: OrchestrationDb + +function startWorker(spec: string): { taskId: string; dispatchId: string; capability: string } { + const task = db.createTask({ spec }) + const started = db.createStartingWorkerDispatch({ + creator: { kind: 'system' }, + maxDepth: Number.MAX_SAFE_INTEGER, + taskId: task.id, + startOptions: {} + }) + const capability = db.mintDispatchCapability({ + dispatchId: started.dispatch.id, + paneKey: WORKER_PANE_KEY, + processIncarnation: INCARNATION + }) + return { taskId: task.id, dispatchId: started.dispatch.id, capability } +} + +function verify(dispatchId: string, capability: string): { valid: boolean; reason?: string } { + return db.verifyDispatchCapability({ + dispatchId, + capability, + paneKey: WORKER_PANE_KEY, + processIncarnation: INCARNATION + }) +} + +describe('worker start settled by an unobserved prompt', () => { + afterEach(() => db?.close()) + + it('keeps the capability and lets the worker report correct the record', () => { + db = new OrchestrationDb(':memory:') + const { taskId, dispatchId, capability } = startWorker('run to completion') + + db.failWorkerStart(dispatchId, 'dispatch_input', 'agent_prompt_stalled', { + retainCapability: true + }) + + expect(db.getDispatchContextById(dispatchId)).toMatchObject({ + status: 'failed', + last_failure: 'agent_prompt_stalled', + capability_revoked_at: null + }) + expect(verify(dispatchId, capability)).toEqual({ valid: true }) + + expect( + db.settleWorkerReport({ + taskId, + dispatchId, + outcome: 'succeeded', + result: 'done the work' + }) + ).toEqual({ action: 'settled', outcome: 'succeeded', duplicate: false }) + expect(db.getTask(taskId)).toMatchObject({ status: 'completed', result: 'done the work' }) + expect(db.getDispatchContextById(dispatchId)?.status).toBe('completed') + expect(db.getWorkerDispatch(dispatchId)).toMatchObject({ state: 'succeeded', stage: 'settled' }) + }) + + it('revokes and stays settled when the start failed for any other cause', () => { + db = new OrchestrationDb(':memory:') + const { taskId, dispatchId, capability } = startWorker('never became ready') + + db.failWorkerStart(dispatchId, 'agent_readiness', 'Agent did not become ready (idle).') + + expect(db.getDispatchContextById(dispatchId)?.capability_revoked_at).toEqual(expect.any(String)) + expect(verify(dispatchId, capability).valid).toBe(false) + expect( + db.settleWorkerReport({ taskId, dispatchId, outcome: 'succeeded', result: 'done' }) + ).toMatchObject({ action: 'rejected', code: 'inactive_dispatch' }) + expect(db.getTask(taskId)?.status).toBe('failed') + }) + + it('lets a failure report replace the unobserved-prompt cause with the real one', () => { + db = new OrchestrationDb(':memory:') + const { taskId, dispatchId } = startWorker('reports its own failure') + + db.failWorkerStart(dispatchId, 'dispatch_input', 'agent_prompt_stalled', { + retainCapability: true + }) + + expect( + db.settleWorkerReport({ taskId, dispatchId, outcome: 'failed', result: 'build broke on X' }) + ).toEqual({ action: 'settled', outcome: 'failed', duplicate: false }) + expect(db.getTask(taskId)).toMatchObject({ status: 'failed', result: 'build broke on X' }) + expect(db.getDispatchContextById(dispatchId)).toMatchObject({ + status: 'failed', + last_failure: 'build broke on X' + }) + expect(db.getWorkerDispatch(dispatchId)).toMatchObject({ state: 'failed', stage: 'settled' }) + + // The stalled cause is gone, so a repeat report has nothing left to correct. + expect( + db.settleWorkerReport({ taskId, dispatchId, outcome: 'failed', result: 'again' }) + ).toEqual({ action: 'settled', outcome: 'failed', duplicate: true }) + expect(db.getTask(taskId)?.result).toBe('build broke on X') + }) +}) diff --git a/src/main/runtime/rpc/methods/orchestration-worker-start-outcome-classification.test.ts b/src/main/runtime/rpc/methods/orchestration-worker-start-outcome-classification.test.ts new file mode 100644 index 00000000000..112df925f3e --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration-worker-start-outcome-classification.test.ts @@ -0,0 +1,32 @@ +import { describe, expect, it } from 'vitest' +import { isUnknownWorkerStartOutcome } from './orchestration-worker-topology' + +describe('worker start outcome classification', () => { + it('treats an explicit operation_unknown code as unknown at any stage', () => { + const error = Object.assign(new Error('relay dropped'), { code: 'operation_unknown' }) + + expect(isUnknownWorkerStartOutcome(error, 'dispatch_input')).toBe(true) + expect(isUnknownWorkerStartOutcome(error, 'worktree_create')).toBe(true) + }) + + it('treats a lost connection during worktree create as unknown', () => { + expect(isUnknownWorkerStartOutcome(new Error('connection reset'), 'worktree_create')).toBe(true) + expect(isUnknownWorkerStartOutcome(new Error('request timed out'), 'worktree_create')).toBe( + true + ) + }) + + it('keeps a definite failure definite', () => { + expect(isUnknownWorkerStartOutcome(new Error('connection reset'), 'dispatch_input')).toBe(false) + expect(isUnknownWorkerStartOutcome(new Error('worktree exists'), 'worktree_create')).toBe(false) + }) + + // Why: a stalled prompt still reports a definite failure to the caller — the correction path is + // the worker's own report, which keeps its capability and can re-settle the dispatch (see + // worker-start-unobserved-prompt-settlement.test.ts), not an outcome_unknown receipt. + it('does not class a stalled dispatch prompt as unknown', () => { + expect(isUnknownWorkerStartOutcome(new Error('agent_prompt_stalled'), 'dispatch_input')).toBe( + false + ) + }) +}) diff --git a/src/main/runtime/rpc/methods/orchestration-worker-start-prompt-contract.test.ts b/src/main/runtime/rpc/methods/orchestration-worker-start-prompt-contract.test.ts index 89455f9cc0a..826eb57ea55 100644 --- a/src/main/runtime/rpc/methods/orchestration-worker-start-prompt-contract.test.ts +++ b/src/main/runtime/rpc/methods/orchestration-worker-start-prompt-contract.test.ts @@ -254,10 +254,12 @@ describe('orchestration worker-start prompt contract', () => { expect(harness.writes.filter((data) => data === '\r')).toHaveLength(1) const persisted = reopenPromptContractDb(harness) expect(persisted.getTask(harness.taskId)?.status).toBe('failed') + // Why (#16095): the receipt still reports the failure, but Enter was written before it was + // verified — so the capability survives and the worker's own report can correct the record. expect(persisted.getDispatchContextById(dispatchId)).toMatchObject({ status: 'failed', last_failure: 'agent_prompt_stalled', - capability_revoked_at: expect.any(String) + capability_revoked_at: null }) expect(persisted.getWorkerDispatch(dispatchId)).toMatchObject({ state: 'failed', diff --git a/src/main/runtime/rpc/methods/orchestration-worker-start-receipt.ts b/src/main/runtime/rpc/methods/orchestration-worker-start-receipt.ts index e58f9fff01e..cde2ea9a22f 100644 --- a/src/main/runtime/rpc/methods/orchestration-worker-start-receipt.ts +++ b/src/main/runtime/rpc/methods/orchestration-worker-start-receipt.ts @@ -1,4 +1,5 @@ import type { OrchestrationDb } from '../../orchestration/db' +import { isAgentPromptStalledError } from '../../agent-prompt-submission-verification' import { isUnknownWorkerStartOutcome, type WorkerSetupReceipt @@ -19,7 +20,11 @@ export function failWorkerStartWithReceipt(args: { const unknown = isUnknownWorkerStartOutcome(args.error, args.failedStage) const worker = unknown ? args.db.markWorkerStartUnknown(args.dispatchId, args.failedStage, reason) - : args.db.failWorkerStart(args.dispatchId, args.failedStage, reason) + : args.db.failWorkerStart(args.dispatchId, args.failedStage, reason, { + // Why (#16095): the preamble is written before submission is verified, so a stalled + // verdict never means the worker lacks its task — keep the authority its report needs. + retainCapability: isAgentPromptStalledError(args.error) + }) return { runId: args.runId, taskId: args.taskId, From 26721bd6328103e09b79faba1d40bb124db13536 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 26 Aug 2026 16:44:55 -0700 Subject: [PATCH 17/19] fix(codex): stop blocking the main thread on trust grants (#16441) (#16594) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(codex): stop blocking the main thread on trust grants (#16441) Codex hook trust was granted by blocking the Electron main thread on `spawnSync` of a bundled ELECTRON_RUN_AS_NODE entry for the whole app-server deadline: 15s native, 35s WSL, ~45s on the real-home path (rebase inspect + repair + grant). Cold start and every Codex pane launch showed "Not Responding"; the reported event-loop gap was 15,049 ms. The subprocess only ever existed to donate an event loop to a deliberately blocked parent — `runCodexHookTrustGrantSession` was already the real async implementation. Make the callers async and the fork is unnecessary, so the bridge, the forked entry and its envelope are deleted along with their build/knip/tsconfig registrations. The CLI `agent hooks prepare-codex` handler is already async, so it awaits the in-process session and saves a process spawn per managed-home shell. `resolveCodexTrustGrantHost` is async too; the WSL identity probe moves from `execFileSync` to `runProcess`, dropping that file from the child-process import allowlist. Status reads keep a synchronous native-only stamp path. Two invariants that held only because the lane blocked: - Overlapping capability probes were impossible by construction. `GitCapabilityCache`'s dedupe engine is extracted to a shared `CapabilityProbeCache` and `CodexAppServerCapabilityCache` now inherits it, so concurrent launches against a cold host share one app-server session instead of one each. - Two grants on one `config.toml` could not interleave capture and restore. A reentrant per-file lane now serializes the whole install sequence (managed, WSL runtime, real-home ensure, legacy sweep) and the grant and rebase inside it. Cold-start work moves off the critical path: retained-home reconciliation (N sequential sessions) is fire-and-forget behind the daemon provider, and the startup real-home ensure chains into managed hook reconciliation instead of blocking app init. Every preserved semantic is unchanged: never throws, the ORCA_DISABLE_CODEX_TRUST_RPC kill switch, ledger hits, backfill-pending and cooldown fallbacks, config rollback on every failure path, pre-grant self-computed trust removal, the verify-failure taxonomy, diagnostics and telemetry. * fix(codex): widen the trust-config lane to every config.toml writer Review follow-ups on #16441's async trust grant: - `markCodexProjectTrusted` now runs inside the runtime+system config.toml lanes, so a project-trust write can no longer land inside a hook grant's capture->restore window and be silently reverted. Its callers await it. - `install`/`refreshRuntimeUserHooks`/`remove` hold the system config.toml lane as well as the runtime one — they promote approvals into ~/.codex/config.toml and mirror it back. Lock order is runtime-before-system everywhere. - The real-home ensure chain resumes after a rejection instead of returning the same rejected promise to every later pane launch, and resolving the real home is now inside the module's never-throws boundary. - `buildSpawnEnv` awaits inside a cancelable pending-spawn registration, so shutdown during the (now long) env build stops the PTY from launching. `prepareLocalPtySpawn` generalizes into `awaitCancelableLocalPtySpawn`. - CapabilityProbeCache drops the test-only `nowMs` passthrough; its probe backstop comment now describes what it actually guards. - Preflight is a plain async function; the trust dispatch in orca-runtime collapses into one `markWorkspaceTrustedForAgent`. * test(codex): exercise the trust-config lane under real concurrency The async grant makes two pane launches overlap for the first time. These drive the real modules end to end on real files: a rollback swallowing a sibling's grant, a markCodexProjectTrusted write landing inside a capture -> restore window, shared capability-probe dedupe on a cold host, the host-scoped transient cooldown, and reentrancy from inside an installer. Each was verified to fail against a deliberately broken implementation (lane removed, dedupe disabled, cooldown made global, reentrancy pass- through disabled). * test(codex): stop hook-service suites spawning the developer's real codex The forked grant bundle never existed under vitest, so the RPC lane was unreachable in tests on main. Running it in-process makes these suites spawn a real `codex app-server` when one is installed: 38 spawns and two failures in hook-service-runtime-trust-repair on a machine with codex, green in CI where there is none. Stand in for the missing binary so both environments exercise the same fallback lane. * docs(codex): scope the trust-RPC kill switch comment to what it actually gates The comment read as though the flag forces the fallback lane everywhere. It gates the managed grant only: the real-home rebase still runs its own inspect/repair app-server sessions when Orca's insertion shifts a user's hook positions, and never reads the flag. Verified by exercise, not by reading — with the flag set, both inspect-user-hook-trust and repair-user-hook-trust still ran. Pre-existing: main has no check there either, it just blocked the main thread while doing it. Widening the flag to cover the rebase is a follow-up; this only stops the comment promising something the constant does not do. --- .../build-plugins/plain-node-entry-guard.ts | 3 +- config/knip.json | 1 - config/tsconfig.cli.json | 4 +- electron.vite.config.ts | 6 - src/cli/handlers/agent-hooks.ts | 2 +- .../managed-agent-hook-controls.ts | 38 +- .../managed-agent-hook-registry.ts | 12 +- .../managed-hook-stdin-lifecycle.test.ts | 15 +- .../windows-hook-post-interpreter.test.ts | 13 +- src/main/agent-trust-presets.test.ts | 45 +- src/main/agent-trust-presets.ts | 17 +- ...home-service-per-account-migration.test.ts | 2 +- .../codex-app-server-capability-cache.test.ts | 201 ++++++--- .../codex-app-server-capability-cache.ts | 72 +-- .../codex/codex-app-server-client.test.ts | 107 ----- src/main/codex/codex-app-server-client.ts | 4 +- .../codex/codex-app-server-grant-bridge.ts | 144 ------ .../codex/codex-app-server-grant-entry.ts | 79 ---- .../codex/codex-app-server-grant-envelope.ts | 35 -- src/main/codex/codex-hook-trust-grant.test.ts | 256 +++++++---- src/main/codex/codex-hook-trust-grant.ts | 359 +++++++-------- .../codex/codex-managed-trust-grant-plan.ts | 90 ++++ .../codex-real-home-hook-install.test.ts | 240 +++++----- .../codex/codex-real-home-hook-install.ts | 159 ++----- src/main/codex/codex-real-home-hooks-json.ts | 115 +++++ ...dex-trust-config-concurrent-launch.test.ts | 410 ++++++++++++++++++ .../codex-trust-config-mutation-queue.test.ts | 111 +++++ .../codex-trust-config-mutation-queue.ts | 46 ++ src/main/codex/codex-trust-grant-host.test.ts | 65 ++- src/main/codex/codex-trust-grant-host.ts | 44 +- ...x-trust-grant-main-thread-boundary.test.ts | 60 +++ src/main/codex/codex-trust-grant-telemetry.ts | 7 +- .../codex-user-hook-trust-rebase.test.ts | 44 +- .../codex/codex-user-hook-trust-rebase.ts | 38 +- .../codex/hook-service-legacy-cleanup.test.ts | 28 +- .../hook-service-managed-install.test.ts | 84 +++- .../hook-service-runtime-trust-repair.test.ts | 32 +- src/main/codex/hook-service-test-harness.ts | 26 ++ .../codex/hook-service-trust-grant.test.ts | 62 +-- .../hook-service-user-hook-mirroring.test.ts | 52 +-- .../codex/hook-service-wsl-runtime.test.ts | 58 +-- src/main/codex/hook-service.ts | 135 ++++-- src/main/codex/hook-trust-promotion.test.ts | 86 ++-- .../managed-home-shell-preflight.test.ts | 16 +- .../codex/managed-home-shell-preflight.ts | 14 +- .../codex/retained-codex-hook-state.test.ts | 8 +- src/main/codex/retained-codex-hook-state.ts | 23 +- src/main/index.ts | 88 ++-- src/main/ipc/pty/host-env/codex-home.ts | 10 +- src/main/ipc/pty/host-env/codex-resume.ts | 8 +- src/main/ipc/pty/host-env/types.ts | 5 +- src/main/ipc/pty/ipc/spawn-env-codex.ts | 29 +- src/main/ipc/pty/ipc/spawn-types.ts | 4 +- src/main/ipc/pty/provider/local-configure.ts | 6 +- src/main/ipc/pty/runtime/controller-deps.ts | 4 +- src/main/ipc/pty/runtime/spawn-preflight.ts | 29 +- .../local-pty-provider-spawn-session.test.ts | 24 + src/main/providers/local-pty-provider.ts | 39 +- src/main/runtime/orca-runtime.ts | 65 +-- .../commit-message-agent-environment.ts | 6 +- src/shared/capability-probe-cache.ts | 128 ++++++ .../child-process-import-allowlist.txt | 2 - src/shared/git-capability-cache.ts | 107 +---- 63 files changed, 2464 insertions(+), 1558 deletions(-) delete mode 100644 src/main/codex/codex-app-server-grant-bridge.ts delete mode 100644 src/main/codex/codex-app-server-grant-entry.ts delete mode 100644 src/main/codex/codex-app-server-grant-envelope.ts create mode 100644 src/main/codex/codex-managed-trust-grant-plan.ts create mode 100644 src/main/codex/codex-real-home-hooks-json.ts create mode 100644 src/main/codex/codex-trust-config-concurrent-launch.test.ts create mode 100644 src/main/codex/codex-trust-config-mutation-queue.test.ts create mode 100644 src/main/codex/codex-trust-config-mutation-queue.ts create mode 100644 src/main/codex/codex-trust-grant-main-thread-boundary.test.ts create mode 100644 src/shared/capability-probe-cache.ts diff --git a/config/build-plugins/plain-node-entry-guard.ts b/config/build-plugins/plain-node-entry-guard.ts index 441536e7edb..c87e2548a53 100644 --- a/config/build-plugins/plain-node-entry-guard.ts +++ b/config/build-plugins/plain-node-entry-guard.ts @@ -24,8 +24,7 @@ const PLAIN_NODE_ENTRY_NAMES = [ 'parcel-watcher-process-entry', 'computer-sidecar', 'wsl-transcript-fs-process-entry', - 'agent-hooks/managed-agent-hook-controls', - 'codex/codex-app-server-grant-entry' + 'agent-hooks/managed-agent-hook-controls' ] as const // Entries executed as worker threads of the main process. Electron's module is diff --git a/config/knip.json b/config/knip.json index 565fb68bc0a..9af0b8d1a74 100644 --- a/config/knip.json +++ b/config/knip.json @@ -14,7 +14,6 @@ "src/main/ports/port-scan-command-worker-entry.ts", "src/main/ipc/parcel-watcher-process-entry.ts", "src/main/hang-watchdog/main-thread-hang-watchdog-entry.ts", - "src/main/codex/codex-app-server-grant-entry.ts", "src/main/agent-hooks/managed-agent-hook-controls.ts", "src/main/claude-accounts/keychain.ts", "src/renderer/src/main.tsx", diff --git a/config/tsconfig.cli.json b/config/tsconfig.cli.json index e5cf60dd5a7..eb278257159 100644 --- a/config/tsconfig.cli.json +++ b/config/tsconfig.cli.json @@ -30,8 +30,6 @@ "../src/main/codex/codex-app-server-capability-cache.ts", "../src/main/codex/codex-app-server-capability-signal.ts", "../src/main/codex/codex-app-server-client.ts", - "../src/main/codex/codex-app-server-grant-bridge.ts", - "../src/main/codex/codex-app-server-grant-envelope.ts", "../src/main/codex/codex-app-server-session.ts", "../src/main/codex/codex-config-mirror.ts", "../src/main/codex/codex-config-path-reference-rewrite.ts", @@ -40,6 +38,7 @@ "../src/main/codex/codex-config-settings-upsert.ts", "../src/main/codex/codex-home-paths.ts", "../src/main/codex/codex-managed-home-resource-copy-marker.ts", + "../src/main/codex/codex-managed-trust-grant-plan.ts", "../src/main/codex/codex-path-observation.ts", "../src/main/codex/codex-hook-identity.ts", "../src/main/codex/codex-hook-trust-grant.ts", @@ -48,6 +47,7 @@ "../src/main/codex/codex-state-db.ts", "../src/main/codex/codex-trust-identity.ts", "../src/main/codex/codex-trust-config-rollback.ts", + "../src/main/codex/codex-trust-config-mutation-queue.ts", "../src/main/codex/codex-trust-grant-telemetry.ts", "../src/main/codex/codex-trust-grant-host.ts", "../src/main/codex/codex-trust-grant-ledger.ts", diff --git a/electron.vite.config.ts b/electron.vite.config.ts index 982677811c5..f81dbb93041 100644 --- a/electron.vite.config.ts +++ b/electron.vite.config.ts @@ -239,12 +239,6 @@ export const electronViteConfig: UserConfig = { 'main-thread-hang-watchdog-entry': resolve( 'src/main/hang-watchdog/main-thread-hang-watchdog-entry.ts' ), - // Why: run under ELECTRON_RUN_AS_NODE while the caller blocks on - // spawnSync — codex app-server trust grants need a live event loop - // but must finish before a Codex pane launch proceeds. - 'codex/codex-app-server-grant-entry': resolve( - 'src/main/codex/codex-app-server-grant-entry.ts' - ), // Why: electron-vite cleans out/main in dev. The dev CLI imports // this path for `orca agent hooks ...`, so it must survive rebuilds. 'agent-hooks/managed-agent-hook-controls': resolve( diff --git a/src/cli/handlers/agent-hooks.ts b/src/cli/handlers/agent-hooks.ts index 7e21c4973a2..57419518d4b 100644 --- a/src/cli/handlers/agent-hooks.ts +++ b/src/cli/handlers/agent-hooks.ts @@ -208,7 +208,7 @@ async function setAgentHooksEnabled( export const AGENT_HOOK_HANDLERS: Record = { 'agent hooks prepare-codex': async ({ client }) => { const settings = await readHookSettings(client) - prepareManagedCodexHomeBeforeShellLaunch({ + await prepareManagedCodexHomeBeforeShellLaunch({ userDataPath: getDefaultUserDataPath(), hooksEnabled: settings.agentStatusHooksEnabled && !settings.disabledTuiAgents.includes('codex') diff --git a/src/main/agent-hooks/managed-agent-hook-controls.ts b/src/main/agent-hooks/managed-agent-hook-controls.ts index 043cc1bb1ea..64d426ae072 100644 --- a/src/main/agent-hooks/managed-agent-hook-controls.ts +++ b/src/main/agent-hooks/managed-agent-hook-controls.ts @@ -86,14 +86,14 @@ function selectedInstallers(options: InstallOptions): readonly ManagedAgentHookI return MANAGED_AGENT_HOOK_INSTALLERS.filter(([agent]) => allowed.has(agent)) } -function runInstaller( +async function runInstaller( entry: ManagedAgentHookInstaller, onInstallError: InstallOptions['onInstallError'], userInitiated?: boolean -): AgentHookInstallStatus { +): Promise { const [agent, install] = entry try { - return install({ userInitiated }) + return await install({ userInitiated }) } catch (error) { console.error(`[agent-hooks] Failed to install ${agent} managed hooks:`, error) try { @@ -177,22 +177,27 @@ export async function installManagedAgentHooks( ) continue } - results.push(runInstaller(entry, options.onInstallError, options.userInitiated)) + results.push(await runInstaller(entry, options.onInstallError, options.userInitiated)) } return results } -export function removeManagedAgentHooks(options: RemoveOptions = {}): AgentHookInstallStatus[] { +export async function removeManagedAgentHooks( + options: RemoveOptions = {} +): Promise { const allowed = options.agents ? new Set(options.agents) : null - return MANAGED_AGENT_HOOK_REMOVERS.filter( - ([agent]) => allowed === null || allowed.has(agent) - ).map(([agent, remove]) => { - try { - return remove() - } catch (error) { - return errorStatus(agent, error) + const results: AgentHookInstallStatus[] = [] + for (const [agent, remove] of MANAGED_AGENT_HOOK_REMOVERS) { + if (allowed !== null && !allowed.has(agent)) { + continue } - }) + try { + results.push(await remove()) + } catch (error) { + results.push(errorStatus(agent, error)) + } + } + return results } export async function removeManagedAgentHooksAsync( @@ -228,7 +233,7 @@ export async function applyAgentStatusHooksEnabled( options: InstallOptions = {} ): Promise { if (!enabled) { - return removeManagedAgentHooks() + return await removeManagedAgentHooks() } const disabled = normalizeDisabledTuiAgents(settings?.disabledTuiAgents).filter( isManagedAgentHookTarget @@ -241,7 +246,10 @@ export async function applyAgentStatusHooksEnabled( return installed } const removed = new Map( - removeManagedAgentHooks({ agents: disabledToRemove }).map((status) => [status.agent, status]) + (await removeManagedAgentHooks({ agents: disabledToRemove })).map((status) => [ + status.agent, + status + ]) ) return installed.map((status) => removed.get(status.agent) ?? status) } diff --git a/src/main/agent-hooks/managed-agent-hook-registry.ts b/src/main/agent-hooks/managed-agent-hook-registry.ts index c94d367bd79..5c462f1794a 100644 --- a/src/main/agent-hooks/managed-agent-hook-registry.ts +++ b/src/main/agent-hooks/managed-agent-hook-registry.ts @@ -15,13 +15,21 @@ import { hermesHookService } from '../hermes/hook-service' import { kimiHookService } from '../kimi/hook-service' import { openClaudeHookService } from '../openclaude/hook-service' +// Why (#16441): Codex's installer awaits a codex app-server trust-grant session +// instead of blocking the main thread on spawnSync. Widening the tuple keeps the +// other thirteen agent services synchronous — the shared loop already awaits. export type ManagedAgentHookInstallOptions = { userInitiated?: boolean } export type ManagedAgentHookInstaller = readonly [ HookInstallAgent, - (options?: ManagedAgentHookInstallOptions) => AgentHookInstallStatus + ( + options?: ManagedAgentHookInstallOptions + ) => AgentHookInstallStatus | Promise ] export type ManagedAgentHookScriptRefresher = readonly [HookInstallAgent, () => Promise] -export type ManagedAgentHookRemover = readonly [HookInstallAgent, () => AgentHookInstallStatus] +export type ManagedAgentHookRemover = readonly [ + HookInstallAgent, + () => AgentHookInstallStatus | Promise +] export type ManagedAgentHookAsyncRemover = readonly [ HookInstallAgent, () => Promise diff --git a/src/main/agent-hooks/managed-hook-stdin-lifecycle.test.ts b/src/main/agent-hooks/managed-hook-stdin-lifecycle.test.ts index c52d6656013..40237141d24 100644 --- a/src/main/agent-hooks/managed-hook-stdin-lifecycle.test.ts +++ b/src/main/agent-hooks/managed-hook-stdin-lifecycle.test.ts @@ -209,11 +209,14 @@ async function generatePosixScripts(): Promise> { return scripts } -function withPlatform(platform: NodeJS.Platform, run: () => T): T { +// Why: the Codex installer awaits an app-server trust-grant session, so the +// override has to stay pinned across the await instead of being restored by a +// synchronous `finally` while the install is still running. +async function withPlatform(platform: NodeJS.Platform, run: () => T | Promise): Promise { const original = Object.getOwnPropertyDescriptor(process, 'platform') Object.defineProperty(process, 'platform', { configurable: true, value: platform }) try { - return run() + return await run() } finally { if (original) { Object.defineProperty(process, 'platform', original) @@ -222,7 +225,7 @@ function withPlatform(platform: NodeJS.Platform, run: () => T): T { } describe('Windows managed hook stdin structure', () => { - it('exits immediately when Orca env is missing and keeps drain for other failures', () => { + it('exits immediately when Orca env is missing and keeps drain for other failures', async () => { const home = mkdtempSync(join(tmpdir(), 'orca-hook-stdin-windows-')) homedirMock.mockReturnValue(home) const previousGrokHome = process.env.GROK_HOME @@ -230,9 +233,9 @@ describe('Windows managed hook stdin structure', () => { delete process.env.GROK_HOME delete process.env.KIMI_CODE_HOME try { - withPlatform('win32', () => { + await withPlatform('win32', async () => { for (const entry of LOCAL_INSTALLERS) { - expect(entry.install().state, `${entry.agent} install status`).toBe('installed') + expect((await entry.install()).state, `${entry.agent} install status`).toBe('installed') } }) const hooksDir = join(home, '.orca', 'agent-hooks') @@ -317,7 +320,7 @@ describe('Windows managed hook stdin structure', () => { try { const gitBash = findGitBash() for (const entry of LOCAL_INSTALLERS) { - expect(entry.install().state, `${entry.agent} install status`).toBe('installed') + expect((await entry.install()).state, `${entry.agent} install status`).toBe('installed') } const hooksDir = join(home, '.orca', 'agent-hooks') const mainScripts = readdirSync(hooksDir).filter( diff --git a/src/main/agent-hooks/windows-hook-post-interpreter.test.ts b/src/main/agent-hooks/windows-hook-post-interpreter.test.ts index 0e79c125c2e..09377f3adf5 100644 --- a/src/main/agent-hooks/windows-hook-post-interpreter.test.ts +++ b/src/main/agent-hooks/windows-hook-post-interpreter.test.ts @@ -56,11 +56,14 @@ const BATCH_SCRIPT_INSTALLERS = [ { agent: 'grok', install: () => new GrokHookService().install() } ] as const -function withPlatform(platform: NodeJS.Platform, run: () => T): T { +// Why: the Codex installer awaits an app-server trust-grant session, so the +// override has to stay pinned across the await instead of being restored by a +// synchronous `finally` while the install is still running. +async function withPlatform(platform: NodeJS.Platform, run: () => T | Promise): Promise { const originalPlatform = Object.getOwnPropertyDescriptor(process, 'platform') Object.defineProperty(process, 'platform', { configurable: true, value: platform }) try { - return run() + return await run() } finally { if (originalPlatform) { Object.defineProperty(process, 'platform', originalPlatform) @@ -93,10 +96,10 @@ describe('Windows managed hook post interpreter', () => { home = '' }) - it('posts through curl.exe from every managed batch script, spawning no interpreter', () => { - const scripts = withPlatform('win32', () => { + it('posts through curl.exe from every managed batch script, spawning no interpreter', async () => { + const scripts = await withPlatform('win32', async () => { for (const entry of BATCH_SCRIPT_INSTALLERS) { - expect(entry.install().state, `${entry.agent} install status`).toBe('installed') + expect((await entry.install()).state, `${entry.agent} install status`).toBe('installed') } const hooksDir = join(home, '.orca', 'agent-hooks') return readdirSync(hooksDir) diff --git a/src/main/agent-trust-presets.test.ts b/src/main/agent-trust-presets.test.ts index 2a979c60e27..f5377be86c4 100644 --- a/src/main/agent-trust-presets.test.ts +++ b/src/main/agent-trust-presets.test.ts @@ -40,6 +40,8 @@ vi.mock('node:os', async () => { const { markCodexProjectTrusted, markCopilotFolderTrusted, markCursorWorkspaceTrusted } = await import('./agent-trust-presets') +const { runExclusivelyForCodexTrustConfig } = + await import('./codex/codex-trust-config-mutation-queue') beforeEach(() => { testState.fakeHomeDir = mkdtempSync(join(tmpdir(), 'orca-trust-presets-')) @@ -137,7 +139,32 @@ describe('markCopilotFolderTrusted', () => { }) describe('markCodexProjectTrusted', () => { - it('trusts the main repository root for a linked worktree without reading commondir', () => { + // Why (#16441): a hook install/grant holds this file across an awaited + // app-server session; an unqueued write here lands inside its + // capture->restore window and is silently reverted. + it('queues behind an in-flight Codex trust-config mutation', async () => { + const workspace = mkdtempSync(join(tmpdir(), 'orca-codex-ws-')) + const configPath = join(testState.fakeHomeDir, '.codex', 'config.toml') + let releaseGrant!: () => void + const grantHoldingTheFile = new Promise((resolve) => { + releaseGrant = resolve + }) + try { + const held = runExclusivelyForCodexTrustConfig(configPath, () => grantHoldingTheFile) + const marked = markCodexProjectTrusted(workspace) + await Promise.resolve() + expect(existsSync(configPath)).toBe(false) + + releaseGrant() + await held + await marked + expect(readFileSync(configPath, 'utf-8')).toContain('trust_level = "trusted"') + } finally { + rmSync(workspace, { recursive: true, force: true }) + } + }) + + it('trusts the main repository root for a linked worktree without reading commondir', async () => { const fixtureRoot = mkdtempSync(join(tmpdir(), 'orca-codex-linked-ws-')) const repository = join(fixtureRoot, 'repo') const workspace = join(fixtureRoot, 'worktrees', 'feature') @@ -148,7 +175,7 @@ describe('markCodexProjectTrusted', () => { writeFileSync(join(workspace, '.git'), `gitdir: ${worktreeGitDir}\n`, 'utf-8') writeFileSync(join(worktreeGitDir, 'gitdir'), join(workspace, '.git'), 'utf-8') - markCodexProjectTrusted(workspace) + await markCodexProjectTrusted(workspace) const repositoryRoot = realpathSync.native(repository) const workspaceRoot = realpathSync.native(workspace) @@ -171,7 +198,7 @@ describe('markCodexProjectTrusted', () => { } }) - it('does not broaden trust through arbitrary or adversarial Git metadata', () => { + it('does not broaden trust through arbitrary or adversarial Git metadata', async () => { const fixtureRoot = mkdtempSync(join(tmpdir(), 'orca-codex-untrusted-gitdir-')) const workspace = join(fixtureRoot, 'workspace') const arbitraryGitDir = join(fixtureRoot, 'metadata', 'feature') @@ -183,12 +210,12 @@ describe('markCodexProjectTrusted', () => { writeFileSync(join(workspace, '.git'), `gitdir: ${arbitraryGitDir}\n`, 'utf-8') writeFileSync(join(arbitraryGitDir, 'commondir'), join(unrelatedRoot, '.git'), 'utf-8') - markCodexProjectTrusted(workspace) + await markCodexProjectTrusted(workspace) const structuredGitDir = join(unrelatedRoot, '.git', 'worktrees', 'feature') mkdirSync(structuredGitDir, { recursive: true }) writeFileSync(join(workspace, '.git'), `gitdir: ${structuredGitDir}\n`, 'utf-8') writeFileSync(join(structuredGitDir, 'gitdir'), join(unrelatedRoot, '.git'), 'utf-8') - markCodexProjectTrusted(workspace) + await markCodexProjectTrusted(workspace) const written = readFileSync(join(testState.fakeHomeDir, '.codex', 'config.toml'), 'utf-8') expect(written).toContain( @@ -202,11 +229,11 @@ describe('markCodexProjectTrusted', () => { } }) - it('writes ~/.codex/config.toml with the project marked trusted', () => { + it('writes ~/.codex/config.toml with the project marked trusted', async () => { const workspace = mkdtempSync(join(tmpdir(), 'orca-codex-ws-')) try { const realpath = realpathSync.native(workspace) - markCodexProjectTrusted(workspace) + await markCodexProjectTrusted(workspace) const configPath = join(testState.fakeHomeDir, '.codex', 'config.toml') const runtimeConfigPath = join( testState.userDataDir, @@ -227,7 +254,7 @@ describe('markCodexProjectTrusted', () => { } }) - it('preserves existing config keys and updates an existing project block', () => { + it('preserves existing config keys and updates an existing project block', async () => { const workspace = mkdtempSync(join(tmpdir(), 'orca-codex-ws-')) const realpath = realpathSync.native(workspace) try { @@ -260,7 +287,7 @@ describe('markCodexProjectTrusted', () => { 'utf-8' ) - markCodexProjectTrusted(workspace) + await markCodexProjectTrusted(workspace) const written = readFileSync(join(codexDir, 'config.toml'), 'utf-8') const runtimeWritten = readFileSync(join(runtimeCodexDir, 'config.toml'), 'utf-8') diff --git a/src/main/agent-trust-presets.ts b/src/main/agent-trust-presets.ts index 3b441d6d1fe..16c728eba55 100644 --- a/src/main/agent-trust-presets.ts +++ b/src/main/agent-trust-presets.ts @@ -4,6 +4,7 @@ import { basename, dirname, join, resolve } from 'node:path' import { writeFileAtomically } from './codex-accounts/fs-utils' import { getOrcaManagedCodexHomePath } from './codex/codex-home-paths' import { upsertProjectTrustLevel } from './codex/config-toml-trust' +import { runExclusivelyForCodexTrustConfig } from './codex/codex-trust-config-mutation-queue' export type AgentTrustPreset = 'cursor' | 'copilot' | 'codex' @@ -108,13 +109,21 @@ export function markCopilotFolderTrusted(workspacePath: string): void { * Verified against codex-rs/tui/src/onboarding/trust_directory.rs and * codex-rs/core/src/config/config_tests.rs in the Codex CLI source. */ -export function markCodexProjectTrusted(workspacePath: string): void { +export function markCodexProjectTrusted(workspacePath: string): Promise { const absPath = resolveCodexProjectTrustRoot(workspacePath) - const configPath = join(homedir(), '.codex', 'config.toml') - upsertProjectTrustLevel(configPath, absPath, 'trusted') + const systemTomlPath = join(homedir(), '.codex', 'config.toml') // Why: Orca-launched Codex runs with an Orca-owned CODEX_HOME, so the trust // preset must also update the runtime config Codex will actually read. - upsertProjectTrustLevel(join(getOrcaManagedCodexHomePath(), 'config.toml'), absPath, 'trusted') + const runtimeTomlPath = join(getOrcaManagedCodexHomePath(), 'config.toml') + // Why (#16441): hook installs now await a codex app-server grant, so an + // unqueued write here can land inside their capture->restore window and be + // reverted. Same runtime-before-system lock order the installer takes. + return runExclusivelyForCodexTrustConfig(runtimeTomlPath, () => + runExclusivelyForCodexTrustConfig(systemTomlPath, async () => { + upsertProjectTrustLevel(systemTomlPath, absPath, 'trusted') + upsertProjectTrustLevel(runtimeTomlPath, absPath, 'trusted') + }) + ) } function resolveCodexProjectTrustRoot(workspacePath: string): string { diff --git a/src/main/codex-accounts/runtime-home-service-per-account-migration.test.ts b/src/main/codex-accounts/runtime-home-service-per-account-migration.test.ts index 20b2ef75839..53736a11beb 100644 --- a/src/main/codex-accounts/runtime-home-service-per-account-migration.test.ts +++ b/src/main/codex-accounts/runtime-home-service-per-account-migration.test.ts @@ -95,7 +95,7 @@ describe('CodexRuntimeHomeService per-account takeover composition', () => { const config = readFileSync(join(account.managedHomePath, 'config.toml'), 'utf8') expect(config).toContain('model = "fixture-model"') expect(config).not.toContain('[hooks.state') - expect(hookService.install(account.managedHomePath).state).toBe('installed') + expect((await hookService.install(account.managedHomePath)).state).toBe('installed') expect(readFileSync(join(account.managedHomePath, 'hooks.json'), 'utf8')).toContain( process.platform === 'win32' ? 'codex-hook.cmd' : 'codex-hook.sh' ) diff --git a/src/main/codex/codex-app-server-capability-cache.test.ts b/src/main/codex/codex-app-server-capability-cache.test.ts index c6dea6e0031..676c3f8191b 100644 --- a/src/main/codex/codex-app-server-capability-cache.test.ts +++ b/src/main/codex/codex-app-server-capability-cache.test.ts @@ -21,30 +21,40 @@ describe('CodexAppServerCapabilityCache', () => { ) }) - it('falls back on the first unsupported probe and skips the probe on later calls', () => { + it('falls back on the first unsupported probe and skips the probe on later calls', async () => { const cache = new CodexAppServerCapabilityCache() - const firstPreferred = vi.fn(() => { - throw unsupportedError - }) - expect( - cache.runWithFallbackSync('native', firstPreferred, () => 'first-fallback', isUnsupported, 5) - ).toBe('first-fallback') + const firstPreferred = vi.fn(() => Promise.reject(unsupportedError)) + await expect( + cache.runWithFallback( + 'native', + firstPreferred, + () => Promise.resolve('first-fallback'), + isUnsupported + ) + ).resolves.toBe('first-fallback') expect(firstPreferred).toHaveBeenCalledTimes(1) - // Why: probes are synchronous on the main thread, so they can never - // overlap — back-to-back calls inside the retry window are the - // "concurrent probe" equivalent and must share the first probe's result. - const laterPreferred = vi.fn(() => 'unexpected-preferred') - expect( - cache.runWithFallbackSync('native', laterPreferred, () => 'cached-fallback', isUnsupported, 6) - ).toBe('cached-fallback') - expect( - cache.runWithFallbackSync('native', laterPreferred, () => 'cached-fallback', isUnsupported, 7) - ).toBe('cached-fallback') + const laterPreferred = vi.fn(() => Promise.resolve('unexpected-preferred')) + await expect( + cache.runWithFallback( + 'native', + laterPreferred, + () => Promise.resolve('cached-fallback'), + isUnsupported + ) + ).resolves.toBe('cached-fallback') + await expect( + cache.runWithFallback( + 'native', + laterPreferred, + () => Promise.resolve('cached-fallback'), + isUnsupported + ) + ).resolves.toBe('cached-fallback') expect(laterPreferred).not.toHaveBeenCalled() }) - it('isolates capability state per execution host', () => { + it('isolates capability state per execution host', async () => { const cache = new CodexAppServerCapabilityCache() cache.rememberUnsupported('wsl:Ubuntu', 1_000) @@ -52,63 +62,148 @@ describe('CodexAppServerCapabilityCache', () => { expect(cache.shouldTry('native', 1_001)).toBe(true) expect(cache.shouldTry('wsl:Debian', 1_001)).toBe(true) - const nativePreferred = vi.fn(() => 'native-result') - expect( - cache.runWithFallbackSync('native', nativePreferred, () => 'unexpected', isUnsupported, 1_001) - ).toBe('native-result') + const nativePreferred = vi.fn(() => Promise.resolve('native-result')) + await expect( + cache.runWithFallback( + 'native', + nativePreferred, + () => Promise.resolve('unexpected'), + isUnsupported + ) + ).resolves.toBe('native-result') expect(nativePreferred).toHaveBeenCalledTimes(1) }) - it('drops known support when a later call reports the capability unsupported', () => { + it('drops known support when a later call reports the capability unsupported', async () => { const cache = new CodexAppServerCapabilityCache() - expect( - cache.runWithFallbackSync( + await expect( + cache.runWithFallback( 'native', - () => 'supported', - () => 'unexpected', - isUnsupported, - 1 + () => Promise.resolve('supported'), + () => Promise.resolve('unexpected'), + isUnsupported ) - ).toBe('supported') + ).resolves.toBe('supported') expect(cache.isKnownSupported('native')).toBe(true) - expect( - cache.runWithFallbackSync( + await expect( + cache.runWithFallback( 'native', - () => { - throw unsupportedError - }, - () => 'fallback', - isUnsupported, - 2 + () => Promise.reject(unsupportedError), + () => Promise.resolve('fallback'), + isUnsupported ) - ).toBe('fallback') + ).resolves.toBe('fallback') expect(cache.isKnownSupported('native')).toBe(false) - const laterPreferred = vi.fn(() => 'unexpected-preferred') - expect( - cache.runWithFallbackSync('native', laterPreferred, () => 'cached-fallback', isUnsupported, 3) - ).toBe('cached-fallback') + const laterPreferred = vi.fn(() => Promise.resolve('unexpected-preferred')) + await expect( + cache.runWithFallback( + 'native', + laterPreferred, + () => Promise.resolve('cached-fallback'), + isUnsupported + ) + ).resolves.toBe('cached-fallback') expect(laterPreferred).not.toHaveBeenCalled() }) - it('rethrows transient errors without marking the host unsupported', () => { + it('rethrows transient errors without marking the host unsupported', async () => { const cache = new CodexAppServerCapabilityCache() const transient = new Error('spawn ETIMEDOUT') - expect(() => - cache.runWithFallbackSync( + await expect( + cache.runWithFallback( 'native', - () => { - throw transient - }, - () => 'unexpected-fallback', - isUnsupported, - 1 + () => Promise.reject(transient), + () => Promise.resolve('unexpected-fallback'), + isUnsupported ) - ).toThrow(transient) + ).rejects.toBe(transient) expect(cache.shouldTry('native', 2)).toBe(true) }) + // Why (#16441): grants no longer block the main thread, so two pane launches + // can reach a cold host at once. Without dedupe each one pays its own + // app-server session against a codex that has no such RPC surface. + it('dedupes concurrent probes on one host to a single app-server session', async () => { + const cache = new CodexAppServerCapabilityCache() + let releaseProbe!: (error: unknown) => void + const preferred = vi.fn( + () => + new Promise((_resolve, reject) => { + releaseProbe = reject + }) + ) + const first = cache.runWithFallback( + 'native', + preferred, + () => Promise.resolve('fallback'), + isUnsupported + ) + const second = cache.runWithFallback( + 'native', + preferred, + () => Promise.resolve('fallback'), + isUnsupported + ) + await Promise.resolve() + releaseProbe(unsupportedError) + + await expect(first).resolves.toBe('fallback') + await expect(second).resolves.toBe('fallback') + expect(preferred).toHaveBeenCalledTimes(1) + }) + + it('lets a waiter run its own work once the in-flight probe reports support', async () => { + const cache = new CodexAppServerCapabilityCache() + let releaseProbe!: (value: string) => void + const firstPreferred = vi.fn( + () => + new Promise((resolve) => { + releaseProbe = resolve + }) + ) + const secondPreferred = vi.fn(() => Promise.resolve('second')) + const first = cache.runWithFallback( + 'native', + firstPreferred, + () => Promise.resolve('fallback'), + isUnsupported + ) + const second = cache.runWithFallback( + 'native', + secondPreferred, + () => Promise.resolve('fallback'), + isUnsupported + ) + await Promise.resolve() + releaseProbe('first') + + await expect(first).resolves.toBe('first') + await expect(second).resolves.toBe('second') + expect(secondPreferred).toHaveBeenCalledTimes(1) + }) + + it('isolates in-flight probes per host so a cold WSL distro never waits on native', async () => { + const cache = new CodexAppServerCapabilityCache() + const nativePreferred = vi.fn(() => new Promise(() => {})) + void cache.runWithFallback( + 'native', + nativePreferred, + () => Promise.resolve('fallback'), + isUnsupported + ) + const wslPreferred = vi.fn(() => Promise.resolve('wsl-result')) + await expect( + cache.runWithFallback( + 'wsl:Ubuntu', + wslPreferred, + () => Promise.resolve('fallback'), + isUnsupported + ) + ).resolves.toBe('wsl-result') + }) + it('builds host keys that keep WSL distros apart', () => { expect(getCodexAppServerHostKey({ kind: 'native' })).toBe('native') expect(getCodexAppServerHostKey({ kind: 'wsl', distro: 'Ubuntu' })).toBe('wsl:Ubuntu') diff --git a/src/main/codex/codex-app-server-capability-cache.ts b/src/main/codex/codex-app-server-capability-cache.ts index 85816c9cf0c..ae6c175f043 100644 --- a/src/main/codex/codex-app-server-capability-cache.ts +++ b/src/main/codex/codex-app-server-capability-cache.ts @@ -1,3 +1,5 @@ +import { CapabilityProbeCache } from '../../shared/capability-probe-cache' + // Why: suppress a known-missing RPC surface without pinning it forever — an // in-place codex upgrade during a long Orca session self-heals after the // interval, mirroring GitCapabilityCache's rationale. @@ -14,70 +16,14 @@ export function getCodexAppServerHostKey( } /** - * Capability cache for the codex app-server trust-grant RPC pair, modeled on - * GitCapabilityCache but with a synchronous runner: the grant client blocks - * the main thread by design (launch prep), so probes cannot overlap — the - * unsupported mark alone is what keeps later installs off the dead probe. + * Capability cache for the codex app-server trust-grant RPC pair. The grant + * client runs off the main thread's critical path, so two pane launches can + * probe the same host at once; the shared probe dedupe is what keeps a cold + * host to one app-server session instead of one per concurrent launch. */ -export class CodexAppServerCapabilityCache { - private readonly retryAfterByHost = new Map() - private readonly supportedHosts = new Set() - - shouldTry(hostKey: CodexAppServerHostKey, nowMs = Date.now()): boolean { - const retryAfterMs = this.retryAfterByHost.get(hostKey) - if (retryAfterMs === undefined) { - return true - } - if (nowMs < retryAfterMs) { - return false - } - this.retryAfterByHost.delete(hostKey) - return true - } - - isKnownSupported(hostKey: CodexAppServerHostKey): boolean { - return this.supportedHosts.has(hostKey) - } - - rememberUnsupported(hostKey: CodexAppServerHostKey, nowMs = Date.now()): void { - this.supportedHosts.delete(hostKey) - this.retryAfterByHost.set(hostKey, nowMs + CODEX_APP_SERVER_CAPABILITY_RETRY_INTERVAL_MS) - } - - rememberSupported(hostKey: CodexAppServerHostKey): void { - this.retryAfterByHost.delete(hostKey) - this.supportedHosts.add(hostKey) - } - - runWithFallbackSync( - hostKey: CodexAppServerHostKey, - runPreferred: () => T, - runFallback: () => T, - isUnsupportedError: (error: unknown) => boolean, - nowMs = Date.now() - ): T { - if (!this.supportedHosts.has(hostKey) && !this.shouldTry(hostKey, nowMs)) { - return runFallback() - } - try { - const result = runPreferred() - this.rememberSupported(hostKey) - return result - } catch (error) { - // Why: only a positive absence signal (unknown method / missing - // subcommand) marks unsupported. Transient spawn failures, timeouts, - // and RPC errors fall back once without poisoning the capability. - if (!isUnsupportedError(error)) { - throw error - } - this.rememberUnsupported(hostKey, nowMs) - return runFallback() - } - } - - clear(): void { - this.retryAfterByHost.clear() - this.supportedHosts.clear() +export class CodexAppServerCapabilityCache extends CapabilityProbeCache { + constructor() { + super(CODEX_APP_SERVER_CAPABILITY_RETRY_INTERVAL_MS) } } diff --git a/src/main/codex/codex-app-server-client.test.ts b/src/main/codex/codex-app-server-client.test.ts index 121aebe7255..30d4941ad57 100644 --- a/src/main/codex/codex-app-server-client.test.ts +++ b/src/main/codex/codex-app-server-client.test.ts @@ -13,10 +13,6 @@ import { type CodexHookTrustGrantRequest } from './codex-app-server-client' import { killCodexAppServerProcessTree, runCodexAppServerSession } from './codex-app-server-session' -import { - resolveCodexGrantEntryPath, - runCodexHookTrustGrantSessionSync -} from './codex-app-server-grant-bridge' // Stub codex app-server speaking the same JSONL protocol: initialize → // initialized → hooks/list → config/batchWrite → hooks/list. Scenario-driven @@ -470,106 +466,3 @@ describe('runCodexHookTrustGrantSession', () => { expect(isCodexAppServerUnsupportedError(error)).toBe(false) }) }) - -describe('runCodexHookTrustGrantSessionSync', () => { - function writeEntryFixture(source: string): string { - const root = mkdtempSync(join(tmpdir(), 'orca-codex-entry-')) - tempRoots.push(root) - const entryPath = join(root, 'grant-entry.cjs') - writeFileSync(entryPath, source) - return entryPath - } - - const baseRequest: CodexHookTrustGrantRequest = { - invocation: { command: 'codex', cliPath: null, args: ['app-server'], timeoutMs: 1_000 }, - hooksListCwd: '/tmp', - expectedTrustKeys: ['k'], - managedCommand: MANAGED_COMMAND - } - - it('returns the entry envelope result and passes the request over stdin', () => { - const entryPath = writeEntryFixture(` - let input = '' - process.stdin.setEncoding('utf8') - process.stdin.on('data', (chunk) => { input += chunk }) - process.stdin.on('end', () => { - const request = JSON.parse(input) - process.stdout.write(JSON.stringify({ - ok: true, - result: { - outcome: 'granted', - wroteTrust: true, - entries: [{ key: request.expectedTrustKeys[0], normalizedKey: request.expectedTrustKeys[0], trustedHash: 'sha256:x' }] - } - }) + '\\n') - }) - `) - const result = runCodexHookTrustGrantSessionSync(baseRequest, { entryPath }) - expect(result).toMatchObject({ outcome: 'granted', wroteTrust: true }) - }) - - it('rethrows unsupported envelopes as the unsupported error class', () => { - const entryPath = writeEntryFixture(` - process.stdin.resume() - process.stdin.on('end', () => { - process.stdout.write(JSON.stringify({ ok: false, errorName: 'CodexAppServerUnsupportedError', message: 'no app-server', unsupported: true }) + '\\n') - }) - `) - expect(() => runCodexHookTrustGrantSessionSync(baseRequest, { entryPath })).toThrow( - CodexAppServerUnsupportedError - ) - }) - - it('fails with a clear error when the entry produces no result', () => { - const entryPath = writeEntryFixture( - `process.stdin.resume(); process.stdin.on('end', () => process.exit(7))` - ) - expect(() => runCodexHookTrustGrantSessionSync(baseRequest, { entryPath })).toThrow( - /produced no result \(exit 7\)/ - ) - }) - - it('classifies the spawnSync deadline as a typed timeout', () => { - const entryPath = writeEntryFixture(`setInterval(() => {}, 1000)`) - const request = { - ...baseRequest, - invocation: { ...baseRequest.invocation, timeoutMs: 20 } - } - expect(() => - runCodexHookTrustGrantSessionSync(request, { entryPath, timeoutMarginMs: 20 }) - ).toThrow(CodexAppServerTimeoutError) - }) -}) - -describe('resolveCodexGrantEntryPath', () => { - const entryName = 'codex-app-server-grant-entry.js' - - it('finds the sibling entry from emitted main and chunk directories', () => { - const mainDir = join('/opt', 'orca', 'out', 'main') - expect( - resolveCodexGrantEntryPath( - (candidate) => candidate === join(mainDir, 'codex', entryName), - mainDir - ) - ).toBe(join(mainDir, 'codex', entryName)) - - const chunkDir = join(mainDir, 'chunks') - expect( - resolveCodexGrantEntryPath( - (candidate) => candidate === join(mainDir, 'codex', entryName), - chunkDir - ) - ).toBe(join(mainDir, 'codex', entryName)) - }) - - it('redirects app.asar to unpacked without double-unpacking an existing path', () => { - const resourcesDir = join('/Applications', 'Orca.app', 'Contents', 'Resources') - const expected = join(resourcesDir, 'app.asar.unpacked', 'out', 'main', 'codex', entryName) - for (const archiveDir of ['app.asar', 'app.asar.unpacked']) { - const moduleDir = join(resourcesDir, archiveDir, 'out', 'main', 'chunks') - expect(resolveCodexGrantEntryPath((candidate) => candidate === expected, moduleDir)).toBe( - expected - ) - } - }) -}) diff --git a/src/main/codex/codex-app-server-client.ts b/src/main/codex/codex-app-server-client.ts index cdc2c4d20ff..8c95562e66c 100644 --- a/src/main/codex/codex-app-server-client.ts +++ b/src/main/codex/codex-app-server-client.ts @@ -41,8 +41,8 @@ export type CodexGrantedHookTrust = { trustedHash: string } -/** Closed verify-failure taxonomy — crosses the grant-bridge JSON envelope, so - * telemetry never has to parse the free-form `reason` diagnostics string. */ +/** Closed verify-failure taxonomy, so telemetry never has to parse the + * free-form `reason` diagnostics string. */ export type CodexTrustGrantSessionVerifyClass = | 'list-mismatch' | 'post-grant-untrusted' diff --git a/src/main/codex/codex-app-server-grant-bridge.ts b/src/main/codex/codex-app-server-grant-bridge.ts deleted file mode 100644 index c85c4a1c479..00000000000 --- a/src/main/codex/codex-app-server-grant-bridge.ts +++ /dev/null @@ -1,144 +0,0 @@ -import { spawnSync } from 'node:child_process' -import { existsSync } from 'node:fs' -import { join } from 'node:path' -import { - CodexAppServerTimeoutError, - CodexAppServerUnsupportedError, - type CodexHookTrustGrantRequest, - type CodexHookTrustGrantSessionResult -} from './codex-app-server-client' -import type { - CodexAppServerEntryRequest, - CodexAppServerEntryResult, - GrantEntryEnvelope -} from './codex-app-server-grant-envelope' -import type { - CodexUserHookTrustRebaseRequest, - CodexUserHookTrustRebaseResult -} from './codex-user-hook-trust-rebase-client' - -// Why: hook install/refresh is synchronous launch prep — a Codex pane must -// not start before its trust is settled — but a stdio JSON-RPC session needs -// a live event loop. This bridge blocks the caller on spawnSync of a bundled -// ELECTRON_RUN_AS_NODE entry (same pattern as the daemon and parcel-watcher -// entries) that runs the session and reports one JSON envelope on stdout. - -const GRANT_ENTRY_FILE_NAME = 'codex-app-server-grant-entry.js' -// Why: spawnSync must outlive the session deadline so the entry's own timeout -// (and its result envelope) win the race; the margin only reaps a hung entry. -const GRANT_ENTRY_TIMEOUT_MARGIN_MS = 5_000 -const GRANT_ENTRY_MAX_BUFFER_BYTES = 16 * 1024 * 1024 - -export function resolveCodexGrantEntryPath( - pathExists: (candidate: string) => boolean = existsSync, - moduleDir = __dirname -): string | null { - // Why: resolved from __dirname (not electron's app paths) so this module - // stays loadable in plain-node CLI entries — the build guard rejects any - // electron require reachable from them. The emitted bridge chunk sits in - // out/main or out/main/chunks, so the entry is one or two levels up. - // ELECTRON_RUN_AS_NODE bypasses asar integration, so packaged builds must - // run the copy under app.asar.unpacked (out/main/codex/** is asarUnpacked). - const toUnpackedDir = (dir: string): string => - dir.replace(/([\\/])app\.asar(?=([\\/]|$))/, '$1app.asar.unpacked') - const baseDirs = [moduleDir, join(moduleDir, '..')].map(toUnpackedDir) - for (const baseDir of baseDirs) { - const candidate = join(baseDir, 'codex', GRANT_ENTRY_FILE_NAME) - if (pathExists(candidate)) { - return candidate - } - } - return null -} - -export type RunGrantSessionSyncOptions = { - entryPath?: string - nodeCommand?: string - /** Test-only override; production keeps enough margin for child cleanup. */ - timeoutMarginMs?: number -} - -/** - * Blocking wrapper for the grant session. Hook install/refresh is synchronous - * launch prep (pane launch must not proceed until trust is settled), and a - * stdio JSON-RPC session needs a live event loop — so the session runs in a - * short-lived ELECTRON_RUN_AS_NODE child (same pattern as the daemon and - * parcel-watcher entries) while the caller blocks on spawnSync. spawnSync - * always reaps the entry; a killed entry closes the codex child's stdin, - * which makes codex app-server exit on EOF. - */ -export function runCodexHookTrustGrantSessionSync( - request: CodexHookTrustGrantRequest, - options: RunGrantSessionSyncOptions = {} -): CodexHookTrustGrantSessionResult { - return runCodexAppServerEntrySync(request, options) as CodexHookTrustGrantSessionResult -} - -export function runCodexUserHookTrustRebaseSessionSync( - request: CodexUserHookTrustRebaseRequest, - options: RunGrantSessionSyncOptions = {} -): CodexUserHookTrustRebaseResult { - return runCodexAppServerEntrySync(request, options) as CodexUserHookTrustRebaseResult -} - -function runCodexAppServerEntrySync( - request: CodexAppServerEntryRequest, - options: RunGrantSessionSyncOptions -): CodexAppServerEntryResult { - const entryPath = options.entryPath ?? resolveCodexGrantEntryPath() - if (!entryPath) { - throw new Error('codex trust-grant entry bundle not found') - } - const spawned = spawnSync(options.nodeCommand ?? process.execPath, [entryPath], { - input: JSON.stringify(request), - encoding: 'utf8', - timeout: - request.invocation.timeoutMs + (options.timeoutMarginMs ?? GRANT_ENTRY_TIMEOUT_MARGIN_MS), - killSignal: 'SIGKILL', - maxBuffer: GRANT_ENTRY_MAX_BUFFER_BYTES, - windowsHide: true, - env: { ...process.env, ELECTRON_RUN_AS_NODE: '1' } - }) - if ((spawned.error as NodeJS.ErrnoException | undefined)?.code === 'ETIMEDOUT') { - // Why: spawnSync reports its own deadline through error.code before the - // signal field; preserve the typed timeout so cooldown diagnostics work. - throw new CodexAppServerTimeoutError( - `codex trust-grant entry exceeded ${request.invocation.timeoutMs}ms session deadline` - ) - } - if (spawned.error) { - throw spawned.error - } - if (spawned.signal) { - throw new CodexAppServerTimeoutError( - `codex trust-grant entry killed by ${spawned.signal} after ${request.invocation.timeoutMs}ms deadline` - ) - } - const lines = (spawned.stdout ?? '').split('\n').filter((line) => line.trim().length > 0) - const lastLine = lines.at(-1) - let envelope: GrantEntryEnvelope | null = null - if (lastLine) { - try { - envelope = JSON.parse(lastLine) as GrantEntryEnvelope - } catch { - envelope = null - } - } - if (!envelope) { - throw new Error( - `codex trust-grant entry produced no result (exit ${spawned.status ?? 'unknown'})${ - spawned.stderr ? `: ${spawned.stderr.trim().slice(0, 400)}` : '' - }` - ) - } - if (!envelope.ok) { - if (envelope.unsupported) { - throw new CodexAppServerUnsupportedError(envelope.message) - } - if (envelope.errorName === 'CodexAppServerTimeoutError') { - throw new CodexAppServerTimeoutError(envelope.message) - } - throw new Error(envelope.message) - } - return envelope.result -} diff --git a/src/main/codex/codex-app-server-grant-entry.ts b/src/main/codex/codex-app-server-grant-entry.ts deleted file mode 100644 index 8e16892d279..00000000000 --- a/src/main/codex/codex-app-server-grant-entry.ts +++ /dev/null @@ -1,79 +0,0 @@ -// Forked (ELECTRON_RUN_AS_NODE) child that runs one codex app-server -// trust-grant session. The parent blocks on spawnSync because hook -// install/refresh must finish before a Codex pane launch proceeds, while the -// JSONL RPC session itself needs a live event loop. Reads the request JSON -// from stdin, writes a single result-envelope JSON line to stdout, and never -// imports electron (see PLAIN_NODE_ENTRY_NAMES in the build guard). -import { - buildGrantEntryEnvelope, - type CodexAppServerEntryRequest -} from './codex-app-server-grant-envelope' -import { writeSync } from 'node:fs' -import { runCodexHookTrustGrantSession } from './codex-app-server-client' -import { runCodexUserHookTrustRebaseSession } from './codex-user-hook-trust-rebase-client' - -const HARD_EXIT_MARGIN_MS = 2_000 - -async function readStdin(): Promise { - const chunks: Buffer[] = [] - for await (const chunk of process.stdin) { - chunks.push(chunk as Buffer) - } - return Buffer.concat(chunks).toString('utf8') -} - -async function main(): Promise { - const raw = await readStdin() - let request: CodexAppServerEntryRequest - try { - request = JSON.parse(raw) as CodexAppServerEntryRequest - } catch (error) { - process.stdout.write( - `${JSON.stringify({ - ok: false, - errorName: 'Error', - message: `invalid trust-grant request JSON: ${error instanceof Error ? error.message : String(error)}` - })}\n` - ) - return - } - // Why: backstop for a session whose own deadline failed to fire (clock - // suspend mid-session); exiting closes the codex child's stdio so it - // exits on EOF instead of orphaning. - const hardExit = setTimeout(() => { - // Why: process.exit() does not flush asynchronous stdout pipes; write the - // timeout envelope synchronously so the parent can classify the fallback. - writeSync( - process.stdout.fd, - `${JSON.stringify({ - ok: false, - errorName: 'CodexAppServerTimeoutError', - message: `trust-grant entry hard deadline (${request.invocation.timeoutMs + HARD_EXIT_MARGIN_MS}ms) elapsed` - })}\n` - ) - process.exit(3) - }, request.invocation.timeoutMs + HARD_EXIT_MARGIN_MS) - const run = - 'operation' in request - ? runCodexUserHookTrustRebaseSession(request) - : runCodexHookTrustGrantSession(request) - const envelope = await buildGrantEntryEnvelope(run) - clearTimeout(hardExit) - process.stdout.write(`${JSON.stringify(envelope)}\n`) -} - -void main().then( - () => { - process.exitCode = 0 - }, - (error: unknown) => { - process.stdout.write( - `${JSON.stringify({ - ok: false, - errorName: error instanceof Error ? error.name : 'Error', - message: error instanceof Error ? error.message : String(error) - })}\n` - ) - process.exitCode = 0 - } -) diff --git a/src/main/codex/codex-app-server-grant-envelope.ts b/src/main/codex/codex-app-server-grant-envelope.ts deleted file mode 100644 index d5b133c2975..00000000000 --- a/src/main/codex/codex-app-server-grant-envelope.ts +++ /dev/null @@ -1,35 +0,0 @@ -import { - isCodexAppServerUnsupportedError, - type CodexHookTrustGrantRequest, - type CodexHookTrustGrantSessionResult -} from './codex-app-server-client' -import type { - CodexUserHookTrustRebaseRequest, - CodexUserHookTrustRebaseResult -} from './codex-user-hook-trust-rebase-client' - -export type CodexAppServerEntryRequest = - | CodexHookTrustGrantRequest - | CodexUserHookTrustRebaseRequest - -export type CodexAppServerEntryResult = - | CodexHookTrustGrantSessionResult - | CodexUserHookTrustRebaseResult - -export type GrantEntryEnvelope = - | { ok: true; result: CodexAppServerEntryResult } - | { ok: false; errorName: string; message: string; unsupported?: boolean } - -export function buildGrantEntryEnvelope( - run: Promise -): Promise { - return run.then( - (result) => ({ ok: true as const, result }), - (error: unknown) => ({ - ok: false as const, - errorName: error instanceof Error ? error.name : 'Error', - message: error instanceof Error ? error.message : String(error), - ...(isCodexAppServerUnsupportedError(error) ? { unsupported: true as const } : {}) - }) - ) -} diff --git a/src/main/codex/codex-hook-trust-grant.test.ts b/src/main/codex/codex-hook-trust-grant.test.ts index e1ca7c41076..5292be120c1 100644 --- a/src/main/codex/codex-hook-trust-grant.test.ts +++ b/src/main/codex/codex-hook-trust-grant.test.ts @@ -45,7 +45,7 @@ beforeEach(() => { afterEach(() => { vi.useRealTimers() - _internals.setGrantSessionRunnerSync(null) + _internals.setGrantSessionRunner(null) setCodexTrustGrantTelemetry(() => {}) codexAppServerCapabilityCache.clear() if (previousUserDataPath === undefined) { @@ -97,28 +97,30 @@ function grantedSessionResult(entries: CodexTrustEntry[], hashPrefix = 'sha256:c } describe('grantManagedCodexHookTrust', () => { - it('does not let a short trust RPC claim an incomplete session index', () => { + it('does not let a short trust RPC claim an incomplete session index', async () => { const sessions = join(runtimeHomeDir, 'sessions') mkdirSync(sessions, { recursive: true }) for (let index = 0; index < 100; index += 1) { writeFileSync(join(sessions, `${index}.jsonl`), '{}\n') } const runner = vi.fn() - _internals.setGrantSessionRunnerSync(runner) + _internals.setGrantSessionRunner(runner) - expect(grantManagedCodexHookTrust(buildPlan([managedEntry('stop')]))).toMatchObject({ + expect(await grantManagedCodexHookTrust(buildPlan([managedEntry('stop')]))).toMatchObject({ lane: 'fallback', reason: 'retry-cached' }) expect(runner).not.toHaveBeenCalled() }) - it('returns granted entries with codex-verbatim hashes and records the ledger', () => { + it('returns granted entries with codex-verbatim hashes and records the ledger', async () => { const entries = [managedEntry('session_start'), managedEntry('stop')] - const runner = vi.fn((_request: CodexHookTrustGrantRequest) => grantedSessionResult(entries)) - _internals.setGrantSessionRunnerSync(runner) + const runner = vi.fn(async (_request: CodexHookTrustGrantRequest) => + grantedSessionResult(entries) + ) + _internals.setGrantSessionRunner(runner) - const outcome = grantManagedCodexHookTrust(buildPlan(entries)) + const outcome = await grantManagedCodexHookTrust(buildPlan(entries)) expect(outcome.lane).toBe('rpc') if (outcome.lane !== 'rpc') { return @@ -139,20 +141,22 @@ describe('grantManagedCodexHookTrust', () => { expect(getCodexTrustGrantDiagnostics()).toMatchObject({ granted: 1, fellBack: 0 }) }) - it('builds a default-home grant invocation without an inherited CODEX_HOME', () => { + it('builds a default-home grant invocation without an inherited CODEX_HOME', async () => { const entries = [managedEntry('stop')] - const runner = vi.fn((_request: CodexHookTrustGrantRequest) => grantedSessionResult(entries)) - _internals.setGrantSessionRunnerSync(runner) + const runner = vi.fn(async (_request: CodexHookTrustGrantRequest) => + grantedSessionResult(entries) + ) + _internals.setGrantSessionRunner(runner) expect( - grantManagedCodexHookTrust({ ...buildPlan(entries), useDefaultCodexHome: true }) + await grantManagedCodexHookTrust({ ...buildPlan(entries), useDefaultCodexHome: true }) ).toMatchObject({ lane: 'rpc' }) const invocation = runner.mock.calls[0]![0]!.invocation expect(invocation.env?.CODEX_HOME).toBeUndefined() expect(invocation.envToDelete).toContain('CODEX_HOME') }) - it('removes equivalent Windows fallback keys before the RPC writes canonical trust', () => { + it('removes equivalent Windows fallback keys before the RPC writes canonical trust', async () => { const entry: CodexTrustEntry = { ...managedEntry('stop'), sourcePath: String.raw`C:\Users\Alice\.codex\hooks.json` @@ -162,23 +166,23 @@ describe('grantManagedCodexHookTrust', () => { expect(readHookTrustEntries(plan.tomlPath).get(computeTrustKey(entry))?.trustedHash).toBe( computeTrustedHash(entry) ) - const runner = vi.fn(() => { + const runner = vi.fn(async () => { expect(readHookTrustEntries(plan.tomlPath).has(computeTrustKey(entry))).toBe(false) return grantedSessionResult([entry]) }) - _internals.setGrantSessionRunnerSync(runner) + _internals.setGrantSessionRunner(runner) - expect(grantManagedCodexHookTrust(plan)).toMatchObject({ lane: 'rpc' }) + expect(await grantManagedCodexHookTrust(plan)).toMatchObject({ lane: 'rpc' }) expect(runner).toHaveBeenCalledTimes(1) }) - it('skips the RPC session while the ledger grant still holds, and re-grants on config drift', () => { + it('skips the RPC session while the ledger grant still holds, and re-grants on config drift', async () => { const entries = [managedEntry('session_start')] - const runner = vi.fn(() => grantedSessionResult(entries)) - _internals.setGrantSessionRunnerSync(runner) + const runner = vi.fn(async () => grantedSessionResult(entries)) + _internals.setGrantSessionRunner(runner) const plan = buildPlan(entries) - const first = grantManagedCodexHookTrust(plan) + const first = await grantManagedCodexHookTrust(plan) expect(first.lane).toBe('rpc') expect(runner).toHaveBeenCalledTimes(1) @@ -187,92 +191,95 @@ describe('grantManagedCodexHookTrust', () => { upsertHookTrustEntries(plan.tomlPath, [ { ...entries[0], trustedHash: 'sha256:codex-session_start' } ]) - const second = grantManagedCodexHookTrust(plan) + const second = await grantManagedCodexHookTrust(plan) expect(second.lane).toBe('rpc') expect(runner).toHaveBeenCalledTimes(1) expect(getCodexTrustGrantDiagnostics()).toMatchObject({ granted: 1, ledgerHits: 1 }) // Config drift (user wiped the trust entry) must re-run the session. upsertHookTrustEntries(plan.tomlPath, [{ ...entries[0], trustedHash: 'sha256:wiped' }]) - const third = grantManagedCodexHookTrust(plan) + const third = await grantManagedCodexHookTrust(plan) expect(third.lane).toBe('rpc') expect(runner).toHaveBeenCalledTimes(2) }) - it('re-grants when the managed hook identity changes', () => { + it('re-grants when the managed hook identity changes', async () => { const entries = [managedEntry('session_start')] - const runner = vi.fn(() => grantedSessionResult(entries)) - _internals.setGrantSessionRunnerSync(runner) + const runner = vi.fn(async () => grantedSessionResult(entries)) + _internals.setGrantSessionRunner(runner) const plan = buildPlan(entries) - grantManagedCodexHookTrust(plan) + await grantManagedCodexHookTrust(plan) upsertHookTrustEntries(plan.tomlPath, [ { ...entries[0], trustedHash: 'sha256:codex-session_start' } ]) const changedEntries = [{ ...entries[0], timeoutSec: 99 }] - const changedRunner = vi.fn(() => grantedSessionResult(changedEntries)) - _internals.setGrantSessionRunnerSync(changedRunner) - const outcome = grantManagedCodexHookTrust(buildPlan(changedEntries)) + const changedRunner = vi.fn(async () => grantedSessionResult(changedEntries)) + _internals.setGrantSessionRunner(changedRunner) + const outcome = await grantManagedCodexHookTrust(buildPlan(changedEntries)) expect(outcome.lane).toBe('rpc') expect(changedRunner).toHaveBeenCalledTimes(1) }) - it('marks the host unsupported only for the unsupported error class', () => { + it('marks the host unsupported only for the unsupported error class', async () => { const entries = [managedEntry('session_start')] - const runner = vi.fn((): CodexHookTrustGrantSessionResult => { + const runner = vi.fn((): Promise => { throw new CodexAppServerUnsupportedError('no such method') }) - _internals.setGrantSessionRunnerSync(runner) + _internals.setGrantSessionRunner(runner) const plan = buildPlan(entries) - expect(grantManagedCodexHookTrust(plan)).toMatchObject({ + expect(await grantManagedCodexHookTrust(plan)).toMatchObject({ lane: 'fallback', reason: 'unsupported' }) expect(runner).toHaveBeenCalledTimes(1) // Cached: the second install skips the probe entirely. - expect(grantManagedCodexHookTrust(plan)).toMatchObject({ + expect(await grantManagedCodexHookTrust(plan)).toMatchObject({ lane: 'fallback', reason: 'unsupported-cached' }) expect(runner).toHaveBeenCalledTimes(1) }) - it('backs off transient failures without poisoning the capability', () => { + it('backs off transient failures without poisoning the capability', async () => { vi.useFakeTimers() vi.setSystemTime(1_000) const entries = [managedEntry('session_start')] - const runner = vi.fn((): CodexHookTrustGrantSessionResult => { + const runner = vi.fn((): Promise => { throw new Error('spawn ETIMEDOUT') }) - _internals.setGrantSessionRunnerSync(runner) + _internals.setGrantSessionRunner(runner) const plan = buildPlan(entries) - expect(grantManagedCodexHookTrust(plan)).toMatchObject({ lane: 'fallback', reason: 'error' }) - expect(grantManagedCodexHookTrust(plan)).toMatchObject({ + expect(await grantManagedCodexHookTrust(plan)).toMatchObject({ + lane: 'fallback', + reason: 'error' + }) + expect(await grantManagedCodexHookTrust(plan)).toMatchObject({ lane: 'fallback', reason: 'retry-cached' }) expect(runner).toHaveBeenCalledTimes(1) expect(codexAppServerCapabilityCache.shouldTry('native')).toBe(true) - runner.mockImplementation(() => grantedSessionResult(entries)) + runner.mockImplementation(async () => grantedSessionResult(entries)) vi.setSystemTime(1_000 + CODEX_TRUST_GRANT_TRANSIENT_RETRY_INTERVAL_MS) - expect(grantManagedCodexHookTrust(plan)).toMatchObject({ lane: 'rpc' }) + expect(await grantManagedCodexHookTrust(plan)).toMatchObject({ lane: 'rpc' }) expect(runner).toHaveBeenCalledTimes(2) }) - it('falls back on verify-failed without marking unsupported', () => { + it('falls back on verify-failed without marking unsupported', async () => { const entries = [managedEntry('session_start')] - const runner = vi.fn(() => ({ + const runner = vi.fn(async () => ({ outcome: 'verify-failed' as const, reason: 'missing entries', reasonClass: 'list-mismatch' as const })) - _internals.setGrantSessionRunnerSync(runner) + _internals.setGrantSessionRunner(runner) - expect(grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ + expect(await grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ lane: 'fallback', reason: 'verify-failed' }) @@ -280,53 +287,56 @@ describe('grantManagedCodexHookTrust', () => { expect(getCodexTrustGrantDiagnostics()).toMatchObject({ verifyFailed: 1 }) }) - it('rejects duplicate granted keys instead of treating another key as covered', () => { + it('rejects duplicate granted keys instead of treating another key as covered', async () => { const entries = [managedEntry('session_start'), managedEntry('stop')] const duplicated = grantedSessionResult([entries[0]!, entries[0]!]) - _internals.setGrantSessionRunnerSync(() => duplicated) + _internals.setGrantSessionRunner(async () => duplicated) - expect(grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ + expect(await grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ lane: 'fallback', reason: 'verify-failed' }) expect(readCodexTrustGrantLedgerHome(runtimeHomeDir)).toBeNull() }) - it('keeps grant and fallback outcomes stable when telemetry throws', () => { + it('keeps grant and fallback outcomes stable when telemetry throws', async () => { const entries = [managedEntry('session_start')] setCodexTrustGrantTelemetry(() => { throw new Error('telemetry unavailable') }) - _internals.setGrantSessionRunnerSync(() => grantedSessionResult(entries)) + _internals.setGrantSessionRunner(async () => grantedSessionResult(entries)) - expect(grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ lane: 'rpc' }) + expect(await grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ lane: 'rpc' }) process.env.ORCA_DISABLE_CODEX_TRUST_RPC = '1' - expect(grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ + expect(await grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ lane: 'fallback', reason: 'disabled' }) }) - it('restores exact config bytes before fallback after a mutating RPC error', () => { + it('restores exact config bytes before fallback after a mutating RPC error', async () => { const entries = [managedEntry('session_start')] const plan = buildPlan(entries) const original = '# user formatting\r\n[hooks]\r\n' mkdirSync(runtimeHomeDir, { recursive: true }) writeFileSync(plan.tomlPath, original) - _internals.setGrantSessionRunnerSync(() => { + _internals.setGrantSessionRunner(async () => { writeFileSync(plan.tomlPath, '[hooks.state."rpc-partial"]\ntrusted_hash = "changed"\n') throw new Error('post-write transport failure') }) - expect(grantManagedCodexHookTrust(plan)).toMatchObject({ lane: 'fallback', reason: 'error' }) + expect(await grantManagedCodexHookTrust(plan)).toMatchObject({ + lane: 'fallback', + reason: 'error' + }) expect(readFileSync(plan.tomlPath, 'utf8')).toBe(original) }) - it('removes an RPC-created config before fallback when none existed', () => { + it('removes an RPC-created config before fallback when none existed', async () => { const entries = [managedEntry('session_start')] const plan = buildPlan(entries) mkdirSync(runtimeHomeDir, { recursive: true }) - _internals.setGrantSessionRunnerSync(() => { + _internals.setGrantSessionRunner(async () => { writeFileSync(plan.tomlPath, '[hooks.state."rpc-partial"]\ntrusted_hash = "changed"\n') return { outcome: 'verify-failed', @@ -335,32 +345,116 @@ describe('grantManagedCodexHookTrust', () => { } }) - expect(grantManagedCodexHookTrust(plan)).toMatchObject({ + expect(await grantManagedCodexHookTrust(plan)).toMatchObject({ lane: 'fallback', reason: 'verify-failed' }) expect(existsSync(plan.tomlPath)).toBe(false) }) - it('honors the ops kill switch env flag', () => { + it('honors the ops kill switch env flag', async () => { process.env.ORCA_DISABLE_CODEX_TRUST_RPC = '1' const entries = [managedEntry('session_start')] - const runner = vi.fn(() => grantedSessionResult(entries)) - _internals.setGrantSessionRunnerSync(runner) + const runner = vi.fn(async () => grantedSessionResult(entries)) + _internals.setGrantSessionRunner(runner) - expect(grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ + expect(await grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ lane: 'fallback', reason: 'disabled' }) expect(runner).not.toHaveBeenCalled() }) - it('builds a WSL invocation that runs codex inside the distro', () => { + // Why (#16441): the grant used to run through spawnSync, so two grants on one + // config.toml were impossible by construction. Now they must queue — an + // interleaved capture/restore pair resurrects trust the other run removed. + it('serializes concurrent grants that share one config.toml', async () => { const entries = [managedEntry('session_start')] - const runner = vi.fn((_request: CodexHookTrustGrantRequest) => grantedSessionResult(entries)) - _internals.setGrantSessionRunnerSync(runner) + const plan = buildPlan(entries) + let inFlight = 0 + let maxInFlight = 0 + const releases: (() => void)[] = [] + _internals.setGrantSessionRunner(async () => { + inFlight += 1 + maxInFlight = Math.max(maxInFlight, inFlight) + await new Promise((resolve) => releases.push(resolve)) + inFlight -= 1 + return grantedSessionResult(entries) + }) - const outcome = grantManagedCodexHookTrust({ + const first = grantManagedCodexHookTrust(plan) + const second = grantManagedCodexHookTrust(plan) + await vi.waitFor(() => expect(releases).toHaveLength(1)) + releases[0]!() + await first + await vi.waitFor(() => expect(releases).toHaveLength(2)) + releases[1]!() + await second + + expect(maxInFlight).toBe(1) + }) + + it('lets grants on different config.toml paths overlap', async () => { + const entries = [managedEntry('session_start')] + const otherHome = join(userDataDir, 'codex-accounts', 'other', 'home') + mkdirSync(otherHome, { recursive: true }) + // Why: the probe dedupe only holds the first session on an unproven host. + // A known-supported host must keep its intended launch concurrency. + codexAppServerCapabilityCache.rememberSupported('native') + let inFlight = 0 + let maxInFlight = 0 + const releases: (() => void)[] = [] + _internals.setGrantSessionRunner(async () => { + inFlight += 1 + maxInFlight = Math.max(maxInFlight, inFlight) + await new Promise((resolve) => releases.push(resolve)) + inFlight -= 1 + return grantedSessionResult(entries) + }) + + const first = grantManagedCodexHookTrust(buildPlan(entries)) + const second = grantManagedCodexHookTrust({ + ...buildPlan(entries), + runtimeHomePath: otherHome, + tomlPath: join(otherHome, 'config.toml') + }) + await vi.waitFor(() => expect(releases).toHaveLength(2)) + releases.forEach((release) => release()) + await Promise.all([first, second]) + + expect(maxInFlight).toBe(2) + }) + + it('dedupes the capability probe when concurrent grants hit an unsupported host', async () => { + const entries = [managedEntry('session_start')] + const otherHome = join(userDataDir, 'codex-accounts', 'other', 'home') + mkdirSync(otherHome, { recursive: true }) + const releases: ((error: unknown) => void)[] = [] + const runner = vi.fn(() => new Promise((_resolve, reject) => releases.push(reject))) + _internals.setGrantSessionRunner(runner) + + const first = grantManagedCodexHookTrust(buildPlan(entries)) + const second = grantManagedCodexHookTrust({ + ...buildPlan(entries), + runtimeHomePath: otherHome, + tomlPath: join(otherHome, 'config.toml') + }) + await vi.waitFor(() => expect(releases).toHaveLength(1)) + releases[0]!(new CodexAppServerUnsupportedError('no such method')) + + expect(await first).toMatchObject({ lane: 'fallback', reason: 'unsupported' }) + expect(await second).toMatchObject({ lane: 'fallback', reason: 'unsupported-cached' }) + expect(runner).toHaveBeenCalledTimes(1) + }) + + it('builds a WSL invocation that runs codex inside the distro', async () => { + const entries = [managedEntry('session_start')] + const runner = vi.fn(async (_request: CodexHookTrustGrantRequest) => + grantedSessionResult(entries) + ) + _internals.setGrantSessionRunner(runner) + + const outcome = await grantManagedCodexHookTrust({ ...buildPlan(entries), host: { kind: 'wsl', distro: 'Ubuntu', linuxRuntimeHome: '/home/alice/.codex-runtime' } }) @@ -384,32 +478,32 @@ describe('trust-grant telemetry detail', () => { return events } - it('attributes the plan lane on granted events', () => { + it('attributes the plan lane on granted events', async () => { const events = captureTelemetry() const entries = [managedEntry('session_start')] - _internals.setGrantSessionRunnerSync(() => grantedSessionResult(entries)) + _internals.setGrantSessionRunner(async () => grantedSessionResult(entries)) - expect(grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ lane: 'rpc' }) + expect(await grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ lane: 'rpc' }) expect(events).toEqual([{ outcome: 'granted', hostKind: 'native', lane: 'real-home' }]) }) - it('reports the managed lane independently of host kind', () => { + it('reports the managed lane independently of host kind', async () => { const events = captureTelemetry() const entries = [managedEntry('session_start')] - _internals.setGrantSessionRunnerSync(() => grantedSessionResult(entries)) + _internals.setGrantSessionRunner(async () => grantedSessionResult(entries)) - grantManagedCodexHookTrust({ ...buildPlan(entries), telemetryLane: 'managed' }) + await grantManagedCodexHookTrust({ ...buildPlan(entries), telemetryLane: 'managed' }) expect(events).toEqual([{ outcome: 'granted', hostKind: 'native', lane: 'managed' }]) }) - it('classifies error fallbacks on the wire', () => { + it('classifies error fallbacks on the wire', async () => { const events = captureTelemetry() const entries = [managedEntry('session_start')] - _internals.setGrantSessionRunnerSync(() => { + _internals.setGrantSessionRunner(async () => { throw new Error('spawn codex ENOENT') }) - expect(grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ + expect(await grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ lane: 'fallback', reason: 'error' }) @@ -424,16 +518,16 @@ describe('trust-grant telemetry detail', () => { ]) }) - it('carries the session verify class through the fallback event', () => { + it('carries the session verify class through the fallback event', async () => { const events = captureTelemetry() const entries = [managedEntry('session_start')] - _internals.setGrantSessionRunnerSync(() => ({ + _internals.setGrantSessionRunner(async () => ({ outcome: 'verify-failed' as const, reason: 'post-grant verify left 1 entries untrusted', reasonClass: 'post-grant-untrusted' as const })) - grantManagedCodexHookTrust(buildPlan(entries)) + await grantManagedCodexHookTrust(buildPlan(entries)) expect(events).toEqual([ { outcome: 'verify_failed', @@ -445,12 +539,12 @@ describe('trust-grant telemetry detail', () => { ]) }) - it('classifies module-detected verify failures', () => { + it('classifies module-detected verify failures', async () => { const events = captureTelemetry() const entries = [managedEntry('session_start'), managedEntry('stop')] - _internals.setGrantSessionRunnerSync(() => grantedSessionResult([entries[0]!, entries[0]!])) + _internals.setGrantSessionRunner(async () => grantedSessionResult([entries[0]!, entries[0]!])) - grantManagedCodexHookTrust(buildPlan(entries)) + await grantManagedCodexHookTrust(buildPlan(entries)) expect(events).toEqual([ { outcome: 'verify_failed', diff --git a/src/main/codex/codex-hook-trust-grant.ts b/src/main/codex/codex-hook-trust-grant.ts index 9142a590ab3..f886dbb6344 100644 --- a/src/main/codex/codex-hook-trust-grant.ts +++ b/src/main/codex/codex-hook-trust-grant.ts @@ -1,5 +1,6 @@ import { isCodexAppServerUnsupportedError, + runCodexHookTrustGrantSession, type CodexHookTrustGrantRequest, type CodexHookTrustGrantSessionResult } from './codex-app-server-client' @@ -10,55 +11,50 @@ import { type CodexTrustGrantTelemetryLane, type CodexTrustGrantVerifyClass } from './codex-trust-grant-telemetry' -import { runCodexHookTrustGrantSessionSync } from './codex-app-server-grant-bridge' import { codexAppServerCapabilityCache, - getCodexAppServerHostKey + getCodexAppServerHostKey, + type CodexAppServerHostKey } from './codex-app-server-capability-cache' import { writeCodexTrustGrantLedgerHome, type CodexTrustGrantBinaryStamp, type CodexTrustGrantLedgerEntry } from './codex-trust-grant-ledger' -import { - computeTrustKey, - computeTrustedHash, - normalizeHookTrustKeyForLookup, - readHookTrustEntries, - removeHookTrustEntries, - type CodexTrustEntry -} from './config-toml-trust' -import { getCodexHookTrustSignature } from './codex-hook-identity' +import type { CodexTrustEntry } from './config-toml-trust' import { captureCodexTrustConfig, restoreCodexTrustConfig } from './codex-trust-config-rollback' +import { runExclusivelyForCodexTrustConfig } from './codex-trust-config-mutation-queue' import { - readCodexTrustGrantLedgerHomeMatchingStamp, resolveCodexTrustGrantHost, - type CodexTrustGrantHost + type ResolvedCodexTrustGrantHost } from './codex-trust-grant-host' +import { + buildExpectedEntries, + findLedgerGrant, + removeSelfComputedTrustBeforeGrant, + type CodexManagedTrustGrantPlan, + type ExpectedManagedEntry +} from './codex-managed-trust-grant-plan' import { isCodexStateDbBackfillPending } from './codex-state-db' // Why: a transiently hung app-server must not block launch prep on every pane. // The legacy lane remains available while a short, host-scoped cooldown runs. export const CODEX_TRUST_GRANT_TRANSIENT_RETRY_INTERVAL_MS = 5 * 60_000 -/** Ops escape hatch (not a setting): forces the unchanged fallback lane. */ +/** + * Ops escape hatch (not a setting): forces the unchanged fallback lane for the + * *managed* grant only. + * + * Scope, because the name reads broader than it is: the real-home rebase + * (`mutateRealHomeHooksPreservingUserTrust`) still runs its own inspect/repair + * app-server sessions when Orca's insertion shifts a user's hook positions, and + * does not read this flag. That is unchanged from before the grant went async — + * those sessions simply used to block the main thread instead. Widening the flag + * to cover the rebase is a follow-up, not something this constant already does. + */ const DISABLE_ENV_FLAG = 'ORCA_DISABLE_CODEX_TRUST_RPC' -export type CodexManagedTrustGrantPlan = { - /** Host-visible runtime home path (UNC for WSL) — ledger key + config reads. */ - runtimeHomePath: string - /** Host-visible config.toml path holding the trust entries. */ - tomlPath: string - /** Exact command string written to the managed hooks.json entries. */ - managedCommand: string - /** Managed trust identities Orca just wrote (no trustedHash). */ - managedEntries: readonly CodexTrustEntry[] - host: CodexTrustGrantHost - telemetryLane: CodexTrustGrantTelemetryLane - /** Match a pane where CODEX_HOME is absent instead of an explicit managed home. */ - useDefaultCodexHome?: boolean -} - +export type { CodexManagedTrustGrantPlan } export type { CodexTrustGrantFallbackReason, CodexTrustGrantTelemetryLane } export type CodexManagedTrustGrantOutcome = @@ -77,11 +73,14 @@ const transientRetryAfterByHost = new Map() export const getCodexTrustGrantDiagnostics = (): CodexTrustGrantDiagnostics => ({ ...diagnostics }) -type GrantSessionRunnerSync = ( +type GrantSessionRunner = ( request: CodexHookTrustGrantRequest -) => CodexHookTrustGrantSessionResult +) => Promise -let runSessionSync: GrantSessionRunnerSync = runCodexHookTrustGrantSessionSync +// Why (#16441): the session runs in-process on the main thread's event loop. +// It used to be forked through spawnSync purely to donate an event loop to a +// deliberately-blocked parent, which froze the window for the whole deadline. +let runSession: GrantSessionRunner = runCodexHookTrustGrantSession function fallback( plan: CodexManagedTrustGrantPlan, @@ -109,60 +108,141 @@ function fallback( return { lane: 'fallback', reason } } -type ExpectedManagedEntry = { - entry: CodexTrustEntry - normalizedKey: string - signature: string +function startTransientCooldown(hostKey: CodexAppServerHostKey): void { + transientRetryAfterByHost.set(hostKey, Date.now() + CODEX_TRUST_GRANT_TRANSIENT_RETRY_INTERVAL_MS) } -function buildExpectedEntries(plan: CodexManagedTrustGrantPlan): ExpectedManagedEntry[] { - return plan.managedEntries.map((entry) => ({ - entry, - normalizedKey: normalizeHookTrustKeyForLookup(computeTrustKey(entry)), - signature: getCodexHookTrustSignature(entry) - })) +type GrantAttempt = { + plan: CodexManagedTrustGrantPlan + expected: ExpectedManagedEntry[] + hostKey: CodexAppServerHostKey + currentStamp: CodexTrustGrantBinaryStamp | null + configSnapshot: ReturnType + startedAtMs: number } -function removeSelfComputedTrustBeforeGrant(plan: CodexManagedTrustGrantPlan): void { - const trustStates = readHookTrustEntries(plan.tomlPath) - const ownedKeys = plan.managedEntries - .map((entry) => { - const key = computeTrustKey(entry) - return trustStates.get(key)?.trustedHash === computeTrustedHash(entry) ? key : null - }) - .filter((key): key is string => key !== null) - if (ownedKeys.length > 0) { - removeHookTrustEntries(plan.tomlPath, ownedKeys) +/** Post-session verification, ledger persistence and telemetry. Never throws for + * a verify failure — every rejection is a rolled-back fallback. */ +function completeGrant( + attempt: GrantAttempt, + result: CodexHookTrustGrantSessionResult +): CodexManagedTrustGrantOutcome { + const { plan, expected, hostKey, configSnapshot } = attempt + const rejectGrant = ( + detail: unknown, + verifyClass: CodexTrustGrantVerifyClass + ): CodexManagedTrustGrantOutcome => { + restoreCodexTrustConfig(plan.tomlPath, configSnapshot) + startTransientCooldown(hostKey) + return fallback(plan, 'verify-failed', detail, verifyClass) } + if (result.outcome === 'verify-failed') { + return rejectGrant(result.reason, result.reasonClass) + } + + const byNormalizedKey = new Map(expected.map((item) => [item.normalizedKey, item])) + const seenNormalizedKeys = new Set() + const grantedEntries: CodexTrustEntry[] = [] + const ledgerRecord: Record = {} + for (const granted of result.entries) { + const match = byNormalizedKey.get(granted.normalizedKey) + if (!match) { + return rejectGrant(`unexpected granted key ${granted.key}`, 'unexpected-key') + } + if (seenNormalizedKeys.has(granted.normalizedKey)) { + return rejectGrant(`duplicate granted key ${granted.key}`, 'duplicate-key') + } + seenNormalizedKeys.add(granted.normalizedKey) + grantedEntries.push({ ...match.entry, trustedHash: granted.trustedHash }) + ledgerRecord[granted.normalizedKey] = { + signature: match.signature, + trustedHash: granted.trustedHash + } + } + if (seenNormalizedKeys.size !== expected.length) { + return rejectGrant('granted entry set did not cover expected entries', 'coverage') + } + transientRetryAfterByHost.delete(hostKey) + try { + writeCodexTrustGrantLedgerHome(plan.runtimeHomePath, { + binary: attempt.currentStamp, + entries: ledgerRecord + }) + } catch (error) { + // Why: a ledger write failure only costs an extra session next launch. + console.warn('[codex-trust-grant] failed to persist grant ledger', error) + } + diagnostics.granted += 1 + console.log( + `[codex-trust-grant] granted ${grantedEntries.length} managed hook entries via codex app-server ` + + `(host=${plan.host.kind}, wrote=${result.wroteTrust}, ${Date.now() - attempt.startedAtMs}ms)` + ) + emitCodexTrustGrantTelemetry({ + outcome: 'granted', + hostKind: plan.host.kind, + lane: plan.telemetryLane + }) + return { lane: 'rpc', entries: grantedEntries } } -function findLedgerGrant( +async function runGrantAttempt( plan: CodexManagedTrustGrantPlan, expected: ExpectedManagedEntry[], - currentStamp: CodexTrustGrantBinaryStamp | null -): CodexTrustEntry[] | null { - const home = readCodexTrustGrantLedgerHomeMatchingStamp(plan.runtimeHomePath, currentStamp) - if (!home) { - return null + resolvedHost: ResolvedCodexTrustGrantHost, + hostKey: CodexAppServerHostKey +): Promise { + // Why: the RPC may rewrite config.toml before a later RPC fails. Restore its + // exact pre-session bytes before the legacy lane runs so every fallback has + // the same input and output as the pre-RPC implementation. + const attempt: GrantAttempt = { + plan, + expected, + hostKey, + currentStamp: resolvedHost.binaryStamp, + configSnapshot: captureCodexTrustConfig(plan.tomlPath), + startedAtMs: Date.now() } - let trustStates: ReturnType + let unsupportedError: unknown try { - trustStates = readHookTrustEntries(plan.tomlPath) - } catch { - return null + return await codexAppServerCapabilityCache.runWithFallback( + hostKey, + async () => { + removeSelfComputedTrustBeforeGrant(plan) + return completeGrant( + attempt, + await runSession( + resolvedHost.buildRequest({ + runtimeHomePath: plan.runtimeHomePath, + managedCommand: plan.managedCommand, + expectedTrustKeys: expected.map(({ normalizedKey }) => normalizedKey), + useDefaultCodexHome: plan.useDefaultCodexHome + }) + ) + ) + }, + async () => { + if (unsupportedError === undefined) { + // Why: a concurrent launch's probe proved the surface missing while + // this one waited behind it; nothing was mutated, so nothing to undo. + return fallback(plan, 'unsupported-cached') + } + restoreCodexTrustConfig(plan.tomlPath, attempt.configSnapshot) + transientRetryAfterByHost.delete(hostKey) + return fallback(plan, 'unsupported', unsupportedError) + }, + (error) => { + if (!isCodexAppServerUnsupportedError(error)) { + return false + } + unsupportedError = error + return true + } + ) + } catch (error) { + restoreCodexTrustConfig(plan.tomlPath, attempt.configSnapshot) + startTransientCooldown(hostKey) + return fallback(plan, 'error', error) } - const entries: CodexTrustEntry[] = [] - for (const { entry, normalizedKey, signature } of expected) { - const recorded = home.entries[normalizedKey] - if (!recorded || recorded.signature !== signature) { - return null - } - if (trustStates.get(normalizedKey)?.trustedHash !== recorded.trustedHash) { - return null - } - entries.push({ ...entry, trustedHash: recorded.trustedHash }) - } - return entries } /** @@ -173,9 +253,9 @@ function findLedgerGrant( * throws: any unexpected failure is a fallback, because hook install is * best-effort launch prep. */ -export function grantManagedCodexHookTrust( +export async function grantManagedCodexHookTrust( plan: CodexManagedTrustGrantPlan -): CodexManagedTrustGrantOutcome { +): Promise { try { if (process.env[DISABLE_ENV_FLAG] === '1') { return fallback(plan, 'disabled') @@ -184,9 +264,8 @@ export function grantManagedCodexHookTrust( return fallback(plan, 'no-managed-entries') } const expected = buildExpectedEntries(plan) - const resolvedHost = resolveCodexTrustGrantHost(plan.host) - const currentStamp = resolvedHost.binaryStamp - const ledgerEntries = findLedgerGrant(plan, expected, currentStamp) + const resolvedHost = await resolveCodexTrustGrantHost(plan.host) + const ledgerEntries = findLedgerGrant(plan, expected, resolvedHost.binaryStamp) if (ledgerEntries !== null) { diagnostics.ledgerHits += 1 return { lane: 'rpc', entries: ledgerEntries } @@ -207,131 +286,17 @@ export function grantManagedCodexHookTrust( } transientRetryAfterByHost.delete(hostKey) } - - const startedAtMs = Date.now() - // Why: the RPC may rewrite config.toml before a later RPC fails. Restore - // its exact pre-session bytes before the legacy lane runs so every fallback - // has the same input and output as the pre-RPC implementation. - const configSnapshot = captureCodexTrustConfig(plan.tomlPath) - let result: CodexHookTrustGrantSessionResult - try { - // Why: Windows fallback writes equivalent separator variants that Codex's - // canonical RPC key may not overwrite, leaving conflicting logical trust. - removeSelfComputedTrustBeforeGrant(plan) - result = runSessionSync( - resolvedHost.buildRequest({ - runtimeHomePath: plan.runtimeHomePath, - managedCommand: plan.managedCommand, - expectedTrustKeys: expected.map(({ normalizedKey }) => normalizedKey), - useDefaultCodexHome: plan.useDefaultCodexHome - }) - ) - } catch (error) { - restoreCodexTrustConfig(plan.tomlPath, configSnapshot) - if (isCodexAppServerUnsupportedError(error)) { - transientRetryAfterByHost.delete(hostKey) - codexAppServerCapabilityCache.rememberUnsupported(hostKey) - return fallback(plan, 'unsupported', error) - } - transientRetryAfterByHost.set( - hostKey, - Date.now() + CODEX_TRUST_GRANT_TRANSIENT_RETRY_INTERVAL_MS - ) - return fallback(plan, 'error', error) - } - // Why: the RPC surface answered, even if our entries were not verifiable — - // remember support so a later drift event retries the preferred lane. - codexAppServerCapabilityCache.rememberSupported(hostKey) - if (result.outcome === 'verify-failed') { - restoreCodexTrustConfig(plan.tomlPath, configSnapshot) - transientRetryAfterByHost.set( - hostKey, - Date.now() + CODEX_TRUST_GRANT_TRANSIENT_RETRY_INTERVAL_MS - ) - return fallback(plan, 'verify-failed', result.reason, result.reasonClass) - } - - const byNormalizedKey = new Map(expected.map((item) => [item.normalizedKey, item])) - const seenNormalizedKeys = new Set() - const grantedEntries: CodexTrustEntry[] = [] - const ledgerRecord: Record = {} - for (const granted of result.entries) { - const match = byNormalizedKey.get(granted.normalizedKey) - if (!match) { - restoreCodexTrustConfig(plan.tomlPath, configSnapshot) - transientRetryAfterByHost.set( - hostKey, - Date.now() + CODEX_TRUST_GRANT_TRANSIENT_RETRY_INTERVAL_MS - ) - return fallback( - plan, - 'verify-failed', - `unexpected granted key ${granted.key}`, - 'unexpected-key' - ) - } - if (seenNormalizedKeys.has(granted.normalizedKey)) { - restoreCodexTrustConfig(plan.tomlPath, configSnapshot) - transientRetryAfterByHost.set( - hostKey, - Date.now() + CODEX_TRUST_GRANT_TRANSIENT_RETRY_INTERVAL_MS - ) - return fallback( - plan, - 'verify-failed', - `duplicate granted key ${granted.key}`, - 'duplicate-key' - ) - } - seenNormalizedKeys.add(granted.normalizedKey) - grantedEntries.push({ ...match.entry, trustedHash: granted.trustedHash }) - ledgerRecord[granted.normalizedKey] = { - signature: match.signature, - trustedHash: granted.trustedHash - } - } - if (seenNormalizedKeys.size !== expected.length) { - restoreCodexTrustConfig(plan.tomlPath, configSnapshot) - transientRetryAfterByHost.set( - hostKey, - Date.now() + CODEX_TRUST_GRANT_TRANSIENT_RETRY_INTERVAL_MS - ) - return fallback( - plan, - 'verify-failed', - 'granted entry set did not cover expected entries', - 'coverage' - ) - } - transientRetryAfterByHost.delete(hostKey) - try { - writeCodexTrustGrantLedgerHome(plan.runtimeHomePath, { - binary: currentStamp, - entries: ledgerRecord - }) - } catch (error) { - // Why: a ledger write failure only costs an extra session next launch. - console.warn('[codex-trust-grant] failed to persist grant ledger', error) - } - diagnostics.granted += 1 - console.log( - `[codex-trust-grant] granted ${grantedEntries.length} managed hook entries via codex app-server ` + - `(host=${plan.host.kind}, wrote=${result.wroteTrust}, ${Date.now() - startedAtMs}ms)` + return await runExclusivelyForCodexTrustConfig(plan.tomlPath, () => + runGrantAttempt(plan, expected, resolvedHost, hostKey) ) - emitCodexTrustGrantTelemetry({ - outcome: 'granted', - hostKind: plan.host.kind, - lane: plan.telemetryLane - }) - return { lane: 'rpc', entries: grantedEntries } } catch (error) { return fallback(plan, 'error', error) } } export const _internals = { - setGrantSessionRunnerSync(runner: GrantSessionRunnerSync | null): void { - runSessionSync = runner ?? runCodexHookTrustGrantSessionSync + setGrantSessionRunner(runner: GrantSessionRunner | null): void { + runSession = runner ?? runCodexHookTrustGrantSession }, resetDiagnostics(): void { diagnostics.granted = 0 diff --git a/src/main/codex/codex-managed-trust-grant-plan.ts b/src/main/codex/codex-managed-trust-grant-plan.ts new file mode 100644 index 00000000000..e816b0cbff5 --- /dev/null +++ b/src/main/codex/codex-managed-trust-grant-plan.ts @@ -0,0 +1,90 @@ +import type { CodexTrustGrantTelemetryLane } from './codex-trust-grant-telemetry' +import { + readCodexTrustGrantLedgerHomeMatchingStamp, + type CodexTrustGrantHost +} from './codex-trust-grant-host' +import type { CodexTrustGrantBinaryStamp } from './codex-trust-grant-ledger' +import { getCodexHookTrustSignature } from './codex-hook-identity' +import { + computeTrustKey, + computeTrustedHash, + normalizeHookTrustKeyForLookup, + readHookTrustEntries, + removeHookTrustEntries, + type CodexTrustEntry +} from './config-toml-trust' + +export type CodexManagedTrustGrantPlan = { + /** Host-visible runtime home path (UNC for WSL) — ledger key + config reads. */ + runtimeHomePath: string + /** Host-visible config.toml path holding the trust entries. */ + tomlPath: string + /** Exact command string written to the managed hooks.json entries. */ + managedCommand: string + /** Managed trust identities Orca just wrote (no trustedHash). */ + managedEntries: readonly CodexTrustEntry[] + host: CodexTrustGrantHost + telemetryLane: CodexTrustGrantTelemetryLane + /** Match a pane where CODEX_HOME is absent instead of an explicit managed home. */ + useDefaultCodexHome?: boolean +} + +export type ExpectedManagedEntry = { + entry: CodexTrustEntry + normalizedKey: string + signature: string +} + +export function buildExpectedEntries(plan: CodexManagedTrustGrantPlan): ExpectedManagedEntry[] { + return plan.managedEntries.map((entry) => ({ + entry, + normalizedKey: normalizeHookTrustKeyForLookup(computeTrustKey(entry)), + signature: getCodexHookTrustSignature(entry) + })) +} + +/** Windows fallback writes equivalent separator variants that Codex's canonical + * RPC key may not overwrite, leaving conflicting logical trust behind. */ +export function removeSelfComputedTrustBeforeGrant(plan: CodexManagedTrustGrantPlan): void { + const trustStates = readHookTrustEntries(plan.tomlPath) + const ownedKeys = plan.managedEntries + .map((entry) => { + const key = computeTrustKey(entry) + return trustStates.get(key)?.trustedHash === computeTrustedHash(entry) ? key : null + }) + .filter((key): key is string => key !== null) + if (ownedKeys.length > 0) { + removeHookTrustEntries(plan.tomlPath, ownedKeys) + } +} + +/** Entries a prior grant already recorded for this exact binary and config + * state, or null when the RPC session has to run again. */ +export function findLedgerGrant( + plan: CodexManagedTrustGrantPlan, + expected: ExpectedManagedEntry[], + currentStamp: CodexTrustGrantBinaryStamp | null +): CodexTrustEntry[] | null { + const home = readCodexTrustGrantLedgerHomeMatchingStamp(plan.runtimeHomePath, currentStamp) + if (!home) { + return null + } + let trustStates: ReturnType + try { + trustStates = readHookTrustEntries(plan.tomlPath) + } catch { + return null + } + const entries: CodexTrustEntry[] = [] + for (const { entry, normalizedKey, signature } of expected) { + const recorded = home.entries[normalizedKey] + if (!recorded || recorded.signature !== signature) { + return null + } + if (trustStates.get(normalizedKey)?.trustedHash !== recorded.trustedHash) { + return null + } + entries.push({ ...entry, trustedHash: recorded.trustedHash }) + } + return entries +} diff --git a/src/main/codex/codex-real-home-hook-install.test.ts b/src/main/codex/codex-real-home-hook-install.test.ts index b108a0b5250..ed0a184163c 100644 --- a/src/main/codex/codex-real-home-hook-install.test.ts +++ b/src/main/codex/codex-real-home-hook-install.test.ts @@ -87,7 +87,7 @@ beforeEach(() => { }) afterEach(() => { - rebaseInternals.setSessionRunnerSync(null) + rebaseInternals.setSessionRunner(null) rebaseInternals.resetRetryState() rmSync(fakeHomeDir, { recursive: true, force: true }) rmSync(userDataDir, { recursive: true, force: true }) @@ -100,10 +100,30 @@ afterEach(() => { }) describe('ensureRealHomeCodexHookState (install)', () => { - it('creates hooks.json with the Orca entry in every managed event for a fresh home', () => { + // Why (#16441): the ensure chain is process-wide; a rejection that escapes it + // would return the same rejected promise to every later pane launch, with no + // retry and no cooldown recovery. + it('recovers from a home-resolution failure instead of poisoning later ensures', async () => { + grantSucceeds() + homedirMock.mockImplementationOnce(() => { + throw new Error('home unavailable') + }) + + await expect( + ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + ).resolves.toBe('unavailable') + await expect( + ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir }) + ).resolves.toBe('removed') + }) + + it('creates hooks.json with the Orca entry in every managed event for a fresh home', async () => { grantSucceeds() - const lane = ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + const lane = await ensureRealHomeCodexHookState({ + hooksEnabled: true, + userDataPath: userDataDir + }) expect(lane).toBe('installed') const material = getCodexManagedHookInstallMaterial() @@ -121,7 +141,7 @@ describe('ensureRealHomeCodexHookState (install)', () => { expect(plan.managedEntries.every((entry) => entry.groupIndex === 0)).toBe(true) }) - it('keeps a symlinked default home logical in the keys sent to Codex', () => { + it('keeps a symlinked default home logical in the keys sent to Codex', async () => { grantSucceeds() const logicalHome = join(fakeHomeDir, '.codex') const targetHome = join(fakeHomeDir, 'dotfiles-codex') @@ -129,9 +149,9 @@ describe('ensureRealHomeCodexHookState (install)', () => { mkdirSync(targetHome) symlinkSync(targetHome, logicalHome, process.platform === 'win32' ? 'junction' : 'dir') - expect(ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir })).toBe( - 'installed' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + ).toBe('installed') const plan = grantMock.mock.calls[0]![0] as CodexManagedTrustGrantPlan expect( @@ -139,7 +159,7 @@ describe('ensureRealHomeCodexHookState (install)', () => { ).toBe(true) }) - it('keeps the managed lane for unknown top-level fields Codex cannot load', () => { + it('keeps the managed lane for unknown top-level fields Codex cannot load', async () => { grantSucceeds() const userConfig = { hooks: { @@ -151,7 +171,10 @@ describe('ensureRealHomeCodexHookState (install)', () => { const original = `${JSON.stringify(userConfig, null, 2)}\n` writeFileSync(getRealHooksJsonPath(), original, 'utf-8') - const lane = ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + const lane = await ensureRealHomeCodexHookState({ + hooksEnabled: true, + userDataPath: userDataDir + }) expect(lane).toBe('unavailable') expect(readFileSync(getRealHooksJsonPath(), 'utf-8')).toBe(original) @@ -161,7 +184,7 @@ describe('ensureRealHomeCodexHookState (install)', () => { ) }) - it('appends LAST and preserves user entries and trust positions', () => { + it('appends LAST and preserves user entries and trust positions', async () => { grantSucceeds() const userConfig = { hooks: { @@ -172,9 +195,9 @@ describe('ensureRealHomeCodexHookState (install)', () => { const original = `${JSON.stringify(userConfig, null, 2)}\n` writeFileSync(getRealHooksJsonPath(), original, 'utf-8') - expect(ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir })).toBe( - 'installed' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + ).toBe('installed') const config = readRealHooksJson() expect(config.hooks?.Stop).toHaveLength(2) @@ -190,7 +213,7 @@ describe('ensureRealHomeCodexHookState (install)', () => { // Why: ordinary Windows CI tokens cannot create file symlinks without Developer Mode. it.skipIf(process.platform === 'win32')( 'updates a symlinked hooks.json target without replacing the symlink', - () => { + async () => { grantSucceeds() const dotfilesDir = join(fakeHomeDir, 'dotfiles') const targetPath = join(dotfilesDir, 'hooks.json') @@ -202,78 +225,87 @@ describe('ensureRealHomeCodexHookState (install)', () => { ) symlinkSync(targetPath, getRealHooksJsonPath()) - expect(ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir })).toBe( - 'installed' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + ).toBe('installed') expect(lstatSync(getRealHooksJsonPath()).isSymbolicLink()).toBe(true) expect(JSON.parse(readFileSync(targetPath, 'utf-8')).hooks.Stop).toHaveLength(2) } ) - it('keeps the managed lane and original bytes when the pristine backup cannot be created', () => { + it('keeps the managed lane and original bytes when the pristine backup cannot be created', async () => { grantSucceeds() const original = `${JSON.stringify({ hooks: { Stop: [] } }, null, 2)}\n` writeFileSync(getRealHooksJsonPath(), original, 'utf-8') writeFileSync(join(userDataDir, 'codex-real-home-hooks'), 'blocks backup directory', 'utf-8') - expect(ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir })).toBe( - 'unavailable' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + ).toBe('unavailable') expect(readFileSync(getRealHooksJsonPath(), 'utf-8')).toBe(original) expect(grantMock).not.toHaveBeenCalled() }) - it.skipIf(process.platform === 'win32')('preserves restrictive hooks.json permissions', () => { - grantSucceeds() - writeFileSync(getRealHooksJsonPath(), '{ "hooks": {} }\n', 'utf-8') - chmodSync(getRealHooksJsonPath(), 0o600) - - expect(ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir })).toBe( - 'installed' - ) - - expect(statSync(getRealHooksJsonPath()).mode & 0o777).toBe(0o600) - }) - it.skipIf(process.platform === 'win32')( - 'restores restrictive hooks.json permissions after grant fallback', - () => { - grantUnavailable() + 'preserves restrictive hooks.json permissions', + async () => { + grantSucceeds() writeFileSync(getRealHooksJsonPath(), '{ "hooks": {} }\n', 'utf-8') chmodSync(getRealHooksJsonPath(), 0o600) - expect(ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir })).toBe( - 'unavailable' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + ).toBe('installed') expect(statSync(getRealHooksJsonPath()).mode & 0o777).toBe(0o600) } ) - it('rolls the file back byte-exactly when the grant lane is unavailable', () => { + it.skipIf(process.platform === 'win32')( + 'restores restrictive hooks.json permissions after grant fallback', + async () => { + grantUnavailable() + writeFileSync(getRealHooksJsonPath(), '{ "hooks": {} }\n', 'utf-8') + chmodSync(getRealHooksJsonPath(), 0o600) + + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + ).toBe('unavailable') + + expect(statSync(getRealHooksJsonPath()).mode & 0o777).toBe(0o600) + } + ) + + it('rolls the file back byte-exactly when the grant lane is unavailable', async () => { grantUnavailable() const userRaw = `${JSON.stringify({ hooks: { Stop: [{ hooks: [{ type: 'command', command: 'mine.sh' }] }] } }, null, 2)}\n` writeFileSync(getRealHooksJsonPath(), userRaw, 'utf-8') - const lane = ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + const lane = await ensureRealHomeCodexHookState({ + hooksEnabled: true, + userDataPath: userDataDir + }) expect(lane).toBe('unavailable') expect(getRealHomeCodexHookLane()).toBe('unavailable') expect(readFileSync(getRealHooksJsonPath(), 'utf-8')).toBe(userRaw) }) - it('removes a freshly created hooks.json when the grant lane is unavailable', () => { + it('removes a freshly created hooks.json when the grant lane is unavailable', async () => { grantUnavailable() - const lane = ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + const lane = await ensureRealHomeCodexHookState({ + hooksEnabled: true, + userDataPath: userDataDir + }) expect(lane).toBe('unavailable') expect(existsSync(getRealHooksJsonPath())).toBe(false) }) - it('surfaces rollback failures to the retry boundary', () => { + it('surfaces rollback failures to the retry boundary', async () => { const warning = vi.spyOn(console, 'warn').mockImplementation(() => {}) grantMock.mockImplementation(() => { rmSync(getRealHooksJsonPath()) @@ -281,9 +313,9 @@ describe('ensureRealHomeCodexHookState (install)', () => { return { lane: 'fallback', reason: 'unsupported' } }) - expect(ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir })).toBe( - 'unavailable' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + ).toBe('unavailable') expect(warning).toHaveBeenCalledWith( '[codex-real-home-hooks] ensure failed; staying on managed lane:', @@ -291,43 +323,49 @@ describe('ensureRealHomeCodexHookState (install)', () => { ) }) - it('does no hook-file or grant work on repeated unsupported launches', () => { + it('does no hook-file or grant work on repeated unsupported launches', async () => { grantUnavailable() - expect(ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir })).toBe( - 'unavailable' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + ).toBe('unavailable') expect(existsSync(getRealHooksJsonPath())).toBe(false) - expect(ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir })).toBe( - 'unavailable' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + ).toBe('unavailable') expect(grantMock).toHaveBeenCalledTimes(1) expect(existsSync(getRealHooksJsonPath())).toBe(false) }) - it('leaves an unparseable hooks.json untouched and keeps the managed lane', () => { + it('leaves an unparseable hooks.json untouched and keeps the managed lane', async () => { writeFileSync(getRealHooksJsonPath(), '{not json', 'utf-8') - const lane = ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + const lane = await ensureRealHomeCodexHookState({ + hooksEnabled: true, + userDataPath: userDataDir + }) expect(lane).toBe('unavailable') expect(readFileSync(getRealHooksJsonPath(), 'utf-8')).toBe('{not json') expect(grantMock).not.toHaveBeenCalled() }) - it('is idempotent: a second ensure keeps a single appended entry per event', () => { + it('is idempotent: a second ensure keeps a single appended entry per event', async () => { grantSucceeds() - ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) const firstRaw = readFileSync(getRealHooksJsonPath(), 'utf-8') - const lane = ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + const lane = await ensureRealHomeCodexHookState({ + hooksEnabled: true, + userDataPath: userDataDir + }) expect(lane).toBe('installed') expect(readFileSync(getRealHooksJsonPath(), 'utf-8')).toBe(firstRaw) }) - it('keeps later user hook trust positions stable when reconciling an existing install', () => { + it('keeps later user hook trust positions stable when reconciling an existing install', async () => { grantSucceeds() const userBefore = { hooks: [{ type: 'command', command: 'before.sh' }] } writeFileSync( @@ -335,15 +373,15 @@ describe('ensureRealHomeCodexHookState (install)', () => { `${JSON.stringify({ hooks: { Stop: [userBefore] } }, null, 2)}\n`, 'utf-8' ) - ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) const installed = readRealHooksJson() const userAfter = { hooks: [{ type: 'command', command: 'after.sh' }] } installed.hooks!.Stop!.push(userAfter) writeFileSync(getRealHooksJsonPath(), `${JSON.stringify(installed, null, 2)}\n`, 'utf-8') - expect(ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir })).toBe( - 'installed' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + ).toBe('installed') const reconciled = readRealHooksJson().hooks?.Stop expect(reconciled?.[0]).toEqual(userBefore) @@ -352,17 +390,17 @@ describe('ensureRealHomeCodexHookState (install)', () => { expect(plan.managedEntries.find((entry) => entry.eventLabel === 'stop')?.groupIndex).toBe(1) }) - it("keeps later user handler trust positions stable inside Orca's hook group", () => { + it("keeps later user handler trust positions stable inside Orca's hook group", async () => { grantSucceeds() - ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) const installed = readRealHooksJson() const userAfter = { type: 'command', command: 'after.sh' } installed.hooks!.Stop![0]!.hooks!.push(userAfter) writeFileSync(getRealHooksJsonPath(), `${JSON.stringify(installed, null, 2)}\n`, 'utf-8') - expect(ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir })).toBe( - 'installed' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + ).toBe('installed') expect(readRealHooksJson().hooks?.Stop?.[0]?.hooks?.[1]).toEqual(userAfter) const plan = grantMock.mock.calls.at(-1)![0] as CodexManagedTrustGrantPlan @@ -372,37 +410,37 @@ describe('ensureRealHomeCodexHookState (install)', () => { }) describe('ensureRealHomeCodexHookState (opt-out sweep)', () => { - it('keeps the managed lane when hooks.json cannot be read', () => { + it('keeps the managed lane when hooks.json cannot be read', async () => { mkdirSync(getRealHooksJsonPath()) - expect(ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir })).toBe( - 'unavailable' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir }) + ).toBe('unavailable') }) - it('keeps the managed lane when hooks.json is malformed', () => { + it('keeps the managed lane when hooks.json is malformed', async () => { writeFileSync(getRealHooksJsonPath(), '{ not json', 'utf-8') - expect(ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir })).toBe( - 'unavailable' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir }) + ).toBe('unavailable') expect(readFileSync(getRealHooksJsonPath(), 'utf-8')).toBe('{ not json') }) - it('rebases trust when a user appended hooks after Orca installed', () => { + it('rebases trust when a user appended hooks after Orca installed', async () => { grantSucceeds() const before = { type: 'command', command: 'before.sh' } writeFileSync( getRealHooksJsonPath(), `${JSON.stringify({ hooks: { Stop: [{ hooks: [before] }] } }, null, 2)}\n` ) - ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) const installed = readRealHooksJson() const after = { type: 'command', command: 'after.sh' } installed.hooks!.Stop!.push({ hooks: [after] }) writeFileSync(getRealHooksJsonPath(), `${JSON.stringify(installed, null, 2)}\n`) const operations: string[] = [] - rebaseInternals.setSessionRunnerSync((request) => { + rebaseInternals.setSessionRunner(async (request) => { operations.push(request.operation) if (request.operation === 'inspect-user-hook-trust') { expect(readRealHooksJson().hooks?.Stop?.[2]?.hooks?.[0]?.command).toBe('after.sh') @@ -420,21 +458,21 @@ describe('ensureRealHomeCodexHookState (opt-out sweep)', () => { return { outcome: 'repaired', repaired: 1 } }) - expect(ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir })).toBe( - 'removed' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir }) + ).toBe('removed') expect(operations).toEqual(['inspect-user-hook-trust', 'repair-user-hook-trust']) expect(readRealHooksJson().hooks?.Stop).toEqual([{ hooks: [before] }, { hooks: [after] }]) }) - it('aborts without writing when hooks.json changes during the trust inspection', () => { + it('aborts without writing when hooks.json changes during the trust inspection', async () => { grantSucceeds() const before = { type: 'command', command: 'before.sh' } writeFileSync( getRealHooksJsonPath(), `${JSON.stringify({ hooks: { Stop: [{ hooks: [before] }] } }, null, 2)}\n` ) - ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) const installed = readRealHooksJson() const after = { type: 'command', command: 'after.sh' } installed.hooks!.Stop!.push({ hooks: [after] }) @@ -443,7 +481,7 @@ describe('ensureRealHomeCodexHookState (opt-out sweep)', () => { writeFileSync(getRealConfigTomlPath(), userTrustToml, 'utf-8') const concurrentSave = `${JSON.stringify({ hooks: { Stop: [{ hooks: [before] }] } }, null, 2)}\n` const operations: string[] = [] - rebaseInternals.setSessionRunnerSync((request) => { + rebaseInternals.setSessionRunner(async (request) => { operations.push(request.operation) // A user save (or a second Orca instance) lands while the RPC runs. writeFileSync(getRealHooksJsonPath(), concurrentSave, 'utf-8') @@ -458,16 +496,16 @@ describe('ensureRealHomeCodexHookState (opt-out sweep)', () => { } }) - expect(ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir })).toBe( - 'unavailable' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir }) + ).toBe('unavailable') expect(operations).toEqual(['inspect-user-hook-trust']) expect(readFileSync(getRealHooksJsonPath(), 'utf-8')).toBe(concurrentSave) expect(readFileSync(getRealConfigTomlPath(), 'utf-8')).toBe(userTrustToml) }) - it('removes only Orca entries and reports the removed lane', () => { + it('removes only Orca entries and reports the removed lane', async () => { grantSucceeds() const userStop = { matcher: 'deploy-*', @@ -478,10 +516,13 @@ describe('ensureRealHomeCodexHookState (opt-out sweep)', () => { `${JSON.stringify({ hooks: { Stop: [userStop] } }, null, 2)}\n`, 'utf-8' ) - ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) + await ensureRealHomeCodexHookState({ hooksEnabled: true, userDataPath: userDataDir }) expect(readRealHooksJson().hooks?.Stop).toHaveLength(2) - const lane = ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir }) + const lane = await ensureRealHomeCodexHookState({ + hooksEnabled: false, + userDataPath: userDataDir + }) expect(lane).toBe('removed') const config = readRealHooksJson() @@ -495,14 +536,17 @@ describe('ensureRealHomeCodexHookState (opt-out sweep)', () => { } }) - it('no-ops the sweep when the real home has no hooks.json', () => { - const lane = ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir }) + it('no-ops the sweep when the real home has no hooks.json', async () => { + const lane = await ensureRealHomeCodexHookState({ + hooksEnabled: false, + userDataPath: userDataDir + }) expect(lane).toBe('removed') expect(existsSync(getRealHooksJsonPath())).toBe(false) }) - it('removes only hash-proven Orca trust from a mixed hook group', () => { + it('removes only hash-proven Orca trust from a mixed hook group', async () => { const material = getCodexManagedHookInstallMaterial() const userCommand = 'my-user-hook.sh' writeFileSync( @@ -544,9 +588,9 @@ describe('ensureRealHomeCodexHookState (opt-out sweep)', () => { ] writeFileSync(getRealConfigTomlPath(), upsertHookTrustEntriesInContent('', entries), 'utf-8') - expect(ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir })).toBe( - 'removed' - ) + expect( + await ensureRealHomeCodexHookState({ hooksEnabled: false, userDataPath: userDataDir }) + ).toBe('removed') expect(readRealHooksJson().hooks?.Stop).toEqual([ { hooks: [{ type: 'command', command: userCommand }] } diff --git a/src/main/codex/codex-real-home-hook-install.ts b/src/main/codex/codex-real-home-hook-install.ts index 678526e4e60..f5ab1b279c7 100644 --- a/src/main/codex/codex-real-home-hook-install.ts +++ b/src/main/codex/codex-real-home-hook-install.ts @@ -1,8 +1,5 @@ -import { existsSync, mkdirSync, readFileSync, statSync, unlinkSync } from 'node:fs' -import { join } from 'node:path' -import { writeFileAtomically } from '../codex-accounts/fs-utils' +import { statSync } from 'node:fs' import { - buildManagedCommandHook, createManagedCommandMatcher, MANAGED_HOOK_TIMEOUT_SECONDS, readHooksJsonWithRaw, @@ -13,6 +10,14 @@ import { type HooksConfig } from '../agent-hooks/installer-utils' import { resolveHooksJsonWritePath } from '../agent-hooks/hook-config-write-path' +import { + assertHooksJsonGeneration, + backupRealHomeHooksJsonOnce, + getRealHomeConfigTomlPath, + getRealHomeHooksJsonPath, + reconcileManagedHookDefinition, + restoreRealHomeHooksJson +} from './codex-real-home-hooks-json' import { getCodexManagedScriptFileName } from './codex-hook-identity' import { CODEX_TRUST_GRANT_TRANSIENT_RETRY_INTERVAL_MS, @@ -25,6 +30,7 @@ import { getSystemCodexHomePath } from './codex-home-paths' import type { CodexTrustEntry } from './config-toml-trust' import { restoreCodexTrustConfig } from './codex-trust-config-rollback' import { mutateRealHomeHooksPreservingUserTrust } from './codex-user-hook-trust-rebase' +import { runExclusivelyForCodexTrustConfig } from './codex-trust-config-mutation-queue' /** * Real-home Codex hook lane for the system-default selection (flag ON). @@ -42,6 +48,7 @@ export type RealHomeCodexHookLane = 'pending' | 'installed' | 'unavailable' | 'r let currentLane: RealHomeCodexHookLane = 'pending' let installRetryAfterMs = 0 +let ensureInFlight: Promise = Promise.resolve(currentLane) export function getRealHomeCodexHookLane(): RealHomeCodexHookLane { return currentLane @@ -56,52 +63,43 @@ export function isRealHomeCodexHookLaneUsable(): boolean { return currentLane !== 'unavailable' } -function getRealHomeHooksJsonPath(): string { - return join(getSystemCodexHomePath(), 'hooks.json') -} - -function getRealHomeConfigTomlPath(): string { - return join(getSystemCodexHomePath(), 'config.toml') -} - -/** Orca-side state dir; nothing extra is ever written into the user's ~/.codex. */ -function getRealHomeHookStateDir(userDataPath: string): string { - return join(userDataPath, 'codex-real-home-hooks') -} - -function assertHooksJsonGeneration( - hooksJsonPath: string, - hooksWritePath: string, - expectedRaw: string | null -): void { - const currentRaw = existsSync(hooksJsonPath) ? readFileSync(hooksJsonPath, 'utf-8') : null - if (currentRaw !== expectedRaw || resolveHooksJsonWritePath(hooksJsonPath) !== hooksWritePath) { - // Why: the pre-mutation RPC can overlap a user's editor save. Abort rather - // than atomically replacing a newer file with the stale parsed snapshot. - throw new Error('Codex hooks.json changed while Orca prepared its trust repair') - } -} - /** * Ensures the real-home hook state matches the settings: installs and trusts - * the Orca status hook when enabled, sweeps it when opted out. Idempotent and - * synchronous (launch prep); repeat calls are cheap — an unchanged hooks.json - * write no-ops and a valid grant ledger skips the RPC session entirely. + * the Orca status hook when enabled, sweeps it when opted out. Idempotent; + * repeat calls are cheap — an unchanged hooks.json write no-ops and a valid + * grant ledger skips the RPC session entirely. * Never throws: any failure logs and leaves the host on the managed lane. */ export function ensureRealHomeCodexHookState(args: { hooksEnabled: boolean userDataPath: string -}): RealHomeCodexHookLane { +}): Promise { // Why: the grant client caches failed probes, but mutating and rolling back - // hooks.json before consulting it still adds synchronous work to every pane. + // hooks.json before consulting it still adds work to every pane launch. if (args.hooksEnabled && currentLane === 'unavailable' && Date.now() < installRetryAfterMs) { - return currentLane + return Promise.resolve(currentLane) } + // Why: this mutates the user's real ~/.codex and the module's lane state. + // Concurrent pane launches must not interleave two of them, and the shared + // config.toml lane keeps the rebase + grant pair atomic against the managed + // installer's legacy sweep of the same file. + const run = (): Promise => runRealHomeCodexHookEnsure(args) + // Why both handlers: a rejected predecessor must not poison every later + // ensure for the process' lifetime. + ensureInFlight = ensureInFlight.then(run, run) + return ensureInFlight +} + +async function runRealHomeCodexHookEnsure(args: { + hooksEnabled: boolean + userDataPath: string +}): Promise { try { - currentLane = args.hooksEnabled - ? installRealHomeCodexHook(args.userDataPath) - : sweepRealHomeCodexHook() + // Why inside the try: resolving the real home can throw too, and this + // function is the module's "never throws" boundary. + currentLane = await runExclusivelyForCodexTrustConfig(getRealHomeConfigTomlPath(), () => + args.hooksEnabled ? installRealHomeCodexHook(args.userDataPath) : sweepRealHomeCodexHook() + ) if (!args.hooksEnabled || currentLane === 'installed') { installRetryAfterMs = 0 } @@ -115,7 +113,7 @@ export function ensureRealHomeCodexHookState(args: { return currentLane } -function installRealHomeCodexHook(userDataPath: string): RealHomeCodexHookLane { +async function installRealHomeCodexHook(userDataPath: string): Promise { const material = getCodexManagedHookInstallMaterial() const hooksJsonPath = getRealHomeHooksJsonPath() const hooksWritePath = resolveHooksJsonWritePath(hooksJsonPath) @@ -175,7 +173,7 @@ function installRealHomeCodexHook(userDataPath: string): RealHomeCodexHookLane { backupRealHomeHooksJsonOnce(userDataPath, previousRaw) // Why: unknown top-level fields belong to the user (other managers' // metadata); unlike the managed-home writer, preserve them verbatim. - const trustConfigSnapshot = mutateRealHomeHooksPreservingUserTrust({ + const trustConfigSnapshot = await mutateRealHomeHooksPreservingUserTrust({ sourcePath: hooksJsonPath, runtimeHomePath: getSystemCodexHomePath(), tomlPath: getRealHomeConfigTomlPath(), @@ -190,7 +188,7 @@ function installRealHomeCodexHook(userDataPath: string): RealHomeCodexHookLane { restoreHooks: () => restoreRealHomeHooksJson(hooksWritePath, previousRaw, previousMode) }) - const grant = grantManagedCodexHookTrust({ + const grant = await grantManagedCodexHookTrust({ runtimeHomePath: getSystemCodexHomePath(), tomlPath: getRealHomeConfigTomlPath(), managedCommand: material.command, @@ -223,53 +221,13 @@ function installRealHomeCodexHook(userDataPath: string): RealHomeCodexHookLane { return 'unavailable' } -function reconcileManagedHookDefinition( - current: HookDefinition[], - isManagedCommand: (command: string | undefined) => boolean, - command: string -): { definitions: HookDefinition[]; groupIndex: number; handlerIndex: number } { - const directCommandKeys = ['command', 'bash', 'powershell'] as const - const hasManagedDirectCommand = current.some((definition) => - directCommandKeys.some((key) => isManagedCommand(definition[key])) - ) - const nestedLocations = current.flatMap((definition, groupIndex) => - Array.isArray(definition.hooks) - ? definition.hooks.flatMap((hook, handlerIndex) => - isManagedCommand(hook.command) ? [{ groupIndex, handlerIndex }] : [] - ) - : [] - ) - if (!hasManagedDirectCommand && nestedLocations.length === 1) { - const { groupIndex, handlerIndex } = nestedLocations[0]! - const definition = current[groupIndex]! - const hasDirectCommand = directCommandKeys.some((key) => typeof definition[key] === 'string') - if (definition.matcher === undefined && !hasDirectCommand) { - const definitions = [...current] - // Why: users can append groups or handlers after Orca's first install. - // Reusing the exact slot preserves all later positional trust keys. - const hooks = [...definition.hooks!] - hooks[handlerIndex] = buildManagedCommandHook(command) - definitions[groupIndex] = { ...definition, hooks } - return { definitions, groupIndex, handlerIndex } - } - } - - const cleaned = removeManagedCommands(current, isManagedCommand) - // Why: first install appends LAST so no existing user trust position shifts. - return { - definitions: [...cleaned, { hooks: [buildManagedCommandHook(command)] }], - groupIndex: cleaned.length, - handlerIndex: 0 - } -} - function getInstallRetryAfterMs(reason: CodexTrustGrantFallbackReason): number { return reason === 'unsupported' || reason === 'unsupported-cached' || reason === 'disabled' ? Number.POSITIVE_INFINITY : Date.now() + CODEX_TRUST_GRANT_TRANSIENT_RETRY_INTERVAL_MS } -function sweepRealHomeCodexHook(): RealHomeCodexHookLane { +async function sweepRealHomeCodexHook(): Promise { const hooksJsonPath = getRealHomeHooksJsonPath() // Why: single read — the pre-write generation guard must compare against // the exact bytes this sweep's parse came from. @@ -306,7 +264,7 @@ function sweepRealHomeCodexHook(): RealHomeCodexHookLane { if (removedAny) { const hooksWritePath = resolveHooksJsonWritePath(hooksJsonPath) const previousMode = statSync(hooksWritePath).mode - mutateRealHomeHooksPreservingUserTrust({ + await mutateRealHomeHooksPreservingUserTrust({ sourcePath: hooksJsonPath, runtimeHomePath: getSystemCodexHomePath(), tomlPath: getRealHomeConfigTomlPath(), @@ -345,41 +303,10 @@ function sweepRealHomeCodexHook(): RealHomeCodexHookLane { return 'removed' } -/** One-time pristine copy of the user's file, kept under Orca's userData. */ -function backupRealHomeHooksJsonOnce(userDataPath: string, previousRaw: string | null): void { - if (previousRaw === null) { - return - } - const backupDir = getRealHomeHookStateDir(userDataPath) - const backupPath = join(backupDir, 'hooks.json.pre-orca') - if (existsSync(backupPath)) { - return - } - // Why: this lane mutates the user's real Codex home. If the required - // pristine recovery copy cannot be created, keep the managed lane intact. - mkdirSync(backupDir, { recursive: true }) - writeFileAtomically(backupPath, previousRaw, { mode: 0o600 }) -} - -function restoreRealHomeHooksJson( - hooksJsonPath: string, - previousRaw: string | null, - previousMode?: number -): void { - if (previousRaw === null) { - if (existsSync(hooksJsonPath)) { - unlinkSync(hooksJsonPath) - } - return - } - // Why: rollback is part of the safety boundary. Use the shared atomic - // writer so Windows file-lock retries and failed-temp cleanup are covered. - writeFileAtomically(hooksJsonPath, previousRaw, { mode: previousMode }) -} - export const _internals = { setLaneForTesting(lane: RealHomeCodexHookLane): void { currentLane = lane installRetryAfterMs = 0 + ensureInFlight = Promise.resolve(lane) } } diff --git a/src/main/codex/codex-real-home-hooks-json.ts b/src/main/codex/codex-real-home-hooks-json.ts new file mode 100644 index 00000000000..0b9f63003bb --- /dev/null +++ b/src/main/codex/codex-real-home-hooks-json.ts @@ -0,0 +1,115 @@ +import { existsSync, mkdirSync, readFileSync, unlinkSync } from 'node:fs' +import { join } from 'node:path' +import { writeFileAtomically } from '../codex-accounts/fs-utils' +import { + buildManagedCommandHook, + removeManagedCommands, + type HookDefinition +} from '../agent-hooks/installer-utils' +import { resolveHooksJsonWritePath } from '../agent-hooks/hook-config-write-path' +import { getSystemCodexHomePath } from './codex-home-paths' + +/** The user's real `~/.codex` hook files, plus the guards and rollback the + * real-home lane needs before it is allowed to mutate them. */ +export function getRealHomeHooksJsonPath(): string { + return join(getSystemCodexHomePath(), 'hooks.json') +} + +export function getRealHomeConfigTomlPath(): string { + return join(getSystemCodexHomePath(), 'config.toml') +} + +/** Orca-side state dir; nothing extra is ever written into the user's ~/.codex. */ +function getRealHomeHookStateDir(userDataPath: string): string { + return join(userDataPath, 'codex-real-home-hooks') +} + +export function assertHooksJsonGeneration( + hooksJsonPath: string, + hooksWritePath: string, + expectedRaw: string | null +): void { + const currentRaw = existsSync(hooksJsonPath) ? readFileSync(hooksJsonPath, 'utf-8') : null + if (currentRaw !== expectedRaw || resolveHooksJsonWritePath(hooksJsonPath) !== hooksWritePath) { + // Why: the pre-mutation RPC can overlap a user's editor save. Abort rather + // than atomically replacing a newer file with the stale parsed snapshot. + throw new Error('Codex hooks.json changed while Orca prepared its trust repair') + } +} + +/** One-time pristine copy of the user's file, kept under Orca's userData. */ +export function backupRealHomeHooksJsonOnce( + userDataPath: string, + previousRaw: string | null +): void { + if (previousRaw === null) { + return + } + const backupDir = getRealHomeHookStateDir(userDataPath) + const backupPath = join(backupDir, 'hooks.json.pre-orca') + if (existsSync(backupPath)) { + return + } + // Why: this lane mutates the user's real Codex home. If the required + // pristine recovery copy cannot be created, keep the managed lane intact. + mkdirSync(backupDir, { recursive: true }) + writeFileAtomically(backupPath, previousRaw, { mode: 0o600 }) +} + +export function restoreRealHomeHooksJson( + hooksJsonPath: string, + previousRaw: string | null, + previousMode?: number +): void { + if (previousRaw === null) { + if (existsSync(hooksJsonPath)) { + unlinkSync(hooksJsonPath) + } + return + } + // Why: rollback is part of the safety boundary. Use the shared atomic + // writer so Windows file-lock retries and failed-temp cleanup are covered. + writeFileAtomically(hooksJsonPath, previousRaw, { mode: previousMode }) +} + +/** Places Orca's managed hook in `definitions`, reusing its existing slot when + * one is unambiguous so no later user trust position shifts. */ +export function reconcileManagedHookDefinition( + current: HookDefinition[], + isManagedCommand: (command: string | undefined) => boolean, + command: string +): { definitions: HookDefinition[]; groupIndex: number; handlerIndex: number } { + const directCommandKeys = ['command', 'bash', 'powershell'] as const + const hasManagedDirectCommand = current.some((definition) => + directCommandKeys.some((key) => isManagedCommand(definition[key])) + ) + const nestedLocations = current.flatMap((definition, groupIndex) => + Array.isArray(definition.hooks) + ? definition.hooks.flatMap((hook, handlerIndex) => + isManagedCommand(hook.command) ? [{ groupIndex, handlerIndex }] : [] + ) + : [] + ) + if (!hasManagedDirectCommand && nestedLocations.length === 1) { + const { groupIndex, handlerIndex } = nestedLocations[0]! + const definition = current[groupIndex]! + const hasDirectCommand = directCommandKeys.some((key) => typeof definition[key] === 'string') + if (definition.matcher === undefined && !hasDirectCommand) { + const definitions = [...current] + // Why: users can append groups or handlers after Orca's first install. + // Reusing the exact slot preserves all later positional trust keys. + const hooks = [...definition.hooks!] + hooks[handlerIndex] = buildManagedCommandHook(command) + definitions[groupIndex] = { ...definition, hooks } + return { definitions, groupIndex, handlerIndex } + } + } + + const cleaned = removeManagedCommands(current, isManagedCommand) + // Why: first install appends LAST so no existing user trust position shifts. + return { + definitions: [...cleaned, { hooks: [buildManagedCommandHook(command)] }], + groupIndex: cleaned.length, + handlerIndex: 0 + } +} diff --git a/src/main/codex/codex-trust-config-concurrent-launch.test.ts b/src/main/codex/codex-trust-config-concurrent-launch.test.ts new file mode 100644 index 00000000000..d08d7a54969 --- /dev/null +++ b/src/main/codex/codex-trust-config-concurrent-launch.test.ts @@ -0,0 +1,410 @@ +import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + CodexHookTrustGrantRequest, + CodexHookTrustGrantSessionResult +} from './codex-app-server-client' +import type { CodexManagedTrustGrantPlan } from './codex-hook-trust-grant' +import type { CodexTrustEntry } from './config-toml-trust' + +const testState = { + fakeHomeDir: '', + userDataDir: '', + previousUserDataPath: undefined as string | undefined +} + +vi.mock('node:os', async () => { + // eslint-disable-next-line @typescript-eslint/consistent-type-imports -- vi.importActual requires inline import() + const actual = await vi.importActual('node:os') + return { ...actual, homedir: () => testState.fakeHomeDir } +}) + +const { CodexAppServerUnsupportedError } = await import('./codex-app-server-client') +const { codexAppServerCapabilityCache } = await import('./codex-app-server-capability-cache') +const { _internals, grantManagedCodexHookTrust } = await import('./codex-hook-trust-grant') +const { markCodexProjectTrusted } = await import('../agent-trust-presets') +const { setCodexTrustGrantTelemetry } = await import('./codex-trust-grant-telemetry') +const { + computeTrustKey, + computeTrustedHash, + normalizeHookTrustKeyForLookup, + readHookTrustEntries, + upsertHookTrustEntries +} = await import('./config-toml-trust') + +let runtimeHomeDir: string + +beforeEach(() => { + testState.fakeHomeDir = mkdtempSync(join(tmpdir(), 'orca-concurrent-home-')) + testState.userDataDir = mkdtempSync(join(tmpdir(), 'orca-concurrent-userdata-')) + testState.previousUserDataPath = process.env.ORCA_USER_DATA_PATH + process.env.ORCA_USER_DATA_PATH = testState.userDataDir + runtimeHomeDir = join(testState.userDataDir, 'codex-runtime-home', 'home') + mkdirSync(runtimeHomeDir, { recursive: true }) + writeFileSync(join(runtimeHomeDir, 'hooks.json'), '{"hooks":{}}\n', 'utf-8') + mkdirSync(join(testState.fakeHomeDir, '.codex'), { recursive: true }) + codexAppServerCapabilityCache.clear() + _internals.resetDiagnostics() +}) + +afterEach(() => { + _internals.setGrantSessionRunner(null) + setCodexTrustGrantTelemetry(() => {}) + codexAppServerCapabilityCache.clear() + if (testState.previousUserDataPath === undefined) { + delete process.env.ORCA_USER_DATA_PATH + } else { + process.env.ORCA_USER_DATA_PATH = testState.previousUserDataPath + } + delete process.env.ORCA_DISABLE_CODEX_TRUST_RPC + rmSync(testState.fakeHomeDir, { recursive: true, force: true }) + rmSync(testState.userDataDir, { recursive: true, force: true }) +}) + +const MANAGED_COMMAND = "/bin/sh '/tmp/orca/codex-hook.sh'" + +function managedEntry(eventLabel: CodexTrustEntry['eventLabel']): CodexTrustEntry { + return { + sourcePath: join(runtimeHomeDir, 'hooks.json'), + eventLabel, + groupIndex: 0, + handlerIndex: 0, + command: MANAGED_COMMAND, + timeoutSec: 10 + } +} + +function buildPlan( + entries: CodexTrustEntry[], + overrides: Partial = {} +): CodexManagedTrustGrantPlan { + return { + runtimeHomePath: runtimeHomeDir, + tomlPath: join(runtimeHomeDir, 'config.toml'), + managedCommand: MANAGED_COMMAND, + managedEntries: entries, + host: { kind: 'native' }, + telemetryLane: 'real-home', + ...overrides + } +} + +const tick = (): Promise => new Promise((resolve) => setTimeout(resolve, 0)) + +/** Stands in for codex app-server: really writes the trust entries into + * config.toml across an await, like the RPC does. */ +function writingSessionRunner(args: { + tomlPath: string + entries: CodexTrustEntry[] + hashPrefix: string + gate?: Promise + outcome?: 'granted' | 'verify-failed' +}) { + return async ( + _request: CodexHookTrustGrantRequest + ): Promise => { + const granted = args.entries.map((entry) => { + const key = computeTrustKey(entry) + return { + key, + normalizedKey: normalizeHookTrustKeyForLookup(key), + trustedHash: `${args.hashPrefix}${entry.eventLabel}` + } + }) + await tick() + upsertHookTrustEntries( + args.tomlPath, + args.entries.map((entry, index) => ({ ...entry, trustedHash: granted[index].trustedHash })) + ) + if (args.gate) { + await args.gate + } + if (args.outcome === 'verify-failed') { + return { + outcome: 'verify-failed', + reason: 'listed hash mismatch', + reasonClass: 'post-grant-mismatch' + } + } + return { outcome: 'granted', wroteTrust: true, entries: granted } + } +} + +describe('two Codex pane launches against one config.toml', () => { + it('does not let a failing launch roll back a concurrent launch that already succeeded', async () => { + // Why warm: on a cold host the shared capability probe incidentally + // serializes the two launches. Once the host is known-supported that + // dedupe is bypassed and the per-file lane is the only thing left. + codexAppServerCapabilityCache.rememberSupported('native') + const tomlPath = join(runtimeHomeDir, 'config.toml') + const entries = [managedEntry('session_start')] + let sessionsInFlight = 0 + let maxSessionsInFlight = 0 + let call = 0 + let releaseFirst!: () => void + const firstGate = new Promise((resolve) => { + releaseFirst = resolve + }) + + _internals.setGrantSessionRunner(async (request) => { + sessionsInFlight += 1 + maxSessionsInFlight = Math.max(maxSessionsInFlight, sessionsInFlight) + call += 1 + const isFirst = call === 1 + try { + return await writingSessionRunner({ + tomlPath, + entries, + hashPrefix: isFirst ? 'sha256:doomed-' : 'sha256:survivor-', + gate: isFirst ? firstGate : undefined, + outcome: isFirst ? 'verify-failed' : 'granted' + })(request) + } finally { + sessionsInFlight -= 1 + } + }) + + const doomed = grantManagedCodexHookTrust(buildPlan(entries)) + const survivor = grantManagedCodexHookTrust(buildPlan(entries)) + await tick() + await tick() + releaseFirst() + + expect(await doomed).toMatchObject({ lane: 'fallback', reason: 'verify-failed' }) + expect(await survivor).toMatchObject({ lane: 'rpc' }) + // The doomed run's rollback must not resurrect the pre-grant file over + // the entries the survivor legitimately wrote. + const trust = readHookTrustEntries(tomlPath) + const key = normalizeHookTrustKeyForLookup(computeTrustKey(entries[0])) + expect(trust.get(key)?.trustedHash).toBe('sha256:survivor-session_start') + expect(maxSessionsInFlight).toBe(1) + }) + + it('keeps a concurrent markCodexProjectTrusted write out of a grant rollback window', async () => { + codexAppServerCapabilityCache.rememberSupported('native') + const tomlPath = join(runtimeHomeDir, 'config.toml') + const entries = [managedEntry('session_start')] + const workspace = mkdtempSync(join(tmpdir(), 'orca-concurrent-ws-')) + let releaseSession!: () => void + const sessionGate = new Promise((resolve) => { + releaseSession = resolve + }) + + _internals.setGrantSessionRunner( + writingSessionRunner({ + tomlPath, + entries, + hashPrefix: 'sha256:doomed-', + gate: sessionGate, + outcome: 'verify-failed' + }) + ) + + try { + const grant = grantManagedCodexHookTrust(buildPlan(entries)) + // Let the grant capture config.toml and start its session. + await tick() + await tick() + const marked = markCodexProjectTrusted(workspace) + await tick() + // The lane must hold the preset write back until rollback has run. + expect(readFileSync(tomlPath, 'utf-8')).not.toContain('trust_level') + + releaseSession() + expect(await grant).toMatchObject({ lane: 'fallback', reason: 'verify-failed' }) + await marked + + expect(readFileSync(tomlPath, 'utf-8')).toContain('trust_level = "trusted"') + } finally { + rmSync(workspace, { recursive: true, force: true }) + } + }) +}) + +describe('concurrent capability probes against a cold host', () => { + it('shares one app-server session between two launches on different config files', async () => { + const secondHome = join(testState.userDataDir, 'second-runtime-home') + mkdirSync(secondHome, { recursive: true }) + writeFileSync(join(secondHome, 'hooks.json'), '{"hooks":{}}\n', 'utf-8') + const entries = [managedEntry('session_start')] + let sessions = 0 + let releaseProbe!: () => void + const probeGate = new Promise((resolve) => { + releaseProbe = resolve + }) + _internals.setGrantSessionRunner(async () => { + sessions += 1 + await probeGate + throw new CodexAppServerUnsupportedError('hooks/grantTrust: method not found') + }) + + const first = grantManagedCodexHookTrust(buildPlan(entries)) + const second = grantManagedCodexHookTrust( + buildPlan([{ ...entries[0], sourcePath: join(secondHome, 'hooks.json') }], { + runtimeHomePath: secondHome, + tomlPath: join(secondHome, 'config.toml') + }) + ) + await tick() + await tick() + expect(sessions).toBe(1) + releaseProbe() + + expect(await first).toMatchObject({ lane: 'fallback', reason: 'unsupported' }) + expect(await second).toMatchObject({ lane: 'fallback', reason: 'unsupported-cached' }) + expect(sessions).toBe(1) + }) + + it('leaves the waiter config.toml untouched when the shared probe reports unsupported', async () => { + const secondHome = join(testState.userDataDir, 'second-runtime-home') + mkdirSync(secondHome, { recursive: true }) + writeFileSync(join(secondHome, 'hooks.json'), '{"hooks":{}}\n', 'utf-8') + const waiterToml = join(secondHome, 'config.toml') + const waiterEntry = { + ...managedEntry('session_start'), + sourcePath: join(secondHome, 'hooks.json') + } + // Self-computed trust the fallback lane already wrote for this pane. + upsertHookTrustEntries(waiterToml, [ + { ...waiterEntry, trustedHash: computeTrustedHash(waiterEntry) } + ]) + const before = readFileSync(waiterToml, 'utf-8') + + let releaseProbe!: () => void + const probeGate = new Promise((resolve) => { + releaseProbe = resolve + }) + _internals.setGrantSessionRunner(async () => { + await probeGate + throw new CodexAppServerUnsupportedError('hooks/grantTrust: method not found') + }) + + const first = grantManagedCodexHookTrust(buildPlan([managedEntry('session_start')])) + const waiter = grantManagedCodexHookTrust( + buildPlan([waiterEntry], { runtimeHomePath: secondHome, tomlPath: waiterToml }) + ) + await tick() + releaseProbe() + await first + expect(await waiter).toMatchObject({ lane: 'fallback', reason: 'unsupported-cached' }) + expect(readFileSync(waiterToml, 'utf-8')).toBe(before) + }) +}) + +describe('host-scoped transient cooldown', () => { + // Why: the cooldown lives outside the per-file lane, so a failure on one + // pane's config.toml has to suppress every other pane on that host and + // nothing on a different one. + it('suppresses a second config.toml on the same host but not another host', async () => { + const secondHome = join(testState.userDataDir, 'second-runtime-home') + mkdirSync(secondHome, { recursive: true }) + writeFileSync(join(secondHome, 'hooks.json'), '{"hooks":{}}\n', 'utf-8') + const entries = [managedEntry('session_start')] + let calls = 0 + _internals.setGrantSessionRunner(() => { + calls += 1 + throw new Error('spawn ETIMEDOUT') + }) + + expect(await grantManagedCodexHookTrust(buildPlan(entries))).toMatchObject({ + lane: 'fallback', + reason: 'error' + }) + expect( + await grantManagedCodexHookTrust( + buildPlan([{ ...entries[0], sourcePath: join(secondHome, 'hooks.json') }], { + runtimeHomePath: secondHome, + tomlPath: join(secondHome, 'config.toml') + }) + ) + ).toMatchObject({ lane: 'fallback', reason: 'retry-cached' }) + expect(calls).toBe(1) + + // A WSL distro runs its own codex binary; the native cooldown must not reach it. + expect( + await grantManagedCodexHookTrust( + buildPlan(entries, { + host: { kind: 'wsl', distro: 'Ubuntu', linuxRuntimeHome: '/home/u/.codex' } + }) + ) + ).toMatchObject({ lane: 'fallback', reason: 'error' }) + expect(calls).toBe(2) + }) + + // Why: the cooldown check runs before the lane, so a launch already admitted + // can succeed after a sibling failed. That proof of health must clear the + // sibling's cooldown instead of suppressing the host for five more minutes. + it('lets a concurrent success clear a cooldown a sibling failure just set', async () => { + codexAppServerCapabilityCache.rememberSupported('native') + const secondHome = join(testState.userDataDir, 'second-runtime-home') + mkdirSync(secondHome, { recursive: true }) + writeFileSync(join(secondHome, 'hooks.json'), '{"hooks":{}}\n', 'utf-8') + const entries = [managedEntry('session_start')] + const okEntry = { ...entries[0], sourcePath: join(secondHome, 'hooks.json') } + const okToml = join(secondHome, 'config.toml') + const okPlan = buildPlan([okEntry], { runtimeHomePath: secondHome, tomlPath: okToml }) + + let releaseFailure!: () => void + const failureGate = new Promise((resolve) => { + releaseFailure = resolve + }) + let sessions = 0 + _internals.setGrantSessionRunner(async (request) => { + sessions += 1 + if (request.hooksListCwd === runtimeHomeDir) { + await failureGate + throw new Error('spawn ETIMEDOUT') + } + return writingSessionRunner({ + tomlPath: okToml, + entries: [okEntry], + hashPrefix: 'sha256:ok-' + })(request) + }) + + const failing = grantManagedCodexHookTrust(buildPlan(entries)) + const succeeding = grantManagedCodexHookTrust(okPlan) + releaseFailure() + expect(await failing).toMatchObject({ lane: 'fallback', reason: 'error' }) + expect(await succeeding).toMatchObject({ lane: 'rpc' }) + + // A later launch on the same host must reach the RPC, not the cooldown. + // The ledger is shared across runtime homes, so clear it to force a session. + rmSync(join(testState.userDataDir, 'codex-runtime-home', 'trust-grant-ledger.json'), { + force: true + }) + expect(sessions).toBe(2) + expect(await grantManagedCodexHookTrust(okPlan)).toMatchObject({ lane: 'rpc' }) + expect(sessions).toBe(3) + }) +}) + +describe('reentrancy under concurrency', () => { + it('completes a grant nested inside an installer that already holds both lanes', async () => { + const { runExclusivelyForCodexTrustConfig } = + await import('./codex-trust-config-mutation-queue') + const entries = [managedEntry('session_start')] + const tomlPath = join(runtimeHomeDir, 'config.toml') + const systemToml = join(testState.fakeHomeDir, '.codex', 'config.toml') + _internals.setGrantSessionRunner( + writingSessionRunner({ tomlPath, entries, hashPrefix: 'sha256:nested-' }) + ) + const workspace = mkdtempSync(join(tmpdir(), 'orca-nested-ws-')) + try { + // Installer lock order: runtime then system, with a grant and a preset + // write nested inside both. + const outcome = await runExclusivelyForCodexTrustConfig(tomlPath, () => + runExclusivelyForCodexTrustConfig(systemToml, async () => { + await markCodexProjectTrusted(workspace) + return grantManagedCodexHookTrust(buildPlan(entries)) + }) + ) + expect(outcome).toMatchObject({ lane: 'rpc' }) + expect(readFileSync(tomlPath, 'utf-8')).toContain('trust_level = "trusted"') + } finally { + rmSync(workspace, { recursive: true, force: true }) + } + }, 5000) +}) diff --git a/src/main/codex/codex-trust-config-mutation-queue.test.ts b/src/main/codex/codex-trust-config-mutation-queue.test.ts new file mode 100644 index 00000000000..e2d775b4ed7 --- /dev/null +++ b/src/main/codex/codex-trust-config-mutation-queue.test.ts @@ -0,0 +1,111 @@ +import { describe, expect, it } from 'vitest' +import { runExclusivelyForCodexTrustConfig } from './codex-trust-config-mutation-queue' + +function deferred(): { promise: Promise; resolve: () => void; reject: (e: unknown) => void } { + let resolve!: () => void + let reject!: (e: unknown) => void + const promise = new Promise((res, rej) => { + resolve = res + reject = rej + }) + return { promise, resolve, reject } +} + +describe('runExclusivelyForCodexTrustConfig', () => { + // Why: the grant lane runs inside the installer that already owns the file; + // a non-reentrant lane would queue it behind itself and never settle. + it('passes through a nested acquire of a lane the caller already holds', async () => { + const nested = await runExclusivelyForCodexTrustConfig('/a/config.toml', () => + runExclusivelyForCodexTrustConfig('/a/config.toml', () => Promise.resolve('inner')) + ) + expect(nested).toBe('inner') + }) + + it('still queues an unrelated lane acquired from inside another lane', async () => { + const gate = deferred() + let innerRan = false + const blocking = runExclusivelyForCodexTrustConfig('/b/config.toml', () => gate.promise) + const nested = runExclusivelyForCodexTrustConfig('/a/config.toml', () => + runExclusivelyForCodexTrustConfig('/b/config.toml', () => { + innerRan = true + return Promise.resolve() + }) + ) + await Promise.resolve() + expect(innerRan).toBe(false) + gate.resolve() + await blocking + await nested + expect(innerRan).toBe(true) + }) + + it('runs one mutation at a time per config.toml', async () => { + const order: string[] = [] + const first = deferred() + const second = deferred() + + const a = runExclusivelyForCodexTrustConfig('/home/.codex/config.toml', async () => { + order.push('a:start') + await first.promise + order.push('a:end') + return 'a' + }) + const b = runExclusivelyForCodexTrustConfig('/home/.codex/config.toml', async () => { + order.push('b:start') + await second.promise + order.push('b:end') + return 'b' + }) + + await Promise.resolve() + expect(order).toEqual(['a:start']) + first.resolve() + await a + second.resolve() + await b + expect(order).toEqual(['a:start', 'a:end', 'b:start', 'b:end']) + }) + + it('keeps distinct config.toml paths independent', async () => { + const gate = deferred() + let secondRan = false + const blocked = runExclusivelyForCodexTrustConfig('/a/config.toml', () => gate.promise) + await runExclusivelyForCodexTrustConfig('/b/config.toml', async () => { + secondRan = true + }) + expect(secondRan).toBe(true) + gate.resolve() + await blocked + }) + + it('keeps the queue alive after a rejected mutation', async () => { + const failing = runExclusivelyForCodexTrustConfig('/a/config.toml', () => + Promise.reject(new Error('grant blew up')) + ) + await expect(failing).rejects.toThrow('grant blew up') + await expect( + runExclusivelyForCodexTrustConfig('/a/config.toml', () => Promise.resolve('next')) + ).resolves.toBe('next') + }) + + // Why: normalized keys, so a Windows caller passing the other separator or + // case must still land in the same lane as the run it has to wait for. + it('serializes equivalent paths that differ only in normalization', async () => { + const gate = deferred() + let secondStarted = false + const blocked = runExclusivelyForCodexTrustConfig( + String.raw`C:\Users\Alice\.codex\config.toml`, + () => gate.promise + ) + const queued = runExclusivelyForCodexTrustConfig('C:/Users/Alice/.codex/config.toml', () => { + secondStarted = true + return Promise.resolve() + }) + await Promise.resolve() + expect(secondStarted).toBe(false) + gate.resolve() + await blocked + await queued + expect(secondStarted).toBe(true) + }) +}) diff --git a/src/main/codex/codex-trust-config-mutation-queue.ts b/src/main/codex/codex-trust-config-mutation-queue.ts new file mode 100644 index 00000000000..08aa3db4316 --- /dev/null +++ b/src/main/codex/codex-trust-config-mutation-queue.ts @@ -0,0 +1,46 @@ +import { AsyncLocalStorage } from 'node:async_hooks' +import { normalizeRuntimePathForComparison } from '../../shared/cross-platform-path' + +const tailByTomlPath = new Map>() +// Why: the grant lane runs inside the installer that already owns the file. +// AsyncLocalStorage survives awaits, so the inner acquire can see the outer +// one and pass through instead of queueing behind itself forever. +const heldKeys = new AsyncLocalStorage>() + +/** + * Serializes everything that mutates one Codex `config.toml` — hook installs, + * trust grants, and user-hook rebases — as a single lane per file. + * + * Why (#16441): these used to block the main thread, so two of them could + * never be in flight at once. Now that they await, a second run could write + * the file between another run's capture and its restore-on-failure, undoing + * a mutation that run never made and resurrecting trust it deliberately + * removed. + */ +export function runExclusivelyForCodexTrustConfig( + tomlPath: string, + run: () => Promise +): Promise { + const key = normalizeRuntimePathForComparison(tomlPath) + const held = heldKeys.getStore() + if (held?.has(key)) { + return run() + } + const owned = new Set(held ?? []) + owned.add(key) + const enter = (): Promise => heldKeys.run(owned, run) + const previous = tailByTomlPath.get(key) ?? Promise.resolve() + // Why both handlers: a rejected predecessor must not cancel the queue. + const result = previous.then(enter, enter) + const tail = result.then( + () => undefined, + () => undefined + ) + tailByTomlPath.set(key, tail) + void tail.then(() => { + if (tailByTomlPath.get(key) === tail) { + tailByTomlPath.delete(key) + } + }) + return result +} diff --git a/src/main/codex/codex-trust-grant-host.test.ts b/src/main/codex/codex-trust-grant-host.test.ts index 5e23add099e..5ad415e9779 100644 --- a/src/main/codex/codex-trust-grant-host.test.ts +++ b/src/main/codex/codex-trust-grant-host.test.ts @@ -1,11 +1,11 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' -const { execFileSyncMock, resolveCodexCommandMock } = vi.hoisted(() => ({ - execFileSyncMock: vi.fn(), +const { runProcessMock, resolveCodexCommandMock } = vi.hoisted(() => ({ + runProcessMock: vi.fn(), resolveCodexCommandMock: vi.fn() })) -vi.mock('node:child_process', () => ({ execFileSync: execFileSyncMock })) +vi.mock('../../shared/child-process/run-process', () => ({ runProcess: runProcessMock })) vi.mock('../codex-cli/command', () => ({ resolveCodexCommand: resolveCodexCommandMock @@ -14,23 +14,28 @@ vi.mock('../codex-cli/command', () => ({ import { resolveCodexTrustGrantHost } from './codex-trust-grant-host' beforeEach(() => { - execFileSyncMock.mockReset() + runProcessMock.mockReset() // Stand in for the guest shell: rc banner first, then the payload inside the // command's own fence. The identity script execs, so no closing fence is written. - execFileSyncMock.mockImplementation((_command: string, args: string[]) => { - const nonce = /__ORCA_WSL_CAPTURE_BEGIN_([^_]+)__/.exec(String(args.at(-1)))?.[1] ?? '' - return ( - 'To run a command as administrator (user "root"), use "sudo ".\n\n' + - `__ORCA_WSL_CAPTURE_BEGIN_${nonce}__/home/alice/.local/bin/codex\ncodex-cli 1.2.3\n` - ) + runProcessMock.mockImplementation((spec: { args: string[] }) => { + const nonce = /__ORCA_WSL_CAPTURE_BEGIN_([^_]+)__/.exec(String(spec.args.at(-1)))?.[1] ?? '' + return Promise.resolve({ + code: 0, + signal: null, + timedOut: false, + stderr: '', + stdout: + 'To run a command as administrator (user "root"), use "sudo ".\n\n' + + `__ORCA_WSL_CAPTURE_BEGIN_${nonce}__/home/alice/.local/bin/codex\ncodex-cli 1.2.3\n` + }) }) resolveCodexCommandMock.mockReset() resolveCodexCommandMock.mockReturnValue(process.execPath) }) describe('resolveCodexTrustGrantHost', () => { - it('resolves the native command once for both the binary stamp and request', () => { - const host = resolveCodexTrustGrantHost({ kind: 'native' }) + it('resolves the native command once for both the binary stamp and request', async () => { + const host = await resolveCodexTrustGrantHost({ kind: 'native' }) const input = { runtimeHomePath: '/tmp/codex-home', managedCommand: '/bin/sh codex-hook.sh', @@ -43,11 +48,11 @@ describe('resolveCodexTrustGrantHost', () => { // Why: PATH/version-manager scans are synchronous launch-path I/O. Reusing // the resolved command keeps one grant at one scan regardless of consumers. expect(resolveCodexCommandMock).toHaveBeenCalledTimes(1) - expect(execFileSyncMock).not.toHaveBeenCalled() + expect(runProcessMock).not.toHaveBeenCalled() }) - it('builds WSL requests without scanning the native PATH', () => { - const host = resolveCodexTrustGrantHost({ + it('builds WSL requests without scanning the native PATH', async () => { + const host = await resolveCodexTrustGrantHost({ kind: 'wsl', distro: 'Ubuntu', linuxRuntimeHome: '/home/alice/.codex-runtime' @@ -65,11 +70,33 @@ describe('resolveCodexTrustGrantHost', () => { version: 'codex-cli 1.2.3' }) expect(request.invocation.command).toBe('wsl.exe') - expect(execFileSyncMock).toHaveBeenCalledWith( - 'wsl.exe', - expect.arrayContaining(['-d', 'Ubuntu', '--exec', 'sh', '-c']), - expect.objectContaining({ encoding: 'utf-8', timeout: 5_000, windowsHide: true }) + // Why (#16441): the identity probe runs through the shared async runner — + // an execFileSync here froze the Electron main thread for its full timeout. + expect(runProcessMock).toHaveBeenCalledWith( + expect.objectContaining({ + program: 'wsl.exe', + args: expect.arrayContaining(['-d', 'Ubuntu', '--exec', 'sh', '-c']), + timeoutMs: 5_000 + }) ) expect(resolveCodexCommandMock).not.toHaveBeenCalled() }) + + it('drops the stamp when the guest probe fails instead of trusting partial stdout', async () => { + runProcessMock.mockResolvedValue({ + code: 127, + signal: null, + timedOut: false, + stdout: '', + stderr: 'codex not found' + }) + + const host = await resolveCodexTrustGrantHost({ + kind: 'wsl', + distro: 'Ubuntu', + linuxRuntimeHome: '/home/alice/.codex-runtime' + }) + + expect(host.binaryStamp).toBeNull() + }) }) diff --git a/src/main/codex/codex-trust-grant-host.ts b/src/main/codex/codex-trust-grant-host.ts index c74f8f19af9..9228fad43ce 100644 --- a/src/main/codex/codex-trust-grant-host.ts +++ b/src/main/codex/codex-trust-grant-host.ts @@ -1,4 +1,4 @@ -import { execFileSync } from 'node:child_process' +import { runProcess } from '../../shared/child-process/run-process' import { resolveCodexCommand } from '../codex-cli/command' import { getSpawnArgsForWindows } from '../win32-utils' import { @@ -36,10 +36,18 @@ export type ResolvedCodexTrustGrantHost = { buildRequest: (input: CodexTrustGrantRequestInput) => CodexHookTrustGrantRequest } -export function resolveCodexTrustGrantHost(host: CodexTrustGrantHost): ResolvedCodexTrustGrantHost { +/** + * Resolves the host that runs the codex binary for a grant session. + * + * Async because the WSL identity probe shells into the distro; #16441 measured + * a 15s main-thread stall when launch prep did this work synchronously. + */ +export async function resolveCodexTrustGrantHost( + host: CodexTrustGrantHost +): Promise { if (host.kind === 'wsl') { return { - binaryStamp: buildWslCodexBinaryStamp(host.distro), + binaryStamp: await buildWslCodexBinaryStamp(host.distro), buildRequest: (input) => ({ invocation: { command: 'wsl.exe', @@ -55,6 +63,10 @@ export function resolveCodexTrustGrantHost(host: CodexTrustGrantHost): ResolvedC } } + return resolveNativeCodexTrustGrantHost() +} + +export function resolveNativeCodexTrustGrantHost(): ResolvedCodexTrustGrantHost { // Why: command resolution scans PATH/version-manager directories. Resolve // once per grant and reuse it for both the binary stamp and invocation. const command = resolveCodexCommand() @@ -81,19 +93,24 @@ export function resolveCodexTrustGrantHost(host: CodexTrustGrantHost): ResolvedC } } -function buildWslCodexBinaryStamp(distro: string): CodexTrustGrantBinaryStamp | null { +async function buildWslCodexBinaryStamp( + distro: string +): Promise { try { // Why: WSL PATH resolution happens inside the distro's login shell. The // resolved path plus CLI version detects upgrades without assuming UNC access. const probe = buildWslCodexIdentityProbe(distro) - const stdout = execFileSync('wsl.exe', probe.args, { - encoding: 'utf-8', - timeout: WSL_CODEX_AVAILABILITY_TIMEOUT_MS, - windowsHide: true + const result = await runProcess({ + program: 'wsl.exe', + args: probe.args, + timeoutMs: WSL_CODEX_AVAILABILITY_TIMEOUT_MS }) + if (result.code !== 0 || result.timedOut) { + return null + } // Why: the split below is positional, so login-shell rc output ahead of the // payload would silently become the "path" and destabilize the stamp. - const output = probe.readStdout(stdout) + const output = probe.readStdout(result.stdout) if (output === null) { return null } @@ -114,9 +131,10 @@ export function readCodexTrustGrantLedgerHomeMatchingStamp( return home && binaryStampsMatch(home.binary, currentStamp) ? home : null } -export function readCurrentCodexTrustGrantLedgerHome( - runtimeHomePath: string, - host: CodexTrustGrantHost +/** Native-only: the WSL stamp needs a subprocess, and status reads must stay + * synchronous for the hook-status readers that never target a distro. */ +export function readCurrentNativeCodexTrustGrantLedgerHome( + runtimeHomePath: string ): CodexTrustGrantLedgerHome | null { try { const home = readCodexTrustGrantLedgerHome(runtimeHomePath) @@ -125,7 +143,7 @@ export function readCurrentCodexTrustGrantLedgerHome( // and version-manager scan when there is no recorded stamp to validate. return null } - return binaryStampsMatch(home.binary, resolveCodexTrustGrantHost(host).binaryStamp) + return binaryStampsMatch(home.binary, resolveNativeCodexTrustGrantHost().binaryStamp) ? home : null } catch { diff --git a/src/main/codex/codex-trust-grant-main-thread-boundary.test.ts b/src/main/codex/codex-trust-grant-main-thread-boundary.test.ts new file mode 100644 index 00000000000..4b88d95991c --- /dev/null +++ b/src/main/codex/codex-trust-grant-main-thread-boundary.test.ts @@ -0,0 +1,60 @@ +import { existsSync, readFileSync, readdirSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * Ratchet for stablyai/orca#16441. + * + * Codex hook trust used to be granted by blocking the Electron main thread on + * `spawnSync` of a bundled ELECTRON_RUN_AS_NODE entry, for the whole + * app-server deadline: 15s native, 35s WSL, ~45s on the three-session real-home + * path. The window showed "Not Responding" during cold start and pane launch. + * + * The subprocess only ever existed to donate an event loop to a deliberately + * blocked parent, so this guards the shape of the fix rather than one call + * site: nothing on the trust-grant lane may start a child process + * synchronously, and the forked entry must stay gone. + */ +const CODEX_DIR = __dirname + +const SYNC_SPAWN_PATTERN = /\b(?:spawnSync|execSync|execFileSync|runProcessSync)\s*[(<]/ + +/** Drop comments so the prose explaining the old idiom is not an offender. */ +function codeText(contents: string): string { + return contents + .split('\n') + .filter((line) => !/^\s*(?:\/\/|\/\*|\*)/.test(line)) + .join('\n') +} + +function listCodexSourceFiles(): string[] { + return readdirSync(CODEX_DIR).filter((name) => name.endsWith('.ts') && !name.endsWith('.test.ts')) +} + +describe('codex trust grant main-thread boundary', () => { + it('starts no child process synchronously anywhere in the codex module', () => { + const offenders = listCodexSourceFiles().filter((name) => + SYNC_SPAWN_PATTERN.test(codeText(readFileSync(join(CODEX_DIR, name), 'utf8'))) + ) + expect(offenders).toEqual([]) + }) + + it('keeps the forked grant entry and its blocking bridge deleted', () => { + for (const name of [ + 'codex-app-server-grant-bridge.ts', + 'codex-app-server-grant-entry.ts', + 'codex-app-server-grant-envelope.ts' + ]) { + expect(existsSync(join(CODEX_DIR, name))).toBe(false) + } + }) + + it('keeps the trust-grant lane on async entry points', () => { + const grant = readFileSync(join(CODEX_DIR, 'codex-hook-trust-grant.ts'), 'utf8') + expect(grant).toContain('export async function grantManagedCodexHookTrust(') + const host = readFileSync(join(CODEX_DIR, 'codex-trust-grant-host.ts'), 'utf8') + expect(host).toContain('export async function resolveCodexTrustGrantHost(') + const realHome = readFileSync(join(CODEX_DIR, 'codex-real-home-hook-install.ts'), 'utf8') + expect(realHome).toContain('}): Promise {') + }) +}) diff --git a/src/main/codex/codex-trust-grant-telemetry.ts b/src/main/codex/codex-trust-grant-telemetry.ts index 25265219ba8..aa40c17cba3 100644 --- a/src/main/codex/codex-trust-grant-telemetry.ts +++ b/src/main/codex/codex-trust-grant-telemetry.ts @@ -16,10 +16,9 @@ export type CodexTrustGrantFallbackReason = | 'retry-cached' | 'error' -/** Closed classification of `reason: 'error'` fallbacks. Errors cross the - * grant-bridge envelope as message text (only timeout/unsupported keep their - * name), so classes are matched on the bounded message shapes each layer - * produces — never forwarded raw. */ +/** Closed classification of `reason: 'error'` fallbacks. Only timeout and + * unsupported carry a stable error name, so the rest are matched on the + * bounded message shapes each layer produces — never forwarded raw. */ export type CodexTrustGrantErrorClass = | 'binary-missing' | 'timeout' diff --git a/src/main/codex/codex-user-hook-trust-rebase.test.ts b/src/main/codex/codex-user-hook-trust-rebase.test.ts index 89f25176d50..c3e0f760f69 100644 --- a/src/main/codex/codex-user-hook-trust-rebase.test.ts +++ b/src/main/codex/codex-user-hook-trust-rebase.test.ts @@ -27,7 +27,7 @@ beforeEach(() => { }) afterEach(() => { - _internals.setSessionRunnerSync(null) + _internals.setSessionRunner(null) _internals.resetRetryState() codexAppServerCapabilityCache.clear() rmSync(root, { recursive: true, force: true }) @@ -38,18 +38,18 @@ function command(command: string): HookCommandConfig { } describe('real-home user hook trust rebasing', () => { - it('writes directly without reading config or spawning Codex when user positions stay stable', () => { + it('writes directly without reading config or spawning Codex when user positions stay stable', async () => { const user = command('user-hook') const orca = command('orca-hook') const before = { Stop: [{ hooks: [user] }] } const after = { Stop: [{ hooks: [user] }, { hooks: [orca] }] } let wroteHooks = false - _internals.setSessionRunnerSync(() => { + _internals.setSessionRunner(() => { throw new Error('stable positions must not open an app-server session') }) expect( - mutateRealHomeHooksPreservingUserTrust({ + await mutateRealHomeHooksPreservingUserTrust({ sourcePath: hooksPath, runtimeHomePath: root, tomlPath: configPath, @@ -67,7 +67,7 @@ describe('real-home user hook trust rebasing', () => { expect(existsSync(configPath)).toBe(false) }) - it('finds multiple shifted user hooks, including a handler from a mixed group', () => { + it('finds multiple shifted user hooks, including a handler from a mixed group', async () => { const orca = command('orca-hook') const first = command('first-user') const second = command('second-user') @@ -98,7 +98,7 @@ describe('real-home user hook trust rebasing', () => { ]) }) - it('carries only previously trusted states into the repair request', () => { + it('carries only previously trusted states into the repair request', async () => { const orca = command('orca-hook') const trusted = command('trusted-user') const untrusted = command('untrusted-user') @@ -107,7 +107,7 @@ describe('real-home user hook trust rebasing', () => { writeFileSync(hooksPath, `${JSON.stringify({ hooks: before }, null, 2)}\n`) writeFileSync(configPath, '# original config\n') const requests: CodexUserHookTrustRebaseRequest[] = [] - _internals.setSessionRunnerSync((request) => { + _internals.setSessionRunner(async (request) => { requests.push(request) if (request.operation === 'inspect-user-hook-trust') { return { @@ -123,7 +123,7 @@ describe('real-home user hook trust rebasing', () => { return { outcome: 'repaired', repaired: 1 } }) - mutateRealHomeHooksPreservingUserTrust({ + await mutateRealHomeHooksPreservingUserTrust({ sourcePath: hooksPath, runtimeHomePath: root, tomlPath: configPath, @@ -147,14 +147,14 @@ describe('real-home user hook trust rebasing', () => { } }) - it('marks the host unsupported and skips further codex sessions', () => { + it('marks the host unsupported and skips further codex sessions', async () => { const orca = command('orca-hook') const user = command('user-hook') const before = { Stop: [{ hooks: [orca] }, { hooks: [user] }] } const after = { Stop: [{ hooks: [user] }] } writeFileSync(configPath, '# config\n') let sessions = 0 - _internals.setSessionRunnerSync(() => { + _internals.setSessionRunner(async () => { sessions += 1 throw new CodexAppServerUnsupportedError('unrecognized subcommand app-server') }) @@ -172,20 +172,22 @@ describe('real-home user hook trust rebasing', () => { } } - expect(() => mutateRealHomeHooksPreservingUserTrust(args)).toThrow('unrecognized subcommand') - expect(() => mutateRealHomeHooksPreservingUserTrust(args)).toThrow('marked unsupported') + await expect(mutateRealHomeHooksPreservingUserTrust(args)).rejects.toThrow( + 'unrecognized subcommand' + ) + await expect(mutateRealHomeHooksPreservingUserTrust(args)).rejects.toThrow('marked unsupported') expect(sessions).toBe(1) expect(codexAppServerCapabilityCache.shouldTry('native')).toBe(false) }) - it('cools down after a transient session failure instead of retrying every launch prep', () => { + it('cools down after a transient session failure instead of retrying every launch prep', async () => { const orca = command('orca-hook') const user = command('user-hook') const before = { Stop: [{ hooks: [orca] }, { hooks: [user] }] } const after = { Stop: [{ hooks: [user] }] } writeFileSync(configPath, '# config\n') let sessions = 0 - _internals.setSessionRunnerSync(() => { + _internals.setSessionRunner(async () => { sessions += 1 throw new Error('pre-mutation hooks/list reported 0 of 1 moved user hooks') }) @@ -203,14 +205,16 @@ describe('real-home user hook trust rebasing', () => { } } - expect(() => mutateRealHomeHooksPreservingUserTrust(args)).toThrow('0 of 1 moved user hooks') - expect(() => mutateRealHomeHooksPreservingUserTrust(args)).toThrow('cooling down') + await expect(mutateRealHomeHooksPreservingUserTrust(args)).rejects.toThrow( + '0 of 1 moved user hooks' + ) + await expect(mutateRealHomeHooksPreservingUserTrust(args)).rejects.toThrow('cooling down') expect(sessions).toBe(1) // Why: a transient failure must not poison the shared capability signal. expect(codexAppServerCapabilityCache.shouldTry('native')).toBe(true) }) - it('restores both files byte-exactly when post-mutation repair fails', () => { + it('restores both files byte-exactly when post-mutation repair fails', async () => { const orca = command('orca-hook') const user = command('user-hook') const before = { Stop: [{ hooks: [orca] }, { hooks: [user] }] } @@ -220,7 +224,7 @@ describe('real-home user hook trust rebasing', () => { const originalConfig = '# user formatting\r\nmodel = "x"\r\n' writeFileSync(hooksPath, originalHooks) writeFileSync(configPath, originalConfig) - _internals.setSessionRunnerSync((request) => { + _internals.setSessionRunner(async (request) => { if (request.operation === 'inspect-user-hook-trust') { return { outcome: 'inspected', @@ -236,7 +240,7 @@ describe('real-home user hook trust rebasing', () => { throw new Error('repair transport failed') }) - expect(() => + await expect( mutateRealHomeHooksPreservingUserTrust({ sourcePath: hooksPath, runtimeHomePath: root, @@ -246,7 +250,7 @@ describe('real-home user hook trust rebasing', () => { writeHooks: () => writeFileSync(hooksPath, `${JSON.stringify({ hooks: after })}\n`), restoreHooks: () => writeFileSync(hooksPath, originalHooks) }) - ).toThrow('repair transport failed') + ).rejects.toThrow('repair transport failed') expect(readFileSync(hooksPath, 'utf-8')).toBe(originalHooks) expect(readFileSync(configPath, 'utf-8')).toBe(originalConfig) }) diff --git a/src/main/codex/codex-user-hook-trust-rebase.ts b/src/main/codex/codex-user-hook-trust-rebase.ts index 0020e106c2f..67a69afce6e 100644 --- a/src/main/codex/codex-user-hook-trust-rebase.ts +++ b/src/main/codex/codex-user-hook-trust-rebase.ts @@ -4,7 +4,7 @@ import { getCodexAppServerHostKey, type CodexAppServerHostKey } from './codex-app-server-capability-cache' -import { runCodexUserHookTrustRebaseSessionSync } from './codex-app-server-grant-bridge' +import { runCodexUserHookTrustRebaseSession } from './codex-user-hook-trust-rebase-client' import { isCodexAppServerUnsupportedError } from './codex-app-server-session' import { CODEX_TRUST_GRANT_TRANSIENT_RETRY_INTERVAL_MS } from './codex-hook-trust-grant' import { createCodexHookTrustEntry } from './codex-hook-identity' @@ -14,6 +14,7 @@ import { restoreCodexTrustConfig, type CodexTrustConfigSnapshot } from './codex-trust-config-rollback' +import { runExclusivelyForCodexTrustConfig } from './codex-trust-config-mutation-queue' import { computeTrustKey, type CodexTrustEntry } from './config-toml-trust' import type { CodexUserHookTrustRebaseRequest, @@ -23,11 +24,13 @@ import type { type HooksByEvent = Record -type RebaseSessionRunnerSync = ( +type RebaseSessionRunner = ( request: CodexUserHookTrustRebaseRequest -) => CodexUserHookTrustRebaseResult +) => Promise -let runSessionSync: RebaseSessionRunnerSync = runCodexUserHookTrustRebaseSessionSync +// Why (#16441): the session runs in-process; forking it through spawnSync +// froze the main thread for the whole app-server deadline on every install. +let runSession: RebaseSessionRunner = runCodexUserHookTrustRebaseSession // Why: launch prep re-runs the callers on every pane spawn. A host stuck // without a usable rebase lane (old CLI, unmatched keys) must not pay a codex @@ -129,12 +132,25 @@ export function mutateRealHomeHooksPreservingUserTrust(args: { afterHooks: HooksByEvent writeHooks: () => void restoreHooks: () => void -}): CodexTrustConfigSnapshot | null { +}): Promise { const moves = getMovedCodexUserHookTrust(args.sourcePath, args.beforeHooks, args.afterHooks) if (moves.length === 0) { args.writeHooks() - return null + return Promise.resolve(null) } + // Why: capture/mutate/restore on one config.toml is not reentrant. + return runExclusivelyForCodexTrustConfig(args.tomlPath, () => rebaseMovedUserTrust(args, moves)) +} + +async function rebaseMovedUserTrust( + args: { + runtimeHomePath: string + tomlPath: string + writeHooks: () => void + restoreHooks: () => void + }, + moves: CodexUserHookTrustMove[] +): Promise { const hostKey = getCodexAppServerHostKey({ kind: 'native' }) if (!codexAppServerCapabilityCache.shouldTry(hostKey)) { throw new Error('codex app-server is marked unsupported on this host; trust rebase skipped') @@ -148,7 +164,7 @@ export function mutateRealHomeHooksPreservingUserTrust(args: { } const snapshot = captureCodexTrustConfig(args.tomlPath) - const baseRequest = resolveCodexTrustGrantHost({ kind: 'native' }).buildRequest({ + const baseRequest = (await resolveCodexTrustGrantHost({ kind: 'native' })).buildRequest({ runtimeHomePath: args.runtimeHomePath, managedCommand: '', expectedTrustKeys: [], @@ -158,7 +174,7 @@ export function mutateRealHomeHooksPreservingUserTrust(args: { // without shifting a user's positional trust key. let inspected: CodexUserHookTrustRebaseResult try { - inspected = runSessionSync({ + inspected = await runSession({ operation: 'inspect-user-hook-trust', invocation: baseRequest.invocation, hooksListCwd: baseRequest.hooksListCwd, @@ -177,7 +193,7 @@ export function mutateRealHomeHooksPreservingUserTrust(args: { try { args.writeHooks() hooksWritten = true - const repaired = runSessionSync({ + const repaired = await runSession({ operation: 'repair-user-hook-trust', invocation: baseRequest.invocation, hooksListCwd: baseRequest.hooksListCwd, @@ -197,8 +213,8 @@ export function mutateRealHomeHooksPreservingUserTrust(args: { } export const _internals = { - setSessionRunnerSync(runner: RebaseSessionRunnerSync | null): void { - runSessionSync = runner ?? runCodexUserHookTrustRebaseSessionSync + setSessionRunner(runner: RebaseSessionRunner | null): void { + runSession = runner ?? runCodexUserHookTrustRebaseSession }, resetRetryState(): void { rebaseRetryAfterByHost.clear() diff --git a/src/main/codex/hook-service-legacy-cleanup.test.ts b/src/main/codex/hook-service-legacy-cleanup.test.ts index 7093e0242d4..f2b941944c0 100644 --- a/src/main/codex/hook-service-legacy-cleanup.test.ts +++ b/src/main/codex/hook-service-legacy-cleanup.test.ts @@ -54,7 +54,7 @@ function legacyManagedHookCommand(): string { } describe('CodexHookService', () => { - it('removes legacy Orca-managed hooks from system ~/.codex during install', () => { + it('removes legacy Orca-managed hooks from system ~/.codex during install', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const systemHooksPath = join(systemCodexHome, 'hooks.json') const legacyCommand = legacyManagedHookCommand() @@ -102,7 +102,7 @@ describe('CodexHookService', () => { 'utf-8' ) - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') const systemHooks = JSON.parse(readFileSync(systemHooksPath, 'utf-8')) as { hooks: Record @@ -117,7 +117,7 @@ describe('CodexHookService', () => { expect(systemToml).not.toContain(':session_start:0:0') }) - it('removes very large legacy Orca-managed hook lists from system ~/.codex', () => { + it('removes very large legacy Orca-managed hook lists from system ~/.codex', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const systemHooksPath = join(systemCodexHome, 'hooks.json') const legacyCommand = legacyManagedHookCommand() @@ -136,7 +136,7 @@ describe('CodexHookService', () => { const warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {}) try { - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') expect(warnSpy).not.toHaveBeenCalledWith( '[codex-hook-service] failed to clean legacy Codex hooks', @@ -151,18 +151,18 @@ describe('CodexHookService', () => { expect(systemHooks.hooks.Stop).toBeUndefined() }, 30_000) - it('removes the legacy Orca Codex profile file when it only contains managed hooks', () => { + it('removes the legacy Orca Codex profile file when it only contains managed hooks', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const profilePath = join(systemCodexHome, 'orca-agent-status.config.toml') mkdirSync(systemCodexHome, { recursive: true }) writeFileSync(profilePath, LEGACY_ORCA_PROFILE_LINES.join('\n'), 'utf-8') - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') expect(existsSync(profilePath)).toBe(false) }) - it('removes only the legacy Orca block from a user-edited Codex profile file', () => { + it('removes only the legacy Orca block from a user-edited Codex profile file', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const profilePath = join(systemCodexHome, 'orca-agent-status.config.toml') mkdirSync(systemCodexHome, { recursive: true }) @@ -172,7 +172,7 @@ describe('CodexHookService', () => { 'utf-8' ) - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') const profileConfig = readFileSync(profilePath, 'utf-8') expect(profileConfig).toContain('model = "gpt-5.5"') @@ -180,7 +180,7 @@ describe('CodexHookService', () => { expect(profileConfig).not.toContain('codex-hook') }) - it('cleans legacy system and profile hooks when runtime hooks.json is malformed during remove', () => { + it('cleans legacy system and profile hooks when runtime hooks.json is malformed during remove', async () => { const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') mkdirSync(managedCodexHome, { recursive: true }) writeFileSync(join(managedCodexHome, 'hooks.json'), '{not json', 'utf-8') @@ -209,7 +209,7 @@ describe('CodexHookService', () => { ) writeFileSync(profilePath, LEGACY_ORCA_PROFILE_LINES.join('\n'), 'utf-8') - const status = new CodexHookService().remove() + const status = await new CodexHookService().remove() expect(status.state).toBe('error') expect(status.detail).toBe('Could not parse Codex hooks.json') @@ -221,7 +221,7 @@ describe('CodexHookService', () => { expect(existsSync(profilePath)).toBe(false) }) - it('sanitizes runtime hooks.json metadata during remove even without managed hooks', () => { + it('sanitizes runtime hooks.json metadata during remove even without managed hooks', async () => { const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') const managedHooksPath = join(managedCodexHome, 'hooks.json') mkdirSync(managedCodexHome, { recursive: true }) @@ -244,7 +244,7 @@ describe('CodexHookService', () => { 'utf-8' ) - const status = new CodexHookService().remove() + const status = await new CodexHookService().remove() expect(status.state).toBe('not_installed') const hooksConfig = JSON.parse(readFileSync(managedHooksPath, 'utf-8')) as { @@ -256,7 +256,7 @@ describe('CodexHookService', () => { expect(hooksConfig.hooks.Stop).toEqual([{ hooks: [{ type: 'command', command: 'user-hook' }] }]) }) - it('cleans duplicate Codex hook representations while keeping status hooks in runtime CODEX_HOME', () => { + it('cleans duplicate Codex hook representations while keeping status hooks in runtime CODEX_HOME', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const systemHooksPath = join(systemCodexHome, 'hooks.json') const systemTomlPath = join(systemCodexHome, 'config.toml') @@ -307,7 +307,7 @@ describe('CodexHookService', () => { writeFileSync(legacyProfilePath, LEGACY_ORCA_PROFILE_LINES.join('\n'), 'utf-8') const service = new CodexHookService() - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') const managedHooksPath = join(managedCodexHome, 'hooks.json') diff --git a/src/main/codex/hook-service-managed-install.test.ts b/src/main/codex/hook-service-managed-install.test.ts index 72852e45b55..5d0021f9499 100644 --- a/src/main/codex/hook-service-managed-install.test.ts +++ b/src/main/codex/hook-service-managed-install.test.ts @@ -28,6 +28,7 @@ vi.mock('os', async (importOriginal) => { }) import { CodexHookService } from './hook-service' +import { runExclusivelyForCodexTrustConfig } from './codex-trust-config-mutation-queue' const WINDOWS_POWERSHELL_LAUNCHER = /^[A-Za-z]:\/[^"]*\/System32\/WindowsPowerShell\/v1\.0\/powershell\.exe -NoProfile -WindowStyle Hidden -EncodedCommand \S+$/ @@ -48,7 +49,58 @@ function localManagedCodexEvents(): string[] { } describe('CodexHookService', () => { - it('installs PermissionRequest with trust so Codex approval prompts reach Orca', () => { + // Why (#16441): install promotes in-Orca approvals into ~/.codex/config.toml + // and mirrors that file into the managed home, so holding only the runtime + // lane still lets it land inside a real-home grant's capture->restore window. + it('waits for an in-flight mutation of the system config.toml', async () => { + const systemCodexHome = join(homes.tmpHome, '.codex') + mkdirSync(systemCodexHome, { recursive: true }) + writeFileSync(join(systemCodexHome, 'config.toml'), 'approval_policy = "on-request"\n', 'utf-8') + const managedHooksJsonPath = join(homes.userDataDir, 'codex-runtime-home', 'home', 'hooks.json') + let releaseGrant!: () => void + const grantHoldingSystemConfig = new Promise((resolve) => { + releaseGrant = resolve + }) + const held = runExclusivelyForCodexTrustConfig( + join(systemCodexHome, 'config.toml'), + () => grantHoldingSystemConfig + ) + + const install = new CodexHookService().install() + await new Promise((resolve) => setImmediate(resolve)) + expect(existsSync(managedHooksJsonPath)).toBe(false) + + releaseGrant() + await held + await expect(install).resolves.toMatchObject({ state: 'installed' }) + expect(existsSync(managedHooksJsonPath)).toBe(true) + }) + + it('makes the user-hook refresh wait for the system config.toml too', async () => { + const systemCodexHome = join(homes.tmpHome, '.codex') + mkdirSync(systemCodexHome, { recursive: true }) + writeFileSync(join(systemCodexHome, 'config.toml'), 'approval_policy = "on-request"\n', 'utf-8') + const managedHooksJsonPath = join(homes.userDataDir, 'codex-runtime-home', 'home', 'hooks.json') + let releaseGrant!: () => void + const held = runExclusivelyForCodexTrustConfig( + join(systemCodexHome, 'config.toml'), + () => + new Promise((resolve) => { + releaseGrant = resolve + }) + ) + + const refresh = new CodexHookService().refreshRuntimeUserHooks() + await new Promise((resolve) => setImmediate(resolve)) + expect(existsSync(managedHooksJsonPath)).toBe(false) + + releaseGrant() + await held + await refresh + expect(existsSync(managedHooksJsonPath)).toBe(true) + }) + + it('installs PermissionRequest with trust so Codex approval prompts reach Orca', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') mkdirSync(systemCodexHome, { recursive: true }) writeFileSync( @@ -57,7 +109,7 @@ describe('CodexHookService', () => { 'utf-8' ) - const status = new CodexHookService().install() + const status = await new CodexHookService().install() expect(status.state).toBe('installed') @@ -77,7 +129,7 @@ describe('CodexHookService', () => { expect(trustConfig).toContain(':permission_request:0:0') }) - it('installs managed hooks + trust into a per-account self-contained home, not the shared mirror', () => { + it('installs managed hooks + trust into a per-account self-contained home, not the shared mirror', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') mkdirSync(systemCodexHome, { recursive: true }) writeFileSync(join(systemCodexHome, 'config.toml'), 'approval_policy = "on-request"\n', 'utf-8') @@ -86,7 +138,7 @@ describe('CodexHookService', () => { mkdirSync(perAccountHome, { recursive: true }) writeFileSync(join(perAccountHome, '.orca-managed-home'), 'account-1\n', 'utf-8') - const status = new CodexHookService().install(perAccountHome) + const status = await new CodexHookService().install(perAccountHome) expect(status.state).toBe('installed') // Hooks + trust land in THIS account's home. @@ -106,7 +158,7 @@ describe('CodexHookService', () => { expect(existsSync(join(systemCodexHome, 'hooks.json'))).toBe(false) }) - it('drops plugin manager metadata from runtime hooks.json during install', () => { + it('drops plugin manager metadata from runtime hooks.json during install', async () => { const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') mkdirSync(managedCodexHome, { recursive: true }) writeFileSync( @@ -122,7 +174,7 @@ describe('CodexHookService', () => { 'utf-8' ) - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') const hooksConfig = JSON.parse(readFileSync(join(managedCodexHome, 'hooks.json'), 'utf-8')) as { hooks: Record @@ -138,7 +190,7 @@ describe('CodexHookService', () => { // `cmd.exe /C` never sees the raw script path. it.skipIf(process.platform !== 'win32')( 'wraps the managed hook command when the profile path contains a space (#6078)', - () => { + async () => { const spaceHome = join(tmpdir(), 'orca home with spaces') mkdirSync(spaceHome, { recursive: true }) homedirMock.mockReturnValue(spaceHome) @@ -146,7 +198,7 @@ describe('CodexHookService', () => { const systemCodexHome = join(spaceHome, '.codex') mkdirSync(systemCodexHome, { recursive: true }) - const status = new CodexHookService().install() + const status = await new CodexHookService().install() expect(status.state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') @@ -168,7 +220,7 @@ describe('CodexHookService', () => { // plausible paths. Keep those rare cases on the encoded launcher from #6078. it.skipIf(process.platform !== 'win32')( 'keeps the encoded launcher when the profile path contains cmd metacharacters', - () => { + async () => { const metacharHome = join(tmpdir(), 'orca %ORCA_TEST% ^ home') mkdirSync(metacharHome, { recursive: true }) homedirMock.mockReturnValue(metacharHome) @@ -176,7 +228,7 @@ describe('CodexHookService', () => { const systemCodexHome = join(metacharHome, '.codex') mkdirSync(systemCodexHome, { recursive: true }) - const status = new CodexHookService().install() + const status = await new CodexHookService().install() expect(status.state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') @@ -199,8 +251,8 @@ describe('CodexHookService', () => { // speed that Codex 0.140's synchronous "Running hook" rows expose. it.skipIf(process.platform !== 'win32')( 'launches the managed .cmd directly when the profile path is cmd-safe', - () => { - const status = new CodexHookService().install() + async () => { + const status = await new CodexHookService().install() expect(status.state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') @@ -227,7 +279,7 @@ describe('CodexHookService', () => { it.skipIf(process.platform !== 'win32')( 'posts hook payloads via the curl-based managed script preserving UTF-8 and spaced metadata', async () => { - new CodexHookService().install() + await new CodexHookService().install() const scriptPath = join(homedir(), '.orca', 'agent-hooks', 'codex-hook.cmd') expect(existsSync(scriptPath)).toBe(true) @@ -297,7 +349,7 @@ describe('CodexHookService', () => { } ) - it('keeps hooks isolated by Orca userData instead of mutating system ~/.codex', () => { + it('keeps hooks isolated by Orca userData instead of mutating system ~/.codex', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const systemHooksPath = join(systemCodexHome, 'hooks.json') const existingSystemHooks = '{"hooks":{"Stop":[{"hooks":[{"command":"user-hook"}]}]}}\n' @@ -314,7 +366,7 @@ describe('CodexHookService', () => { throw new Error(`unexpected app.getPath(${name})`) }) process.env.ORCA_USER_DATA_PATH = devUserDataDir - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') getPathMock.mockImplementation((name: string) => { if (name === 'userData') { @@ -323,7 +375,7 @@ describe('CodexHookService', () => { throw new Error(`unexpected app.getPath(${name})`) }) process.env.ORCA_USER_DATA_PATH = prodUserDataDir - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') const devHooksPath = join(devUserDataDir, 'codex-runtime-home', 'home', 'hooks.json') const prodHooksPath = join(prodUserDataDir, 'codex-runtime-home', 'home', 'hooks.json') diff --git a/src/main/codex/hook-service-runtime-trust-repair.test.ts b/src/main/codex/hook-service-runtime-trust-repair.test.ts index ce8c660af67..5adc4b202b4 100644 --- a/src/main/codex/hook-service-runtime-trust-repair.test.ts +++ b/src/main/codex/hook-service-runtime-trust-repair.test.ts @@ -33,7 +33,7 @@ import { CodexHookService } from './hook-service' const homes = setupCodexHookHomes(homedirMock, getPathMock) describe('CodexHookService', () => { - it('removes managed trust entries when userData resolves through a symlink', () => { + it('removes managed trust entries when userData resolves through a symlink', async () => { const linkedUserDataDir = join(homes.tmpHome, 'linked-user-data') symlinkSync( homes.userDataDir, @@ -43,14 +43,14 @@ describe('CodexHookService', () => { process.env.ORCA_USER_DATA_PATH = linkedUserDataDir const service = new CodexHookService() - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const linkedManagedCodexHome = join(linkedUserDataDir, 'codex-runtime-home', 'home') const linkedHooksPath = join(linkedManagedCodexHome, 'hooks.json') let runtimeToml = readFileSync(join(linkedManagedCodexHome, 'config.toml'), 'utf-8') expect(runtimeToml).toContain(hookTrustHeader(`${linkedHooksPath}:permission_request:0:0`)) - const status = service.remove() + const status = await service.remove() expect(status.state).toBe('not_installed') runtimeToml = readFileSync(join(linkedManagedCodexHome, 'config.toml'), 'utf-8') @@ -58,9 +58,9 @@ describe('CodexHookService', () => { expect(runtimeToml).not.toContain(':stop:0:0') }) - it('removes legacy managed trust entries hashed before hook timeouts existed', () => { + it('removes legacy managed trust entries hashed before hook timeouts existed', async () => { const service = new CodexHookService() - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') const managedHooksPath = join(managedCodexHome, 'hooks.json') @@ -88,13 +88,13 @@ describe('CodexHookService', () => { 'utf-8' ) - expect(service.remove().state).toBe('not_installed') + expect((await service.remove()).state).toBe('not_installed') const runtimeToml = readFileSync(runtimeTomlPath, 'utf-8') expect(runtimeToml).not.toContain(':permission_request:0:0') }) - it('mirrors system Codex config while preserving runtime hook trust on hook install', () => { + it('mirrors system Codex config while preserving runtime hook trust on hook install', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') mkdirSync(systemCodexHome, { recursive: true }) writeFileSync(join(systemCodexHome, 'config.toml'), 'model = "system-model"\n', 'utf-8') @@ -114,7 +114,7 @@ describe('CodexHookService', () => { 'utf-8' ) - const status = new CodexHookService().install() + const status = await new CodexHookService().install() expect(status.state).toBe('installed') const trustConfig = readFileSync(join(managedCodexHome, 'config.toml'), 'utf-8') @@ -128,9 +128,9 @@ describe('CodexHookService', () => { it.skipIf(process.platform !== 'win32')( 'treats legacy forward-slash runtime trust keys as installed before canonicalizing on reinstall', - () => { + async () => { const service = new CodexHookService() - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') const managedHooksPath = join(managedCodexHome, 'hooks.json') @@ -154,7 +154,7 @@ describe('CodexHookService', () => { expect(legacyToml).toContain(legacyPermissionHeader) expect(service.getStatus().state).toBe('installed') - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const repairedToml = readFileSync(runtimeTomlPath, 'utf-8') expect(repairedToml).not.toContain(legacyPermissionHeader) @@ -163,13 +163,13 @@ describe('CodexHookService', () => { } ) - it('repairs duplicate managed PermissionRequest trust tables on restart install', () => { + it('repairs duplicate managed PermissionRequest trust tables on restart install', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') mkdirSync(systemCodexHome, { recursive: true }) writeFileSync(join(systemCodexHome, 'config.toml'), 'model = "system-model"\n', 'utf-8') const service = new CodexHookService() - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') const managedHooksPath = join(managedCodexHome, 'hooks.json') @@ -209,7 +209,7 @@ describe('CodexHookService', () => { // Why: preserving `enabled = false` is the repair contract; status can be // partial because the user-disabled managed hook remains disabled. - expect(['installed', 'partial']).toContain(service.install().state) + expect(['installed', 'partial']).toContain((await service.install()).state) const repairedToml = readFileSync(runtimeTomlPath, 'utf-8') expect(repairedToml.split(permissionRequestHeader)).toHaveLength(2) @@ -219,7 +219,7 @@ describe('CodexHookService', () => { expect(repairedToml).toContain('model = "system-model"') }) - it('preserves runtime-only project trust while honoring system project untrust', () => { + it('preserves runtime-only project trust while honoring system project untrust', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') mkdirSync(systemCodexHome, { recursive: true }) writeFileSync( @@ -247,7 +247,7 @@ describe('CodexHookService', () => { 'utf-8' ) - const status = new CodexHookService().install() + const status = await new CodexHookService().install() expect(status.state).toBe('installed') const trustConfig = readFileSync(join(managedCodexHome, 'config.toml'), 'utf-8') diff --git a/src/main/codex/hook-service-test-harness.ts b/src/main/codex/hook-service-test-harness.ts index 2893d3a12ba..f6a450a263d 100644 --- a/src/main/codex/hook-service-test-harness.ts +++ b/src/main/codex/hook-service-test-harness.ts @@ -7,6 +7,17 @@ import { getCodexExplicitHomeHookSourcePath, normalizeCodexHookSourcePath } from './config-toml-trust' +import { _internals as grantInternals } from './codex-hook-trust-grant' +import { _internals as rebaseInternals } from './codex-user-hook-trust-rebase' + +// Why (#16441): the grant/rebase sessions now run in-process instead of in a +// forked bundle that never existed under vitest. Without this stub these +// suites spawn the developer's real `codex app-server`, so they pass in CI +// (no codex installed) and fail on any machine that has one. Stand in for the +// missing binary so the fallback lane is exercised either way. +function stubMissingCodexBinary(): never { + throw Object.assign(new Error('spawn codex ENOENT'), { code: 'ENOENT' }) +} export type CodexHookHomes = { tmpHome: string @@ -14,6 +25,19 @@ export type CodexHookHomes = { } /** Mutable holder: fields are re-pointed at fresh temp dirs by the registered beforeEach. */ +/** Applies the stub above; for suites that build their own temp homes. */ +export function stubCodexTrustSessionsForTests(): void { + grantInternals.setGrantSessionRunner(stubMissingCodexBinary) + rebaseInternals.setSessionRunner(stubMissingCodexBinary) +} + +export function restoreCodexTrustSessionsForTests(): void { + grantInternals.setGrantSessionRunner(null) + grantInternals.resetDiagnostics() + rebaseInternals.setSessionRunner(null) + rebaseInternals.resetRetryState() +} + export function setupCodexHookHomes( homedirMock: Mock<() => string>, getPathMock: Mock<(name: string) => string> @@ -27,6 +51,7 @@ export function setupCodexHookHomes( previousUserDataPath = process.env.ORCA_USER_DATA_PATH process.env.ORCA_USER_DATA_PATH = homes.userDataDir homedirMock.mockReturnValue(homes.tmpHome) + stubCodexTrustSessionsForTests() getPathMock.mockImplementation((name: string) => { if (name === 'userData') { return homes.userDataDir @@ -36,6 +61,7 @@ export function setupCodexHookHomes( }) afterEach(() => { + restoreCodexTrustSessionsForTests() rmSync(homes.tmpHome, { recursive: true, force: true }) rmSync(homes.userDataDir, { recursive: true, force: true }) if (previousUserDataPath === undefined) { diff --git a/src/main/codex/hook-service-trust-grant.test.ts b/src/main/codex/hook-service-trust-grant.test.ts index 7814994c758..1b4ab0cd7ac 100644 --- a/src/main/codex/hook-service-trust-grant.test.ts +++ b/src/main/codex/hook-service-trust-grant.test.ts @@ -75,8 +75,8 @@ beforeEach(() => { }) afterEach(() => { - rebaseInternals.setSessionRunnerSync(null) - trustGrantInternals.setGrantSessionRunnerSync(null) + rebaseInternals.setSessionRunner(null) + trustGrantInternals.setGrantSessionRunner(null) trustGrantInternals.resetDiagnostics() codexAppServerCapabilityCache.clear() if (previousDisableTrustRpc === undefined) { @@ -123,7 +123,7 @@ function writeCodexLikeTrust(configPath: string, entries: CodexTrustEntry[]): vo function installCodexLikeGrantRunner(): ReturnType { const codexHash = (key: string): string => `sha256:codex-${parseTrustKey(key)?.eventLabel ?? 'unknown'}` - const runner = vi.fn((request: CodexHookTrustGrantRequest) => { + const runner = vi.fn(async (request: CodexHookTrustGrantRequest) => { const codexHome = request.invocation.env?.CODEX_HOME expect(codexHome).toBeTruthy() const entries: CodexTrustEntry[] = request.expectedTrustKeys.map((key) => { @@ -145,7 +145,7 @@ function installCodexLikeGrantRunner(): ReturnType { })) } }) - trustGrantInternals.setGrantSessionRunnerSync(runner) + trustGrantInternals.setGrantSessionRunner(runner) return runner } @@ -154,11 +154,11 @@ function prepareSystemHome(): void { } describe('CodexHookService app-server trust grant lane', () => { - it('treats Codex hashes as authoritative and records the verified grant', () => { + it('treats Codex hashes as authoritative and records the verified grant', async () => { prepareSystemHome() const runner = installCodexLikeGrantRunner() - const status = new CodexHookService().install() + const status = await new CodexHookService().install() expect(status.state).toBe('installed') expect(runner).toHaveBeenCalledTimes(1) @@ -177,15 +177,15 @@ describe('CodexHookService app-server trust grant lane', () => { expect(Object.keys(readCodexTrustGrantLedgerHome(managedHome)!.entries)).toHaveLength(8) }) - it('keeps config byte-stable and skips the session on a repeat ledger hit', () => { + it('keeps config byte-stable and skips the session on a repeat ledger hit', async () => { prepareSystemHome() const runner = installCodexLikeGrantRunner() const service = new CodexHookService() - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const managedHome = join(userDataDir, 'codex-runtime-home', 'home') const firstToml = readFileSync(join(managedHome, 'config.toml')) - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') expect(runner).toHaveBeenCalledTimes(1) // Why: each launch validates the binary stamp once; getStatus reuses the // just-verified grant instead of repeating PATH/version-manager scans. @@ -193,7 +193,7 @@ describe('CodexHookService app-server trust grant lane', () => { expect(readFileSync(join(managedHome, 'config.toml'))).toEqual(firstToml) }) - it('retries ledger-proven real-home trust cleanup after the hook is already gone', () => { + it('retries ledger-proven real-home trust cleanup after the hook is already gone', async () => { prepareSystemHome() const systemHome = join(tmpHome, '.codex') const hooksPath = join(systemHome, 'hooks.json') @@ -223,7 +223,7 @@ describe('CodexHookService app-server trust grant lane', () => { }) installCodexLikeGrantRunner() - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') expect(readHookTrustEntries(configPath).has(trustKey)).toBe(false) expect(readCodexTrustGrantLedgerHome(systemHome)).toBeNull() @@ -232,7 +232,7 @@ describe('CodexHookService app-server trust grant lane', () => { // Why: ordinary Windows CI tokens cannot create file symlinks without Developer Mode. it.skipIf(process.platform === 'win32')( 'keeps a real-home symlink and rebases later user trust during flag-off cleanup', - () => { + async () => { prepareSystemHome() const systemHome = join(tmpHome, '.codex') const hooksPath = join(systemHome, 'hooks.json') @@ -256,7 +256,7 @@ describe('CodexHookService app-server trust grant lane', () => { ) symlinkSync(targetPath, hooksPath) const operations: string[] = [] - rebaseInternals.setSessionRunnerSync((request) => { + rebaseInternals.setSessionRunner(async (request) => { operations.push(request.operation) if (request.operation === 'inspect-user-hook-trust') { return { @@ -273,7 +273,7 @@ describe('CodexHookService app-server trust grant lane', () => { }) installCodexLikeGrantRunner() - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') expect(lstatSync(hooksPath).isSymbolicLink()).toBe(true) expect(JSON.parse(readFileSync(targetPath, 'utf-8')).hooks.Stop).toEqual([ @@ -285,7 +285,7 @@ describe('CodexHookService app-server trust grant lane', () => { it.skipIf(process.platform === 'win32')( 'preserves restrictive real-home hooks permissions during flag-off cleanup', - () => { + async () => { prepareSystemHome() const hooksPath = join(tmpHome, '.codex', 'hooks.json') const material = getCodexManagedHookInstallMaterial() @@ -296,17 +296,17 @@ describe('CodexHookService app-server trust grant lane', () => { chmodSync(hooksPath, 0o600) installCodexLikeGrantRunner() - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') expect(statSync(hooksPath).mode & 0o777).toBe(0o600) } ) - it('does not accept a ledger hash after the recorded Codex binary stamp changes', () => { + it('does not accept a ledger hash after the recorded Codex binary stamp changes', async () => { prepareSystemHome() installCodexLikeGrantRunner() const service = new CodexHookService() - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const managedHome = join(userDataDir, 'codex-runtime-home', 'home') const ledger = readCodexTrustGrantLedgerHome(managedHome)! writeCodexTrustGrantLedgerHome(managedHome, { @@ -320,16 +320,16 @@ describe('CodexHookService app-server trust grant lane', () => { }) }) - it('upgrades self-computed trust in place without duplicate logical entries', () => { + it('upgrades self-computed trust in place without duplicate logical entries', async () => { prepareSystemHome() const service = new CodexHookService() process.env.ORCA_DISABLE_CODEX_TRUST_RPC = '1' - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const managedHome = join(userDataDir, 'codex-runtime-home', 'home') delete process.env.ORCA_DISABLE_CODEX_TRUST_RPC installCodexLikeGrantRunner() - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const upgraded = readFileSync(join(managedHome, 'config.toml'), 'utf-8') // Why: the legacy Windows fallback intentionally writes slash variants; // duplicate detection is about the normalized trust identity. @@ -350,7 +350,7 @@ describe('CodexHookService app-server trust grant lane', () => { expect(upgraded).toContain('sha256:codex-session_start') }) - it('leaves user trust byte-untouched while granting managed entries', () => { + it('leaves user trust byte-untouched while granting managed entries', async () => { prepareSystemHome() const managedHome = join(userDataDir, 'codex-runtime-home', 'home') mkdirSync(managedHome, { recursive: true }) @@ -362,35 +362,35 @@ describe('CodexHookService app-server trust grant lane', () => { writeFileSync(join(managedHome, 'config.toml'), `${userBlock}\n`) installCodexLikeGrantRunner() - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') expect(readFileSync(join(managedHome, 'config.toml'), 'utf-8')).toContain(userBlock) }) - it('keeps the forced fallback on self-computed writes', () => { + it('keeps the forced fallback on self-computed writes', async () => { prepareSystemHome() process.env.ORCA_DISABLE_CODEX_TRUST_RPC = '1' const runner = vi.fn() - trustGrantInternals.setGrantSessionRunnerSync(runner) + trustGrantInternals.setGrantSessionRunner(runner) const service = new CodexHookService() - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') expect(service.getStatus().state).toBe('installed') expect(runner).not.toHaveBeenCalled() expect(resolveCodexCommandMock).not.toHaveBeenCalled() }) - it('restores exact config bytes before fallback after a mutating RPC failure', () => { + it('restores exact config bytes before fallback after a mutating RPC failure', async () => { prepareSystemHome() const service = new CodexHookService() process.env.ORCA_DISABLE_CODEX_TRUST_RPC = '1' - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const managedHome = join(userDataDir, 'codex-runtime-home', 'home') const baseline = readFileSync(join(managedHome, 'config.toml')) delete process.env.ORCA_DISABLE_CODEX_TRUST_RPC rmSync(managedHome, { recursive: true, force: true }) trustGrantInternals.resetDiagnostics() - const runner = vi.fn((request: CodexHookTrustGrantRequest) => { + const runner = vi.fn(async (request: CodexHookTrustGrantRequest) => { const codexHome = request.invocation.env?.CODEX_HOME writeFileSync( join(codexHome!, 'config.toml'), @@ -398,9 +398,9 @@ describe('CodexHookService app-server trust grant lane', () => { ) throw new Error('transport failed after config/batchWrite') }) - trustGrantInternals.setGrantSessionRunnerSync(runner) + trustGrantInternals.setGrantSessionRunner(runner) - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') expect(runner).toHaveBeenCalledTimes(1) expect(readFileSync(join(managedHome, 'config.toml'))).toEqual(baseline) }) diff --git a/src/main/codex/hook-service-user-hook-mirroring.test.ts b/src/main/codex/hook-service-user-hook-mirroring.test.ts index d06fec0b3a7..f13953a6051 100644 --- a/src/main/codex/hook-service-user-hook-mirroring.test.ts +++ b/src/main/codex/hook-service-user-hook-mirroring.test.ts @@ -73,10 +73,10 @@ function markHookTrustDisabled(toml: string, header: string): string { } describe('CodexHookService', () => { - it('preserves mirrored user hooks when the system hooks file cannot be read', () => { + it('preserves mirrored user hooks when the system hooks file cannot be read', async () => { const service = new CodexHookService() const { systemHooksPath, managedHooksPath } = seedSystemUserHook('user-hook') - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const systemBefore = readFileSync(systemHooksPath, 'utf-8') const before = readFileSync(managedHooksPath, 'utf-8') @@ -84,7 +84,7 @@ describe('CodexHookService', () => { mkdirSync(systemHooksPath) for (const retry of [() => service.install(), () => service.refreshRuntimeUserHooks()]) { - expect(retry()).toMatchObject({ + expect(await retry()).toMatchObject({ state: 'error', detail: 'Could not read system Codex hooks.json' }) @@ -93,16 +93,16 @@ describe('CodexHookService', () => { rmSync(systemHooksPath, { recursive: true }) writeFileSync(systemHooksPath, systemBefore, 'utf-8') - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') expect(readRuntimeHookCommands(managedHooksPath)).toContain('user-hook') }) it.each(['absent', 'malformed'] as const)( 'rebuilds mirrored user hooks when the system source is %s', - (sourceState) => { + async (sourceState) => { const service = new CodexHookService() const { systemHooksPath, managedHooksPath } = seedSystemUserHook('stale-user-hook') - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') if (sourceState === 'absent') { rmSync(systemHooksPath) @@ -110,12 +110,12 @@ describe('CodexHookService', () => { writeFileSync(systemHooksPath, '{ not json', 'utf-8') } - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') expect(readRuntimeHookCommands(managedHooksPath)).not.toContain('stale-user-hook') } ) - it('mirrors trusted system user hook approvals into the runtime CODEX_HOME', () => { + it('mirrors trusted system user hook approvals into the runtime CODEX_HOME', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const systemHooksPath = join(systemCodexHome, 'hooks.json') mkdirSync(systemCodexHome, { recursive: true }) @@ -163,7 +163,7 @@ describe('CodexHookService', () => { 'utf-8' ) - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') const managedHooksPath = join(managedCodexHome, 'hooks.json') @@ -183,7 +183,7 @@ describe('CodexHookService', () => { expect(runtimeToml).not.toContain(hookTrustHeader(`${systemHooksPath}:stop:0:0`, true)) }) - it('runs managed PostToolUse status before mirrored user hooks', () => { + it('runs managed PostToolUse status before mirrored user hooks', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const systemHooksPath = join(systemCodexHome, 'hooks.json') mkdirSync(systemCodexHome, { recursive: true }) @@ -210,7 +210,7 @@ describe('CodexHookService', () => { 'utf-8' ) - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') const managedHooksPath = join(managedCodexHome, 'hooks.json') @@ -231,7 +231,7 @@ describe('CodexHookService', () => { expect(runtimeToml).not.toContain(hookTrustHeader(`${systemHooksPath}:post_tool_use:0:0`, true)) }) - it('mirrors system user hook approvals when the system trust indices are stale', () => { + it('mirrors system user hook approvals when the system trust indices are stale', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const systemHooksPath = join(systemCodexHome, 'hooks.json') mkdirSync(systemCodexHome, { recursive: true }) @@ -272,7 +272,7 @@ describe('CodexHookService', () => { 'utf-8' ) - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') const managedHooksPath = join(managedCodexHome, 'hooks.json') @@ -284,7 +284,7 @@ describe('CodexHookService', () => { expect(runtimeToml).not.toContain(hookTrustHeader(`${systemHooksPath}:stop:1:0`, true)) }) - it('skips plugin-placeholder system hooks when mirroring into runtime CODEX_HOME', () => { + it('skips plugin-placeholder system hooks when mirroring into runtime CODEX_HOME', async () => { const pluginCommands = [ 'node "${CLAUDE_PLUGIN_ROOT}/scripts/on-stop.mjs"', 'node "${CLAUDE_PLUGIN_DATA}/scripts/on-stop.mjs"', @@ -340,7 +340,7 @@ describe('CodexHookService', () => { 'utf-8' ) - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') const managedHooksPath = join(managedCodexHome, 'hooks.json') @@ -368,7 +368,7 @@ describe('CodexHookService', () => { } }) - it('mirrors compact-event user hook approvals and disabled trust entries', () => { + it('mirrors compact-event user hook approvals and disabled trust entries', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const systemHooksPath = join(systemCodexHome, 'hooks.json') mkdirSync(systemCodexHome, { recursive: true }) @@ -409,7 +409,7 @@ describe('CodexHookService', () => { 'utf-8' ) - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') const managedHooksPath = join(managedCodexHome, 'hooks.json') @@ -430,7 +430,7 @@ describe('CodexHookService', () => { expect(runtimeToml).not.toContain(hookTrustHeader(`${systemHooksPath}:post_compact:0:0`, true)) }) - it('removes runtime user hook trust after system approval is revoked', () => { + it('removes runtime user hook trust after system approval is revoked', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const systemHooksPath = join(systemCodexHome, 'hooks.json') mkdirSync(systemCodexHome, { recursive: true }) @@ -456,7 +456,7 @@ describe('CodexHookService', () => { ) const service = new CodexHookService() - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') const managedHooksPath = join(managedCodexHome, 'hooks.json') @@ -466,14 +466,14 @@ describe('CodexHookService', () => { ) writeFileSync(join(systemCodexHome, 'config.toml'), 'model = "system-model"\n', 'utf-8') - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const runtimeToml = readFileSync(join(managedCodexHome, 'config.toml'), 'utf-8') expect(runtimeToml).not.toContain(runtimeUserTrustHeader) expect(runtimeToml).toContain(hookTrustHeader(`${managedHooksPath}:stop:0:0`)) }) - it('refreshes mirrored system user hooks when the system hooks file changes', () => { + it('refreshes mirrored system user hooks when the system hooks file changes', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const systemHooksPath = join(systemCodexHome, 'hooks.json') mkdirSync(systemCodexHome, { recursive: true }) @@ -486,7 +486,7 @@ describe('CodexHookService', () => { ) const service = new CodexHookService() - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') writeFileSync( systemHooksPath, @@ -495,7 +495,7 @@ describe('CodexHookService', () => { })}\n`, 'utf-8' ) - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const managedHooksPath = join(homes.userDataDir, 'codex-runtime-home', 'home', 'hooks.json') const runtimeHooks = JSON.parse(readFileSync(managedHooksPath, 'utf-8')) as { @@ -509,7 +509,7 @@ describe('CodexHookService', () => { expect(stopCommands).not.toContain('user-hook-old') }) - it('refreshes runtime user hooks without installing Orca-managed hooks', () => { + it('refreshes runtime user hooks without installing Orca-managed hooks', async () => { const systemCodexHome = join(homes.tmpHome, '.codex') const systemHooksPath = join(systemCodexHome, 'hooks.json') mkdirSync(systemCodexHome, { recursive: true }) @@ -537,7 +537,7 @@ describe('CodexHookService', () => { ) const service = new CodexHookService() - expect(service.install().state).toBe('installed') + expect((await service.install()).state).toBe('installed') const managedCodexHome = join(homes.userDataDir, 'codex-runtime-home', 'home') const managedHooksPath = join(managedCodexHome, 'hooks.json') const runtimeTomlPath = join(managedCodexHome, 'config.toml') @@ -559,7 +559,7 @@ describe('CodexHookService', () => { 'utf-8' ) - const status = service.refreshRuntimeUserHooks() + const status = await service.refreshRuntimeUserHooks() expect(status.state).toBe('not_installed') expect(status.managedHooksPresent).toBe(false) diff --git a/src/main/codex/hook-service-wsl-runtime.test.ts b/src/main/codex/hook-service-wsl-runtime.test.ts index dbc51f39b35..8e892e69147 100644 --- a/src/main/codex/hook-service-wsl-runtime.test.ts +++ b/src/main/codex/hook-service-wsl-runtime.test.ts @@ -82,7 +82,7 @@ function expectedManagedCommand(scriptPath: string): string { } describe('Codex WSL runtime hook install', () => { - it('plans WSL hook files with Linux command and trust paths', () => { + it('plans WSL hook files with Linux command and trust paths', async () => { const runtimeHome = '\\\\wsl.localhost\\Ubuntu\\home\\alice\\.local\\share\\orca\\codex-runtime-home\\home' @@ -100,7 +100,7 @@ describe('Codex WSL runtime hook install', () => { }) }) - it('plans WSL hooks when the distro home is mounted on a Windows drive', () => { + it('plans WSL hooks when the distro home is mounted on a Windows drive', async () => { const runtimeHome = 'D:\\wsl-home\\.local\\share\\orca\\codex-runtime-home\\home' expect( @@ -121,7 +121,7 @@ describe('Codex WSL runtime hook install', () => { }) }) - it('uses WSL-canonical paths for hook commands and trust keys', () => { + it('uses WSL-canonical paths for hook commands and trust keys', async () => { const runtimeHome = '\\\\wsl.localhost\\Ubuntu\\home\\alias\\.local\\share\\orca\\codex-runtime-home\\home' const canonicalHome = '/home/alice/.local/share/orca/codex-runtime-home/home' @@ -141,7 +141,7 @@ describe('Codex WSL runtime hook install', () => { expect(plan?.configPath).toBe(pathWin32.join(runtimeHome, 'hooks.json')) }) - it('removes managed trust when the WSL canonical path changes', () => { + it('removes managed trust when the WSL canonical path changes', async () => { const plan = createTestPlan() writeFileSync(plan.configPath, '{"hooks":{}}\n', 'utf-8') writeFileSync(plan.tomlPath, '', 'utf-8') @@ -151,7 +151,7 @@ describe('Codex WSL runtime hook install', () => { commandScriptPath: '/old/home/.orca/agent-hooks/codex-hook.sh', trustConfigPath: '/old/home/hooks.json' } - expect(_internals.installManagedHooksIntoWslRuntime(oldPlan).state).toBe('installed') + expect((await _internals.installManagedHooksIntoWslRuntime(oldPlan)).state).toBe('installed') const oldCommand = expectedManagedCommand(oldPlan.commandScriptPath) const oldKey = computeTrustKey(getManagedTrustEntry(oldPlan, oldCommand)) @@ -160,7 +160,7 @@ describe('Codex WSL runtime hook install', () => { commandScriptPath: '/new/home/.orca/agent-hooks/codex-hook.sh', trustConfigPath: '/new/home/hooks.json' } - expect(_internals.installManagedHooksIntoWslRuntime(newPlan).state).toBe('installed') + expect((await _internals.installManagedHooksIntoWslRuntime(newPlan)).state).toBe('installed') const newCommand = expectedManagedCommand(newPlan.commandScriptPath) const newKey = computeTrustKey(getManagedTrustEntry(newPlan, newCommand)) const trustEntries = readHookTrustEntries(plan.tomlPath) @@ -171,7 +171,7 @@ describe('Codex WSL runtime hook install', () => { it.skipIf(process.platform === 'win32')( 'drains stdin when the WSL runtime script is missing', - () => { + async () => { const basePlan = createTestPlan() const plan = { ...basePlan, @@ -180,7 +180,7 @@ describe('Codex WSL runtime hook install', () => { writeFileSync(plan.configPath, '{"hooks":{}}\n', 'utf-8') writeFileSync(plan.tomlPath, '', 'utf-8') - expect(_internals.installManagedHooksIntoWslRuntime(plan).state).toBe('installed') + expect((await _internals.installManagedHooksIntoWslRuntime(plan)).state).toBe('installed') const installed = JSON.parse(readFileSync(plan.configPath, 'utf-8')) as HooksConfig const command = installed.hooks.UserPromptSubmit[0]?.hooks?.[0]?.command expect(command).toBe(expectedManagedCommand(plan.commandScriptPath)) @@ -193,20 +193,20 @@ describe('Codex WSL runtime hook install', () => { } ) - it('sweeps all managed WSL trust for disable or confirmed absence', () => { + it('sweeps all managed WSL trust for disable or confirmed absence', async () => { // Why: disable and confirmed absence intentionally pass []. Transient // unavailability must NOT use this path — last known-good trust remains. const plan = createTestPlan() writeFileSync(plan.configPath, '{"hooks":{}}\n', 'utf-8') writeFileSync(plan.tomlPath, '', 'utf-8') - expect(_internals.installManagedHooksIntoWslRuntime(plan).state).toBe('installed') + expect((await _internals.installManagedHooksIntoWslRuntime(plan)).state).toBe('installed') _internals.removeStaleWslRuntimeManagedHookTrustEntries(plan.tomlPath, []) expect(readHookTrustEntries(plan.tomlPath).size).toBe(0) }) - it('reconciles only current, conclusive WSL path settlements', () => { + it('reconciles only current, conclusive WSL path settlements', async () => { expect( _internals.getWslHookReconciliationAction({ settlement: { status: 'unavailable' }, @@ -272,7 +272,7 @@ describe('Codex WSL runtime hook install', () => { ).toBe('reinstall') }) - it('generates a POSIX hook that bridges WSL loopback failures through Windows curl', () => { + it('generates a POSIX hook that bridges WSL loopback failures through Windows curl', async () => { const script = _internals.getManagedScript('posix') expect(script).toContain('load_hook_endpoint()') expect(script).toContain('"set ORCA_AGENT_HOOK_TOKEN="*)') @@ -287,7 +287,7 @@ describe('Codex WSL runtime hook install', () => { it.skipIf(process.platform === 'win32')( 'refreshes stale hook coordinates from a Windows endpoint file', - () => { + async () => { const plan = createTestPlan() const root = dirname(plan.configPath) const endpointPath = join(root, 'endpoint.cmd') @@ -339,7 +339,7 @@ describe('Codex WSL runtime hook install', () => { it.skipIf(process.platform === 'win32')( 'uses the Windows curl discovered from the WSL PATH after loopback fails', - () => { + async () => { const plan = createTestPlan() const root = dirname(plan.configPath) const binDir = join(root, 'bin') @@ -378,7 +378,7 @@ describe('Codex WSL runtime hook install', () => { } ) - it('installs trusted WSL hooks and removes only Orca entries when disabled', () => { + it('installs trusted WSL hooks and removes only Orca entries when disabled', async () => { const plan = createTestPlan() const userCommand = '/bin/sh /home/alice/user-hook.sh' writeFileSync( @@ -408,7 +408,7 @@ describe('Codex WSL runtime hook install', () => { 'utf-8' ) - expect(_internals.installManagedHooksIntoWslRuntime(plan).state).toBe('installed') + expect((await _internals.installManagedHooksIntoWslRuntime(plan)).state).toBe('installed') const installed = JSON.parse(readFileSync(plan.configPath, 'utf-8')) as HooksConfig expect(Object.keys(installed.hooks).sort()).toEqual([...managedEvents].sort()) @@ -457,7 +457,7 @@ describe('Codex WSL runtime hook install app-server grant lane', () => { }) afterEach(() => { - trustGrantInternals.setGrantSessionRunnerSync(null) + trustGrantInternals.setGrantSessionRunner(null) trustGrantInternals.resetDiagnostics() codexAppServerCapabilityCache.clear() if (previousUserDataPath === undefined) { @@ -467,12 +467,12 @@ describe('Codex WSL runtime hook install app-server grant lane', () => { } }) - it('grants WSL managed trust through codex inside the distro instead of self-computed writes', () => { + it('grants WSL managed trust through codex inside the distro instead of self-computed writes', async () => { const plan = createTestPlan() writeFileSync(plan.configPath, '{"hooks":{}}\n', 'utf-8') writeFileSync(plan.tomlPath, '', 'utf-8') - const runner = vi.fn((request: CodexHookTrustGrantRequest) => { + const runner = vi.fn(async (request: CodexHookTrustGrantRequest) => { // Simulate codex's side: write trusted_hash blocks the way its config // writer would, then report the entries trusted. upsertHookTrustEntries( @@ -496,9 +496,9 @@ describe('Codex WSL runtime hook install app-server grant lane', () => { })) } }) - trustGrantInternals.setGrantSessionRunnerSync(runner) + trustGrantInternals.setGrantSessionRunner(runner) - expect(_internals.installManagedHooksIntoWslRuntime(plan).state).toBe('installed') + expect((await _internals.installManagedHooksIntoWslRuntime(plan)).state).toBe('installed') expect(runner).toHaveBeenCalledTimes(1) const request = runner.mock.calls[0]![0]! @@ -519,7 +519,7 @@ describe('Codex WSL runtime hook install app-server grant lane', () => { ) }) - it('keeps the unchanged self-computed lane when the WSL grant falls back', () => { + it('keeps the unchanged self-computed lane when the WSL grant falls back', async () => { const plan = createTestPlan() writeFileSync(plan.configPath, '{"hooks":{}}\n', 'utf-8') writeFileSync(plan.tomlPath, '', 'utf-8') @@ -527,9 +527,9 @@ describe('Codex WSL runtime hook install app-server grant lane', () => { const runner = vi.fn(() => { throw new Error('wsl.exe not reachable') }) - trustGrantInternals.setGrantSessionRunnerSync(runner) + trustGrantInternals.setGrantSessionRunner(runner) - expect(_internals.installManagedHooksIntoWslRuntime(plan).state).toBe('installed') + expect((await _internals.installManagedHooksIntoWslRuntime(plan)).state).toBe('installed') expect(runner).toHaveBeenCalledTimes(1) const command = expectedManagedCommand(plan.commandScriptPath) @@ -540,12 +540,12 @@ describe('Codex WSL runtime hook install app-server grant lane', () => { }) }) - it('uses the previous ledger to remove stale Codex hashes after a canonical path change', () => { + it('uses the previous ledger to remove stale Codex hashes after a canonical path change', async () => { const basePlan = createTestPlan() writeFileSync(basePlan.configPath, '{"hooks":{}}\n', 'utf-8') writeFileSync(basePlan.tomlPath, '', 'utf-8') let staleKeyExpectedRemoved: string | null = null - const runner = vi.fn((request: CodexHookTrustGrantRequest) => { + const runner = vi.fn(async (request: CodexHookTrustGrantRequest) => { if (staleKeyExpectedRemoved) { expect(readHookTrustEntries(basePlan.tomlPath).has(staleKeyExpectedRemoved)).toBe(false) } @@ -571,7 +571,7 @@ describe('Codex WSL runtime hook install app-server grant lane', () => { })) } }) - trustGrantInternals.setGrantSessionRunnerSync(runner) + trustGrantInternals.setGrantSessionRunner(runner) const oldPlan = { ...basePlan, @@ -579,7 +579,7 @@ describe('Codex WSL runtime hook install app-server grant lane', () => { trustConfigPath: '/old/home/hooks.json', linuxRuntimeHome: '/old/home' } - expect(_internals.installManagedHooksIntoWslRuntime(oldPlan).state).toBe('installed') + expect((await _internals.installManagedHooksIntoWslRuntime(oldPlan)).state).toBe('installed') const oldKey = computeTrustKey( getManagedTrustEntry(oldPlan, expectedManagedCommand(oldPlan.commandScriptPath)) ) @@ -591,7 +591,7 @@ describe('Codex WSL runtime hook install app-server grant lane', () => { trustConfigPath: '/new/home/hooks.json', linuxRuntimeHome: '/new/home' } - expect(_internals.installManagedHooksIntoWslRuntime(newPlan).state).toBe('installed') + expect((await _internals.installManagedHooksIntoWslRuntime(newPlan)).state).toBe('installed') const newKey = computeTrustKey( getManagedTrustEntry(newPlan, expectedManagedCommand(newPlan.commandScriptPath)) ) diff --git a/src/main/codex/hook-service.ts b/src/main/codex/hook-service.ts index 270906012b5..93dfaf12f41 100644 --- a/src/main/codex/hook-service.ts +++ b/src/main/codex/hook-service.ts @@ -73,7 +73,8 @@ import { snapshotCodexRuntimeHookTrustProvenance } from './hook-trust-promotion' import { grantManagedCodexHookTrust } from './codex-hook-trust-grant' -import { readCurrentCodexTrustGrantLedgerHome } from './codex-trust-grant-host' +import { runExclusivelyForCodexTrustConfig } from './codex-trust-config-mutation-queue' +import { readCurrentNativeCodexTrustGrantLedgerHome } from './codex-trust-grant-host' import { getCodexLedgerTrustedHash, readCodexTrustGrantLedgerHomeForReconciliation, @@ -585,7 +586,30 @@ function removeSystemManagedHookTrustEntries(systemHomePath: string, hooksJsonPa }) } -function cleanupLegacySystemManagedHooks(): void { +// Why (#16441): these sequences mutate the runtime config.toml *and* the +// system one — approval promotion, the system-config sync and the legacy sweep +// all touch ~/.codex/config.toml — so holding only the runtime lane still lets +// a real-home grant's capture->restore window swallow their writes. Lock order +// is always runtime-before-system; every other holder acquires it that way too. +function runExclusivelyForRuntimeAndSystemTrustConfig( + runtimeHomePath: string, + run: () => Promise +): Promise { + return runExclusivelyForCodexTrustConfig(getCodexConfigTomlPath(runtimeHomePath), () => + runExclusivelyForCodexTrustConfig(getSystemCodexConfigTomlPath(), run) + ) +} + +function cleanupLegacySystemManagedHooks(): Promise { + // Why: shares the real-home lane with ensureRealHomeCodexHookState — both + // capture, mutate and roll back the user's ~/.codex/config.toml. + return runExclusivelyForCodexTrustConfig( + getSystemCodexConfigTomlPath(), + sweepLegacySystemManagedHooks + ) +} + +async function sweepLegacySystemManagedHooks(): Promise { if (systemCodexHomeHookSweepSuppressed()) { return } @@ -650,7 +674,7 @@ function cleanupLegacySystemManagedHooks(): void { // Remove only stale Orca hook entries and preserve other managers' metadata. const hooksWritePath = resolveHooksJsonWritePath(legacyConfigPath) const previousMode = statSync(hooksWritePath).mode - mutateRealHomeHooksPreservingUserTrust({ + await mutateRealHomeHooksPreservingUserTrust({ sourcePath: legacyConfigPath, runtimeHomePath: systemHomePath, tomlPath: getSystemCodexConfigTomlPath(), @@ -717,9 +741,9 @@ function cleanupLegacyCodexProfileHooks(): void { } } -function cleanupLegacyManagedHookRepresentations(): void { +async function cleanupLegacyManagedHookRepresentations(): Promise { try { - cleanupLegacySystemManagedHooks() + await cleanupLegacySystemManagedHooks() cleanupLegacyCodexProfileHooks() } catch (error) { console.warn('[codex-hook-service] failed to clean legacy Codex hooks', error) @@ -859,9 +883,20 @@ function getManagedScript(target: 'local' | 'posix' = 'local'): string { ].join('\n') } +// Why (#16441): the grant inside awaits a codex app-server session, so a +// concurrent pane launch could write this config.toml between this run's +// capture and its restore. One lane per file keeps the sequence atomic. function installManagedHooksIntoWslRuntime( plan: CodexWslRuntimeHookInstallPlan -): AgentHookInstallStatus { +): Promise { + return runExclusivelyForCodexTrustConfig(plan.tomlPath, () => + installManagedHooksIntoWslRuntimeExclusively(plan) + ) +} + +async function installManagedHooksIntoWslRuntimeExclusively( + plan: CodexWslRuntimeHookInstallPlan +): Promise { const config = readHooksJson(plan.configPath) if (!config) { return { @@ -924,7 +959,7 @@ function installManagedHooksIntoWslRuntime( trustEntries, previousLedgerHome ? [previousLedgerHome] : [] ) - const grant = grantManagedCodexHookTrust({ + const grant = await grantManagedCodexHookTrust({ runtimeHomePath, tomlPath: plan.tomlPath, managedCommand: command, @@ -1050,17 +1085,24 @@ export class CodexHookService { return generation } - installForRuntimeHome( + async installForRuntimeHome( runtimeHomePath: string | null | undefined, target?: CodexWslRuntimeHookTarget - ): AgentHookInstallStatus | null { + ): Promise { const generation = this.supersedeWslReconciliation(runtimeHomePath) let installedTrustConfigPath: string | null = null - // Why: JS is single-threaded, so the synchronous install below finishes - // before any async `wsl.exe` settlement callback runs — this flag is - // always set by the time the callback reads it. let installSucceeded = false - const onCanonicalPathSettled = (settlement: WslCanonicalPathSettlement): void => { + // Why: the install below now awaits a codex app-server session, so a + // settlement callback can land mid-install. This gate keeps reconciliation + // reading the finished install's flags, as it did when the install was + // synchronous and no callback could interleave with it. + let markPrimaryInstallSettled!: () => void + let reconciliationChain = new Promise((resolve) => { + markPrimaryInstallSettled = resolve + }) + const reconcileSettledWslCanonicalPath = async ( + settlement: WslCanonicalPathSettlement + ): Promise => { if (!runtimeHomePath) { return } @@ -1097,7 +1139,7 @@ export class CodexHookService { if (!resolvedPlan) { return } - const status = installManagedHooksIntoWslRuntime(resolvedPlan) + const status = await installManagedHooksIntoWslRuntime(resolvedPlan) if (status.state === 'error') { console.warn('[codex-hook-service] failed to reconcile WSL hook path', status.detail) return @@ -1105,6 +1147,13 @@ export class CodexHookService { installedTrustConfigPath = resolvedPlan.trustConfigPath installSucceeded = status.state === 'installed' } + const onCanonicalPathSettled = (settlement: WslCanonicalPathSettlement): void => { + const run = (): Promise => reconcileSettledWslCanonicalPath(settlement) + reconciliationChain = reconciliationChain.then(run, run) + void reconciliationChain.catch((error: unknown) => { + console.warn('[codex-hook-service] failed to reconcile WSL hook path', error) + }) + } const wslPlan = createCodexWslRuntimeHookInstallPlan( runtimeHomePath, target, @@ -1112,9 +1161,13 @@ export class CodexHookService { onCanonicalPathSettled ) installedTrustConfigPath = wslPlan?.trustConfigPath ?? null - const status = wslPlan ? installManagedHooksIntoWslRuntime(wslPlan) : null - installSucceeded = status?.state === 'installed' - return status + try { + const status = wslPlan ? await installManagedHooksIntoWslRuntime(wslPlan) : null + installSucceeded = status?.state === 'installed' + return status + } finally { + markPrimaryInstallSettled() + } } refreshRuntimeUserHooksForRuntimeHome( @@ -1169,7 +1222,7 @@ export class CodexHookService { // hashes or wrote fallback hashes. Re-resolving PATH here doubles sync launch work. const ledgerHome = recentGrantEntries === null - ? readCurrentCodexTrustGrantLedgerHome(runtimeHomePath, { kind: 'native' }) + ? readCurrentNativeCodexTrustGrantLedgerHome(runtimeHomePath) : null const recentGrantHashes = new Map() for (const entry of recentGrantEntries ?? []) { @@ -1273,7 +1326,16 @@ export class CodexHookService { // Why: runtimeHomePath defaults to the shared managed mirror, but a managed // account launching against its own self-contained CODEX_HOME passes that // per-account home so hooks.json/config.toml/trust land where codex reads. - install(runtimeHomePath: string = getOrcaManagedCodexHomePath()): AgentHookInstallStatus { + install( + runtimeHomePath: string = getOrcaManagedCodexHomePath() + ): Promise { + // Why: same lane as the grant it performs — see installManagedHooksIntoWslRuntime. + return runExclusivelyForRuntimeAndSystemTrustConfig(runtimeHomePath, () => + this.installExclusively(runtimeHomePath) + ) + } + + private async installExclusively(runtimeHomePath: string): Promise { const configPath = getConfigPath(runtimeHomePath) const scriptPath = getManagedScriptPath() // Why: must run before this install rewrites hooks.json/config.toml — @@ -1375,7 +1437,7 @@ export class CodexHookService { // then carry Codex's verbatim hashes into stale cleanup so it cannot // delete what Codex just wrote. Mirrored user trust keeps its existing // verbatim-carry lane either way. - const grant = grantManagedCodexHookTrust({ + const grant = await grantManagedCodexHookTrust({ runtimeHomePath, tomlPath, managedCommand: command, @@ -1410,12 +1472,7 @@ export class CodexHookService { } } snapshotCodexRuntimeHookTrustProvenance(runtimeHomePath) - try { - cleanupLegacySystemManagedHooks() - cleanupLegacyCodexProfileHooks() - } catch (error) { - console.warn('[codex-hook-service] failed to clean legacy Codex hooks', error) - } + await cleanupLegacyManagedHookRepresentations() return this.getStatusAfterInstall(recentGrantEntries, runtimeHomePath) } @@ -1535,7 +1592,15 @@ export class CodexHookService { refreshRuntimeUserHooks( runtimeHomePath: string = getOrcaManagedCodexHomePath() - ): AgentHookInstallStatus { + ): Promise { + return runExclusivelyForRuntimeAndSystemTrustConfig(runtimeHomePath, () => + this.refreshRuntimeUserHooksExclusively(runtimeHomePath) + ) + } + + private async refreshRuntimeUserHooksExclusively( + runtimeHomePath: string + ): Promise { const configPath = getConfigPath(runtimeHomePath) // Why: same as install() — capture in-Orca approvals before this refresh // rewrites the runtime files they are keyed against. @@ -1543,7 +1608,7 @@ export class CodexHookService { const config = readHooksJson(configPath) if (!config) { // Why: disabled launch prep once called remove(); preserve that legacy cleanup even when runtime hooks.json is malformed. - cleanupLegacyManagedHookRepresentations() + await cleanupLegacyManagedHookRepresentations() return { agent: 'codex', state: 'error', @@ -1592,17 +1657,23 @@ export class CodexHookService { } snapshotCodexRuntimeHookTrustProvenance(runtimeHomePath) - cleanupLegacyManagedHookRepresentations() + await cleanupLegacyManagedHookRepresentations() return this.getStatus(runtimeHomePath) } - remove(): AgentHookInstallStatus { + remove(): Promise { + return runExclusivelyForRuntimeAndSystemTrustConfig(getOrcaManagedCodexHomePath(), () => + this.removeExclusively() + ) + } + + private async removeExclusively(): Promise { const configPath = getConfigPath() const configExists = existsSync(configPath) const config = readHooksJson(configPath) if (!config) { // Why: a malformed hooks.json shouldn't strand old hooks in ~/.codex or the legacy profile after disabling. - cleanupLegacyManagedHookRepresentations() + await cleanupLegacyManagedHookRepresentations() return { agent: 'codex', state: 'error', @@ -1635,7 +1706,7 @@ export class CodexHookService { // Why: drop trust entries so config.toml doesn't accumulate dead [hooks.state] blocks across install/remove cycles. removeRuntimeManagedHookTrustEntries(configPath) - cleanupLegacyManagedHookRepresentations() + await cleanupLegacyManagedHookRepresentations() return this.getStatus() } diff --git a/src/main/codex/hook-trust-promotion.test.ts b/src/main/codex/hook-trust-promotion.test.ts index 0f65b73b515..56a772bc1f4 100644 --- a/src/main/codex/hook-trust-promotion.test.ts +++ b/src/main/codex/hook-trust-promotion.test.ts @@ -39,6 +39,10 @@ vi.mock('os', async (importOriginal) => { } }) +import { + restoreCodexTrustSessionsForTests, + stubCodexTrustSessionsForTests +} from './hook-service-test-harness' import { CodexHookService } from './hook-service' let tmpHome: string @@ -50,6 +54,7 @@ beforeEach(() => { userDataDir = mkdtempSync(join(tmpdir(), 'orca-codex-user-data-')) previousUserDataPath = process.env.ORCA_USER_DATA_PATH process.env.ORCA_USER_DATA_PATH = userDataDir + stubCodexTrustSessionsForTests() homedirMock.mockReturnValue(tmpHome) getPathMock.mockImplementation((name: string) => { if (name === 'userData') { @@ -60,6 +65,7 @@ beforeEach(() => { }) afterEach(() => { + restoreCodexTrustSessionsForTests() rmSync(tmpHome, { recursive: true, force: true }) rmSync(userDataDir, { recursive: true, force: true }) if (previousUserDataPath === undefined) { @@ -130,7 +136,7 @@ function readSystemToml(): string { } describe('codex hook trust write-back promotion', () => { - it('mirrors default-home trust when the .codex directory is a symlink', () => { + it('mirrors default-home trust when the .codex directory is a symlink', async () => { const targetHome = join(tmpHome, 'dotfiles-codex') mkdirSync(targetHome) symlinkSync(targetHome, systemCodexDir(), process.platform === 'win32' ? 'junction' : 'dir') @@ -141,7 +147,7 @@ describe('codex hook trust write-back promotion', () => { upsertHookTrustEntriesInContent('', [systemEntry]) ) - expect(new CodexHookService().install().state).toBe('installed') + expect((await new CodexHookService().install()).state).toBe('installed') const runtimeTrust = readHookTrustEntries(join(runtimeHomeDir(), 'config.toml')) expect(runtimeTrust.get(computeTrustKey(runtimeUserStopEntry()))?.trustedHash).toBe( @@ -149,10 +155,10 @@ describe('codex hook trust write-back promotion', () => { ) }) - it('keeps an in-Orca approval of a user hook across launches and promotes it to ~/.codex', () => { + it('keeps an in-Orca approval of a user hook across launches and promotes it to ~/.codex', async () => { writeSystemUserHook() const service = new CodexHookService() - service.install() + await service.install() // The mirrored user hook has no system trust yet, so no runtime trust // entry exists for it — Codex would show it as pending review. @@ -163,7 +169,7 @@ describe('codex hook trust write-back promotion', () => { simulateCodexApproval(runtimeUserStopEntry()) const approvedHash = computeTrustedHash(runtimeUserStopEntry()) - service.install() + await service.install() // Approval survives the relaunch instead of being wiped as stale… expect(readHookTrustEntries(runtimeTomlPath).get(approvalKey)?.trustedHash).toBe(approvedHash) @@ -177,15 +183,15 @@ describe('codex hook trust write-back promotion', () => { // Steady state: another launch with no external changes rewrites nothing. const systemTomlAfterPromotion = readSystemToml() const runtimeTomlAfterPromotion = readFileSync(runtimeTomlPath, 'utf-8') - service.install() + await service.install() expect(readSystemToml()).toBe(systemTomlAfterPromotion) expect(readFileSync(runtimeTomlPath, 'utf-8')).toBe(runtimeTomlAfterPromotion) }) - it('never promotes trust for the Orca-managed status hook into ~/.codex', () => { + it('never promotes trust for the Orca-managed status hook into ~/.codex', async () => { writeSystemUserHook() const service = new CodexHookService() - service.install() + await service.install() // Simulate Codex rewriting the managed Stop hook's trust entry (as an // approval after hash drift would). @@ -206,20 +212,20 @@ describe('codex hook trust write-back promotion', () => { { hash: 'sha256:codex-corrected-managed-hash' } ) - service.install() + await service.install() expect(readSystemToml()).not.toContain(managedCommand) expect(readSystemToml()).not.toContain('codex-corrected-managed-hash') }) - it('does not resurrect trust the user revoked in ~/.codex/config.toml', () => { + it('does not resurrect trust the user revoked in ~/.codex/config.toml', async () => { writeSystemUserHook() // Pre-trust the hook in the system config, as a terminal Codex session would. const systemTomlPath = join(systemCodexDir(), 'config.toml') writeFileSync(systemTomlPath, upsertHookTrustEntriesInContent('', [systemUserStopEntry()])) const service = new CodexHookService() - service.install() + await service.install() const runtimeTomlPath = join(runtimeHomeDir(), 'config.toml') const approvalKey = computeTrustKey(runtimeUserStopEntry()) @@ -227,19 +233,19 @@ describe('codex hook trust write-back promotion', () => { // User revokes in the system config; the runtime copy must not win. writeFileSync(systemTomlPath, '') - service.install() + await service.install() expect(readHookTrustEntries(runtimeTomlPath).get(approvalKey)).toBeUndefined() expect(readSystemToml()).not.toContain('[hooks.state.') }) - it('promotes an in-Orca disable of a mirrored user hook back to the system config', () => { + it('promotes an in-Orca disable of a mirrored user hook back to the system config', async () => { writeSystemUserHook() const systemTomlPath = join(systemCodexDir(), 'config.toml') writeFileSync(systemTomlPath, upsertHookTrustEntriesInContent('', [systemUserStopEntry()])) const service = new CodexHookService() - service.install() + await service.install() // User disables the hook via /hooks inside Orca-launched Codex. const runtimeTomlPath = join(runtimeHomeDir(), 'config.toml') @@ -251,7 +257,7 @@ describe('codex hook trust write-back promotion', () => { upsertHookTrustEntriesInContent(runtimeToml, [{ ...runtimeUserStopEntry(), enabled: false }]) ) - service.install() + await service.install() const systemState = readHookTrustEntries(systemTomlPath).get( computeTrustKey(systemUserStopEntry()) @@ -261,17 +267,17 @@ describe('codex hook trust write-back promotion', () => { expect(readHookTrustEntries(runtimeTomlPath).get(approvalKey)?.enabled).toBe(false) }) - it('carries a Codex-written hash verbatim when it differs from the reproduced hash', () => { + it('carries a Codex-written hash verbatim when it differs from the reproduced hash', async () => { // Simulates Codex changing its trust hash algorithm: the approval hash in // the runtime config no longer matches computeTrustedHash's output. writeSystemUserHook() const service = new CodexHookService() - service.install() + await service.install() const driftedHash = 'sha256:codex-next-gen-hash-orca-cannot-reproduce' simulateCodexApproval(runtimeUserStopEntry(), { hash: driftedHash }) - service.install() + await service.install() const runtimeTomlPath = join(runtimeHomeDir(), 'config.toml') const approvalKey = computeTrustKey(runtimeUserStopEntry()) @@ -283,24 +289,24 @@ describe('codex hook trust write-back promotion', () => { ).toBe(driftedHash) // And the launch after that still keeps it. - service.install() + await service.install() expect(readHookTrustEntries(runtimeTomlPath).get(approvalKey)?.trustedHash).toBe(driftedHash) }) - it('does not touch ~/.codex on the first launch after upgrading (no provenance yet)', () => { + it('does not touch ~/.codex on the first launch after upgrading (no provenance yet)', async () => { // Simulates an existing install: runtime home fully materialized by a // build without provenance snapshots, managed hooks only. const service = new CodexHookService() - service.install() + await service.install() rmSync(join(runtimeHomeDir(), '.orca-hook-trust-provenance.json'), { force: true }) - service.install() + await service.install() expect(readSystemToml()).toBe('') expect(existsSync(join(systemCodexDir(), 'config.toml'))).toBe(false) }) - it('re-promoting mirrored trust without provenance is a no-op on ~/.codex', () => { + it('re-promoting mirrored trust without provenance is a no-op on ~/.codex', async () => { // Existing install with a system-trusted user hook, upgraded to this // build: the mirrored runtime entry has no provenance, so promotion must // sit out this launch and leave the system config byte-identical. @@ -308,16 +314,16 @@ describe('codex hook trust write-back promotion', () => { const systemTomlPath = join(systemCodexDir(), 'config.toml') writeFileSync(systemTomlPath, upsertHookTrustEntriesInContent('', [systemUserStopEntry()])) const service = new CodexHookService() - service.install() + await service.install() rmSync(join(runtimeHomeDir(), '.orca-hook-trust-provenance.json'), { force: true }) const systemTomlBefore = readSystemToml() - service.install() + await service.install() expect(readSystemToml()).toBe(systemTomlBefore) }) - it('does not resurrect trust revoked in ~/.codex before the first provenance snapshot', () => { + it('does not resurrect trust revoked in ~/.codex before the first provenance snapshot', async () => { // Old build mirrored a system-trusted hook into the runtime home; the // user then revoked it in ~/.codex/config.toml and upgraded to this // build. The stale runtime mirror must not be mistaken for an approval. @@ -325,11 +331,11 @@ describe('codex hook trust write-back promotion', () => { const systemTomlPath = join(systemCodexDir(), 'config.toml') writeFileSync(systemTomlPath, upsertHookTrustEntriesInContent('', [systemUserStopEntry()])) const service = new CodexHookService() - service.install() + await service.install() rmSync(join(runtimeHomeDir(), '.orca-hook-trust-provenance.json'), { force: true }) writeFileSync(systemTomlPath, '') - service.install() + await service.install() expect(readSystemToml()).not.toContain('[hooks.state.') expect( @@ -339,21 +345,21 @@ describe('codex hook trust write-back promotion', () => { ).toBeUndefined() }) - it('does not flip a hook the user disabled in ~/.codex back to enabled after upgrading', () => { + it('does not flip a hook the user disabled in ~/.codex back to enabled after upgrading', async () => { // Old build mirrored the hook enabled=true; the user then set // enabled = false in ~/.codex/config.toml and upgraded to this build. writeSystemUserHook() const systemTomlPath = join(systemCodexDir(), 'config.toml') writeFileSync(systemTomlPath, upsertHookTrustEntriesInContent('', [systemUserStopEntry()])) const service = new CodexHookService() - service.install() + await service.install() rmSync(join(runtimeHomeDir(), '.orca-hook-trust-provenance.json'), { force: true }) writeFileSync( systemTomlPath, upsertHookTrustEntriesInContent('', [{ ...systemUserStopEntry(), enabled: false }]) ) - service.install() + await service.install() expect( readHookTrustEntries(systemTomlPath).get(computeTrustKey(systemUserStopEntry()))?.enabled @@ -365,13 +371,13 @@ describe('codex hook trust write-back promotion', () => { ).toBe(false) }) - it('promotes one approval to every identical system hook collapsed by deduping', () => { + it('promotes one approval to every identical system hook collapsed by deduping', async () => { writeSystemUserHook([USER_HOOK_COMMAND, USER_HOOK_COMMAND]) const service = new CodexHookService() - service.install() + await service.install() simulateCodexApproval(runtimeUserStopEntry()) - service.install() + await service.install() const systemTrust = readHookTrustEntries(join(systemCodexDir(), 'config.toml')) const approvedHash = computeTrustedHash(runtimeUserStopEntry()) @@ -379,16 +385,16 @@ describe('codex hook trust write-back promotion', () => { expect(systemTrust.get(computeTrustKey(systemUserStopEntry(1)))?.trustedHash).toBe(approvedHash) }) - it('skips promotion when the approved hook no longer exists in ~/.codex/hooks.json', () => { + it('skips promotion when the approved hook no longer exists in ~/.codex/hooks.json', async () => { writeSystemUserHook() const service = new CodexHookService() - service.install() + await service.install() simulateCodexApproval(runtimeUserStopEntry()) // User deletes the hook from their system hooks.json before relaunching. writeFileSync(join(systemCodexDir(), 'hooks.json'), JSON.stringify({ hooks: {} })) - service.install() + await service.install() expect(readSystemToml()).not.toContain('[hooks.state.') // The runtime copy of the deleted hook (and its approval) is cleaned up. @@ -398,10 +404,10 @@ describe('codex hook trust write-back promotion', () => { ).toBeUndefined() }) - it('promotes approvals recorded while status hooks are disabled (refresh path)', () => { + it('promotes approvals recorded while status hooks are disabled (refresh path)', async () => { writeSystemUserHook() const service = new CodexHookService() - service.refreshRuntimeUserHooks() + await service.refreshRuntimeUserHooks() // Without the managed status hook, the mirrored user hook sits at group 0. const refreshedRuntimeEntry: CodexTrustEntry = { @@ -411,7 +417,7 @@ describe('codex hook trust write-back promotion', () => { simulateCodexApproval(refreshedRuntimeEntry) const approvedHash = computeTrustedHash(refreshedRuntimeEntry) - service.refreshRuntimeUserHooks() + await service.refreshRuntimeUserHooks() expect( readHookTrustEntries(join(systemCodexDir(), 'config.toml')).get( diff --git a/src/main/codex/managed-home-shell-preflight.test.ts b/src/main/codex/managed-home-shell-preflight.test.ts index 38cd9b7d146..8cc519d3c79 100644 --- a/src/main/codex/managed-home-shell-preflight.test.ts +++ b/src/main/codex/managed-home-shell-preflight.test.ts @@ -35,7 +35,7 @@ describe('managed Codex shell preflight', () => { ).toBe(home) }) - it('accepts a marker-proven account home and installs only while hooks are enabled', () => { + it('accepts a marker-proven account home and installs only while hooks are enabled', async () => { const userDataPath = makeRoot() const home = join(userDataPath, 'codex-accounts', 'account-1', 'home') mkdirSync(home, { recursive: true }) @@ -50,12 +50,22 @@ describe('managed Codex shell preflight', () => { const env = { CODEX_HOME: home, ORCA_CODEX_HOME: home } expect( - prepareManagedCodexHomeBeforeShellLaunch({ userDataPath, hooksEnabled: true, env, install }) + await prepareManagedCodexHomeBeforeShellLaunch({ + userDataPath, + hooksEnabled: true, + env, + install + }) ).toMatchObject({ state: 'installed' }) expect(install).toHaveBeenCalledWith(home) expect( - prepareManagedCodexHomeBeforeShellLaunch({ userDataPath, hooksEnabled: false, env, install }) + await prepareManagedCodexHomeBeforeShellLaunch({ + userDataPath, + hooksEnabled: false, + env, + install + }) ).toBeNull() expect(install).toHaveBeenCalledTimes(1) }) diff --git a/src/main/codex/managed-home-shell-preflight.ts b/src/main/codex/managed-home-shell-preflight.ts index 35dd3a68a70..d94b31e23d7 100644 --- a/src/main/codex/managed-home-shell-preflight.ts +++ b/src/main/codex/managed-home-shell-preflight.ts @@ -77,12 +77,20 @@ export function resolveManagedCodexShellPreflightHome( return resolveAccountManagedHome(codexHome, userDataPath) } -export function prepareManagedCodexHomeBeforeShellLaunch(args: { +/** + * Shell-startup preflight for a managed CODEX_HOME. + * + * Async because the Codex install awaits an app-server trust-grant session + * in-process. The old lane forked that session through spawnSync purely to + * borrow an event loop; the CLI already has one, so awaiting here removes a + * whole ELECTRON_RUN_AS_NODE process from every managed-home shell launch. + */ +export async function prepareManagedCodexHomeBeforeShellLaunch(args: { env?: ShellPreflightEnvironment userDataPath: string hooksEnabled: boolean - install?: (runtimeHomePath: string) => AgentHookInstallStatus -}): AgentHookInstallStatus | null { + install?: (runtimeHomePath: string) => AgentHookInstallStatus | Promise +}): Promise { if (!args.hooksEnabled) { return null } diff --git a/src/main/codex/retained-codex-hook-state.test.ts b/src/main/codex/retained-codex-hook-state.test.ts index 642b8bc7476..4dcdb9dc9a0 100644 --- a/src/main/codex/retained-codex-hook-state.test.ts +++ b/src/main/codex/retained-codex-hook-state.test.ts @@ -13,11 +13,11 @@ function status(state: 'installed' | 'not_installed' | 'error'): AgentHookInstal } describe('retained Codex hook state', () => { - it('repairs Orca hooks before a retained shell can launch Codex', () => { + it('repairs Orca hooks before a retained shell can launch Codex', async () => { const install = vi.fn(() => status('installed')) const refreshRuntimeUserHooks = vi.fn(() => status('not_installed')) - reconcileRetainedCodexHookHomes({ + await reconcileRetainedCodexHookHomes({ hookService: { install, refreshRuntimeUserHooks }, hooksEnabled: true, runtimeHomePaths: ['/orca/shared-home', '/orca/account-home'] @@ -29,11 +29,11 @@ describe('retained Codex hook state', () => { expect(refreshRuntimeUserHooks).not.toHaveBeenCalled() }) - it('removes only Orca hooks from retained homes when hooks are disabled', () => { + it('removes only Orca hooks from retained homes when hooks are disabled', async () => { const install = vi.fn(() => status('installed')) const refreshRuntimeUserHooks = vi.fn(() => status('not_installed')) - reconcileRetainedCodexHookHomes({ + await reconcileRetainedCodexHookHomes({ hookService: { install, refreshRuntimeUserHooks }, hooksEnabled: false, runtimeHomePaths: ['/orca/shared-home'] diff --git a/src/main/codex/retained-codex-hook-state.ts b/src/main/codex/retained-codex-hook-state.ts index ccdf8c86af3..f3e6304e390 100644 --- a/src/main/codex/retained-codex-hook-state.ts +++ b/src/main/codex/retained-codex-hook-state.ts @@ -1,20 +1,31 @@ import type { AgentHookInstallStatus } from '../../shared/agent-hook-types' type RetainedCodexHookService = { - install: (runtimeHomePath: string) => AgentHookInstallStatus - refreshRuntimeUserHooks: (runtimeHomePath: string) => AgentHookInstallStatus + install: (runtimeHomePath: string) => AgentHookInstallStatus | Promise + refreshRuntimeUserHooks: ( + runtimeHomePath: string + ) => AgentHookInstallStatus | Promise } -export function reconcileRetainedCodexHookHomes(args: { +/** + * Repairs the hook state of Codex homes that retained shells still point at. + * + * Why not on the startup critical path (#16441): each home can run a codex + * app-server trust-grant session, so N retained homes used to mean N sequential + * multi-second blocks before the first window could paint. Callers start this + * and move on — the repair only matters before a retained shell's next Codex + * invocation, which cannot happen until the daemon provider is already serving. + */ +export async function reconcileRetainedCodexHookHomes(args: { hookService: RetainedCodexHookService hooksEnabled: boolean runtimeHomePaths: readonly string[] -}): void { +}): Promise { for (const runtimeHomePath of args.runtimeHomePaths) { try { const status = args.hooksEnabled - ? args.hookService.install(runtimeHomePath) - : args.hookService.refreshRuntimeUserHooks(runtimeHomePath) + ? await args.hookService.install(runtimeHomePath) + : await args.hookService.refreshRuntimeUserHooks(runtimeHomePath) if (status.state === 'error') { console.warn('[codex-hook-service] failed to reconcile retained Codex home', status.detail) } diff --git a/src/main/index.ts b/src/main/index.ts index 0f7e44742a3..249700bcd16 100644 --- a/src/main/index.ts +++ b/src/main/index.ts @@ -1073,12 +1073,17 @@ function startTerminalRuntimeStartupServices(): WindowsDesktopStartupServices { if (livePtyIds) { reconcileCodexPaneAccountsWithLivePtys(livePtyIds) const settings = store?.getSettings() - reconcileRetainedCodexHookHomes({ + // Why (#16441): each retained home can run a codex app-server grant + // session. Awaiting them here delayed the first window by N sessions; + // a retained shell cannot invoke Codex before this provider serves. + void reconcileRetainedCodexHookHomes({ hookService: codexHookService, hooksEnabled: isAgentStatusHooksEnabled(settings) && settings?.disabledTuiAgents.includes('codex') !== true, runtimeHomePaths: codexRuntimeHome.getRetainedHostCodexHookHomePaths(livePtyIds) + }).catch((error: unknown) => { + console.warn('[codex-hook-service] retained Codex home reconcile failed:', error) }) } } @@ -1139,24 +1144,24 @@ function bindTerminalRuntimeStartupServices( localPtyProviderStartupReady = services.then((value) => value.localPtyProviderReady) } -function prepareCodexRuntimeHomeForLaunch( +async function prepareCodexRuntimeHomeForLaunch( target?: CodexAccountSelectionTarget, launchEnv?: NodeJS.ProcessEnv, launchContext?: CodexHomeLaunchContext -): string | null { +): Promise { if ( target?.runtime !== 'wsl' && launchContext?.launchAgent === 'codex' && launchContext.workspacePath ) { try { - // Why: renderer quick-launch cannot await trust IPC before its PTY mounts; launch prep runs synchronously before every recognized Codex spawn. - markCodexProjectTrusted(launchContext.workspacePath) + // Why: renderer quick-launch cannot await trust IPC before its PTY mounts; launch prep runs before every recognized Codex spawn. + await markCodexProjectTrusted(launchContext.workspacePath) } catch (error) { console.warn('[codex-project-trust] failed to pre-mark launch workspace:', error) } } - const ensureRealHomeHooksIfSelected = (): boolean => { + const ensureRealHomeHooksIfSelected = async (): Promise => { if ( target?.runtime === 'wsl' || !codexRuntimeHome!.isHostSystemDefaultRealHomeSelected(launchEnv) @@ -1167,13 +1172,13 @@ function prepareCodexRuntimeHomeForLaunch( // and trusted by codex's own app-server grant — in the real ~/.codex before // the pane spawns. An incapable grant flips the lane gate so the launch // below falls back to the managed home instead of a status-blind pane. - ensureRealHomeCodexHookState({ + await ensureRealHomeCodexHookState({ hooksEnabled: isAgentStatusHooksEnabled(store?.getSettings()), userDataPath: app.getPath('userData') }) return true } - let realHomeHooksPrepared = ensureRealHomeHooksIfSelected() + let realHomeHooksPrepared = await ensureRealHomeHooksIfSelected() // Why: a ManagedCodexHomeTemporarilyUnavailableError must escape uncaught — // the fallbacks below all key off `null`, which means "system default", so // swallowing the refusal would launch the wrong account (#STA-4422). @@ -1184,7 +1189,7 @@ function prepareCodexRuntimeHomeForLaunch( // Why: launch prep can reject an untrusted managed home and clear its // selection. Establish hook capability for that newly selected lane, then // re-resolve if the capability gate rejects it. - realHomeHooksPrepared = ensureRealHomeHooksIfSelected() + realHomeHooksPrepared = await ensureRealHomeHooksIfSelected() if (realHomeHooksPrepared) { runtimeHomePath = codexRuntimeHome!.prepareForCodexLaunch(target, launchEnv, { unavailableManagedHomePath: launchContext?.unavailableManagedHomePath @@ -1207,12 +1212,12 @@ function prepareCodexRuntimeHomeForLaunch( try { // Why: honor the persisted off switch so post-startup launches can't reinstall removed hooks. const status = hooksEnabled - ? (codexHookService.installForRuntimeHome(runtimeHomePath, hookTarget) ?? + ? ((await codexHookService.installForRuntimeHome(runtimeHomePath, hookTarget)) ?? // Why: a managed account's launch home is its own self-contained // CODEX_HOME, so hooks/trust must install there, not the shared mirror. - codexHookService.install(runtimeHomePath ?? undefined)) + (await codexHookService.install(runtimeHomePath ?? undefined))) : (codexHookService.refreshRuntimeUserHooksForRuntimeHome(runtimeHomePath, hookTarget) ?? - codexHookService.refreshRuntimeUserHooks(runtimeHomePath ?? undefined)) + (await codexHookService.refreshRuntimeUserHooks(runtimeHomePath ?? undefined))) if (status.state === 'error') { console.warn( `[codex-hook-service] failed to ${ @@ -1306,7 +1311,7 @@ async function prepareCodexSessionResumeForLaunch(args: { if (args.workspacePath) { try { - markCodexProjectTrusted(args.workspacePath) + await markCodexProjectTrusted(args.workspacePath) } catch (error) { console.warn('[codex-project-trust] failed to pre-mark resumed workspace:', error) } @@ -1317,11 +1322,14 @@ async function prepareCodexSessionResumeForLaunch(args: { const hooksEnabled = isAgentStatusHooksEnabled(settingsStore.getSettings()) try { if (isSystemHome) { - ensureRealHomeCodexHookState({ hooksEnabled, userDataPath: app.getPath('userData') }) + await ensureRealHomeCodexHookState({ + hooksEnabled, + userDataPath: app.getPath('userData') + }) } else if (hooksEnabled) { - codexHookService.install(resumeHome) + await codexHookService.install(resumeHome) } else { - codexHookService.refreshRuntimeUserHooks(resumeHome) + await codexHookService.refreshRuntimeUserHooks(resumeHome) } } catch (error) { // Why: hook repair is best-effort; session provenance must still win over the currently selected home. @@ -3060,30 +3068,40 @@ void app.whenReady().then(async () => { console.warn('[worktrees] Failed to sweep leftover worktree directories:', error) }) nativeTheme.themeSource = store.getSettings().theme ?? 'system' - if (codexRuntimeHome.isHostSystemDefaultRealHomeSelected()) { - // Why: establish capability before managed-hook reconciliation so an - // incapable host re-arms and completes the legacy real-home sweep now. - ensureRealHomeCodexHookState({ - hooksEnabled: isAgentStatusHooksEnabled(store.getSettings()), - userDataPath: app.getPath('userData') - }) - } + // Why (#16441): the real-home grant runs a codex app-server session. It stays + // ordered before managed-hook reconciliation — an incapable host must re-arm + // and complete the legacy real-home sweep first — but awaiting it inline + // stalled app init behind that session, so chain instead of blocking. + const realHomeCodexHookState = codexRuntimeHome.isHostSystemDefaultRealHomeSelected() + ? ensureRealHomeCodexHookState({ + hooksEnabled: isAgentStatusHooksEnabled(store.getSettings()), + userDataPath: app.getPath('userData') + }).catch((error: unknown) => { + console.warn('[codex-real-home-hooks] startup ensure failed:', error) + }) + : Promise.resolve() if (shouldInstallManagedHooks(is.dev)) { // Why: check the persisted off switch before any auto-install so removed hooks don't silently reappear on launch. if (isAgentStatusHooksEnabled(store.getSettings())) { const managedHookStore = store - void applyAgentStatusHooksEnabled(true, managedHookStore.getSettings(), { - shouldHydrateShellPath: app.isPackaged, - onInstallError: recordManagedHookInstallFailure, - shouldContinue: (agent) => { - const settings = managedHookStore.getSettings() - return shouldContinueManagedHookStartup(isQuitting, settings, agent) - } - }).catch((error) => { - console.warn('[agent-hooks] failed to reconcile managed hooks on startup:', error) - }) + void realHomeCodexHookState + .then(() => + applyAgentStatusHooksEnabled(true, managedHookStore.getSettings(), { + shouldHydrateShellPath: app.isPackaged, + onInstallError: recordManagedHookInstallFailure, + shouldContinue: (agent) => { + const settings = managedHookStore.getSettings() + return shouldContinueManagedHookStartup(isQuitting, settings, agent) + } + }) + ) + .catch((error: unknown) => { + console.warn('[agent-hooks] failed to reconcile managed hooks on startup:', error) + }) } else { - removeManagedAgentHooks() + void removeManagedAgentHooks().catch((error: unknown) => { + console.warn('[agent-hooks] failed to remove managed hooks on startup:', error) + }) } } // Why: process-gone metrics only see survivors; retain a recent whole-app diff --git a/src/main/ipc/pty/host-env/codex-home.ts b/src/main/ipc/pty/host-env/codex-home.ts index ae825e31e78..7abde336181 100644 --- a/src/main/ipc/pty/host-env/codex-home.ts +++ b/src/main/ipc/pty/host-env/codex-home.ts @@ -114,8 +114,10 @@ type ManagedCodexAuthResolutionArgs = { getSettings: () => GlobalSettings | undefined requiredCodexHomePath?: string target: CodexAccountSelectionTarget - resolveCurrent: () => string | null - resolveAfterUnavailable: (unavailableManagedHomePath: string) => string | null + resolveCurrent: () => string | null | Promise + resolveAfterUnavailable: ( + unavailableManagedHomePath: string + ) => string | null | Promise } export function resolveCodexHomeAfterManagedAuthReadiness( @@ -150,7 +152,7 @@ async function continueCodexHomeAfterManagedAuthWait( if (args.requiredCodexHomePath) { return selectedCodexHomePath } - const currentCodexHomePath = args.resolveCurrent() + const currentCodexHomePath = await args.resolveCurrent() if (codexHomeSelectionsEqual(selectedCodexHomePath, currentCodexHomePath)) { return selectedCodexHomePath } @@ -172,7 +174,7 @@ async function continueCodexHomeAfterManagedAuthWait( if (args.requiredCodexHomePath) { throw new Error(CODEX_RESUME_AUTH_UNAVAILABLE_MESSAGE) } - selectedCodexHomePath = args.resolveAfterUnavailable(selectedCodexHomePath!) + selectedCodexHomePath = await args.resolveAfterUnavailable(selectedCodexHomePath!) if (attempt === 1) { break } diff --git a/src/main/ipc/pty/host-env/codex-resume.ts b/src/main/ipc/pty/host-env/codex-resume.ts index a32aefe9057..62e199e09c3 100644 --- a/src/main/ipc/pty/host-env/codex-resume.ts +++ b/src/main/ipc/pty/host-env/codex-resume.ts @@ -103,14 +103,14 @@ export function resolveCodexResumeLaunch( }) } -export function reconcileSharedRuntimeResumeHome( +export async function reconcileSharedRuntimeResumeHome( resumeHome: Extract, - resolveCurrentHome: () => string | null -): string { + resolveCurrentHome: () => string | null | Promise +): Promise { if (!resumeHome.reconcileSharedRuntimeAuth) { return resumeHome.codexHomePath } - const currentHome = resolveCurrentHome() + const currentHome = await resolveCurrentHome() if (!codexHomePathsEqual(currentHome, resumeHome.codexHomePath)) { throw new Error(CODEX_RESUME_AUTH_UNAVAILABLE_MESSAGE) } diff --git a/src/main/ipc/pty/host-env/types.ts b/src/main/ipc/pty/host-env/types.ts index 61ff335b43d..298949d615a 100644 --- a/src/main/ipc/pty/host-env/types.ts +++ b/src/main/ipc/pty/host-env/types.ts @@ -39,11 +39,14 @@ export type CodexHomeLaunchContext = { unavailableManagedHomePath?: string } +// Why (#16441): Codex launch prep grants hook trust through a codex app-server +// session. It resolves asynchronously so the Electron main thread stays +// responsive; every consumer already runs inside an async spawn path. export type GetSelectedCodexHomePath = ( target?: CodexAccountSelectionTarget, launchEnv?: NodeJS.ProcessEnv, launchContext?: CodexHomeLaunchContext -) => string | null +) => string | null | Promise export type PrepareCodexSessionResume = (args: { providerSession: AgentProviderSessionMetadata diff --git a/src/main/ipc/pty/ipc/spawn-env-codex.ts b/src/main/ipc/pty/ipc/spawn-env-codex.ts index 655a848e980..42550f6732d 100644 --- a/src/main/ipc/pty/ipc/spawn-env-codex.ts +++ b/src/main/ipc/pty/ipc/spawn-env-codex.ts @@ -51,24 +51,23 @@ export async function assemblePtyIpcSpawnCodexEnv(ctx: PtyIpcSpawnState): Promis // Why: declared after the strip so a local-provider spawn cannot capture the // pre-strip env — only the daemon branch below re-derives this from baseEnv. ctx.env = ctx.baseEnv + const selectLaunchCodexHome = async (): Promise => + (await ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.baseEnv, { + workspacePath: ctx.cwd, + launchAgent: isTuiAgent(args.launchAgent) ? args.launchAgent : undefined + })) ?? null ctx.selectedCodexHomePath = !ctx.preAdoptedStablePane && !args.connectionId ? getCompatibleSelectedCodexHomePath( ctx.codexSelectionTarget, codexResumeHome - ? ctx.deps.reconcileSharedRuntimeResumeHome(codexResumeHome, () => + ? await ctx.deps.reconcileSharedRuntimeResumeHome(codexResumeHome, async () => getCompatibleSelectedCodexHomePath( ctx.codexSelectionTarget, - ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.baseEnv, { - workspacePath: ctx.cwd, - launchAgent: isTuiAgent(args.launchAgent) ? args.launchAgent : undefined - }) ?? null + await selectLaunchCodexHome() ) ) - : (ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.baseEnv, { - workspacePath: ctx.cwd, - launchAgent: isTuiAgent(args.launchAgent) ? args.launchAgent : undefined - }) ?? null) + : await selectLaunchCodexHome() ) : null if (!ctx.preAdoptedStablePane && args.launchAgent === 'codex' && args.sessionId === undefined) { @@ -77,22 +76,22 @@ export async function assemblePtyIpcSpawnCodexEnv(ctx: PtyIpcSpawnState): Promis getSettings: () => ctx.deps.getSettings?.(), requiredCodexHomePath: codexResumeHome?.codexHomePath, target: ctx.codexSelectionTarget, - resolveCurrent: () => + resolveCurrent: async () => getCompatibleSelectedCodexHomePath( ctx.codexSelectionTarget, - ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.baseEnv, { + (await ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.baseEnv, { workspacePath: ctx.cwd, launchAgent: 'codex' - }) ?? null + })) ?? null ), - resolveAfterUnavailable: (unavailableManagedHomePath) => + resolveAfterUnavailable: async (unavailableManagedHomePath) => getCompatibleSelectedCodexHomePath( ctx.codexSelectionTarget, - ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.baseEnv, { + (await ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.baseEnv, { workspacePath: ctx.cwd, launchAgent: 'codex', unavailableManagedHomePath - }) ?? null + })) ?? null ) }) ctx.selectedCodexHomePath = resolution instanceof Promise ? await resolution : resolution diff --git a/src/main/ipc/pty/ipc/spawn-types.ts b/src/main/ipc/pty/ipc/spawn-types.ts index 68e1cc62227..af5abe564ed 100644 --- a/src/main/ipc/pty/ipc/spawn-types.ts +++ b/src/main/ipc/pty/ipc/spawn-types.ts @@ -110,8 +110,8 @@ export type PtySpawnIpcDeps = { ) => Promise reconcileSharedRuntimeResumeHome: ( resumeHome: Extract, - resolveCurrent: () => string | null - ) => string + resolveCurrent: () => string | null | Promise + ) => Promise stripSequencedStartupResumeArgv: | undefined>( env: T, launch: CodexResumeLaunch diff --git a/src/main/ipc/pty/provider/local-configure.ts b/src/main/ipc/pty/provider/local-configure.ts index 2051d937c94..50f3c3e4016 100644 --- a/src/main/ipc/pty/provider/local-configure.ts +++ b/src/main/ipc/pty/provider/local-configure.ts @@ -38,7 +38,7 @@ export function configureLocalPtyProvider(args: { getWindowsPowerShellImplementation: () => getSettings ? (getSettings()?.terminalWindowsPowerShellImplementation ?? 'auto') : undefined, pwshAvailable: () => isPwshAvailableAsync(), - buildSpawnEnv: (id, baseEnv, ctx) => { + buildSpawnEnv: async (id, baseEnv, ctx) => { const codexSelectionTarget: CodexAccountSelectionTarget = ctx?.isWsl === true ? { runtime: 'wsl', wslDistro: ctx.wslDistro ?? null } @@ -47,10 +47,10 @@ export function configureLocalPtyProvider(args: { codexSelectionTarget, ctx?.codexHomePathOverride ? ctx.codexHomePathOverride.value - : (getSelectedCodexHomePath?.(codexSelectionTarget, baseEnv, { + : ((await getSelectedCodexHomePath?.(codexSelectionTarget, baseEnv, { workspacePath: ctx?.cwd, launchAgent: ctx?.launchAgent - }) ?? null) + })) ?? null) ) const skipCodexHomeEnv = ctx?.isWsl === true && !selectedCodexHomePath const ptySettings = getSettings?.() diff --git a/src/main/ipc/pty/runtime/controller-deps.ts b/src/main/ipc/pty/runtime/controller-deps.ts index 4abebc3b28c..da6c45ce5c5 100644 --- a/src/main/ipc/pty/runtime/controller-deps.ts +++ b/src/main/ipc/pty/runtime/controller-deps.ts @@ -40,8 +40,8 @@ export type PtyRuntimeControllerDeps = { noCodexResumeLaunch: (command: string | undefined) => CodexResumeLaunch reconcileSharedRuntimeResumeHome: ( resumeHome: Extract, - resolveCurrent: () => string | null - ) => string + resolveCurrent: () => string | null | Promise + ) => Promise stripSequencedStartupResumeArgv: | undefined>( env: T, launch: CodexResumeLaunch diff --git a/src/main/ipc/pty/runtime/spawn-preflight.ts b/src/main/ipc/pty/runtime/spawn-preflight.ts index 45c3ea1172f..f1ce7df7633 100644 --- a/src/main/ipc/pty/runtime/spawn-preflight.ts +++ b/src/main/ipc/pty/runtime/spawn-preflight.ts @@ -171,24 +171,23 @@ export async function prepareRuntimePtySpawn( if (args.preAllocatedHandle) { ctx.env = { ...ctx.env, ORCA_TERMINAL_HANDLE: args.preAllocatedHandle } } + const selectLaunchCodexHome = async (): Promise => + (await ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.env, { + workspacePath: ctx.cwd, + launchAgent: isTuiAgent(args.launchAgent) ? args.launchAgent : undefined + })) ?? null ctx.selectedCodexHomePath = !ctx.preAdoptedStablePane && !args.connectionId ? getCompatibleSelectedCodexHomePath( ctx.codexSelectionTarget, codexResumeHome - ? ctx.deps.reconcileSharedRuntimeResumeHome(codexResumeHome, () => + ? await ctx.deps.reconcileSharedRuntimeResumeHome(codexResumeHome, async () => getCompatibleSelectedCodexHomePath( ctx.codexSelectionTarget, - ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.env, { - workspacePath: ctx.cwd, - launchAgent: isTuiAgent(args.launchAgent) ? args.launchAgent : undefined - }) ?? null + await selectLaunchCodexHome() ) ) - : (ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.env, { - workspacePath: ctx.cwd, - launchAgent: isTuiAgent(args.launchAgent) ? args.launchAgent : undefined - }) ?? null) + : await selectLaunchCodexHome() ) : null if ( @@ -201,22 +200,22 @@ export async function prepareRuntimePtySpawn( getSettings: () => ctx.deps.getSettings?.(), requiredCodexHomePath: codexResumeHome?.codexHomePath, target: ctx.codexSelectionTarget, - resolveCurrent: () => + resolveCurrent: async () => getCompatibleSelectedCodexHomePath( ctx.codexSelectionTarget, - ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.env, { + (await ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.env, { workspacePath: ctx.cwd, launchAgent: 'codex' - }) ?? null + })) ?? null ), - resolveAfterUnavailable: (unavailableManagedHomePath) => + resolveAfterUnavailable: async (unavailableManagedHomePath) => getCompatibleSelectedCodexHomePath( ctx.codexSelectionTarget, - ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.env, { + (await ctx.deps.getSelectedCodexHomePath?.(ctx.codexSelectionTarget, ctx.env, { workspacePath: ctx.cwd, launchAgent: 'codex', unavailableManagedHomePath - }) ?? null + })) ?? null ) }) ctx.selectedCodexHomePath = resolution instanceof Promise ? await resolution : resolution diff --git a/src/main/providers/local-pty-provider-spawn-session.test.ts b/src/main/providers/local-pty-provider-spawn-session.test.ts index b5ec1706225..c4ef90816ed 100644 --- a/src/main/providers/local-pty-provider-spawn-session.test.ts +++ b/src/main/providers/local-pty-provider-spawn-session.test.ts @@ -317,6 +317,30 @@ describe('LocalPtyProvider', () => { expect(spawnMock).toHaveBeenCalledOnce() }) + // Why (#16441): the Codex hook install and trust grant moved into + // buildSpawnEnv, so the env build is now the long await before node-pty + // exists — shutdown must be able to cancel the session id during it. + it('does not spawn after shutdown cancels a pending spawn during the env build', async () => { + spawnMock.mockClear() + let finishEnvBuild!: (env: Record) => void + const buildSpawnEnv = vi.fn( + (_id: string, baseEnv: Record) => + new Promise>((resolve) => { + finishEnvBuild = () => resolve(baseEnv) + }) + ) + const envProvider = new LocalPtyProvider({ buildSpawnEnv }) + + const spawn = envProvider.spawn({ cols: 80, rows: 24, sessionId: 'env-build-session' }) + const canceledSpawn = expect(spawn).rejects.toThrow('PTY spawn canceled: env-build-session') + await vi.waitFor(() => expect(buildSpawnEnv).toHaveBeenCalledOnce()) + + await envProvider.shutdown('env-build-session', { immediate: true }) + finishEnvBuild({}) + await canceledSpawn + expect(spawnMock).not.toHaveBeenCalled() + }) + it('coalesces a concurrent same-session-id spawn before launching a redundant shell (F3)', async () => { spawnMock.mockClear() const procA = { ...mockProc, pid: 1001 } diff --git a/src/main/providers/local-pty-provider.ts b/src/main/providers/local-pty-provider.ts index 4cdab4df4af..d8ddc7cd902 100644 --- a/src/main/providers/local-pty-provider.ts +++ b/src/main/providers/local-pty-provider.ts @@ -373,17 +373,19 @@ function allocatePtyId(sessionId: string | undefined): string { return id } -async function prepareLocalPtySpawn(id: string): Promise { +/** Awaits pre-launch work that shutdown must be able to cancel: no node-pty + * process exists yet, so cancellation can only be observed after the await. */ +async function awaitCancelableLocalPtySpawn(id: string, operation: T | Promise): Promise { const pendingSpawn: PendingLocalPtySpawn = { canceled: false } const pending = pendingLocalPtySpawns.get(id) ?? new Set() pending.add(pendingSpawn) pendingLocalPtySpawns.set(id, pending) try { - // Why: shutdown must be able to cancel a stable session id during the async macOS capability probe, before node-pty exists. - await prepareMacosTccLoginShell() + const result = await operation if (pendingSpawn.canceled) { throw new Error(`PTY spawn canceled: ${id}`) } + return result } finally { pending.delete(pendingSpawn) if (pending.size === 0) { @@ -534,7 +536,10 @@ export type LocalPtyProviderOptions = { isWsl?: boolean wslDistro?: string | null } - ) => Record + // Why (#16441): Codex launch prep grants hook trust through a codex + // app-server session. `spawn` already awaits, so returning a promise keeps + // the Electron main thread responsive instead of blocking on spawnSync. + ) => Record | Promise> /** Whether worktree-scoped shell history is enabled; when true (or absent) with a worktreeId, HISTFILE is scoped per-worktree. */ isHistoryEnabled?: () => boolean /** Why: COMSPEC is always cmd.exe, so this callback injects the user's persisted shell preference. Undefined when none set. */ @@ -743,16 +748,21 @@ export class LocalPtyProvider implements IPtyProvider { const isWslShell = Boolean(wslInfo) || pathWin32.basename(shellPath).toLowerCase() === 'wsl.exe' const launchWslDistro = isWslShell ? (launchWslContext?.distro ?? null) : null + // Why (#16441): building the env now awaits Codex hook installs and trust + // grants, so shutdown must be able to cancel this session id here too. const finalEnv = this.opts.buildSpawnEnv - ? this.opts.buildSpawnEnv(id, spawnEnv, { - command: args.command, - launchAgent: args.launchAgent, - codexHomePathOverride: args.codexHomePathOverride, - cwd, - shellPath, - isWsl: isWslShell, - wslDistro: launchWslDistro - }) + ? await awaitCancelableLocalPtySpawn( + id, + this.opts.buildSpawnEnv(id, spawnEnv, { + command: args.command, + launchAgent: args.launchAgent, + codexHomePathOverride: args.codexHomePathOverride, + cwd, + shellPath, + isWsl: isWslShell, + wslDistro: launchWslDistro + }) + ) : spawnEnv // Why: app-level env hooks can re-add scrubbed vars; delete last so shims like Claude Agent Teams keep their PATH. for (const key of args.envToDelete ?? []) { @@ -914,7 +924,8 @@ export class LocalPtyProvider implements IPtyProvider { primaryLaunchEnvKeys = Object.keys(shellLaunch.env) } - await prepareLocalPtySpawn(id) + // Why: the async macOS capability probe runs before node-pty exists. + await awaitCancelableLocalPtySpawn(id, prepareMacosTccLoginShell()) if (args.signal?.aborted) { throw new Error('client_disconnected') } diff --git a/src/main/runtime/orca-runtime.ts b/src/main/runtime/orca-runtime.ts index 54c6b711173..32373ee98cc 100644 --- a/src/main/runtime/orca-runtime.ts +++ b/src/main/runtime/orca-runtime.ts @@ -24483,7 +24483,20 @@ export class OrcaRuntimeService { } } - private markLocalWorkspaceTrustedForAgent(agent: TuiAgent, workspacePath: string): void { + private markWorkspaceTrustedForAgent( + agent: TuiAgent, + connectionId: string | null | undefined, + workspacePath: string + ): Promise { + return connectionId + ? this.markRemoteWorkspaceTrustedForAgent(agent, connectionId, workspacePath) + : this.markLocalWorkspaceTrustedForAgent(agent, workspacePath) + } + + private async markLocalWorkspaceTrustedForAgent( + agent: TuiAgent, + workspacePath: string + ): Promise { const preset = TUI_AGENT_CONFIG[agent].preflightTrust if (!preset) { return @@ -24494,7 +24507,9 @@ export class OrcaRuntimeService { } else if (preset === 'copilot') { markCopilotFolderTrusted(workspacePath) } else if (preset === 'codex') { - markCodexProjectTrusted(workspacePath) + // Why: the Codex write queues behind any in-flight hook grant, so the + // agent must not launch until it has actually landed. + await markCodexProjectTrusted(workspacePath) } } catch { // Best-effort: the user can still accept the agent trust prompt manually. @@ -25019,7 +25034,7 @@ export class OrcaRuntimeService { try { const startupTrustAgent = effectiveDraftPaste?.agent ?? effectiveCreatedWithAgent if (startupTrustAgent) { - this.markLocalWorkspaceTrustedForAgent(startupTrustAgent, worktree.path) + await this.markLocalWorkspaceTrustedForAgent(startupTrustAgent, worktree.path) } const terminal = await this.createTerminal(`id:${worktree.id}`, { command: effectiveStartup.command, @@ -25812,7 +25827,7 @@ export class OrcaRuntimeService { // session later, matching `orca terminal create` background semantics. const startupTrustAgent = effectiveDraftPaste?.agent ?? effectiveCreatedWithAgent if (startupTrustAgent) { - this.markLocalWorkspaceTrustedForAgent(startupTrustAgent, worktreePath) + await this.markLocalWorkspaceTrustedForAgent(startupTrustAgent, worktreePath) } const terminal = await this.createTerminal(`id:${worktree.id}`, { command: sequencedStartup.command, @@ -28298,11 +28313,7 @@ export class OrcaRuntimeService { return opts } - if (workspace.connectionId) { - await this.markRemoteWorkspaceTrustedForAgent(agent, workspace.connectionId, workspace.path) - } else { - this.markLocalWorkspaceTrustedForAgent(agent, workspace.path) - } + await this.markWorkspaceTrustedForAgent(agent, workspace.connectionId, workspace.path) return { ...opts, @@ -28436,15 +28447,7 @@ export class OrcaRuntimeService { if (!startup) { throw new Error('agent_session_identity_required') } - if (workspace.connectionId) { - await this.markRemoteWorkspaceTrustedForAgent( - request.agent, - workspace.connectionId, - workspace.path - ) - } else { - this.markLocalWorkspaceTrustedForAgent(request.agent, workspace.path) - } + await this.markWorkspaceTrustedForAgent(request.agent, workspace.connectionId, workspace.path) if (_caller.signal?.aborted) { throw new Error('client_disconnected') } @@ -28605,15 +28608,7 @@ export class OrcaRuntimeService { if (!startup) { throw new Error('agent_session_identity_required') } - if (workspace.connectionId) { - await this.markRemoteWorkspaceTrustedForAgent( - request.agent, - workspace.connectionId, - workspace.path - ) - } else { - this.markLocalWorkspaceTrustedForAgent(request.agent, workspace.path) - } + await this.markWorkspaceTrustedForAgent(request.agent, workspace.connectionId, workspace.path) if (caller.signal?.aborted) { throw new Error('client_disconnected') } @@ -29245,11 +29240,7 @@ export class OrcaRuntimeService { throw new Error('Repository for the selected workspace is no longer available.') } const startup = this.buildStartupForAgent(repo, opts.agent, opts.prompt) - if (repo.connectionId) { - await this.markRemoteWorkspaceTrustedForAgent(opts.agent, repo.connectionId, worktree.path) - } else { - this.markLocalWorkspaceTrustedForAgent(opts.agent, worktree.path) - } + await this.markWorkspaceTrustedForAgent(opts.agent, repo.connectionId, worktree.path) return await this.createTerminal(`id:${worktree.id}`, { command: startup.startup.command, env: startup.startup.env, @@ -29638,15 +29629,7 @@ export class OrcaRuntimeService { if (opts.agentPrompt && startupPlan.followupPrompt) { throw new Error(`Agent ${opts.agent} does not support startup prompt quick commands.`) } - if (workspace.connectionId) { - await this.markRemoteWorkspaceTrustedForAgent( - opts.agent, - workspace.connectionId, - workspace.path - ) - } else { - this.markLocalWorkspaceTrustedForAgent(opts.agent, workspace.path) - } + await this.markWorkspaceTrustedForAgent(opts.agent, workspace.connectionId, workspace.path) return { command: startupPlan.launchCommand, env: startupPlan.env, diff --git a/src/main/text-generation/commit-message-agent-environment.ts b/src/main/text-generation/commit-message-agent-environment.ts index de83bcc3559..3c5523bc420 100644 --- a/src/main/text-generation/commit-message-agent-environment.ts +++ b/src/main/text-generation/commit-message-agent-environment.ts @@ -4,7 +4,9 @@ import { readShellStartupEnvVar } from '../pty/shell-startup-env' import { parseWslUncPath } from '../../shared/wsl-paths' export type CommitMessageAgentEnvironmentResolvers = { - prepareForCodexLaunch?: (target?: CommitMessageAgentRuntimeTarget) => string | null + prepareForCodexLaunch?: ( + target?: CommitMessageAgentRuntimeTarget + ) => string | null | Promise prepareForClaudeLaunch?: ( target?: CommitMessageAgentRuntimeTarget ) => Promise @@ -100,7 +102,7 @@ export async function prepareLocalCommitMessageAgentEnv( try { if (agentId === 'codex' && resolvers.prepareForCodexLaunch) { - const codexHomePath = resolvers.prepareForCodexLaunch(target) + const codexHomePath = await resolvers.prepareForCodexLaunch(target) const wslCodexHome = codexHomePath ? parseWslUncPath(codexHomePath) : null if (target?.runtime === 'wsl') { const codexHomeForTarget = wslCodexHome?.linuxPath ?? null diff --git a/src/shared/capability-probe-cache.ts b/src/shared/capability-probe-cache.ts new file mode 100644 index 00000000000..0a571dcf4da --- /dev/null +++ b/src/shared/capability-probe-cache.ts @@ -0,0 +1,128 @@ +/** + * Optimistic capability probing with a bounded retry window and in-flight + * probe dedupe. + * + * Extracted from GitCapabilityCache so every host-capability cache in the tree + * gets the same three behaviors: probe once, remember only a positive absence + * signal, and let a concurrent caller wait on the probe already running rather + * than starting a duplicate one. + */ +export type CapabilityProbeOutcome = 'supported' | 'unsupported' | 'unknown' + +export class CapabilityProbeCache { + private readonly retryAfterByCapability = new Map() + private readonly probesByCapability = new Map>() + private readonly supportedCapabilities = new Set() + + constructor(private readonly retryIntervalMs: number) {} + + shouldTry(capability: TCapability, nowMs = Date.now()): boolean { + const retryAfterMs = this.retryAfterByCapability.get(capability) + if (retryAfterMs === undefined) { + return true + } + if (nowMs < retryAfterMs) { + return false + } + this.retryAfterByCapability.delete(capability) + return true + } + + isKnownSupported(capability: TCapability): boolean { + return this.supportedCapabilities.has(capability) + } + + rememberSupported(capability: TCapability): void { + this.retryAfterByCapability.delete(capability) + this.supportedCapabilities.add(capability) + } + + rememberUnsupported(capability: TCapability, nowMs = Date.now()): void { + // Why: optimistic probes preserve newer behavior, but repeating a known + // failure on every poll/search wastes subprocesses and trace space. + this.supportedCapabilities.delete(capability) + this.retryAfterByCapability.set(capability, nowMs + this.retryIntervalMs) + } + + async runWithFallback( + capability: TCapability, + runPreferred: () => Promise, + runFallback: () => Promise, + isUnsupportedError: (error: unknown) => boolean + ): Promise { + if (this.supportedCapabilities.has(capability)) { + // Why: supported commands are real work, not disposable probes. Let + // sibling repo/SSH calls retain their intended concurrency. + return this.runPreferredOrFallback(capability, runPreferred, runFallback, isUnsupportedError) + } + if (!this.shouldTry(capability)) { + return runFallback() + } + + const inFlightProbe = this.probesByCapability.get(capability) + if (inFlightProbe) { + const outcome = await inFlightProbe + if (outcome === 'unsupported' || !this.shouldTry(capability)) { + return runFallback() + } + return this.runPreferredOrFallback(capability, runPreferred, runFallback, isUnsupportedError) + } + + let settleProbe!: (outcome: CapabilityProbeOutcome) => void + const probe = new Promise((resolve) => { + settleProbe = resolve + }) + this.probesByCapability.set(capability, probe) + try { + return await this.runPreferredOrFallback( + capability, + runPreferred, + runFallback, + isUnsupportedError, + settleProbe + ) + } finally { + if (this.probesByCapability.get(capability) === probe) { + this.probesByCapability.delete(capability) + } + // Backstop: `isUnsupportedError` or `rememberUnsupported` can throw + // before the settle below them runs; waiters must not hang behind it. + settleProbe('unknown') + } + } + + clear(): void { + this.retryAfterByCapability.clear() + this.probesByCapability.clear() + this.supportedCapabilities.clear() + } + + private async runPreferredOrFallback( + capability: TCapability, + runPreferred: () => Promise, + runFallback: () => Promise, + isUnsupportedError: (error: unknown) => boolean, + settleProbe?: (outcome: CapabilityProbeOutcome) => void + ): Promise { + try { + const result = await runPreferred() + // A preferred callback can detect a weaker positive signal (old Git's + // exit-zero option echo) and remember it as unsupported, so do not + // overwrite that stronger signal. + const outcome = this.retryAfterByCapability.has(capability) ? 'unsupported' : 'supported' + if (outcome === 'supported') { + this.supportedCapabilities.add(capability) + } + settleProbe?.(outcome) + return result + } catch (error) { + if (!isUnsupportedError(error)) { + settleProbe?.('unknown') + throw error + } + this.rememberUnsupported(capability) + settleProbe?.('unsupported') + return runFallback() + } + } +} diff --git a/src/shared/child-process/__fixtures__/child-process-import-allowlist.txt b/src/shared/child-process/__fixtures__/child-process-import-allowlist.txt index fcf8258d6b3..03e06eb374f 100644 --- a/src/shared/child-process/__fixtures__/child-process-import-allowlist.txt +++ b/src/shared/child-process/__fixtures__/child-process-import-allowlist.txt @@ -44,10 +44,8 @@ src/main/claude-accounts/keychain.ts src/main/codex-accounts/runtime-home-service.ts src/main/codex-accounts/service.ts src/main/codex/codex-app-server-client.ts -src/main/codex/codex-app-server-grant-bridge.ts src/main/codex/codex-app-server-session.ts src/main/codex/codex-state-db-backfill-recovery.ts -src/main/codex/codex-trust-grant-host.ts src/main/codex/codex-wsl-hook-install-plan.ts src/main/computer/desktop-script-provider-bridge.ts src/main/computer/macos-computer-use-permission-status.ts diff --git a/src/shared/git-capability-cache.ts b/src/shared/git-capability-cache.ts index 9e3d4ce89e0..06d0e957c74 100644 --- a/src/shared/git-capability-cache.ts +++ b/src/shared/git-capability-cache.ts @@ -1,3 +1,5 @@ +import { CapabilityProbeCache } from './capability-probe-cache' + // Why: suppress hot-loop failures while still detecting an in-place Git // upgrade during a long Orca session without requiring a restart. export const GIT_CAPABILITY_RETRY_INTERVAL_MS = 30 * 60_000 @@ -10,107 +12,8 @@ export type GitCapability = | 'rev-parse-path-format' | 'worktree-list-z' -type GitCapabilityProbeOutcome = 'supported' | 'unsupported' | 'unknown' - -export class GitCapabilityCache { - private readonly retryAfterByCapability = new Map() - private readonly probesByCapability = new Map>() - private readonly supportedCapabilities = new Set() - - shouldTry(capability: GitCapability, nowMs = Date.now()): boolean { - const retryAfterMs = this.retryAfterByCapability.get(capability) - if (retryAfterMs === undefined) { - return true - } - if (nowMs < retryAfterMs) { - return false - } - this.retryAfterByCapability.delete(capability) - return true - } - - rememberUnsupported(capability: GitCapability, nowMs = Date.now()): void { - // Why: optimistic probes preserve newer Git behavior, but repeating a - // known failure on every poll/search wastes subprocesses and trace space. - this.supportedCapabilities.delete(capability) - this.retryAfterByCapability.set(capability, nowMs + GIT_CAPABILITY_RETRY_INTERVAL_MS) - } - - async runWithFallback( - capability: GitCapability, - runPreferred: () => Promise, - runFallback: () => Promise, - isUnsupportedError: (error: unknown) => boolean - ): Promise { - if (this.supportedCapabilities.has(capability)) { - // Why: supported commands are real work, not disposable probes. Let - // sibling repo/SSH calls retain their intended concurrency. - return this.runPreferredOrFallback(capability, runPreferred, runFallback, isUnsupportedError) - } - if (!this.shouldTry(capability)) { - return runFallback() - } - - const inFlightProbe = this.probesByCapability.get(capability) - if (inFlightProbe) { - const outcome = await inFlightProbe - if (outcome === 'unsupported' || !this.shouldTry(capability)) { - return runFallback() - } - return this.runPreferredOrFallback(capability, runPreferred, runFallback, isUnsupportedError) - } - - let settleProbe!: (outcome: GitCapabilityProbeOutcome) => void - const probe = new Promise((resolve) => { - settleProbe = resolve - }) - this.probesByCapability.set(capability, probe) - try { - return await this.runPreferredOrFallback( - capability, - runPreferred, - runFallback, - isUnsupportedError, - settleProbe - ) - } finally { - if (this.probesByCapability.get(capability) === probe) { - this.probesByCapability.delete(capability) - } - } - } - - clear(): void { - this.retryAfterByCapability.clear() - this.probesByCapability.clear() - this.supportedCapabilities.clear() - } - - private async runPreferredOrFallback( - capability: GitCapability, - runPreferred: () => Promise, - runFallback: () => Promise, - isUnsupportedError: (error: unknown) => boolean, - settleProbe?: (outcome: GitCapabilityProbeOutcome) => void - ): Promise { - try { - const result = await runPreferred() - // A preferred callback can detect old Git's exit-zero option echo and - // remember it as unsupported, so do not overwrite that stronger signal. - const outcome = this.retryAfterByCapability.has(capability) ? 'unsupported' : 'supported' - if (outcome === 'supported') { - this.supportedCapabilities.add(capability) - } - settleProbe?.(outcome) - return result - } catch (error) { - if (!isUnsupportedError(error)) { - settleProbe?.('unknown') - throw error - } - this.rememberUnsupported(capability) - settleProbe?.('unsupported') - return runFallback() - } +export class GitCapabilityCache extends CapabilityProbeCache { + constructor() { + super(GIT_CAPABILITY_RETRY_INTERVAL_MS) } } From 88ecc739e3ee7a261e978f28dd0bb129b832ca1c Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 26 Aug 2026 16:49:31 -0700 Subject: [PATCH 18/19] fix(browser): bound the agent-browser daemon's life instead of hoping teardown runs (#16367) (#16588) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(browser): bound the agent-browser daemon lifetime (#16367) `agent-browser` is a client/daemon CLI. Orca only ever spawns the short-lived client; that client forks a daemon Orca holds no handle on, which reparents to pid 1 immediately. Nothing in Orca reclaimed it, so a crashed or SIGKILL'd run left one daemon per browser tab alive forever — two of them at ~25.6 GiB and ~7.2 GiB RSS saturated a 64 GiB cgroup under headless `orca serve`. Three fixes, in order of how much they cover: 1. Set `AGENT_BROWSER_IDLE_TIMEOUT_MS` on both spawn paths (the bundled-binary bridge and the orcad external-Chromium provider). This is the only bound that survives every way Orca can die, including SIGKILL, where no teardown code ever runs. 10 minutes: >6x the bridge's 90s `EXEC_TIMEOUT_MS`, so it can never cut a command, a retry chain, or an ordinary gap between two user commands, while capping an abandoned daemon at minutes instead of days. Verified against agent-browser 0.27.0: an idle daemon exits and takes its Chromium tree and socket sidecar files with it, and a daemon attached over `--cdp` (the bridge's case) leaves the attached browser running, so an Orca tab is never closed by its daemon idling out. 2. Await `destroyAllSessions()` in the will-quit teardown barrier. It was fire-and-forget and the only browser member missing from `settleTeardownWithinDeadline`; each session's close is its own agent-browser child taking hundreds of ms, so `app.quit()` won. 3. Sweep daemons a previous run left behind, using agent-browser's own `session list` / `close` rather than a pid walk (see `windows-pty-job.ts` for why walking your own orphans is guesswork). `closeStaleAgentBrowserSession` only ever reset the one name a new tab was about to reuse. The sweep runs only when `AGENT_BROWSER_SOCKET_DIR` is set, because that private per-profile directory is what proves the enumeration can only see this Orca profile's daemons; it is never set on Windows, so Windows gets no enumeration rather than a machine-wide sweep that could close a daemon Orca does not own. Windows stays bounded by the idle timeout, which needs no ownership proof. orcad's session name is stable across runs, so it closes that one name at start instead — a killed orcad's daemon would otherwise be reused while still holding the previous run's Chromium on a dead serve port. Where the 25 GiB went is inference from code, not a measurement: `captureStart` sets `activeCapture` and only an explicit `captureStop` ends it, so a HAR capture in a daemon living for days is unbounded. Not claimed as proven; the idle bound caps it either way. Not re-landed: the queue bounds from #10179 (reverted by #10255) bound Orca's own main-process heap, not the daemon's RSS, so they do not address this report. Also true but left alone: the 3-strike breaker's `destroySession` is an unawaited call whose `close` is `catch {}`-swallowed, and it only fires while a command is in flight — an idle-but-bloated daemon is never noticed. The idle timeout now bounds that case. `getOffscreenBrowserBackend()?.destroyAll?.()` is declared `void` and fully synchronous, so unlike `destroyAllSessions` it has no promise to lose and needs no barrier entry. * fix(browser): scope the daemon idle bound and close every daemon Orca owns Review follow-ups on the agent-browser orphan fix. - Never idle-bound the orcad external-Chromium daemon: it owns the user's remote browser, so the 10-minute bound closed a live session and every tab in it. Per-tab helper daemons keep the bound; the stable session name plus the `close` in start() is what reclaims a killed orcad's Chromium tree. - Retire a page's daemon from the headless offscreen backend, which is the only place `orca serve` closes a page and never reached the bridge. Credit to @Jinwoo-H (#16564) for identifying this owner-boundary gap. - Bound the teardown close at 5s so the will-quit barrier member cannot inherit the 90s exec timeout, and close sessions still being created. - Gate the startup sweep on a socket directory Orca derived itself; an inherited AGENT_BROWSER_SOCKET_DIR is no proof of per-profile ownership. - Replay a session's network routes when the daemon idled out between two commands, instead of silently serving unstubbed requests. * fix(browser): give the orphan sweep a kill switch Of the three behaviours this PR adds, two are already recoverable in the field without a build: the idle bound is an env passthrough an operator can raise, and the quit close is bounded by its own timeout inside the teardown deadline. The startup sweep was the exception — it fires unconditionally, and if it closes a daemon it should not, or spawns one process per stale name on a profile holding hundreds, the only remedy was a revert. ORCA_DISABLE_AGENT_BROWSER_SWEEP=1 turns it off, matching the existing ORCA_DISABLE_CODEX_TRUST_RPC / ORCA_DISABLE_HTTP2 convention. Note for anyone reaching for it on macOS: a Finder-launched Orca does not see shell env, so it needs launchctl setenv or a terminal launch. * fix(orcad): reuse a surviving browser session instead of closing the user's start() closed the daemon before every open, killing the Chromium tree with it. That runs on every provider start, not just after a crash — so an `orca serve` restart took the remote user's browser and every tab in it. The justification was borrowed from the pane bridge, which passes --cdp and so really does hold a port that dies with its Orca. This session passes only --session and --profile: nothing binds it to the old process, and the daemon owns its Chromium independently. A survivor is reusable as-is. start() now probes for an active tab first and returns it untouched. Only a name that answers nothing gets closed and reopened — which is still the killed-orcad case the stable session name exists to recover. This matters more as orcad becomes the backend the remote host runs on: the browser it manages belongs to a user, not to the process that happens to be driving it this minute. It is also the same principle that already exempts this path from AGENT_BROWSER_IDLE_TIMEOUT_MS. --- ...t-browser-bridge-command-transport.test.ts | 6 +- ...t-browser-bridge-session-lifecycle.test.ts | 95 +++++++++ src/main/browser/agent-browser-bridge.ts | 81 +++++++- .../agent-browser-orphan-sweep.test.ts | 190 ++++++++++++++++++ .../browser/agent-browser-orphan-sweep.ts | 91 +++++++++ .../agent-browser-process-environment.test.ts | 80 ++++++-- .../agent-browser-process-environment.ts | 43 +++- ...ffscreen-browser-backend-lifecycle.test.ts | 15 ++ src/main/index.ts | 7 +- .../external-chromium-browser-session.test.ts | 130 ++++++++++++ .../external-chromium-browser-session.ts | 59 +++++- ...uit-teardown-agent-browser-daemons.test.ts | 33 +++ 12 files changed, 786 insertions(+), 44 deletions(-) create mode 100644 src/main/browser/agent-browser-orphan-sweep.test.ts create mode 100644 src/main/browser/agent-browser-orphan-sweep.ts create mode 100644 src/main/orcad/external-chromium-browser-session.test.ts create mode 100644 src/main/quit-teardown-agent-browser-daemons.test.ts diff --git a/src/main/browser/agent-browser-bridge-command-transport.test.ts b/src/main/browser/agent-browser-bridge-command-transport.test.ts index e8c5b786a9b..733cd966ce7 100644 --- a/src/main/browser/agent-browser-bridge-command-transport.test.ts +++ b/src/main/browser/agent-browser-bridge-command-transport.test.ts @@ -51,6 +51,7 @@ vi.mock('./cdp-bridge', () => ({ })) import { AgentBrowserBridge } from './agent-browser-bridge' +import { AGENT_BROWSER_IDLE_TIMEOUT_MS } from './agent-browser-process-environment' import { createSucceedWith, mockBrowserManager, @@ -338,7 +339,10 @@ describe('AgentBrowserBridge', () => { expect(args).toContain('wait') expect(args).toContain('#ready') expect(options.timeout).toBe(2200) - expect(options.env).toBe(process.env) + // Why not toBe(process.env): the bridge hands the daemon an idle-lifetime bound (#16367). + const env = options.env as NodeJS.ProcessEnv + expect(env.PATH).toBe(process.env.PATH) + expect(env.AGENT_BROWSER_IDLE_TIMEOUT_MS).toBe(String(AGENT_BROWSER_IDLE_TIMEOUT_MS)) }) it('returns browser_timeout for timed conditional waits without recycling the session', async () => { diff --git a/src/main/browser/agent-browser-bridge-session-lifecycle.test.ts b/src/main/browser/agent-browser-bridge-session-lifecycle.test.ts index 4ef3d24fa25..5eab202541c 100644 --- a/src/main/browser/agent-browser-bridge-session-lifecycle.test.ts +++ b/src/main/browser/agent-browser-bridge-session-lifecycle.test.ts @@ -64,6 +64,12 @@ overrideBridgeWebContentsLookup(AgentBrowserBridge.prototype, webContentsFromIdM const succeedWith = createSucceedWith(execFileMock, stdinWrites) +function closeCallCount(): number { + return execFileMock.mock.calls.filter((call: unknown[]) => + (call[1] as string[]).includes('close') + ).length +} + describe('AgentBrowserBridge', () => { let bridge: AgentBrowserBridge @@ -453,6 +459,52 @@ describe('AgentBrowserBridge', () => { ).toBe(0) }) + // Why: the daemon's own idle timer retires it between commands; a replacement still serves the + // page but has none of the session's network routes, so leaving them dropped is a silent wrong + // answer for the next request the caller expected to be stubbed (#16367). + it('replays intercept routes after the daemon idles out', async () => { + succeedWith({ ok: true }) + await bridge.interceptEnable(['https://api.example/**']) + + const sessions = (bridge as unknown as { sessions: Map }) + .sessions + const session = sessions.get('orca-tab-tab-1')! + session.lastCommandAt = Date.now() - 11 * 60 * 1000 + + const commandCalls: string[][] = [] + execFileMock.mockImplementation( + (_bin: string, args: string[], _opts: unknown, cb: ExecFileCallback) => { + commandCalls.push(args) + cb(null, JSON.stringify({ success: true, data: { snapshot: 'tree' } }), '') + } + ) + await bridge.snapshot() + + const routeCalls = commandCalls.filter( + (args) => args.includes('network') && args.includes('route') + ) + expect(routeCalls).toHaveLength(1) + expect(routeCalls[0]).toContain('https://api.example/**') + }) + + it('leaves a session alone while the daemon is still within its idle bound', async () => { + succeedWith({ ok: true }) + await bridge.interceptEnable(['https://api.example/**']) + + const commandCalls: string[][] = [] + execFileMock.mockImplementation( + (_bin: string, args: string[], _opts: unknown, cb: ExecFileCallback) => { + commandCalls.push(args) + cb(null, JSON.stringify({ success: true, data: { snapshot: 'tree' } }), '') + } + ) + await bridge.snapshot() + + expect( + commandCalls.filter((args) => args.includes('network') && args.includes('route')) + ).toHaveLength(0) + }) + // ── destroyAllSessions ── it('makes runtime-wide session destruction terminal', async () => { @@ -591,4 +643,47 @@ describe('AgentBrowserBridge', () => { expect(CdpWsProxyMock.instances).toHaveLength(1) expect(execFileMock).toHaveBeenCalledTimes(1) }) + + // Why: quit awaits destroyAllSessions inside a 20s barrier, so an unbounded close can hold the + // window up for the whole deadline when the daemon is wedged (#16367). + it('bounds every teardown close well inside the quit barrier', async () => { + succeedWith({ snapshot: 'tree' }) + await bridge.snapshot() + + succeedWith(null) + await bridge.destroyAllSessions() + + const closeCall = execFileMock.mock.calls.findLast((c: unknown[]) => + (c[1] as string[]).includes('close') + ) + expect((closeCall![2] as { timeout: number }).timeout).toBeLessThanOrEqual(5_000) + }) + + // Why: the daemon is already spawned by the time the name reaches pendingSessionCreation, so a + // quit that only walks `sessions` leaves exactly the orphan the barrier was added to prevent. + it('closes a session still being created when everything is torn down', async () => { + let releaseProxyStart: (() => void) | undefined + CdpWsProxyMock.mockImplementationOnce(function (this: Record) { + this.start = vi.fn( + () => + new Promise((resolve) => { + releaseProxyStart = () => resolve('ws://127.0.0.1:9222') + }) + ) + this.stop = vi.fn(async () => {}) + this.getPort = vi.fn(() => 9222) + }) + + succeedWith({ snapshot: 'tree' }) + const inFlight = bridge.snapshot() + await vi.waitFor(() => expect(releaseProxyStart).toBeDefined()) + + // Why the baseline: session creation already spawned a stale-session `close` of its own. + const closesBeforeTeardown = closeCallCount() + const teardown = bridge.destroyAllSessions() + releaseProxyStart!() + await Promise.allSettled([inFlight, teardown]) + + expect(closeCallCount()).toBeGreaterThan(closesBeforeTeardown) + }) }) diff --git a/src/main/browser/agent-browser-bridge.ts b/src/main/browser/agent-browser-bridge.ts index 10a64dc16ce..54611bffa8f 100644 --- a/src/main/browser/agent-browser-bridge.ts +++ b/src/main/browser/agent-browser-bridge.ts @@ -52,12 +52,17 @@ import { normalizeBrowserNavigationUrl } from '../../shared/browser-url' import { mapSettledWithConcurrency } from '../../shared/map-with-concurrency' import { iterateBrowserTextInsertionChunks } from './browser-text-insertion' import { createAgentBrowserProcessEnvironment } from './agent-browser-process-environment' +import { + ORCA_TAB_SESSION_PREFIX, + sweepOrphanedAgentBrowserSessions +} from './agent-browser-orphan-sweep' // Why: must exceed agent-browser's internal timeouts (goto 30s, wait 60s) so the bridge never kills a command before its own timeout fires. const EXEC_TIMEOUT_MS = 90_000 const CONSECUTIVE_TIMEOUT_LIMIT = 3 const WAIT_PROCESS_TIMEOUT_GRACE_MS = 1_000 const STALE_SESSION_CLOSE_TIMEOUT_MS = 3_000 +// Why separate from EXEC_TIMEOUT_MS: a close is a member of the 20s will-quit barrier and must finish well inside it. const AGENT_BROWSER_CLEANUP_TIMEOUT_MS = 5_000 const AGENT_BROWSER_CLEANUP_CONCURRENCY = 4 const EMBEDDED_NAVIGATION_TIMEOUT_MS = 30_000 @@ -72,6 +77,8 @@ type SessionState = { // Why: track active interception patterns so they can be re-enabled after session restart activeInterceptPatterns: string[] activeCapture: boolean + // Why: the daemon retires itself once idle; the gap since the last command is how the bridge notices. + lastCommandAt: number // Why: verify the tab is alive at execution time, not just enqueue time — queue delay can destroy it in between. webContentsId: number activeProcess: ChildProcess | null @@ -348,7 +355,7 @@ function isTabClosedTransportError(message: string): boolean { } function pageUnavailableMessageForSession(sessionName: string): string { - const prefix = 'orca-tab-' + const prefix = ORCA_TAB_SESSION_PREFIX const browserPageId = sessionName.startsWith(prefix) ? sessionName.slice(prefix.length) : null return browserPageId ? `Browser page ${browserPageId} is no longer available` @@ -587,6 +594,9 @@ export class AgentBrowserBridge { private screenshotTurn: Promise = Promise.resolve() private readonly agentBrowserBin: string private readonly agentBrowserEnv: NodeJS.ProcessEnv + private readonly ownsAgentBrowserSocketDirectory: boolean + // Why: null when nothing bounds the daemon, so the bridge never guesses that one was replaced. + private readonly agentBrowserIdleTimeoutMs: number | null // Why: stash intercept patterns from a swap-destroyed session, keyed by name, so the next session restores them. private readonly pendingInterceptRestore = new Map() // Why: promise-lock so two concurrent ensureSession calls don't both create the session entry. @@ -601,11 +611,15 @@ export class AgentBrowserBridge { private readonly options: AgentBrowserBridgeOptions = {} ) { this.agentBrowserBin = resolveAgentBrowserBinary() - this.agentBrowserEnv = createAgentBrowserProcessEnvironment({ + const processEnvironment = createAgentBrowserProcessEnvironment({ inheritedEnv: process.env, platform: process.platform, userDataPath: app.getPath('userData') }) + this.agentBrowserEnv = processEnvironment.env + this.ownsAgentBrowserSocketDirectory = processEnvironment.ownsSocketDirectory + const idleTimeoutMs = Number(this.agentBrowserEnv.AGENT_BROWSER_IDLE_TIMEOUT_MS) + this.agentBrowserIdleTimeoutMs = idleTimeoutMs > 0 ? idleTimeoutMs : null } // ── Tab tracking ── @@ -691,9 +705,15 @@ export class AgentBrowserBridge { this.options.onTabsChanged?.(owningWorktreeId) } - /** Retire a helper by its stable page identity when WebContents mapping is gone. */ + /** + * Retire a page's daemon by page id. + * + * The headless offscreen backend owns pages by id and unregisters the guest + * itself, so `onTabClosed`'s webContentsId lookup can never resolve one — it + * has to say which page closed (#16367). + */ async onPageClosed(browserPageId: string): Promise { - const sessionName = `orca-tab-${browserPageId}` + const sessionName = `${ORCA_TAB_SESSION_PREFIX}${browserPageId}` await this.destroySession(sessionName) this.pendingInterceptRestore.delete(sessionName) } @@ -704,7 +724,7 @@ export class AgentBrowserBridge { previousWebContentsId?: number ): Promise { // Why: an Electron process swap keeps browserPageId but gives a new webContentsId — destroy the session so the next command recreates it. - const sessionName = `orca-tab-${browserPageId}` + const sessionName = `${ORCA_TAB_SESSION_PREFIX}${browserPageId}` const session = this.sessions.get(sessionName) const oldWebContentsId = previousWebContentsId ?? session?.webContentsId const owningWorktreeId = this.browserManager.getWorktreeIdForTab(browserPageId) @@ -898,7 +918,9 @@ export class AgentBrowserBridge { navigationTimeout = null } if (!this.getWebContents(target.webContentsId)) { - throw this.createPageUnavailableError(`orca-tab-${target.browserPageId}`) + throw this.createPageUnavailableError( + `${ORCA_TAB_SESSION_PREFIX}${target.browserPageId}` + ) } // Why: ERR_ABORTED also covers a page vetoing unload; that navigation did not succeed. if ( @@ -1619,7 +1641,9 @@ export class AgentBrowserBridge { throw error } if (!this.getWebContents(target.webContentsId)) { - throw this.createPageUnavailableError(`orca-tab-${target.browserPageId}`) + throw this.createPageUnavailableError( + `${ORCA_TAB_SESSION_PREFIX}${target.browserPageId}` + ) } throw new BrowserError( 'browser_error', @@ -2057,8 +2081,22 @@ export class AgentBrowserBridge { // ── Session lifecycle ── + // Why: a previous run that crashed or was SIGKILL'd left one daemon per open tab with + // nobody holding its name — closeStaleAgentBrowserSession only resets a name being reused. + async sweepOrphanedSessions(): Promise { + return sweepOrphanedAgentBrowserSessions({ + binaryPath: this.agentBrowserBin, + env: this.agentBrowserEnv, + ownsSocketDirectory: this.ownsAgentBrowserSocketDirectory, + isSessionLive: (sessionName) => + this.sessions.has(sessionName) || this.pendingSessionCreation.has(sessionName) + }) + } + async destroyAllSessions(options?: AgentBrowserCleanupOptions): Promise { this.shutdownStarted = true + // Why the union: a session still being created has already spawned its daemon but is not in + // `sessions` yet, so closing only `sessions` lets that daemon outlive the quit (#16367). const sessionNames = new Set([ ...this.sessions.keys(), ...this.pendingSessionCreation.keys(), @@ -2094,7 +2132,7 @@ export class AgentBrowserBridge { ): Promise { this.assertCommandAdmission() const target = this.resolveCommandTarget(worktreeId, browserPageId, options.requireScopedTarget) - const sessionName = `orca-tab-${target.browserPageId}` + const sessionName = `${ORCA_TAB_SESSION_PREFIX}${target.browserPageId}` if (options.ensureSession !== false) { await this.ensureSession(sessionName, target.browserPageId, target.webContentsId) @@ -2360,6 +2398,7 @@ export class AgentBrowserBridge { consecutiveTimeouts: 0, activeInterceptPatterns: [], activeCapture: false, + lastCommandAt: Date.now(), webContentsId, activeProcess: null }) @@ -2471,6 +2510,7 @@ export class AgentBrowserBridge { const destroy = (async (): Promise => { try { // Why: each tab has its own named session — close without --session leaves this tab's daemon running. + // Why bounded: this runs inside the 20s will-quit barrier, so it cannot inherit the 90s exec timeout. await this.runAgentBrowserRaw( sessionName, ['--session', sessionName, 'close'], @@ -2506,6 +2546,26 @@ export class AgentBrowserBridge { } } + /** + * Notice that the daemon retired itself between two commands. + * + * A replacement daemon still serves the page (every call reasserts `--cdp`) + * but carries none of the session's network routes, so without this the + * interception the caller configured is silently gone (#16367). + */ + private reinitializeIfDaemonIdledOut(sessionName: string, session: SessionState): void { + if ( + this.agentBrowserIdleTimeoutMs === null || + Date.now() - session.lastCommandAt < this.agentBrowserIdleTimeoutMs + ) { + return + } + session.initialized = false + if (session.activeInterceptPatterns.length > 0) { + this.pendingInterceptRestore.set(sessionName, [...session.activeInterceptPatterns]) + } + } + private assertCommandAdmission(): void { if (this.shutdownStarted) { throw new BrowserError('browser_owner_unavailable', 'Browser runtime is shutting down') @@ -2529,6 +2589,9 @@ export class AgentBrowserBridge { throw this.createPageUnavailableError(sessionName) } + this.reinitializeIfDaemonIdledOut(sessionName, session) + session.lastCommandAt = Date.now() + const args = ['--session', sessionName] const managesInterceptRoutes = commandArgs[0] === 'network' && (commandArgs[1] === 'route' || commandArgs[1] === 'unroute') @@ -2810,7 +2873,7 @@ export class AgentBrowserBridge { private requireTargetWebContents(target: ResolvedBrowserCommandTarget): WebContents { const wc = this.getWebContents(target.webContentsId) if (!wc || wc.isDestroyed()) { - throw this.createPageUnavailableError(`orca-tab-${target.browserPageId}`) + throw this.createPageUnavailableError(`${ORCA_TAB_SESSION_PREFIX}${target.browserPageId}`) } return wc } diff --git a/src/main/browser/agent-browser-orphan-sweep.test.ts b/src/main/browser/agent-browser-orphan-sweep.test.ts new file mode 100644 index 00000000000..ce565e2fa6e --- /dev/null +++ b/src/main/browser/agent-browser-orphan-sweep.test.ts @@ -0,0 +1,190 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const runProcessMock = vi.fn() +vi.mock('../../shared/child-process/run-process', () => ({ + runProcess: (spec: unknown) => runProcessMock(spec) +})) + +import { sweepOrphanedAgentBrowserSessions } from './agent-browser-orphan-sweep' + +type Spec = { args?: readonly string[] } + +const BIN = '/opt/orca/agent-browser' +const SCOPED = { + env: { AGENT_BROWSER_SOCKET_DIR: '/tmp/orca-ab-0123456789abcdef' }, + ownsSocketDirectory: true +} + +function respond(sessions: string[]): void { + runProcessMock.mockImplementation((spec: Spec) => { + if (spec.args?.[0] === 'session') { + return Promise.resolve({ + code: 0, + signal: null, + stdout: JSON.stringify({ success: true, data: { sessions } }), + stderr: '', + timedOut: false + }) + } + return Promise.resolve({ code: 0, signal: null, stdout: '', stderr: '', timedOut: false }) + }) +} + +function closedArgs(): string[][] { + return runProcessMock.mock.calls + .map((call) => [...((call[0] as Spec).args ?? [])]) + .filter((args) => args.includes('close')) +} + +describe('agent-browser orphan sweep', () => { + beforeEach(() => { + runProcessMock.mockReset() + }) + + it('closes tab daemons left by a previous run', async () => { + respond(['orca-tab-aaa', 'orca-tab-bbb']) + + const closed = await sweepOrphanedAgentBrowserSessions({ binaryPath: BIN, ...SCOPED }) + + expect(closed).toEqual(['orca-tab-aaa', 'orca-tab-bbb']) + expect(closedArgs()).toEqual([ + ['--session', 'orca-tab-aaa', 'close'], + ['--session', 'orca-tab-bbb', 'close'] + ]) + }) + + it('never closes a daemon outside Orca tab naming', async () => { + respond(['default', 'agent1', 'orca-orcad-deadbeef', 'orca-tab-aaa']) + + await sweepOrphanedAgentBrowserSessions({ binaryPath: BIN, ...SCOPED }) + + expect(closedArgs()).toEqual([['--session', 'orca-tab-aaa', 'close']]) + }) + + it('leaves sessions this run already owns alone', async () => { + respond(['orca-tab-live', 'orca-tab-orphan']) + + await sweepOrphanedAgentBrowserSessions({ + binaryPath: BIN, + ...SCOPED, + isSessionLive: (name) => name === 'orca-tab-live' + }) + + expect(closedArgs()).toEqual([['--session', 'orca-tab-orphan', 'close']]) + }) + + // Why: without a socket dir Orca derived itself, `session list` can reach daemons another Orca + // profile owns (Windows named pipes, or an inherited AGENT_BROWSER_SOCKET_DIR). Idle timeout bounds those. + it.each([ + ['no socket directory at all', { PATH: 'C:\\Windows' }], + ['a socket directory Orca inherited', { AGENT_BROWSER_SOCKET_DIR: '/tmp/shared-ab' }] + ])('does not enumerate with %s', async (_label, env) => { + respond(['orca-tab-aaa']) + + const closed = await sweepOrphanedAgentBrowserSessions({ + binaryPath: BIN, + env, + ownsSocketDirectory: false + }) + + expect(closed).toEqual([]) + expect(runProcessMock).not.toHaveBeenCalled() + }) + + it('closes nothing when the listing is unusable', async () => { + runProcessMock.mockResolvedValue({ + code: 1, + signal: null, + stdout: 'not json', + stderr: 'boom', + timedOut: false + }) + + await expect( + sweepOrphanedAgentBrowserSessions({ binaryPath: BIN, ...SCOPED }) + ).resolves.toEqual([]) + expect(closedArgs()).toEqual([]) + }) + + it('survives a listing that never returns', async () => { + runProcessMock.mockRejectedValue(new Error('ENOENT')) + + await expect( + sweepOrphanedAgentBrowserSessions({ binaryPath: BIN, ...SCOPED }) + ).resolves.toEqual([]) + }) + + it('keeps sweeping after one close fails', async () => { + runProcessMock.mockImplementation((spec: Spec) => { + if (spec.args?.[0] === 'session') { + return Promise.resolve({ + code: 0, + signal: null, + stdout: JSON.stringify({ data: { sessions: ['orca-tab-aaa', 'orca-tab-bbb'] } }), + stderr: '', + timedOut: false + }) + } + if (spec.args?.[1] === 'orca-tab-aaa') { + return Promise.reject(new Error('spawn failed')) + } + return Promise.resolve({ code: 0, signal: null, stdout: '', stderr: '', timedOut: false }) + }) + + const closed = await sweepOrphanedAgentBrowserSessions({ binaryPath: BIN, ...SCOPED }) + + expect(closed).toEqual(['orca-tab-bbb']) + }) + + it('bounds every child it starts', async () => { + respond(['orca-tab-aaa']) + + await sweepOrphanedAgentBrowserSessions({ binaryPath: BIN, ...SCOPED }) + + for (const call of runProcessMock.mock.calls) { + expect((call[0] as { timeoutMs?: number | null }).timeoutMs).toBeGreaterThan(0) + } + }) +}) + +describe('sweep kill switch', () => { + const previous = process.env.ORCA_DISABLE_AGENT_BROWSER_SWEEP + + afterEach(() => { + if (previous === undefined) { + delete process.env.ORCA_DISABLE_AGENT_BROWSER_SWEEP + } else { + process.env.ORCA_DISABLE_AGENT_BROWSER_SWEEP = previous + } + }) + + // Why: the idle bound is an env passthrough an operator can raise and the quit close is + // self-bounded, so the sweep is the only new behaviour whose failure would need a revert. + it('enumerates nothing when disabled, even when Orca owns the socket directory', async () => { + process.env.ORCA_DISABLE_AGENT_BROWSER_SWEEP = '1' + runProcessMock.mockClear() + + const closed = await sweepOrphanedAgentBrowserSessions({ + binaryPath: BIN, + env: {}, + ownsSocketDirectory: true + }) + + expect(closed).toEqual([]) + expect(runProcessMock).not.toHaveBeenCalled() + }) + + it('still sweeps when the flag holds any other value', async () => { + process.env.ORCA_DISABLE_AGENT_BROWSER_SWEEP = '0' + runProcessMock.mockClear() + runProcessMock.mockResolvedValue({ code: 0, stdout: '{"data":{"sessions":[]}}', stderr: '' }) + + await sweepOrphanedAgentBrowserSessions({ + binaryPath: BIN, + env: {}, + ownsSocketDirectory: true + }) + + expect(runProcessMock).toHaveBeenCalled() + }) +}) diff --git a/src/main/browser/agent-browser-orphan-sweep.ts b/src/main/browser/agent-browser-orphan-sweep.ts new file mode 100644 index 00000000000..062eb1235db --- /dev/null +++ b/src/main/browser/agent-browser-orphan-sweep.ts @@ -0,0 +1,91 @@ +import { runProcess } from '../../shared/child-process/run-process' + +/** Session-name namespace Orca gives one daemon per browser tab. */ +export const ORCA_TAB_SESSION_PREFIX = 'orca-tab-' + +const SWEEP_TIMEOUT_MS = 5_000 +const SWEEP_MAX_OUTPUT_BYTES = 256 * 1024 + +type SessionListEnvelope = { + data?: { sessions?: unknown } +} + +function parseSessionNames(stdout: string): string[] { + let envelope: SessionListEnvelope + try { + envelope = JSON.parse(stdout) as SessionListEnvelope + } catch { + return [] + } + const sessions = envelope?.data?.sessions + if (!Array.isArray(sessions)) { + return [] + } + return sessions.filter((name): name is string => typeof name === 'string' && name.length > 0) +} + +/** + * Close agent-browser daemons left behind by a previous Orca run. + * + * A crash (or SIGKILL) leaves one daemon per open tab with nobody holding its + * name; `closeStaleAgentBrowserSession` only resets the single name a new tab + * is about to reuse, so the rest persist. This closes them through + * agent-browser's own CLI rather than by walking pids. + * + * Scoping — this only runs when Orca derived the socket directory itself + * (`ownsSocketDirectory`), because that private per-profile directory is what + * proves the enumeration can only see this Orca profile's daemons. An inherited + * `AGENT_BROWSER_SOCKET_DIR` can be shared with a second Orca profile, and + * Windows gets none at all (named pipes make the directory moot); both cases + * skip the sweep rather than run a `session list` that could close a daemon Orca + * does not own, and stay bounded by `AGENT_BROWSER_IDLE_TIMEOUT_MS` instead. + * + * `ORCA_DISABLE_AGENT_BROWSER_SWEEP=1` turns it off in the field. The other two + * behaviours this PR adds are already recoverable without a build — the idle bound + * is an env passthrough an operator can raise, and the quit close is bounded by its + * own timeout — but a sweep that closes the wrong daemon, or spawns one process per + * stale name on a profile with hundreds, would otherwise need a revert. + */ +export async function sweepOrphanedAgentBrowserSessions(options: { + binaryPath: string + env: NodeJS.ProcessEnv + ownsSocketDirectory: boolean + isSessionLive?: (sessionName: string) => boolean +}): Promise { + if (!options.ownsSocketDirectory || process.env.ORCA_DISABLE_AGENT_BROWSER_SWEEP === '1') { + return [] + } + let listed: string[] + try { + const result = await runProcess({ + program: options.binaryPath, + args: ['session', 'list', '--json'], + env: options.env, + timeoutMs: SWEEP_TIMEOUT_MS, + maxOutputBytes: SWEEP_MAX_OUTPUT_BYTES + }) + listed = result.timedOut ? [] : parseSessionNames(result.stdout) + } catch { + return [] + } + + const closed: string[] = [] + for (const sessionName of listed) { + if (!sessionName.startsWith(ORCA_TAB_SESSION_PREFIX) || options.isSessionLive?.(sessionName)) { + continue + } + try { + await runProcess({ + program: options.binaryPath, + args: ['--session', sessionName, 'close'], + env: options.env, + timeoutMs: SWEEP_TIMEOUT_MS, + maxOutputBytes: SWEEP_MAX_OUTPUT_BYTES + }) + closed.push(sessionName) + } catch { + // A daemon that died mid-sweep needs no closing. + } + } + return closed +} diff --git a/src/main/browser/agent-browser-process-environment.test.ts b/src/main/browser/agent-browser-process-environment.test.ts index 739567e49f1..69606d1241d 100644 --- a/src/main/browser/agent-browser-process-environment.test.ts +++ b/src/main/browser/agent-browser-process-environment.test.ts @@ -1,40 +1,86 @@ import { describe, expect, it, vi } from 'vitest' -import { createAgentBrowserProcessEnvironment } from './agent-browser-process-environment' +import { + AGENT_BROWSER_IDLE_TIMEOUT_MS, + createAgentBrowserProcessEnvironment +} from './agent-browser-process-environment' vi.mock('node:fs', () => ({ mkdirSync: vi.fn(), chmodSync: vi.fn() })) describe('agent-browser process environment', () => { it('bounds Unix socket paths independently of a long profile path', () => { - const env = createAgentBrowserProcessEnvironment({ + const { env, ownsSocketDirectory } = createAgentBrowserProcessEnvironment({ inheritedEnv: { PATH: '/bin' }, platform: 'darwin', userDataPath: `/private/var/folders/${'long-profile-segment/'.repeat(12)}` }) const socketDirectory = env.AGENT_BROWSER_SOCKET_DIR + expect(ownsSocketDirectory).toBe(true) expect(socketDirectory).toMatch(/^\/tmp\/orca-ab-[0-9a-f]{16}$/) expect( `${socketDirectory}/orca-tab-00000000-0000-4000-8000-000000000000.sock`.length ).toBeLessThan(104) }) - it('preserves explicit overrides and leaves Windows unchanged', () => { - const configured = { AGENT_BROWSER_SOCKET_DIR: '/custom/socket-dir' } - expect( - createAgentBrowserProcessEnvironment({ - inheritedEnv: configured, - platform: 'linux', + // Why ownsSocketDirectory is false for both: a directory Orca did not derive can be shared with + // another Orca profile, so `session list` under it is no proof of ownership. + it('preserves explicit overrides and leaves Windows socket routing unchanged', () => { + const configured = createAgentBrowserProcessEnvironment({ + inheritedEnv: { AGENT_BROWSER_SOCKET_DIR: '/custom/socket-dir' }, + platform: 'linux', + userDataPath: '/profile' + }) + expect(configured.env.AGENT_BROWSER_SOCKET_DIR).toBe('/custom/socket-dir') + expect(configured.ownsSocketDirectory).toBe(false) + + const windows = createAgentBrowserProcessEnvironment({ + inheritedEnv: { PATH: 'C:\\Windows' }, + platform: 'win32', + userDataPath: 'C:\\Users\\Orca' + }) + expect(windows.env.AGENT_BROWSER_SOCKET_DIR).toBeUndefined() + expect(windows.env.PATH).toBe('C:\\Windows') + expect(windows.ownsSocketDirectory).toBe(false) + }) + + // Why: the only daemon bound that survives a SIGKILL'd Orca, so it must be set on every platform. + it.each(['darwin', 'linux', 'win32'])( + 'bounds daemon idle lifetime on %s', + (platform) => { + const { env } = createAgentBrowserProcessEnvironment({ + inheritedEnv: { PATH: '/bin' }, + platform, userDataPath: '/profile' }) - ).toBe(configured) + expect(env.AGENT_BROWSER_IDLE_TIMEOUT_MS).toBe(String(AGENT_BROWSER_IDLE_TIMEOUT_MS)) + } + ) - const windows = { PATH: 'C:\\Windows' } - expect( - createAgentBrowserProcessEnvironment({ - inheritedEnv: windows, - platform: 'win32', - userDataPath: 'C:\\Users\\Orca' - }) - ).toBe(windows) + it('never cuts a command short: idle timeout exceeds the bridge exec timeout', () => { + expect(AGENT_BROWSER_IDLE_TIMEOUT_MS).toBeGreaterThan(90_000) + }) + + it('honors an explicit idle timeout from the environment', () => { + const { env } = createAgentBrowserProcessEnvironment({ + inheritedEnv: { AGENT_BROWSER_IDLE_TIMEOUT_MS: '5000' }, + platform: 'darwin', + userDataPath: '/profile' + }) + expect(env.AGENT_BROWSER_IDLE_TIMEOUT_MS).toBe('5000') + }) + + it('still bounds the daemon when the socket directory cannot be created', async () => { + const fs = await import('node:fs') + vi.mocked(fs.mkdirSync).mockImplementationOnce(() => { + throw new Error('EACCES') + }) + const { env, ownsSocketDirectory } = createAgentBrowserProcessEnvironment({ + inheritedEnv: { PATH: '/bin' }, + platform: 'linux', + userDataPath: '/profile' + }) + expect(env.AGENT_BROWSER_SOCKET_DIR).toBeUndefined() + expect(ownsSocketDirectory).toBe(false) + expect(env.AGENT_BROWSER_IDLE_TIMEOUT_MS).toBe(String(AGENT_BROWSER_IDLE_TIMEOUT_MS)) }) }) diff --git a/src/main/browser/agent-browser-process-environment.ts b/src/main/browser/agent-browser-process-environment.ts index 9d25236b8dd..ae21104f8d3 100644 --- a/src/main/browser/agent-browser-process-environment.ts +++ b/src/main/browser/agent-browser-process-environment.ts @@ -4,13 +4,45 @@ import { join } from 'node:path' const AGENT_BROWSER_SOCKET_DIRECTORY_PREFIX = 'orca-ab-' +/** + * Lifetime bound for the agent-browser daemon. + * + * `agent-browser` is a client/daemon CLI: Orca only ever spawns the short-lived + * client, which forks a daemon Orca holds no handle on and that reparents to + * pid 1 immediately. Nothing in Orca can reap it — not teardown, not a pid walk + * (see `windows-pty-job.ts` for why walking your own orphans is guesswork) — + * and a SIGKILL'd Orca never runs teardown at all. The daemon's own idle timer + * is the only bound that survives every way Orca can die (#16367). + * + * 10 minutes: >6x `EXEC_TIMEOUT_MS` (90s) so no command, retry chain, or normal + * gap between two user commands can be cut short by it, while capping an + * abandoned daemon at minutes instead of days. Only per-tab helper daemons get + * this: see `externalChromiumAgentBrowserEnvironment` for why the daemon that + * owns a whole Chromium tree must not be idled out. + */ +export const AGENT_BROWSER_IDLE_TIMEOUT_MS = 10 * 60 * 1000 + +export type AgentBrowserProcessEnvironment = { + env: NodeJS.ProcessEnv + /** + * True only when Orca derived the socket directory itself. An inherited + * `AGENT_BROWSER_SOCKET_DIR` can be shared with another Orca profile, so it is + * no proof that `session list` under it sees only this profile's daemons. + */ + ownsSocketDirectory: boolean +} + export function createAgentBrowserProcessEnvironment(options: { inheritedEnv: NodeJS.ProcessEnv platform: NodeJS.Platform userDataPath: string -}): NodeJS.ProcessEnv { - if (options.platform === 'win32' || options.inheritedEnv.AGENT_BROWSER_SOCKET_DIR?.trim()) { - return options.inheritedEnv +}): AgentBrowserProcessEnvironment { + const env = { ...options.inheritedEnv } + if (!env.AGENT_BROWSER_IDLE_TIMEOUT_MS?.trim()) { + env.AGENT_BROWSER_IDLE_TIMEOUT_MS = String(AGENT_BROWSER_IDLE_TIMEOUT_MS) + } + if (options.platform === 'win32' || env.AGENT_BROWSER_SOCKET_DIR?.trim()) { + return { env, ownsSocketDirectory: false } } const profileKey = createHash('sha256').update(options.userDataPath).digest('hex').slice(0, 16) const socketDirectory = join('/tmp', `${AGENT_BROWSER_SOCKET_DIRECTORY_PREFIX}${profileKey}`) @@ -18,7 +50,8 @@ export function createAgentBrowserProcessEnvironment(options: { mkdirSync(socketDirectory, { recursive: true, mode: 0o700 }) chmodSync(socketDirectory, 0o700) } catch { - return options.inheritedEnv + return { env, ownsSocketDirectory: false } } - return { ...options.inheritedEnv, AGENT_BROWSER_SOCKET_DIR: socketDirectory } + env.AGENT_BROWSER_SOCKET_DIR = socketDirectory + return { env, ownsSocketDirectory: true } } diff --git a/src/main/browser/offscreen-browser-backend-lifecycle.test.ts b/src/main/browser/offscreen-browser-backend-lifecycle.test.ts index 52a555c9f5f..da393fa273d 100644 --- a/src/main/browser/offscreen-browser-backend-lifecycle.test.ts +++ b/src/main/browser/offscreen-browser-backend-lifecycle.test.ts @@ -282,4 +282,19 @@ describe('OffscreenBrowserBackend lifecycle', () => { expect(peakRetirements).toBe(4) }) + + it('closes the page even when daemon retirement throws', async () => { + const browserManager = { registerOffscreenGuest: vi.fn(), unregisterGuest: vi.fn() } + const backend = new OffscreenBrowserBackend(browserManager as never, { + getAgentBrowserBridge: () => ({ + onPageClosed: vi.fn(async () => { + throw new Error('daemon gone') + }) + }) + }) + + await backend.createTab({ browserPageId: 'page-1', url: 'about:blank', worktreeId: 'wt' }) + await expect(backend.closeTab('page-1')).resolves.toBeUndefined() + expect(browserManager.unregisterGuest).toHaveBeenCalledWith('page-1') + }) }) diff --git a/src/main/index.ts b/src/main/index.ts index 249700bcd16..a142fc6cf7d 100644 --- a/src/main/index.ts +++ b/src/main/index.ts @@ -3035,6 +3035,8 @@ void app.whenReady().then(async () => { onTabsChanged: (worktreeId) => runtimeService.notifyMobileSessionTabsChanged(worktreeId) }) runtimeService.setAgentBrowserBridge(agentBrowserBridge) + // Why: daemons a crashed or SIGKILL'd previous run left behind answer to nobody; nothing else reclaims them. + void agentBrowserBridge.sweepOrphanedSessions() const browserClientAutomationDispatcher = new RpcDispatcher({ runtime: runtimeService }) configureBrowserClientPageAutomationRuntime({ browserManager, @@ -3534,7 +3536,10 @@ app.on('will-quit', (e) => { // Why: cancels relay restart/reinstall timers and kills wsl.exe children deterministically, not via stdio-pipe teardown. wslHookRelayManager.disposeAll() const statsFlush = stats?.flushAsync() ?? Promise.resolve() - // Why: retire headless page owners first, then sweep residual helper sessions without duplicate close fanout. + // Why: agent-browser daemon processes would otherwise linger after quit, holding ports and stale session state on disk. + // Why the barrier below: each session's close is its own agent-browser child taking hundreds of ms, + // so an unawaited call reaches app.quit() first and every open tab's daemon survives the quit (#16367). + // Why retire headless page owners first: it closes those helpers without a duplicate close fanout. const browserShutdown = (async (): Promise => { await runtime?.getOffscreenBrowserBackend()?.destroyAll?.() await runtime?.getAgentBrowserBridge()?.destroyAllSessions() diff --git a/src/main/orcad/external-chromium-browser-session.test.ts b/src/main/orcad/external-chromium-browser-session.test.ts new file mode 100644 index 00000000000..730a3d36e2d --- /dev/null +++ b/src/main/orcad/external-chromium-browser-session.test.ts @@ -0,0 +1,130 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const runProcessMock = vi.fn() +vi.mock('../../shared/child-process/run-process', () => ({ + runProcess: (spec: unknown) => runProcessMock(spec) +})) +vi.mock('node:fs/promises', () => ({ + mkdir: vi.fn(async () => undefined), + readFile: vi.fn(async () => Buffer.from('')), + rm: vi.fn(async () => undefined) +})) + +import { + ExternalChromiumBrowserSession, + externalChromiumAgentBrowserEnvironment +} from './external-chromium-browser-session' + +const BASE = { + executablePath: '/opt/orca/chromium', + profilePath: '/state/browser-chromium', + sessionName: 'orca-orcad-0123456789abcdef' +} + +type Spec = { args?: readonly string[]; env?: NodeJS.ProcessEnv } + +function commands(): string[][] { + return runProcessMock.mock.calls.map((call) => [...((call[0] as Spec).args ?? [])]) +} + +describe('orcad external-chromium agent-browser environment', () => { + beforeEach(() => { + runProcessMock.mockReset() + }) + + // Why: this daemon owns the user's remote Chromium, so an idle bound would close a live browser. + it('never bounds the daemon that owns the Chromium tree', () => { + const env = externalChromiumAgentBrowserEnvironment({ inheritedEnv: {}, ...BASE }) + + expect(env.AGENT_BROWSER_IDLE_TIMEOUT_MS).toBeUndefined() + expect(env.AGENT_BROWSER_EXECUTABLE_PATH).toBe(BASE.executablePath) + expect(env.AGENT_BROWSER_SESSION).toBe(BASE.sessionName) + }) + + it('passes an operator-set idle timeout through untouched', () => { + const env = externalChromiumAgentBrowserEnvironment({ + inheritedEnv: { AGENT_BROWSER_IDLE_TIMEOUT_MS: '1234' }, + ...BASE + }) + + expect(env.AGENT_BROWSER_IDLE_TIMEOUT_MS).toBe('1234') + }) + + it('keeps launch arguments joined for the daemon', () => { + const env = externalChromiumAgentBrowserEnvironment({ + inheritedEnv: {}, + ...BASE, + browserArgs: ['--headless=new', '--no-sandbox'] + }) + + expect(env.AGENT_BROWSER_ARGS).toBe('--headless=new\n--no-sandbox') + }) + + // Why (#16367): orcad is the backend a remote user's Chromium hangs off, and this session + // never passes --cdp, so a daemon an earlier orcad left behind is not bound to the dead + // process. Closing it on every start would take their browser and every tab with it. + it("reuses a surviving session instead of closing the user's browser", async () => { + runProcessMock.mockImplementation((spec: Spec) => { + const data = spec.args?.includes('tab') + ? { tabs: [{ active: true, tabId: 'tab-live', title: 'x', url: 'https://example.test' }] } + : {} + return Promise.resolve({ + code: 0, + signal: null, + stdout: JSON.stringify({ success: true, data }), + stderr: '', + timedOut: false + }) + }) + + const session = new ExternalChromiumBrowserSession( + '/opt/orca/agent-browser', + { executablePath: BASE.executablePath, provider: 'chromium' }, + '/state' + ) + await expect(session.start()).resolves.toBe('tab-live') + + const issued = commands().map((args) => args.filter((arg) => !arg.startsWith('-'))) + expect(issued.some((args) => args.includes('close'))).toBe(false) + expect(issued.some((args) => args.includes('open'))).toBe(false) + }) + + // Why: a name that answers nothing is wedged or half-dead, so reclaiming it is correct — + // that is the killed-orcad case the stable session name exists to recover. + it('reclaims a session that answers nothing, then opens', async () => { + let listed = 0 + runProcessMock.mockImplementation((spec: Spec) => { + if (spec.args?.includes('tab')) { + listed += 1 + // First probe finds nothing; after `open` the page exists. + const tabs = + listed === 1 ? [] : [{ active: true, tabId: 'tab-1', title: 'x', url: 'about:blank' }] + return Promise.resolve({ + code: 0, + signal: null, + stdout: JSON.stringify({ success: true, data: { tabs } }), + stderr: '', + timedOut: false + }) + } + return Promise.resolve({ + code: 0, + signal: null, + stdout: JSON.stringify({ success: true, data: {} }), + stderr: '', + timedOut: false + }) + }) + + const session = new ExternalChromiumBrowserSession( + '/opt/orca/agent-browser', + { executablePath: BASE.executablePath, provider: 'chromium' }, + '/state' + ) + await expect(session.start()).resolves.toBe('tab-1') + + const issued = commands().map((args) => args.filter((arg) => !arg.startsWith('-'))) + expect(issued.some((args) => args.includes('close'))).toBe(true) + expect(issued.some((args) => args.includes('open'))).toBe(true) + }) +}) diff --git a/src/main/orcad/external-chromium-browser-session.ts b/src/main/orcad/external-chromium-browser-session.ts index a8aadac3ccd..de72ae8a061 100644 --- a/src/main/orcad/external-chromium-browser-session.ts +++ b/src/main/orcad/external-chromium-browser-session.ts @@ -51,6 +51,25 @@ function classifyAgentBrowserError(message: string): string { return 'browser_error' } +// Why no AGENT_BROWSER_IDLE_TIMEOUT_MS here: unlike a per-tab helper daemon, this one owns the +// user's remote Chromium tree, so idling it out would close their live browser and every tab in it. +// The stable session name plus the `close` in start() is what reclaims a killed orcad's tree (#16367). +export function externalChromiumAgentBrowserEnvironment(options: { + inheritedEnv: NodeJS.ProcessEnv + executablePath: string + profilePath: string + sessionName: string + browserArgs?: readonly string[] +}): NodeJS.ProcessEnv { + return { + ...options.inheritedEnv, + AGENT_BROWSER_EXECUTABLE_PATH: options.executablePath, + AGENT_BROWSER_PROFILE: options.profilePath, + AGENT_BROWSER_SESSION: options.sessionName, + AGENT_BROWSER_ARGS: options.browserArgs?.join('\n') ?? '' + } +} + export class ExternalChromiumBrowserSession { private readonly profilePath: string private readonly sessionName: string @@ -70,16 +89,34 @@ export class ExternalChromiumBrowserSession { async start(): Promise { await mkdir(this.profilePath, { recursive: true }) + // Why: the session name is stable across runs, so a daemon an earlier orcad left behind is + // still driving the user's Chromium. Unlike the pane bridge this session never passes --cdp, + // so nothing binds it to the old process — a surviving one is reusable as-is, and closing it + // would take the remote user's browser and every tab with it (#16367). + const reusable = await this.readActiveTabId() + if (reusable) { + return reusable + } + // Nothing answered, so anything under this name is wedged or half-dead; reclaim it. + await this.stop() await this.run(['open', 'about:blank']) - const tabs = await this.readTabs() - const active = tabs.find((tab) => tab.active) ?? tabs[0] - if (!active) { + const opened = await this.readActiveTabId() + if (!opened) { throw new BrowserError( BROWSER_UNAVAILABLE_ERROR_CODE, 'The browser launched without an automation target.' ) } - return active.tabId + return opened + } + + private async readActiveTabId(): Promise { + try { + const tabs = await this.readTabs() + return (tabs.find((tab) => tab.active) ?? tabs[0])?.tabId ?? null + } catch { + return null + } } async stop(): Promise { @@ -121,13 +158,13 @@ export class ExternalChromiumBrowserSession { args.push('--args', this.launch.browserArgs.join('\n')) } args.push(...command, '--json') - const env = { - ...process.env, - AGENT_BROWSER_EXECUTABLE_PATH: this.launch.executablePath, - AGENT_BROWSER_PROFILE: this.profilePath, - AGENT_BROWSER_SESSION: this.sessionName, - AGENT_BROWSER_ARGS: this.launch.browserArgs?.join('\n') ?? '' - } + const env = externalChromiumAgentBrowserEnvironment({ + inheritedEnv: process.env, + executablePath: this.launch.executablePath, + profilePath: this.profilePath, + sessionName: this.sessionName, + browserArgs: this.launch.browserArgs + }) const result = await runProcess({ program: this.agentBrowserPath, args, diff --git a/src/main/quit-teardown-agent-browser-daemons.test.ts b/src/main/quit-teardown-agent-browser-daemons.test.ts new file mode 100644 index 00000000000..cc2c7d7a26e --- /dev/null +++ b/src/main/quit-teardown-agent-browser-daemons.test.ts @@ -0,0 +1,33 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * agent-browser forks a daemon per browser tab that Orca holds no handle on, and + * `destroyAllSessions` closes each one by spawning another agent-browser child — + * hundreds of ms apiece. Left off the will-quit barrier, `app.quit()` fired first and + * every open tab's daemon outlived the app (#16367). + */ +const source = readFileSync(join(__dirname, 'index.ts'), 'utf8') + +function teardownBarrierMembers(): string { + const start = source.indexOf('settleTeardownWithinDeadline([') + expect(start).toBeGreaterThanOrEqual(0) + const end = source.indexOf('])', start) + expect(end).toBeGreaterThan(start) + return source.slice(start, end) +} + +describe('quit teardown of agent-browser daemons', () => { + it('joins the will-quit teardown barrier', () => { + expect(teardownBarrierMembers()).toContain("{ name: 'browser', promise: browserShutdown }") + }) + + it('captures the destroyAllSessions promise instead of firing and forgetting', () => { + expect(source).toMatch( + /const browserShutdown = \(async \(\): Promise => \{[\s\S]*?await runtime\?\.getAgentBrowserBridge\(\)\?\.destroyAllSessions\(\)\s+\}\)\(\)/ + ) + // Why: a second, uncaptured call site is the pre-fix shape — it loses the race to app.quit(). + expect(source.match(/getAgentBrowserBridge\(\)\?\.destroyAllSessions\(\)/g)).toHaveLength(1) + }) +}) From 07f2e14c0801fa3b9b1f341701c383b61d9b442d Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 26 Aug 2026 17:00:21 -0700 Subject: [PATCH 19/19] refactor(codex): make WSL account surfaces direct-home aware (#16499) * refactor(codex): make WSL account surfaces direct-home aware * fix(wsl): keep Codex relay hooks on managed runtime home * test(wsl): assert relay hooks use managed Codex home --- .../remote-hook-service-installers.test.ts | 38 +++++++++---------- .../remote-managed-hook-installers.ts | 18 ++++----- src/main/agent-hooks/wsl-hook-fs-adapter.ts | 9 +++-- .../wsl-hook-relay-live.integration.test.ts | 17 +++------ .../wsl-hook-relay-manager.test.ts | 4 +- .../runtime-home-per-account-homes.test.ts | 29 ++++++++++++++ .../codex-accounts/runtime-home-service.ts | 22 ++++++----- .../codex/codex-legacy-session-resume.test.ts | 21 ++++++++++ src/main/codex/codex-legacy-session-resume.ts | 2 + src/main/codex/hook-service.ts | 7 +--- .../host-readable-transcript-path.test.ts | 12 ++++++ .../host-readable-transcript-path.ts | 16 +++++++- src/main/runtime/orca-runtime.ts | 4 ++ 13 files changed, 134 insertions(+), 65 deletions(-) diff --git a/src/main/agent-hooks/remote-hook-service-installers.test.ts b/src/main/agent-hooks/remote-hook-service-installers.test.ts index 28d11c73f0f..051b8bc48d1 100644 --- a/src/main/agent-hooks/remote-hook-service-installers.test.ts +++ b/src/main/agent-hooks/remote-hook-service-installers.test.ts @@ -241,7 +241,20 @@ describe('remote hook service installers', () => { expect(toml).toContain('trusted_hash = "sha256:') }) - it('installs Codex hooks into an explicit redirected CODEX_HOME (WSL managed runtime home)', async () => { + it('reports Codex trust-write failures without rolling back installed hooks', async () => { + const { sftp, fs } = createFakeSftp() + fs.failRenameTo.add('/home/dev/.codex/config.toml') + + const status = await new CodexHookService().installRemote(sftp, '/home/dev') + + expect(status.state).toBe('error') + expect(status.managedHooksPresent).toBe(true) + expect(status.detail).toContain('trust entries could not be written') + expect(fs.files.get('/home/dev/.codex/hooks.json')).toContain('codex-hook.sh') + expect(fs.files.get('/home/dev/.orca/agent-hooks/codex-hook.sh')).toContain('#!/bin/sh') + }) + + it('installs Codex hooks into an explicit redirected CODEX_HOME', async () => { const runtimeHome = '/home/dev/.local/share/orca/codex-runtime-home/home' const { sftp, fs } = createFakeSftp({ [`${runtimeHome}/config.toml`]: 'model = "gpt-5.2-codex"\n' @@ -261,12 +274,12 @@ describe('remote hook service installers', () => { expect(hooks.hooks.Stop?.[0]?.hooks?.[0]?.command).toContain( '/home/dev/.orca/agent-hooks/codex-hook.sh' ) - const toml = fs.files.get(`${runtimeHome}/config.toml`) - expect(toml).toContain('model = "gpt-5.2-codex"') - expect(toml).toContain(`${runtimeHome}/hooks.json:stop:0:0`) + expect(fs.files.get(`${runtimeHome}/config.toml`)).toContain( + `${runtimeHome}/hooks.json:stop:0:0` + ) }) - it('defers Codex trust writes until the redirected config.toml exists (launch-path seed race)', async () => { + it('defers redirected Codex trust writes until config.toml exists', async () => { const runtimeHome = '/home/dev/.local/share/orca/codex-runtime-home/home' const { sftp, fs } = createFakeSftp() @@ -278,24 +291,9 @@ describe('remote hook service installers', () => { expect(status.state).toBe('installed') expect(status.detail).toContain('deferred') expect(fs.files.get(`${runtimeHome}/hooks.json`)).toContain('codex-hook.sh') - // Why: creating config.toml here would make the launch path's - // only-if-absent seed skip the user's real config. expect(fs.files.has(`${runtimeHome}/config.toml`)).toBe(false) }) - it('reports Codex trust-write failures without rolling back installed hooks', async () => { - const { sftp, fs } = createFakeSftp() - fs.failRenameTo.add('/home/dev/.codex/config.toml') - - const status = await new CodexHookService().installRemote(sftp, '/home/dev') - - expect(status.state).toBe('error') - expect(status.managedHooksPresent).toBe(true) - expect(status.detail).toContain('trust entries could not be written') - expect(fs.files.get('/home/dev/.codex/hooks.json')).toContain('codex-hook.sh') - expect(fs.files.get('/home/dev/.orca/agent-hooks/codex-hook.sh')).toContain('#!/bin/sh') - }) - it('installs remote Gemini, Antigravity, Cursor, Command Code, Grok, and Devin configs using their CLI-specific schemas', async () => { const gemini = createFakeSftp() const antigravity = createFakeSftp() diff --git a/src/main/agent-hooks/remote-managed-hook-installers.ts b/src/main/agent-hooks/remote-managed-hook-installers.ts index 6471ec5cfb4..bf97436f8f8 100644 --- a/src/main/agent-hooks/remote-managed-hook-installers.ts +++ b/src/main/agent-hooks/remote-managed-hook-installers.ts @@ -16,11 +16,10 @@ import { kimiHookService } from '../kimi/hook-service' import { openClaudeHookService } from '../openclaude/hook-service' export type RemoteManagedHookInstallOptions = { - /** Explicit CODEX_HOME dir for redirected runtimes (WSL managed runtime - * home). Codex-only: it is the one agent whose home Orca redirects. Also - * defers the config.toml trust write until that file exists, so the - * launch path's only-if-absent seed is never pre-empted. */ + /** Explicit CODEX_HOME dir for redirected runtimes (for example WSL's managed runtime home). */ codexHomeDir?: string + /** Skip the trust write when a redirected runtime config is seeded by the launch path. */ + deferTrustUntilConfigToml?: boolean /** Explicit GROK_HOME for remote runtimes that redirect Grok's config. */ grokHomeDir?: string /** Stops before starting the next installer when the owning relay request @@ -46,13 +45,10 @@ const REMOTE_MANAGED_HOOK_INSTALLERS: readonly RemoteManagedHookInstaller[] = [ [ 'codex', (sftp, remoteHome, options) => - codexHookService.installRemote( - sftp, - remoteHome, - options?.codexHomeDir - ? { codexHomeDir: options.codexHomeDir, deferTrustUntilConfigToml: true } - : undefined - ) + codexHookService.installRemote(sftp, remoteHome, { + codexHomeDir: options?.codexHomeDir, + deferTrustUntilConfigToml: options?.deferTrustUntilConfigToml + }) ], ['gemini', (sftp, remoteHome) => geminiHookService.installRemote(sftp, remoteHome)], ['antigravity', (sftp, remoteHome) => antigravityHookService.installRemote(sftp, remoteHome)], diff --git a/src/main/agent-hooks/wsl-hook-fs-adapter.ts b/src/main/agent-hooks/wsl-hook-fs-adapter.ts index 9578fda1d10..12563b65bd8 100644 --- a/src/main/agent-hooks/wsl-hook-fs-adapter.ts +++ b/src/main/agent-hooks/wsl-hook-fs-adapter.ts @@ -15,9 +15,7 @@ import type { SshChannelMultiplexer } from '../ssh/ssh-channel-multiplexer' import { wslCodexRuntimeHomeForGuestHome } from '../pty/codex-home-wsl-env' import { WSL_HOOK_FS_METHODS, type WslFsResult } from '../../shared/wsl-hook-relay-contract' -/** Run the shared remote hook installers against a WSL guest over the relay's - * fs bridge. Codex is the one agent whose home Orca redirects for WSL - * sessions, so its hooks go to the managed runtime home. */ +/** Run the shared remote hook installers against a WSL guest over the relay's fs bridge. */ export async function installWslGuestHooks(options: { mux: SshChannelMultiplexer guestHome: string @@ -45,8 +43,11 @@ export async function installWslGuestHooks(options: { return } const results = await installHooks(createWslHookSftpAdapter(mux), guestHome, { + agents, + // WSL Codex launches use Orca's managed runtime CODEX_HOME, not ~/.codex. + // Keep relay hooks in that active home so status callbacks are received. codexHomeDir: wslCodexRuntimeHomeForGuestHome(guestHome), - agents + deferTrustUntilConfigToml: true }) const failed = results.filter((r) => r.state === 'error').length if (failed > 0) { diff --git a/src/main/agent-hooks/wsl-hook-relay-live.integration.test.ts b/src/main/agent-hooks/wsl-hook-relay-live.integration.test.ts index ab6d6a03e45..ea6856e9862 100644 --- a/src/main/agent-hooks/wsl-hook-relay-live.integration.test.ts +++ b/src/main/agent-hooks/wsl-hook-relay-live.integration.test.ts @@ -13,6 +13,7 @@ import { afterEach, beforeAll, describe, expect, it, vi } from 'vitest' import { AgentHookServer } from './server' import { WslHookRelayManager } from './wsl-hook-relay-manager' +import { wslCodexRuntimeHomeForGuestHome } from '../pty/codex-home-wsl-env' const BUNDLE_DIR = join(process.cwd(), 'out', 'relay', 'wsl') const BUNDLE_JS = join(BUNDLE_DIR, 'wsl-agent-hook-relay.js') @@ -113,18 +114,10 @@ describe.skipIf(process.platform === 'win32')( manager.ensureForDistro('LiveDistro') - // Codex hooks land in the redirected managed runtime home. Waiting on - // this artifact (not Claude's, which is written first) keeps the + // Waiting on Codex's artifact (not Claude's, which is written first) keeps the // assertions behind the still-running 14-agent installer loop. - const codexRuntimeHome = join( - fakeHome, - '.local', - 'share', - 'orca', - 'codex-runtime-home', - 'home' - ) - await vi.waitFor(() => expect(existsSync(join(codexRuntimeHome, 'hooks.json'))).toBe(true), { + const codexHome = wslCodexRuntimeHomeForGuestHome(fakeHome) + await vi.waitFor(() => expect(existsSync(join(codexHome, 'hooks.json'))).toBe(true), { timeout: 15_000 }) expect(existsSync(join(fakeHome, '.claude', 'settings.json'))).toBe(true) @@ -135,7 +128,7 @@ describe.skipIf(process.platform === 'win32')( expect(claudeScript).toContain('/hook/claude') // Trust TOML is deferred so the launch-path seed is never pre-empted. - expect(existsSync(join(codexRuntimeHome, 'config.toml'))).toBe(false) + expect(existsSync(join(codexHome, 'config.toml'))).toBe(false) expect(existsSync(join(fakeHome, '.codex', 'hooks.json'))).toBe(false) // Re-coordinate exactly like a hook script: read the relay-written diff --git a/src/main/agent-hooks/wsl-hook-relay-manager.test.ts b/src/main/agent-hooks/wsl-hook-relay-manager.test.ts index e6a3a727fb4..797b461d133 100644 --- a/src/main/agent-hooks/wsl-hook-relay-manager.test.ts +++ b/src/main/agent-hooks/wsl-hook-relay-manager.test.ts @@ -244,10 +244,10 @@ describe('WslHookRelayManager', () => { manager.ensureForDistro('Ubuntu') await vi.waitFor(() => expect(deps.installHooks).toHaveBeenCalledTimes(1)) expect(deps.spawnRelay).toHaveBeenCalledTimes(1) - // Codex is the one agent whose home Orca redirects for WSL sessions. expect(deps.installHooks).toHaveBeenCalledWith(expect.anything(), home, { + agents: ['codex'], codexHomeDir: `${home}/.local/share/orca/codex-runtime-home/home`, - agents: ['codex'] + deferTrustUntilConfigToml: true }) expect(manager.getGuestEndpointFilePath('Ubuntu')).toBe( diff --git a/src/main/codex-accounts/runtime-home-per-account-homes.test.ts b/src/main/codex-accounts/runtime-home-per-account-homes.test.ts index fd4823acfbc..7877482f355 100644 --- a/src/main/codex-accounts/runtime-home-per-account-homes.test.ts +++ b/src/main/codex-accounts/runtime-home-per-account-homes.test.ts @@ -305,6 +305,35 @@ describe('CodexRuntimeHomeService', () => { expect(discovery).toContain(home1) }) + it('includes WSL account homes in session discovery', async () => { + const wslHome = + '\\\\wsl.localhost\\Ubuntu\\home\\me\\.local\\share\\orca\\codex-accounts\\account-1\\home' + const store = createStore( + createSettings({ + codexManagedAccounts: [ + { + id: 'account-1', + email: 'wsl@example.com', + managedHomePath: wslHome, + managedHomeRuntime: 'wsl', + wslDistro: 'Ubuntu', + wslLinuxHomePath: '/home/me/.local/share/orca/codex-accounts/account-1/home', + providerAccountId: null, + workspaceLabel: null, + workspaceAccountId: null, + createdAt: 1, + updatedAt: 1, + lastAuthenticatedAt: 1 + } + ] + }) + ) + const { CodexRuntimeHomeService } = await import('./runtime-home-service') + const service = new CodexRuntimeHomeService(store as never) + + expect(service.getHostCodexHomePathsForSessionDiscovery()).toContain(wslHome) + }) + it('surfaces per-account rollouts for session discovery on the mirror lane', async () => { // A Windows host keeps the shared system-default mirror, but its managed // accounts still launch from their own homes and accumulate rollouts there. diff --git a/src/main/codex-accounts/runtime-home-service.ts b/src/main/codex-accounts/runtime-home-service.ts index 1d6f80dbafc..cf3a0105d63 100644 --- a/src/main/codex-accounts/runtime-home-service.ts +++ b/src/main/codex-accounts/runtime-home-service.ts @@ -356,14 +356,14 @@ export class CodexRuntimeHomeService { return account } - // Why: session discovery must surface a managed account's own rollouts wherever - // they physically live. Every host managed home is a live CODEX_HOME, so scan - // them all. - private getManagedHostAccountHomesForSessionDiscovery(): string[] { + // Why: session discovery must surface every account's own rollouts wherever they live. + private getManagedAccountHomesForSessionDiscovery(): string[] { const settings = this.store.getSettings() const homes: string[] = [] for (const account of settings.codexManagedAccounts) { - if (this.getWslManagedHomePath(account)) { + const wslHome = this.getWslManagedHomePath(account) + if (wslHome) { + homes.push(wslHome) continue } const trustedHome = this.getTrustedSelfContainedManagedHomePath(account) @@ -374,6 +374,12 @@ export class CodexRuntimeHomeService { return homes } + private getManagedHostAccountHomesForSessionDiscovery(): string[] { + return this.getManagedAccountHomesForSessionDiscovery().filter( + (home) => parseWslUncPath(home) === null + ) + } + private prepareSelfContainedManagedHomeForLaunch( account: CodexManagedAccount, unavailableManagedHomePath?: string @@ -583,10 +589,8 @@ export class CodexRuntimeHomeService { // mirror, so include the real root for both directly-routed host lanes. homes.push(getSystemCodexHomePath()) } - // Why: each managed host account runs in its own self-contained home, so - // its rollouts live there rather than in the shared mirror. Scan every such - // home so account-scoped sessions still surface in the AI Vault. - for (const perAccountHome of this.getManagedHostAccountHomesForSessionDiscovery()) { + // Why: account-scoped rollouts live in each account's own home, including WSL. + for (const perAccountHome of this.getManagedAccountHomesForSessionDiscovery()) { homes.push(perAccountHome) } return homes.filter((home, index) => homes.indexOf(home) === index) diff --git a/src/main/codex/codex-legacy-session-resume.test.ts b/src/main/codex/codex-legacy-session-resume.test.ts index ee5f7a61d9a..a895653b047 100644 --- a/src/main/codex/codex-legacy-session-resume.test.ts +++ b/src/main/codex/codex-legacy-session-resume.test.ts @@ -325,6 +325,27 @@ describe('per-account resume repin', () => { expect(result).toEqual({ useRealCodexHome: false }) }) + it('does not consult the host selection while resuming a WSL account session', async () => { + const wslHome = + '\\\\wsl.localhost\\Ubuntu\\home\\me\\.local\\share\\orca\\codex-accounts\\account-1\\home' + const result = await prepareLegacySharedCodexSessionResume( + { + agent: 'codex', + filePath: `${wslHome}\\sessions\\2026\\07\\20\\rollout-session.jsonl`, + codexHome: wslHome, + executionHostId: 'local' + }, + { + ...repinOptions(), + getSelectedHostAccountCodexHomePath: () => { + throw new Error('host lane must not be consulted') + } + } + ) + + expect(result).toEqual({ useRealCodexHome: false }) + }) + it('declines a transcript outside the dated rollout layout', async () => { const straySessionPath = join(peerHome, 'sessions', 'stray.jsonl') writeFileSync(straySessionPath, '{"type":"session_meta"}\n', 'utf-8') diff --git a/src/main/codex/codex-legacy-session-resume.ts b/src/main/codex/codex-legacy-session-resume.ts index f61135b6aff..29112c38d03 100644 --- a/src/main/codex/codex-legacy-session-resume.ts +++ b/src/main/codex/codex-legacy-session-resume.ts @@ -9,6 +9,7 @@ import type { import { isPerAccountManagedCodexHome } from '../../shared/ai-vault-resume-preparation' import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' import { normalizeRuntimePathForComparison } from '../../shared/cross-platform-path' +import { parseWslUncPath } from '../../shared/wsl-paths' import { appendCodexSessionHealAuditRecord, createCodexSessionBackfillAuditWriter @@ -104,6 +105,7 @@ async function resolveSelectedAccountCodexHomeForResume( args.agent !== 'codex' || args.executionHostId !== LOCAL_EXECUTION_HOST_ID || !args.codexHome || + parseWslUncPath(args.codexHome) !== null || !isPerAccountManagedCodexHome(args.codexHome) ) { return null diff --git a/src/main/codex/hook-service.ts b/src/main/codex/hook-service.ts index 93dfaf12f41..6dc78fdabf1 100644 --- a/src/main/codex/hook-service.ts +++ b/src/main/codex/hook-service.ts @@ -1479,12 +1479,7 @@ export class CodexHookService { async installRemote( sftp: SFTPWrapper, remoteHome: string, - options?: { - /** Explicit CODEX_HOME dir (flat layout). WSL sessions read Orca's managed runtime home, not ~/.codex, so the default location leaves them hookless. */ - codexHomeDir?: string - /** Skip the trust write when config.toml is absent — the WSL launch path seeds it only-if-absent, so creating it here would cancel that seed. */ - deferTrustUntilConfigToml?: boolean - } + options?: { codexHomeDir?: string; deferTrustUntilConfigToml?: boolean } ): Promise { const codexHomeBase = options?.codexHomeDir?.replace(/\/$/, '') ?? `${remoteHome.replace(/\/$/, '')}/.codex` diff --git a/src/main/native-chat/host-readable-transcript-path.test.ts b/src/main/native-chat/host-readable-transcript-path.test.ts index 4d19b3fa8e8..cc655dcb9c6 100644 --- a/src/main/native-chat/host-readable-transcript-path.test.ts +++ b/src/main/native-chat/host-readable-transcript-path.test.ts @@ -1,6 +1,7 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' import { + configureHostReadableTranscriptPathSources, isGuestAbsoluteLinuxPath, needsWslHostTranslation, resetHostReadableTranscriptPathCacheForTests, @@ -216,4 +217,15 @@ describe('wslCodexSessionsDirs', () => { `${UBUNTU_HOME}\\.codex\\sessions` ]) }) + + it('includes WSL managed-account session roots supplied by the runtime', async () => { + const accountHome = `${UBUNTU_HOME}\\.local\\share\\orca\\codex-accounts\\account-1\\home` + configureHostReadableTranscriptPathSources({ + getAdditionalCodexHomePaths: () => [accountHome, '/host/account/home'] + }) + + await expect( + wslCodexSessionsDirs({ platform: 'win32', listWslHomeDirs: async () => [UBUNTU_HOME] }) + ).resolves.toContain(`${accountHome}\\sessions`) + }) }) diff --git a/src/main/native-chat/host-readable-transcript-path.ts b/src/main/native-chat/host-readable-transcript-path.ts index 9337281cf7f..8650477e430 100644 --- a/src/main/native-chat/host-readable-transcript-path.ts +++ b/src/main/native-chat/host-readable-transcript-path.ts @@ -72,6 +72,13 @@ const WSL_HOME_DIRS_TTL_MS = 5 * 60_000 let cachedWslHomeDirs: string[] | null = null let cachedWslHomeDirsExpiresAt = 0 let inflightWslHomeDirs: Promise | null = null +let getAdditionalCodexHomePaths: (() => readonly string[]) | undefined + +export function configureHostReadableTranscriptPathSources(options: { + getAdditionalCodexHomePaths?: () => readonly string[] +}): void { + getAdditionalCodexHomePaths = options.getAdditionalCodexHomePaths +} async function defaultListWslHomeDirs(): Promise { const homes = await Promise.all( @@ -103,6 +110,7 @@ export function resetHostReadableTranscriptPathCacheForTests(): void { cachedWslHomeDirs = null cachedWslHomeDirsExpiresAt = 0 inflightWslHomeDirs = null + getAdditionalCodexHomePaths = undefined } /** @@ -189,10 +197,16 @@ export async function wslCodexSessionsDirs( return [] } const homeDirs = await wslHomeDirs(deps.listWslHomeDirs ?? defaultListWslHomeDirs) - return homeDirs.flatMap((home) => [ + const dirs = homeDirs.flatMap((home) => [ joinUnderWslHome(home, ...WSL_CODEX_RUNTIME_HOME_SEGMENTS, 'sessions'), joinUnderWslHome(home, '.codex', 'sessions') ]) + for (const home of getAdditionalCodexHomePaths?.() ?? []) { + if (parseWslUncPath(home)) { + dirs.push(joinUnderWslHome(home, 'sessions')) + } + } + return dirs.filter((dir, index) => dirs.indexOf(dir) === index) } // Why: node:path.join is posix-flavoured off Windows and would mangle the diff --git a/src/main/runtime/orca-runtime.ts b/src/main/runtime/orca-runtime.ts index 32373ee98cc..593a5f1db41 100644 --- a/src/main/runtime/orca-runtime.ts +++ b/src/main/runtime/orca-runtime.ts @@ -662,6 +662,7 @@ import { configureAiVaultSessionSources, listAiVaultSessions } from '../ai-vault/cached-session-list' +import { configureHostReadableTranscriptPathSources } from '../native-chat/host-readable-transcript-path' import { resolveLocalAiVaultSessionTitles } from '../ai-vault/session-title-resolver' import type { AiVaultListArgs, AiVaultListResult } from '../../shared/ai-vault-types' import type { @@ -3900,6 +3901,9 @@ export class OrcaRuntimeService { configureAiVaultSessionSources({ getAdditionalCodexHomePaths: deps.getAdditionalAiVaultCodexHomePaths }) + configureHostReadableTranscriptPathSources({ + getAdditionalCodexHomePaths: deps.getAdditionalAiVaultCodexHomePaths + }) } // Why: the daemon adapter is installed via `setLocalPtyProvider()` during // attachMainWindowServices, AFTER this service is constructed. Capturing